talon-agent 5.0.0 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. package/package.json +2 -2
  2. package/src/app.ts +3 -3
  3. package/src/backend/claude-sdk/factory.ts +21 -41
  4. package/src/backend/claude-sdk/handler.ts +191 -69
  5. package/src/backend/claude-sdk/host/in-process.ts +147 -0
  6. package/src/backend/claude-sdk/one-shot.ts +17 -2
  7. package/src/backend/claude-sdk/stream.ts +9 -0
  8. package/src/backend/codex/mcp-config.ts +1 -1
  9. package/src/backend/codex/one-shot.ts +18 -4
  10. package/src/backend/openai-agents/mcp-pool.ts +1 -1
  11. package/src/backend/remote-server/one-shot.ts +16 -3
  12. package/src/backend/runtime/index.ts +2 -2
  13. package/src/backend/runtime/one-shot-hooks.ts +45 -0
  14. package/src/backend/runtime/turn/delivery.ts +1 -1
  15. package/src/backend/runtime/turn/turn-phases.ts +1 -1
  16. package/src/bootstrap.ts +2 -2
  17. package/src/cli.ts +5 -2
  18. package/src/core/agent-runtime/README.md +15 -0
  19. package/src/core/agent-runtime/agent-host.ts +388 -0
  20. package/src/core/agent-runtime/capabilities.ts +3 -0
  21. package/src/core/background/cron/scheduler.ts +1 -1
  22. package/src/core/background/triggers/command.ts +1 -1
  23. package/src/{util → core/config}/harden.ts +1 -1
  24. package/src/core/config/index.ts +1 -1
  25. package/src/core/daemon/pidfile.ts +1 -1
  26. package/src/core/daemon/resource-sampler.ts +2 -2
  27. package/src/{util → core/daemon}/respawn.ts +1 -1
  28. package/src/core/doctor/index.ts +1 -1
  29. package/src/core/engine/gateway-actions/fetch-url.ts +1 -1
  30. package/src/core/engine/gateway-actions/index.ts +1 -1
  31. package/src/core/engine/gateway-actions/memory.ts +2 -2
  32. package/src/core/engine/gateway-actions/native/exec-background.ts +129 -0
  33. package/src/core/engine/gateway-actions/native/exec-remote.ts +85 -0
  34. package/src/core/engine/gateway-actions/native/exec.ts +182 -0
  35. package/src/core/engine/gateway-actions/native/index.ts +46 -0
  36. package/src/core/engine/gateway-actions/native/params.ts +46 -0
  37. package/src/core/engine/gateway-actions/native/read.ts +182 -0
  38. package/src/core/engine/gateway-actions/native/results.ts +34 -0
  39. package/src/core/engine/gateway-actions/native/search.ts +212 -0
  40. package/src/core/engine/gateway-actions/native/shell.ts +39 -0
  41. package/src/core/engine/gateway-actions/native/teleport.ts +55 -0
  42. package/src/core/engine/gateway-actions/native/write.ts +170 -0
  43. package/src/core/frontend-runtime/builtins.ts +2 -2
  44. package/src/core/mcp-hub/children.ts +21 -5
  45. package/src/core/mcp-hub/index.ts +1 -1
  46. package/src/core/mcp-hub/talon-server.ts +1 -1
  47. package/src/core/memory/taps.ts +1 -1
  48. package/src/core/plugin/mcp.ts +1 -1
  49. package/src/core/scripts/lua.ts +1 -1
  50. package/src/core/tools/{chat.ts → chat/chat.ts} +1 -1
  51. package/src/core/tools/{cross-send.ts → chat/cross-send.ts} +1 -1
  52. package/src/core/tools/{history.ts → chat/history.ts} +2 -2
  53. package/src/core/tools/{media.ts → chat/media.ts} +1 -1
  54. package/src/core/tools/{members.ts → chat/members.ts} +2 -2
  55. package/src/core/tools/{messaging.ts → chat/messaging.ts} +2 -2
  56. package/src/core/tools/{moderation.ts → chat/moderation.ts} +2 -2
  57. package/src/core/tools/{stickers.ts → chat/stickers.ts} +2 -2
  58. package/src/core/tools/{whatsapp.ts → chat/whatsapp.ts} +1 -1
  59. package/src/core/tools/{memory.ts → content/memory.ts} +2 -2
  60. package/src/core/tools/{web.ts → content/web.ts} +1 -1
  61. package/src/core/tools/index.ts +20 -20
  62. package/src/core/tools/{admin.ts → ops/admin.ts} +1 -1
  63. package/src/core/tools/{bridge.ts → ops/bridge.ts} +2 -2
  64. package/src/core/tools/{goals.ts → ops/goals.ts} +2 -2
  65. package/src/core/tools/{mesh.ts → ops/mesh.ts} +1 -1
  66. package/src/core/tools/{models.ts → ops/models.ts} +1 -1
  67. package/src/core/tools/{native.ts → ops/native.ts} +1 -1
  68. package/src/core/tools/{scheduling.ts → ops/scheduling.ts} +1 -1
  69. package/src/core/tools/{scripts.ts → ops/scripts.ts} +1 -1
  70. package/src/core/tools/{skills.ts → ops/skills.ts} +1 -1
  71. package/src/core/tools/{triggers.ts → ops/triggers.ts} +1 -1
  72. package/src/core/types.ts +11 -0
  73. package/src/{util → core/vfs}/workspace.ts +2 -2
  74. package/src/frontend/discord/callbacks/components/effort.ts +1 -1
  75. package/src/frontend/discord/callbacks/components/index.ts +1 -1
  76. package/src/frontend/discord/callbacks/components/settings.ts +1 -1
  77. package/src/frontend/discord/commands/admin.ts +1 -1
  78. package/src/frontend/discord/commands/info.ts +1 -1
  79. package/src/frontend/discord/commands/interaction.ts +1 -1
  80. package/src/frontend/discord/commands/session.ts +3 -3
  81. package/src/frontend/discord/commands/settings.ts +1 -1
  82. package/src/frontend/discord/handlers/messages.ts +1 -1
  83. package/src/frontend/discord/handlers/state.ts +1 -1
  84. package/src/frontend/discord/middleware.ts +1 -1
  85. package/src/frontend/discord/ready.ts +1 -1
  86. package/src/frontend/discord/render.ts +41 -259
  87. package/src/frontend/native/chats/chats.ts +1 -1
  88. package/src/frontend/native/surface/models.ts +1 -1
  89. package/src/frontend/native/turn/context.ts +1 -1
  90. package/src/frontend/native/turn/emit.ts +1 -1
  91. package/src/frontend/{shared → presentation}/format.ts +4 -3
  92. package/src/frontend/{shared → presentation}/reasoning-levels.ts +12 -0
  93. package/src/frontend/presentation/reports.ts +493 -0
  94. package/src/frontend/{shared → presentation}/session-status.ts +1 -1
  95. package/src/frontend/teams/commands.ts +2 -2
  96. package/src/frontend/teams/turn.ts +1 -1
  97. package/src/frontend/telegram/admin/health.ts +1 -1
  98. package/src/frontend/telegram/admin/sessions.ts +1 -1
  99. package/src/frontend/telegram/callbacks/effort.ts +2 -2
  100. package/src/frontend/telegram/callbacks/metrics.ts +1 -1
  101. package/src/frontend/telegram/callbacks/model/views.ts +1 -1
  102. package/src/frontend/telegram/callbacks/query.ts +1 -1
  103. package/src/frontend/telegram/callbacks/settings.ts +2 -2
  104. package/src/frontend/telegram/commands/admin.ts +4 -4
  105. package/src/frontend/telegram/commands/info.ts +2 -2
  106. package/src/frontend/telegram/commands/session.ts +4 -4
  107. package/src/frontend/telegram/commands/settings.ts +4 -2
  108. package/src/frontend/telegram/handlers/state.ts +1 -1
  109. package/src/frontend/telegram/model-menu.ts +1 -1
  110. package/src/frontend/telegram/render/html.ts +30 -0
  111. package/src/frontend/telegram/{helpers → render}/menu.ts +46 -19
  112. package/src/frontend/telegram/render/reports.ts +182 -0
  113. package/src/frontend/terminal/builtins/context.ts +2 -2
  114. package/src/frontend/terminal/builtins/session.ts +1 -1
  115. package/src/frontend/terminal/builtins/status.ts +2 -2
  116. package/src/frontend/terminal/index.ts +2 -2
  117. package/src/frontend/terminal/renderer.ts +1 -1
  118. package/src/frontend/whatsapp/commands.ts +5 -5
  119. package/src/frontend/whatsapp/registry.ts +1 -1
  120. package/src/index.ts +5 -2
  121. package/src/util/runtime.ts +1 -1
  122. package/src/core/engine/gateway-actions/native.ts +0 -1035
  123. package/src/frontend/telegram/helpers/diagnostics.ts +0 -428
  124. package/src/frontend/telegram/helpers/format.ts +0 -42
  125. package/src/frontend/telegram/helpers/index.ts +0 -13
  126. package/src/util/cleanup-registry.ts +0 -36
  127. /package/src/{util → core/daemon}/boot-timer.ts +0 -0
  128. /package/src/{util → core/frontend-runtime}/chat-id.ts +0 -0
  129. /package/src/{util/mcp-launcher.ts → core/mcp-hub/launcher.ts} +0 -0
  130. /package/src/{util → core/tools/content}/web-content.ts +0 -0
  131. /package/src/core/tools/{mcp-env.ts → ops/mcp-env.ts} +0 -0
  132. /package/src/{util → core/weaver}/session-name.ts +0 -0
  133. /package/src/frontend/{shared → presentation}/access.ts +0 -0
  134. /package/src/frontend/{shared → presentation}/model-commands.ts +0 -0
  135. /package/src/frontend/{shared → presentation}/plan-usage-report.ts +0 -0
  136. /package/src/frontend/{shared → presentation}/status-context.ts +0 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.0.0",
3
+ "version": "5.1.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -99,7 +99,7 @@
99
99
  "build:fusefs": "node native/talon-fusefs/build.mjs"
100
100
  },
101
101
  "dependencies": {
102
- "@anthropic-ai/claude-agent-sdk": "^0.3.197",
102
+ "@anthropic-ai/claude-agent-sdk": "^0.3.277",
103
103
  "@anthropic-ai/sdk": "^0.104.1",
104
104
  "@brave/brave-search-mcp-server": "^2.0.75",
105
105
  "@clack/prompts": "^1.2.0",
package/src/app.ts CHANGED
@@ -7,7 +7,7 @@
7
7
  */
8
8
 
9
9
  import { getFrontends } from "./core/config/index.js";
10
- import { startUploadCleanup, stopUploadCleanup } from "./util/workspace.js";
10
+ import { startUploadCleanup, stopUploadCleanup } from "./core/vfs/workspace.js";
11
11
  import { flushDatabase } from "./storage/db.js";
12
12
  import { getActiveCount, stopAllTurns } from "./core/engine/dispatcher.js";
13
13
  import {
@@ -28,9 +28,9 @@ import {
28
28
  import { shutdownTriggers } from "./core/background/triggers/index.js";
29
29
  import { pruneSettledTriggers } from "./storage/triggers.js";
30
30
  import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
31
- import { spawnSuccessor } from "./util/respawn.js";
31
+ import { spawnSuccessor } from "./core/daemon/respawn.js";
32
32
  import { log, logError, logWarn } from "./util/log.js";
33
- import { bootPhase, bootReport } from "./util/boot-timer.js";
33
+ import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
34
34
  import {
35
35
  getVfs,
36
36
  mountNamespaceFs,
@@ -24,23 +24,12 @@ import {
24
24
  type ToolRuntime,
25
25
  type UsageTelemetry,
26
26
  } from "../../core/agent-runtime/capabilities.js";
27
- import { getPlanUsage } from "./plan-usage.js";
28
27
 
29
28
  import {
30
- initAgent as initClaudeAgent,
31
29
  updateSystemPrompt as claudeUpdateSystemPrompt,
32
- warmSession as claudeWarmSession,
33
- getActiveQuery,
34
- buildMcpServers,
35
- buildPluginMcpServers,
36
- runOneShotAgent as claudeRunOneShotAgent,
37
30
  evictOrphanSubprocesses as claudeEvictOrphanSubprocesses,
38
31
  } from "./index.js";
39
- import {
40
- runChatTurn as claudeRunChatTurn,
41
- interruptChatTurn as claudeInterruptChatTurn,
42
- } from "./handler.js";
43
- import { waitForMcpServersReady } from "./mcp-ready.js";
32
+ import { createInProcessAgentHost } from "./host/in-process.js";
44
33
 
45
34
  import * as modelProvider from "./model-provider.js";
46
35
  import { claudeDoctorChecks } from "./doctor.js";
@@ -54,16 +43,25 @@ const claudeSdkFactory: BackendFactory = {
54
43
  doctor: (config, isActive) => claudeDoctorChecks(config, isActive),
55
44
 
56
45
  async init(config, ctx) {
57
- await initClaudeAgent(config, ctx.getBridgePort);
46
+ // Everything SDK-side goes through the agent-host seam
47
+ // (docs/agent-host-sidecar.md). Phase 1 binds the in-process client,
48
+ // which calls the same functions this factory used to call inline;
49
+ // Phase 2 swaps in a process-backed client behind the same interface.
50
+ const host = createInProcessAgentHost(config, ctx.getBridgePort);
51
+ await host.hello();
58
52
  log("bot", "Backend: Claude SDK (@anthropic-ai/claude-agent-sdk)");
59
53
 
60
54
  const chat: ChatBackend = {
61
- runChatTurn: (params) => claudeRunChatTurn(params),
62
- interruptChatTurn: (chatId) => claudeInterruptChatTurn(chatId),
55
+ runChatTurn: (params) => host.runTurn(params),
56
+ interruptChatTurn: (chatId) => host.interrupt(chatId),
63
57
  };
64
58
 
65
59
  const background: BackgroundRunner = {
66
- runOneShotAgent: (p) => claudeRunOneShotAgent(p),
60
+ runOneShotAgent: (p) => host.runOneShot(p),
61
+ // Not a protocol-table row yet: subprocess eviction reaches into
62
+ // the SDK's own children, so it belongs to the host — but the
63
+ // design's message table has no request for it. Tracked as an open
64
+ // question against Phase 2; bound directly until it gains one.
67
65
  evictOrphanSubprocesses: (label) => claudeEvictOrphanSubprocesses(label),
68
66
  };
69
67
 
@@ -81,7 +79,10 @@ const claudeSdkFactory: BackendFactory = {
81
79
  getProviderModels: (p, pg, ps) =>
82
80
  modelProvider.getProviderModels(p, pg, ps),
83
81
  formatModelError: (q, r) => modelProvider.formatModelError(q, r),
84
- listModels: (f) => modelProvider.listModels(f),
82
+ // The one catalog member that asks the host: `list_models` is the
83
+ // protocol's model row. The seven above are daemon-side formatting
84
+ // over `core/models/catalog.ts`, which `hello` populates.
85
+ listModels: (f) => host.listModels(f),
85
86
  };
86
87
 
87
88
  // Claude SDK's per-turn subprocess model has no shared session
@@ -89,32 +90,11 @@ const claudeSdkFactory: BackendFactory = {
89
90
  // dispatcher's `/reset` clears Talon's stored session id via
90
91
  // `storage/sessions.ts:resetSession` regardless.
91
92
  const sessions: SessionBackend = {
92
- warmSession: (chatId) => claudeWarmSession(chatId),
93
+ warmSession: (chatId) => host.warmSession(chatId),
93
94
  };
94
95
 
95
96
  const tools: ToolRuntime = {
96
- refreshTools: async (chatId) => {
97
- const qi = getActiveQuery(chatId);
98
- if (!qi) return null;
99
- // Two-phase teardown: remove all MCP servers first so each
100
- // subprocess receives an OS-agnostic shutdown via stdio, then
101
- // install the fresh set.
102
- await qi.setMcpServers({});
103
- const freshServers = {
104
- ...buildMcpServers(chatId),
105
- ...buildPluginMcpServers(chatId),
106
- };
107
- const result = await qi.setMcpServers(freshServers);
108
- // setMcpServers resolves on REGISTER, not CONNECT — MCP startup is
109
- // non-blocking. A stdio server that dials a slow remote (e.g. the
110
- // playwright plugin connecting to the Camoufox websocket) is still
111
- // 'pending' at this point and its tools are absent from the live
112
- // registry, so the turn would proceed with mcp__playwright-tools__*
113
- // stuck "connecting" until the next refresh. Wait (bounded) for the
114
- // newly-added servers to finish connecting before returning.
115
- await waitForMcpServersReady(qi, result.added);
116
- return result;
117
- },
97
+ refreshTools: (chatId) => host.refreshTools(chatId),
118
98
  };
119
99
 
120
100
  const control: SystemControl = {
@@ -124,7 +104,7 @@ const claudeSdkFactory: BackendFactory = {
124
104
  // No per-session snapshot to offer (each turn is a fresh subprocess),
125
105
  // but the subscription's rate-limit windows are readable.
126
106
  const usage: UsageTelemetry = {
127
- getPlanUsage: () => getPlanUsage(),
107
+ getPlanUsage: () => host.planUsage(),
128
108
  };
129
109
 
130
110
  const backend = composeBackend({
@@ -156,23 +156,48 @@ function createPostResultWatchdog(
156
156
  }
157
157
 
158
158
  // ── Active query store ──────────────────────────────────────────────────────
159
- // Holds the Query reference for each in-flight chat so gateway actions
160
- // (e.g. reload_plugins) can call control methods like setMcpServers().
159
+ // Holds the in-flight turn for each chat so gateway actions (e.g.
160
+ // reload_plugins) can call control methods like setMcpServers(), and so a
161
+ // user-driven stop can mark the very turn it interrupts.
161
162
 
162
- const activeQueries = new Map<string, Query>();
163
+ type ActiveTurn = {
164
+ qi: Query;
165
+ /**
166
+ * Set by `interruptChatTurn` the moment a stop is requested. The turn's
167
+ * close-out reads it to tell a deliberate stop from a fault: whatever
168
+ * the SDK does after an interrupt is the stop.
169
+ *
170
+ * Since SDK 0.3.x that is emphatically not a clean `result`. The CLI
171
+ * emits an `is_error` result whose `errors[]` holds only an
172
+ * `[ede_diagnostic] …` line, `Query.readMessages` keeps it as
173
+ * `lastErrorResultText`, and when the underlying stream then errors the
174
+ * SDK replaces the error with `Error("Claude Code returned an error
175
+ * result: " + lastErrorResultText)`. Unmarked, that reads as a genuine
176
+ * SDK failure: an ERROR log per /stop, failed-turn accounting, and a
177
+ * fallback-model retry of the turn the user just stopped.
178
+ */
179
+ interrupted: boolean;
180
+ };
181
+
182
+ const activeQueries = new Map<string, ActiveTurn>();
163
183
 
164
184
  /**
165
185
  * Best-effort graceful interrupt of a chat's in-flight turn. Uses the SDK's
166
- * native `Query.interrupt()`, which stops the agent loop and closes the stream
167
- * with a `result` (subtype `interrupt`) — so the turn ends as a normal
168
- * completion (turn_end + usage), NOT an error, and never trips the
169
- * model-fallback retry path. No-op (returns false) when no turn is running.
186
+ * native `Query.interrupt()`, which stops the agent loop — so the turn ends
187
+ * as a normal completion (turn_end + usage), NOT an error, and never trips
188
+ * the model-fallback retry path.
189
+ *
190
+ * The turn is marked BEFORE the native interrupt is fired, exactly as the
191
+ * shared `runtime/turn/turn-interrupt.ts` marks `state.turnTerminated`
192
+ * first: the mark, not the SDK's own close-out shape, is what makes the
193
+ * contract above true. No-op (returns false) when no turn is running.
170
194
  */
171
195
  export async function interruptChatTurn(chatId: string): Promise<boolean> {
172
- const qi = activeQueries.get(chatId);
173
- if (!qi) return false;
196
+ const active = activeQueries.get(chatId);
197
+ if (!active) return false;
198
+ active.interrupted = true;
174
199
  try {
175
- await qi.interrupt();
200
+ await active.qi.interrupt();
176
201
  log("agent", `[${chatId}] turn interrupted by user`);
177
202
  incrementCounter("sdk.turn_interrupted");
178
203
  return true;
@@ -187,7 +212,7 @@ export async function interruptChatTurn(chatId: string): Promise<boolean> {
187
212
 
188
213
  /** Get the active Query for a chat, if one is in flight. */
189
214
  export function getActiveQuery(chatId: string): Query | undefined {
190
- return activeQueries.get(chatId);
215
+ return activeQueries.get(chatId)?.qi;
191
216
  }
192
217
 
193
218
  // ── Internal state passed across recursive retry calls ──────────────────────
@@ -428,12 +453,6 @@ function accountFailedClaudeTurn(
428
453
  model: string,
429
454
  durationMs: number,
430
455
  ): void {
431
- const sawResultUsage =
432
- state.sdkInputTokens +
433
- state.sdkOutputTokens +
434
- state.sdkCacheRead +
435
- state.sdkCacheWrite >
436
- 0;
437
456
  accountFailedTurn({
438
457
  backend: "claude",
439
458
  chatId,
@@ -441,7 +460,7 @@ function accountFailedClaudeTurn(
441
460
  durationMs,
442
461
  model,
443
462
  apiCalls: state.numApiCalls || live.calls,
444
- usage: sawResultUsage
463
+ usage: sawResultUsage(state)
445
464
  ? turnUsageSnapshot(state)
446
465
  : {
447
466
  inputTokens: live.input,
@@ -452,6 +471,37 @@ function accountFailedClaudeTurn(
452
471
  });
453
472
  }
454
473
 
474
+ /** Whether the turn's `result` message reported any tokens at all. */
475
+ function sawResultUsage(state: StreamState): boolean {
476
+ return (
477
+ state.sdkInputTokens +
478
+ state.sdkOutputTokens +
479
+ state.sdkCacheRead +
480
+ state.sdkCacheWrite >
481
+ 0
482
+ );
483
+ }
484
+
485
+ /**
486
+ * Close out a turn the user stopped. An interrupt is a completion, so it
487
+ * accounts like one (`accountTurn`, not `accountFailedTurn` — nothing
488
+ * failed) — but the `result` message may never have landed, leaving
489
+ * `state.sdk*` at zero while the per-API-call accumulator holds what the
490
+ * turn really burned. Fold the accumulator in so a stop doesn't lose the
491
+ * tokens, and mark the turn terminated so the flow-violation re-prompt
492
+ * can't resurrect what the user just stopped (the same guarantee
493
+ * `runtime/turn/turn-interrupt.ts` gives the callback backends).
494
+ */
495
+ function closeInterruptedTurn(state: StreamState, live: LiveUsage): void {
496
+ state.turnTerminated = true;
497
+ if (sawResultUsage(state)) return;
498
+ state.sdkInputTokens = live.input;
499
+ state.sdkOutputTokens = live.output;
500
+ state.sdkCacheRead = live.cacheRead;
501
+ state.sdkCacheWrite = live.cacheWrite;
502
+ if (!state.numApiCalls) state.numApiCalls = live.calls;
503
+ }
504
+
455
505
  /**
456
506
  * The aggregate `cache=NN%` can't distinguish a turn that reused the
457
507
  * previous turn's prefix from one that re-wrote it — see
@@ -488,6 +538,105 @@ function reportCacheVerdict(
488
538
  noteLookbackRisk(chatId, state.toolCalls);
489
539
  }
490
540
 
541
+ // ── Stream phase ────────────────────────────────────────────────────────────
542
+
543
+ /** What the stream phase left for the rest of the turn to do. */
544
+ type StreamOutcome =
545
+ /** Run the normal post-stream phases (this includes every user stop). */
546
+ | { kind: "ok" }
547
+ /** A retry already ran to completion and yielded its own events. */
548
+ | { kind: "retried" }
549
+ /** Terminal failure — account for it and yield this `error` event. */
550
+ | { kind: "failed"; event: AgentEvent };
551
+
552
+ /**
553
+ * Drive the SDK stream to exhaustion and decide what its ending means.
554
+ * Owns the error recovery (retry decision, model fallback) and the
555
+ * user-interrupt contract; releases the watchdog timer and the
556
+ * `activeQueries` entry on every exit.
557
+ */
558
+ async function* runTurnStream(inputs: {
559
+ chatId: string;
560
+ params: ChatRunParams;
561
+ internal: InternalState;
562
+ active: ActiveTurn;
563
+ state: StreamState;
564
+ live: LiveUsage;
565
+ watchdog: PostResultWatchdog;
566
+ /** Model the turn is accounted against. */
567
+ activeModel: string;
568
+ /** Model string the SDK attributes the result message's usage to. */
569
+ sdkModel: string;
570
+ }): AsyncGenerator<AgentEvent, StreamOutcome, void> {
571
+ const { chatId, active, state, live, watchdog } = inputs;
572
+ try {
573
+ yield* consumeSdkStream({
574
+ chatId,
575
+ qi: active.qi,
576
+ state,
577
+ live,
578
+ watchdog,
579
+ model: inputs.sdkModel,
580
+ pendingTools: new Map(),
581
+ });
582
+ // The SDK doesn't throw on API errors — it converts them into a
583
+ // synthetic assistant message and finishes the turn with an error-
584
+ // flagged result (usage limits, 429s, auth failures all land here).
585
+ // Rethrow so this turn takes the SAME path as a thrown SDK error
586
+ // instead of tripping the flow-violation re-prompt loop against an
587
+ // already-exhausted limit.
588
+ //
589
+ // …unless the user stopped this turn: an interrupted turn's result is
590
+ // flagged `is_error` with nothing but an `[ede_diagnostic]` line
591
+ // behind it, which `readResultError` then renders as bare trailing
592
+ // text or "Claude SDK turn failed (<subtype>)". That is the stop, not
593
+ // a failure — the turn closes as a completion.
594
+ if (state.resultErrorText && !active.interrupted) {
595
+ throw new Error(state.resultErrorText);
596
+ }
597
+ } catch (err) {
598
+ if (active.interrupted) {
599
+ // Same deal one layer down: after an interrupt the SDK re-labels the
600
+ // stream error with that error-result text. Quiet by design — no
601
+ // `logError`, no retry decision (which would classify the ede text,
602
+ // bump `errors.*`, and could fall back to another model for a turn
603
+ // the user just stopped), no `error` event. The shutdown drain takes
604
+ // this same path: its abort interrupts through here too.
605
+ log(
606
+ "agent",
607
+ `[${chatId}] turn ended by user interrupt: ${
608
+ err instanceof Error ? err.message : String(err)
609
+ }`,
610
+ );
611
+ incrementCounter("sdk.interrupt_closed_stream");
612
+ } else if (!watchdog.forceClosed) {
613
+ const { retried, classified } = yield* applyRetryDecisionStream({
614
+ err,
615
+ chatId,
616
+ activeModel: inputs.activeModel,
617
+ retried: inputs.internal.errorRetried ?? false,
618
+ buildRetryStream: retryStreamBuilder(inputs.params, inputs.internal),
619
+ // No backendLabel — historical claude-sdk log shape was un-prefixed.
620
+ });
621
+ if (retried) return { kind: "retried" };
622
+ logError("agent", `[${chatId}] SDK error: ${classified.message}`);
623
+ // Returning (rather than yielding) here defers the `error` event
624
+ // until after `finally` releases the watchdog timer and the
625
+ // activeQueries entry.
626
+ return {
627
+ kind: "failed",
628
+ event: { type: "error", error: classifiedToAgentError(classified) },
629
+ };
630
+ }
631
+ } finally {
632
+ watchdog.clear();
633
+ if (activeQueries.get(chatId) === active) {
634
+ activeQueries.delete(chatId);
635
+ }
636
+ }
637
+ return { kind: "ok" };
638
+ }
639
+
491
640
  // ── Main chat-turn generator ────────────────────────────────────────────────
492
641
 
493
642
  /**
@@ -498,7 +647,11 @@ function reportCacheVerdict(
498
647
  * recurse via `yield*` and produce the retry's event stream
499
648
  * transparently). On flow violation: `yield* runChatTurn(retry
500
649
  * params)` — the recursive call owns its `incrementTurns`, the
501
- * caller deliberately doesn't increment.
650
+ * caller deliberately doesn't increment. On a user interrupt
651
+ * (`interruptChatTurn`, which also backs the shutdown drain): the turn
652
+ * closes as a completion carrying the partial text and the real usage —
653
+ * never an `error` event, never a retry, whatever shape the SDK chose to
654
+ * end the stream in.
502
655
  */
503
656
  export async function* runChatTurn(
504
657
  params: ChatRunParams,
@@ -552,7 +705,8 @@ export async function* runChatTurn(
552
705
  yield { type: "run_started" };
553
706
 
554
707
  const qi = query({ prompt, options });
555
- activeQueries.set(chatId, qi);
708
+ const active: ActiveTurn = { qi, interrupted: false };
709
+ activeQueries.set(chatId, active);
556
710
 
557
711
  // Cold-start delivery-tool race: on the FIRST turn of a freshly-opened
558
712
  // chat the hub's `${frontend}-tools` binding can still be `pending` when
@@ -567,59 +721,27 @@ export async function* runChatTurn(
567
721
  const live = createLiveUsage();
568
722
  const watchdog = createPostResultWatchdog(chatId, abortController, qi);
569
723
 
570
- let propagateError: AgentEvent | null = null;
571
- try {
572
- yield* consumeSdkStream({
573
- chatId,
574
- qi,
575
- state,
576
- live,
577
- watchdog,
578
- model: options.model ?? activeModel,
579
- pendingTools: new Map(),
580
- });
581
- // The SDK doesn't throw on API errors — it converts them into a
582
- // synthetic assistant message and finishes the turn with an error-
583
- // flagged result (usage limits, 429s, auth failures all land here).
584
- // Rethrow so this turn takes the SAME path as a thrown SDK error
585
- // instead of tripping the flow-violation re-prompt loop against an
586
- // already-exhausted limit.
587
- if (state.resultErrorText) {
588
- throw new Error(state.resultErrorText);
589
- }
590
- } catch (err) {
591
- if (!watchdog.forceClosed) {
592
- const { retried, classified } = yield* applyRetryDecisionStream({
593
- err,
594
- chatId,
595
- activeModel,
596
- retried: _internal.errorRetried ?? false,
597
- buildRetryStream: retryStreamBuilder(params, _internal),
598
- // No backendLabel — historical claude-sdk log shape was un-prefixed.
599
- });
600
- // The recursive stream already yielded its own usage + completed.
601
- if (retried) return;
602
- logError("agent", `[${chatId}] SDK error: ${classified.message}`);
603
- // Defer the yield until after `finally` releases the watchdog timer
604
- // and the activeQueries entry.
605
- propagateError = {
606
- type: "error",
607
- error: classifiedToAgentError(classified),
608
- };
609
- }
610
- } finally {
611
- watchdog.clear();
612
- if (activeQueries.get(chatId) === qi) {
613
- activeQueries.delete(chatId);
614
- }
615
- }
616
-
617
- if (propagateError) {
724
+ const outcome = yield* runTurnStream({
725
+ chatId,
726
+ params,
727
+ internal: _internal,
728
+ active,
729
+ state,
730
+ live,
731
+ watchdog,
732
+ activeModel,
733
+ sdkModel: options.model ?? activeModel,
734
+ });
735
+ // The recursive retry stream already yielded its own usage + completed.
736
+ if (outcome.kind === "retried") return;
737
+ if (outcome.kind === "failed") {
618
738
  accountFailedClaudeTurn(chatId, state, live, activeModel, Date.now() - t0);
619
- yield propagateError;
739
+ yield outcome.event;
620
740
  return;
621
741
  }
622
742
 
743
+ if (active.interrupted) closeInterruptedTurn(state, live);
744
+
623
745
  const durationMs = Date.now() - t0;
624
746
  accountTurn({
625
747
  chatId,
@@ -0,0 +1,147 @@
1
+ /**
2
+ * The in-process agent host — `AgentHostClient` implemented as direct
3
+ * calls into `backend/claude-sdk/`.
4
+ *
5
+ * `docs/agent-host-sidecar.md` Phase 1. This is the implementation that
6
+ * keeps today's behaviour exactly: every method is a call-through to the
7
+ * function the claude-sdk `BackendFactory` used to call itself. There is
8
+ * no process, no framing and no logic of its own — the only code that
9
+ * moved here is `refreshTools`' two-phase MCP teardown, which had to,
10
+ * because it drives a `Query` handle and handles do not cross a process
11
+ * boundary.
12
+ *
13
+ * Phase 2 adds a sibling that speaks the same interface over a child
14
+ * process's stdio. Nothing above this file changes when it does.
15
+ */
16
+
17
+ import type {
18
+ AgentHostClient,
19
+ HostReadyInfo,
20
+ HostSessionInfo,
21
+ HostToolRefresh,
22
+ } from "../../../core/agent-runtime/agent-host.js";
23
+ import { AGENT_HOST_PROTOCOL_VERSION } from "../../../core/agent-runtime/agent-host.js";
24
+ import type { TalonConfig } from "../../../core/config/index.js";
25
+ import { getSession } from "../../../storage/sessions.js";
26
+ import { talonVersion } from "../../../util/version.js";
27
+
28
+ import {
29
+ initAgent as claudeInitAgent,
30
+ warmSession as claudeWarmSession,
31
+ getActiveQuery,
32
+ buildMcpServers,
33
+ buildPluginMcpServers,
34
+ runOneShotAgent as claudeRunOneShotAgent,
35
+ } from "../index.js";
36
+ import {
37
+ runChatTurn as claudeRunChatTurn,
38
+ interruptChatTurn as claudeInterruptChatTurn,
39
+ } from "../handler.js";
40
+ import { waitForMcpServersReady } from "../mcp-ready.js";
41
+ import { listModels as claudeListModels } from "../model-provider.js";
42
+ import { getPlanUsage } from "../plan-usage.js";
43
+
44
+ /**
45
+ * Install an MCP server set on the chat's live query. `null` when the
46
+ * chat has no query in flight — the same "nothing to refresh" answer
47
+ * `ToolRuntime.refreshTools` has always given.
48
+ */
49
+ async function setMcpServers(
50
+ chatId: string,
51
+ servers: Record<string, unknown>,
52
+ ): Promise<HostToolRefresh | null> {
53
+ const qi = getActiveQuery(chatId);
54
+ if (!qi) return null;
55
+ return qi.setMcpServers(servers as Parameters<typeof qi.setMcpServers>[0]);
56
+ }
57
+
58
+ /**
59
+ * Re-derive the chat's MCP config from the live plugin registry. Body
60
+ * lifted verbatim from the claude-sdk factory's `tools.refreshTools`,
61
+ * comments included — the ordering is load-bearing.
62
+ */
63
+ async function refreshTools(chatId: string): Promise<HostToolRefresh | null> {
64
+ const qi = getActiveQuery(chatId);
65
+ if (!qi) return null;
66
+ // Two-phase teardown: remove all MCP servers first so each
67
+ // subprocess receives an OS-agnostic shutdown via stdio, then
68
+ // install the fresh set.
69
+ await qi.setMcpServers({});
70
+ const freshServers = {
71
+ ...buildMcpServers(chatId),
72
+ ...buildPluginMcpServers(chatId),
73
+ };
74
+ const result = await qi.setMcpServers(freshServers);
75
+ // setMcpServers resolves on REGISTER, not CONNECT — MCP startup is
76
+ // non-blocking. A stdio server that dials a slow remote (e.g. the
77
+ // playwright plugin connecting to the Camoufox websocket) is still
78
+ // 'pending' at this point and its tools are absent from the live
79
+ // registry, so the turn would proceed with mcp__playwright-tools__*
80
+ // stuck "connecting" until the next refresh. Wait (bounded) for the
81
+ // newly-added servers to finish connecting before returning.
82
+ await waitForMcpServersReady(qi, result.added);
83
+ return result;
84
+ }
85
+
86
+ /**
87
+ * The context figures `warm_session` populates, read back out of the
88
+ * daemon's session store. In-process the host and the store share a
89
+ * process, so this is a lookup; in Phase 2 it is the reply that carries
90
+ * those numbers home.
91
+ */
92
+ async function sessionInfo(chatId: string): Promise<HostSessionInfo> {
93
+ const session = getSession(chatId);
94
+ return {
95
+ chatId,
96
+ sessionId: session.sessionId,
97
+ turns: session.turns,
98
+ contextTokens: session.usage.contextTokens,
99
+ contextWindow: session.usage.contextWindow,
100
+ };
101
+ }
102
+
103
+ /**
104
+ * The Claude SDK backend keeps no conversation memory of its own — each
105
+ * turn is a fresh subprocess — so the only per-chat state the host holds
106
+ * is the in-flight query handle. Dropping it is interrupting it.
107
+ *
108
+ * Not bound into the `Backend` object: claude-sdk has no `resetChat`
109
+ * slot today and Phase 1 does not add one. `/reset` still clears the
110
+ * stored session id through `storage/sessions.ts`, as it always has.
111
+ */
112
+ async function resetSession(chatId: string): Promise<boolean> {
113
+ return claudeInterruptChatTurn(chatId);
114
+ }
115
+
116
+ /**
117
+ * Build the in-process host. `config` and `getBridgePort` are held only
118
+ * for `hello()`, which performs the same `initAgent(config,
119
+ * getBridgePort)` the factory used to call inline.
120
+ */
121
+ export function createInProcessAgentHost(
122
+ config: TalonConfig,
123
+ getBridgePort?: () => number,
124
+ ): AgentHostClient {
125
+ return {
126
+ async hello(): Promise<HostReadyInfo> {
127
+ await claudeInitAgent(config, getBridgePort);
128
+ // `sdk` is omitted: in-process there is no separately-pinned SDK
129
+ // build to name — the daemon's own lockfile is the answer, and
130
+ // Phase 4 gives the host a `package.json` of its own to report.
131
+ return { protocol: AGENT_HOST_PROTOCOL_VERSION, host: talonVersion() };
132
+ },
133
+ runTurn: (params) => claudeRunChatTurn(params),
134
+ interrupt: (chatId) => claudeInterruptChatTurn(chatId),
135
+ runOneShot: (params) => claudeRunOneShotAgent(params),
136
+ warmSession: (chatId) => claudeWarmSession(chatId),
137
+ setMcpServers,
138
+ refreshTools,
139
+ listModels: (filter) => claudeListModels(filter),
140
+ planUsage: () => getPlanUsage(),
141
+ sessionInfo,
142
+ resetSession,
143
+ // Nothing to drain while the host is this process: the daemon's own
144
+ // shutdown path already aborts in-flight turns.
145
+ shutdown: async () => undefined,
146
+ };
147
+ }