talon-agent 3.33.4 → 3.34.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/package.json +5 -2
  2. package/src/app.ts +16 -8
  3. package/src/backend/claude-sdk/handler.ts +428 -415
  4. package/src/backend/claude-sdk/stream.ts +76 -53
  5. package/src/backend/codex/handler/message.ts +275 -366
  6. package/src/backend/codex/handler/rollout-accounting.ts +137 -0
  7. package/src/backend/openai-agents/handler/message.ts +282 -356
  8. package/src/backend/remote-server/chat-turn.ts +66 -122
  9. package/src/backend/remote-server/index.ts +4 -0
  10. package/src/backend/remote-server/mcp.ts +73 -9
  11. package/src/backend/remote-server/model-catalog/presentation.ts +269 -228
  12. package/src/backend/remote-server/sessions.ts +2 -2
  13. package/src/backend/remote-server/turn.ts +58 -55
  14. package/src/backend/shared/cache-telemetry.ts +17 -1
  15. package/src/backend/shared/handler-to-events.ts +11 -16
  16. package/src/backend/shared/index.ts +14 -15
  17. package/src/backend/shared/result-events.ts +30 -0
  18. package/src/backend/shared/turn-phases.ts +277 -0
  19. package/src/bootstrap.ts +120 -79
  20. package/src/cli/setup.ts +375 -349
  21. package/src/core/background/cron-spec.ts +273 -0
  22. package/src/core/background/heartbeat/agent.ts +167 -104
  23. package/src/core/background/triggers/exit.ts +19 -2
  24. package/src/core/background/triggers/resume.ts +14 -7
  25. package/src/core/background/triggers/state.ts +9 -0
  26. package/src/core/engine/backend-controller/pool.ts +2 -0
  27. package/src/core/engine/backend-controller/state.ts +25 -12
  28. package/src/core/engine/gateway-actions/cron.ts +28 -292
  29. package/src/core/engine/gateway-routes.ts +239 -0
  30. package/src/core/engine/gateway.ts +66 -238
  31. package/src/core/mcp-hub/child-transport.ts +215 -0
  32. package/src/core/mcp-hub/children.ts +72 -13
  33. package/src/core/mcp-hub/index.ts +32 -13
  34. package/src/core/models/active-model.ts +29 -6
  35. package/src/core/vfs/mounts/files.ts +128 -115
  36. package/src/core/weaver/shuttle.ts +33 -2
  37. package/src/core/weaver/weaver.ts +37 -6
  38. package/src/frontend/discord/callbacks/components/agent-buttons.ts +82 -0
  39. package/src/frontend/discord/callbacks/components/backend-select.ts +149 -0
  40. package/src/frontend/discord/callbacks/components/effort.ts +72 -0
  41. package/src/frontend/discord/callbacks/components/index.ts +120 -0
  42. package/src/frontend/discord/callbacks/components/metrics.ts +24 -0
  43. package/src/frontend/discord/callbacks/components/model-nav.ts +93 -0
  44. package/src/frontend/discord/callbacks/components/model-select.ts +118 -0
  45. package/src/frontend/discord/callbacks/components/model.ts +33 -0
  46. package/src/frontend/discord/callbacks/components/pulse.ts +93 -0
  47. package/src/frontend/discord/callbacks/components/settings.ts +243 -0
  48. package/src/frontend/discord/callbacks/components/types.ts +34 -0
  49. package/src/frontend/discord/callbacks/index.ts +4 -4
  50. package/src/frontend/discord/connection.ts +36 -0
  51. package/src/frontend/discord/diagnostics.ts +62 -0
  52. package/src/frontend/discord/guild-policy.ts +89 -0
  53. package/src/frontend/discord/index.ts +44 -305
  54. package/src/frontend/discord/outbound.ts +61 -0
  55. package/src/frontend/discord/ready.ts +73 -0
  56. package/src/frontend/discord/runtime.ts +55 -0
  57. package/src/frontend/native/chat-lifecycle.ts +39 -0
  58. package/src/frontend/native/chat-wire.ts +69 -0
  59. package/src/frontend/native/context.ts +106 -0
  60. package/src/frontend/native/control.ts +78 -0
  61. package/src/frontend/native/emit.ts +167 -0
  62. package/src/frontend/native/empty-chat-sweep.ts +50 -0
  63. package/src/frontend/native/handlers.ts +121 -0
  64. package/src/frontend/native/history.ts +101 -0
  65. package/src/frontend/native/index.ts +90 -1293
  66. package/src/frontend/native/media.ts +43 -0
  67. package/src/frontend/native/models.ts +221 -0
  68. package/src/frontend/native/queue.ts +47 -0
  69. package/src/frontend/native/reset.ts +50 -0
  70. package/src/frontend/native/routes/chats.ts +115 -0
  71. package/src/frontend/native/routes/daemon.ts +70 -0
  72. package/src/frontend/native/routes/host.ts +151 -0
  73. package/src/frontend/native/routes/index.ts +22 -0
  74. package/src/frontend/native/routes/mesh.ts +72 -0
  75. package/src/frontend/native/routes/models.ts +54 -0
  76. package/src/frontend/native/routes/params.ts +29 -0
  77. package/src/frontend/native/routes/pre-auth.ts +94 -0
  78. package/src/frontend/native/routes/table.ts +92 -0
  79. package/src/frontend/native/runtime.ts +109 -0
  80. package/src/frontend/native/server.ts +30 -555
  81. package/src/frontend/native/status.ts +26 -0
  82. package/src/frontend/native/tool-result.ts +48 -0
  83. package/src/frontend/native/turn.ts +341 -0
  84. package/src/frontend/telegram/admin.ts +75 -52
  85. package/src/frontend/whatsapp/access.ts +67 -0
  86. package/src/frontend/whatsapp/connection.ts +280 -0
  87. package/src/frontend/whatsapp/inbound.ts +327 -0
  88. package/src/frontend/whatsapp/index.ts +28 -599
  89. package/src/frontend/whatsapp/runtime.ts +74 -0
  90. package/src/storage/metrics.ts +18 -0
  91. package/src/storage/session-record.ts +29 -0
  92. package/src/storage/sessions.ts +24 -0
  93. package/src/storage/trigger-store.ts +8 -0
  94. package/src/util/boot-timer.ts +31 -0
  95. package/src/util/concurrency.ts +28 -0
  96. package/src/util/watchdog.ts +32 -7
  97. package/src/frontend/discord/callbacks/components.ts +0 -793
  98. package/src/frontend/discord/callbacks/shared.ts +0 -22
@@ -1,11 +1,12 @@
1
1
  /**
2
2
  * Claude SDK chat-turn handler — natively emits `AgentEvent`s.
3
3
  *
4
- * Drives the full lifecycle: prompt formatting, SDK query, native
5
- * event emission per stream message, error recovery (session expired
6
- * / context overflow / model fallback via `applyRetryDecisionStream`),
7
- * token accounting, session persistence, and the flow-violation
8
- * re-prompt loop.
4
+ * Owns the SDK-specific half of a turn: the options build, the `query()`
5
+ * subprocess, the per-message event translation, error recovery
6
+ * (session expired / context overflow / model fallback via
7
+ * `applyRetryDecisionStream`) and the post-result watchdog. The phases
8
+ * after the stream — accounting, the trailing-prose contract, the result
9
+ * events — are the shared ones in `backend/shared/turn-phases.ts`.
9
10
  *
10
11
  * The exported async generator `runChatTurn` is what the factory wires
11
12
  * onto `ChatBackend.runChatTurn` — no wrapper, no callback shim.
@@ -15,9 +16,6 @@ import { query } from "@anthropic-ai/claude-agent-sdk";
15
16
  import {
16
17
  getSession,
17
18
  incrementTurns,
18
- recordUsage,
19
- setSessionId,
20
- setSessionName,
21
19
  updateLiveTurn,
22
20
  } from "../../storage/sessions.js";
23
21
  import { log, logError, logWarn } from "../../util/log.js";
@@ -25,7 +23,11 @@ import { traceMessage } from "../../util/trace.js";
25
23
  import { incrementCounter } from "../../storage/metrics.js";
26
24
  import { isTurnTerminator } from "../../core/tools/index.js";
27
25
 
28
- import type { Query } from "@anthropic-ai/claude-agent-sdk";
26
+ import type {
27
+ Query,
28
+ SDKAssistantMessage,
29
+ SDKUserMessage,
30
+ } from "@anthropic-ai/claude-agent-sdk";
29
31
  import {
30
32
  type AgentEvent,
31
33
  classifiedToAgentError,
@@ -50,27 +52,28 @@ import {
50
52
  processStreamDelta,
51
53
  processAssistantMessage,
52
54
  processResultMessage,
55
+ type StreamState,
53
56
  } from "./stream.js";
54
57
  import {
55
58
  formatUserPrompt,
56
59
  prepareSystemPrompt,
57
- extractSessionName,
58
- detectFlowViolation,
59
- FLOW_VIOLATION_MAX_RETRIES,
60
60
  captureDeliveredText,
61
61
  summarizeUsage,
62
62
  buildDeliveryContract,
63
63
  buildFlowViolationReminder,
64
64
  buildFirstTurnReminder,
65
65
  recordToolCall,
66
- recordTurnMetrics,
67
- recordFailedTurnAccounting,
68
- recordFlowViolation,
69
66
  formatTurnCache,
70
67
  crossTurnVerdict,
71
68
  priorLookbackOverflow,
72
69
  noteLookbackRisk,
73
70
  CACHE_LOOKBACK_BLOCKS,
71
+ accountTurn,
72
+ accountFailedTurn,
73
+ nameSessionFromFirstMessage,
74
+ enforceTrailingProse,
75
+ buildResultEvents,
76
+ turnUsageSnapshot,
74
77
  } from "../shared/index.js";
75
78
 
76
79
  // ── Post-result watchdog ────────────────────────────────────────────────────
@@ -101,13 +104,60 @@ const SDK_POST_RESULT_GRACE_MS = envMs(
101
104
  DEFAULT_SDK_POST_RESULT_GRACE_MS,
102
105
  );
103
106
 
107
+ type PostResultWatchdog = {
108
+ /** Start the grace timer once `result` is processed (idempotent). */
109
+ arm(): void;
110
+ clear(): void;
111
+ /** True once the timer fired and force-closed the iterator. */
112
+ readonly forceClosed: boolean;
113
+ };
114
+
115
+ function createPostResultWatchdog(
116
+ chatId: string,
117
+ abortController: AbortController,
118
+ qi: Query,
119
+ ): PostResultWatchdog {
120
+ let timer: ReturnType<typeof setTimeout> | null = null;
121
+ let forceClosed = false;
122
+ return {
123
+ get forceClosed() {
124
+ return forceClosed;
125
+ },
126
+ arm() {
127
+ if (timer) return;
128
+ const t = setTimeout(() => {
129
+ forceClosed = true;
130
+ logWarn(
131
+ "agent",
132
+ `[${chatId}] SDK iterator stuck ${SDK_POST_RESULT_GRACE_MS}ms after result — aborting`,
133
+ );
134
+ incrementCounter("sdk.iterator_force_close_after_result");
135
+ try {
136
+ abortController.abort();
137
+ } catch {
138
+ /* abort() can throw if already aborted — ignore */
139
+ }
140
+ qi.return(undefined).catch(() => {
141
+ /* the generator may already be in a terminal state — ignore */
142
+ });
143
+ }, SDK_POST_RESULT_GRACE_MS);
144
+ t.unref();
145
+ timer = t;
146
+ },
147
+ clear() {
148
+ if (!timer) return;
149
+ clearTimeout(timer);
150
+ timer = null;
151
+ },
152
+ };
153
+ }
154
+
104
155
  // ── Active query store ──────────────────────────────────────────────────────
105
156
  // Holds the Query reference for each in-flight chat so gateway actions
106
157
  // (e.g. reload_plugins) can call control methods like setMcpServers().
107
158
 
108
159
  const activeQueries = new Map<string, Query>();
109
160
 
110
- /** Get the active Query for a chat, if one is in flight. */
111
161
  /**
112
162
  * Best-effort graceful interrupt of a chat's in-flight turn. Uses the SDK's
113
163
  * native `Query.interrupt()`, which stops the agent loop and closes the stream
@@ -132,6 +182,7 @@ export async function interruptChatTurn(chatId: string): Promise<boolean> {
132
182
  }
133
183
  }
134
184
 
185
+ /** Get the active Query for a chat, if one is in flight. */
135
186
  export function getActiveQuery(chatId: string): Query | undefined {
136
187
  return activeQueries.get(chatId);
137
188
  }
@@ -140,6 +191,282 @@ export function getActiveQuery(chatId: string): Query | undefined {
140
191
 
141
192
  type InternalState = { flowRetries?: number; errorRetried?: boolean };
142
193
 
194
+ // ── Stream translation ──────────────────────────────────────────────────────
195
+
196
+ /**
197
+ * Per-API-call usage accumulator for live mid-turn stats. Each assistant
198
+ * message carries its API call's usage as it lands; the authoritative
199
+ * per-turn totals still come from the final result message
200
+ * (`processResultMessage`) — this only feeds the live-turn overlay so
201
+ * /status moves while a long agentic turn runs, and the failure path so
202
+ * an errored turn's burn isn't lost.
203
+ */
204
+ type LiveUsage = {
205
+ input: number;
206
+ output: number;
207
+ cacheRead: number;
208
+ cacheWrite: number;
209
+ calls: number;
210
+ };
211
+
212
+ function noteAssistantUsage(
213
+ chatId: string,
214
+ state: StreamState,
215
+ live: LiveUsage,
216
+ message: SDKAssistantMessage,
217
+ ): void {
218
+ const u = message.message.usage;
219
+ if (!u) return;
220
+ live.input += u.input_tokens ?? 0;
221
+ live.output += u.output_tokens ?? 0;
222
+ live.cacheRead += u.cache_read_input_tokens ?? 0;
223
+ live.cacheWrite += u.cache_creation_input_tokens ?? 0;
224
+ live.calls += 1;
225
+ updateLiveTurn(chatId, {
226
+ inputTokens: live.input,
227
+ outputTokens: live.output,
228
+ cacheRead: live.cacheRead,
229
+ cacheWrite: live.cacheWrite,
230
+ // This call's full prompt = current context fill.
231
+ contextTokens:
232
+ (u.input_tokens ?? 0) +
233
+ (u.cache_read_input_tokens ?? 0) +
234
+ (u.cache_creation_input_tokens ?? 0),
235
+ contextWindow: state.contextWindow ?? 0,
236
+ numApiCalls: live.calls,
237
+ });
238
+ }
239
+
240
+ type StreamContext = {
241
+ chatId: string;
242
+ qi: Query;
243
+ state: StreamState;
244
+ live: LiveUsage;
245
+ watchdog: PostResultWatchdog;
246
+ /** Model the result message's usage is attributed to. */
247
+ model: string;
248
+ /** tool_use id → tool name for calls announced this turn. */
249
+ pendingTools: Map<string, string>;
250
+ };
251
+
252
+ /**
253
+ * Translate one assistant message: progress text segments BEFORE the tool
254
+ * calls they precede, so a model that says "let me check…" then calls a
255
+ * tool delivers the explanatory text first.
256
+ */
257
+ function* translateAssistantMessage(
258
+ ctx: StreamContext,
259
+ message: SDKAssistantMessage,
260
+ ): Generator<AgentEvent, void, void> {
261
+ const { chatId, state } = ctx;
262
+ const result = processAssistantMessage(message, state);
263
+ state.lastTrailingText = result.trailingText;
264
+ noteAssistantUsage(chatId, state, ctx.live, message);
265
+
266
+ for (const progress of result.progressTexts) {
267
+ yield { type: "assistant_message", text: progress };
268
+ }
269
+
270
+ for (const tool of result.tools) {
271
+ recordToolCall(chatId, tool.name, "claude");
272
+ const norm = captureDeliveredText(tool.name, tool.input);
273
+ if (norm) state.deliveredTextNorms.push(norm);
274
+ if (isTurnTerminator(tool.name, tool.input)) {
275
+ state.turnTerminated = true;
276
+ }
277
+ // Use the SDK's tool_use block id so the later tool_result (same id)
278
+ // correlates — a UI spinner opened on this event can only be closed
279
+ // by an event carrying the same id.
280
+ const toolId =
281
+ tool.id ||
282
+ `${tool.name}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
283
+ ctx.pendingTools.set(toolId, tool.name);
284
+ yield { type: "tool_call", id: toolId, name: tool.name, input: tool.input };
285
+ }
286
+ }
287
+
288
+ /**
289
+ * Tool executions come back as tool_result blocks on synthetic user
290
+ * messages. Emit the matching tool_result event — the other half of the
291
+ * lifecycle a tool_call opens (frontends show a spinner until it arrives).
292
+ */
293
+ function* translateToolResults(
294
+ ctx: StreamContext,
295
+ message: SDKUserMessage,
296
+ ): Generator<AgentEvent, void, void> {
297
+ for (const tr of extractToolResults(message)) {
298
+ const name = ctx.pendingTools.get(tr.toolUseId);
299
+ if (!name) continue; // not a tool we announced (subagent, replay)
300
+ ctx.pendingTools.delete(tr.toolUseId);
301
+ yield {
302
+ type: "tool_result",
303
+ id: tr.toolUseId,
304
+ name,
305
+ ...(tr.error ? { error: tr.error } : {}),
306
+ };
307
+ }
308
+ }
309
+
310
+ /** The `for await` over the SDK's message stream, one event type at a time. */
311
+ async function* consumeSdkStream(
312
+ ctx: StreamContext,
313
+ ): AsyncGenerator<AgentEvent, void, void> {
314
+ const { state } = ctx;
315
+ for await (const message of ctx.qi) {
316
+ if (isSystemInit(message)) {
317
+ state.newSessionId = message.session_id;
318
+ continue;
319
+ }
320
+ if (isStreamEvent(message)) {
321
+ const emit = processStreamDelta(message, state);
322
+ if (emit) {
323
+ yield emit.phase === "text"
324
+ ? { type: "text_delta", text: emit.text }
325
+ : { type: "reasoning", text: emit.text };
326
+ }
327
+ continue;
328
+ }
329
+ if (isAssistant(message)) {
330
+ yield* translateAssistantMessage(ctx, message);
331
+ continue;
332
+ }
333
+ if (isUserMessage(message)) {
334
+ yield* translateToolResults(ctx, message);
335
+ continue;
336
+ }
337
+ // The turn just moved the plan's usage — drop the cached windows so
338
+ // the next /status reads them again instead of showing pre-turn
339
+ // figures.
340
+ if (isRateLimitEvent(message)) {
341
+ invalidatePlanUsage();
342
+ continue;
343
+ }
344
+ if (isResult(message)) {
345
+ processResultMessage(message, state, ctx.model);
346
+ ctx.watchdog.arm();
347
+ }
348
+ }
349
+ }
350
+
351
+ // ── Turn-local helpers ──────────────────────────────────────────────────────
352
+
353
+ function createLiveUsage(): LiveUsage {
354
+ return { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, calls: 0 };
355
+ }
356
+
357
+ /**
358
+ * The recursive retry stream for `applyRetryDecisionStream`. A fallback
359
+ * model id is threaded through the retry's params (`params.model`
360
+ * outranks chat settings, so a transient `setChatModel` flip would be a
361
+ * silent no-op).
362
+ */
363
+ function retryStreamBuilder(
364
+ params: ChatRunParams,
365
+ internal: InternalState,
366
+ ): (fallbackModelId?: string) => AsyncIterable<AgentEvent> {
367
+ return (fallbackModelId) =>
368
+ runChatTurn(
369
+ fallbackModelId
370
+ ? {
371
+ ...params,
372
+ model: makeBareModelRef(
373
+ params.model.backend,
374
+ fallbackModelId,
375
+ "fallback",
376
+ ),
377
+ }
378
+ : params,
379
+ { ...internal, errorRetried: true },
380
+ );
381
+ }
382
+
383
+ /**
384
+ * First turn of a session is where flow violations cluster — the model
385
+ * hasn't seen the contract in action yet. One line appended to the turn-0
386
+ * user message (never the system prompt, so the cached prefix is
387
+ * untouched) pre-empts the 2x-token violation retry. Skipped on flow
388
+ * retries: those already carry the full reminder.
389
+ */
390
+ function buildTurnPrompt(
391
+ params: ChatRunParams,
392
+ frontend: string | undefined,
393
+ previousTurns: number,
394
+ internal: InternalState,
395
+ ): string {
396
+ let prompt = formatUserPrompt({
397
+ text: params.text,
398
+ senderName: params.senderName ?? "user",
399
+ senderHandle: params.senderHandle,
400
+ isGroup: params.isGroup,
401
+ messageId: params.messageId,
402
+ });
403
+ if (frontend && previousTurns === 0 && !internal.flowRetries) {
404
+ prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
405
+ }
406
+ return prompt;
407
+ }
408
+
409
+ /**
410
+ * Terminal failure — account for whatever the turn consumed before dying
411
+ * (failed turns burn real tokens). The result message never arrived, so
412
+ * `state.sdk*` is usually empty; fall back to the per-call accumulator.
413
+ */
414
+ function accountFailedClaudeTurn(
415
+ chatId: string,
416
+ state: StreamState,
417
+ live: LiveUsage,
418
+ model: string,
419
+ durationMs: number,
420
+ ): void {
421
+ const sawResultUsage =
422
+ state.sdkInputTokens +
423
+ state.sdkOutputTokens +
424
+ state.sdkCacheRead +
425
+ state.sdkCacheWrite >
426
+ 0;
427
+ accountFailedTurn({
428
+ backend: "claude",
429
+ chatId,
430
+ state,
431
+ durationMs,
432
+ model,
433
+ apiCalls: state.numApiCalls || live.calls,
434
+ usage: sawResultUsage
435
+ ? turnUsageSnapshot(state)
436
+ : {
437
+ inputTokens: live.input,
438
+ outputTokens: live.output,
439
+ cacheRead: live.cacheRead,
440
+ cacheWrite: live.cacheWrite,
441
+ },
442
+ });
443
+ }
444
+
445
+ /**
446
+ * The aggregate `cache=NN%` can't distinguish a turn that reused the
447
+ * previous turn's prefix from one that re-wrote it — see
448
+ * shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
449
+ * so warn when this turn's verdict proves the previous turn's overflow
450
+ * cost a prefix re-write, then record this turn's overflow for the next.
451
+ */
452
+ function reportCacheVerdict(chatId: string, state: StreamState): void {
453
+ if (state.cacheStats) {
454
+ const overflow = priorLookbackOverflow(chatId);
455
+ if (
456
+ overflow !== undefined &&
457
+ crossTurnVerdict(state.cacheStats) === "miss"
458
+ ) {
459
+ logWarn(
460
+ "agent",
461
+ `[${chatId}] previous turn emitted ~${overflow} content blocks ` +
462
+ `(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
463
+ `missed — that turn's cache write was likely never read`,
464
+ );
465
+ }
466
+ }
467
+ noteLookbackRisk(chatId, state.toolCalls);
468
+ }
469
+
143
470
  // ── Main chat-turn generator ────────────────────────────────────────────────
144
471
 
145
472
  /**
@@ -157,19 +484,17 @@ export async function* runChatTurn(
157
484
  _internal: InternalState = {},
158
485
  ): AsyncIterable<AgentEvent> {
159
486
  const config = getConfig();
160
-
161
- const { chatId, text, senderName, senderHandle, isGroup } = params;
487
+ const { chatId, text } = params;
162
488
  const session = getSession(chatId);
163
489
  const t0 = Date.now();
164
490
 
165
- // The chat's OWNING messaging frontend (falling back to the primary
166
- // for cross-surface chats). Drives the delivery-contract suffix and
167
- // the frontend-aware flow-violation text — the tool NAMES differ per
168
- // frontend (native's send tool is send_message, telegram's is send),
169
- // and with tool servers scoped per chat the wrong contract would
170
- // instruct a tool that doesn't exist. Empty in terminal mode, where
171
- // no delivery tools exist and the strict tool-only contract must not
172
- // be asserted.
491
+ // The chat's OWNING messaging frontend (falling back to the primary for
492
+ // cross-surface chats). Drives the delivery-contract suffix and the
493
+ // frontend-aware flow-violation text — the tool NAMES differ per
494
+ // frontend, and with tool servers scoped per chat the wrong contract
495
+ // would instruct a tool that doesn't exist. Empty in terminal mode,
496
+ // where no delivery tools exist and the strict tool-only contract must
497
+ // not be asserted.
173
498
  const frontend: string | undefined = frontendsForChat(
174
499
  chatId,
175
500
  getActiveFrontends(),
@@ -177,11 +502,8 @@ export async function* runChatTurn(
177
502
 
178
503
  // Frozen per-session prompt (keyed by session epoch) — stable across
179
504
  // turns so the provider's prompt-cache prefix survives other chats'
180
- // session resets. See backend/shared/system-prompt.ts. The delivery
181
- // contract joins as the backend suffix — the tail of the static
182
- // prompt, the highest-salience spot — so models see the tool-only
183
- // flow before their first turn instead of discovering it via a
184
- // [FLOW VIOLATION] retry.
505
+ // session resets. The delivery contract joins as the backend suffix —
506
+ // the tail of the static prompt, the highest-salience spot.
185
507
  const preparedPrompt = prepareSystemPrompt({
186
508
  config,
187
509
  previousTurns: session.turns,
@@ -199,464 +521,155 @@ export async function* runChatTurn(
199
521
  params.model.id,
200
522
  preparedPrompt,
201
523
  );
202
-
203
- let prompt = formatUserPrompt({
204
- text,
205
- senderName: senderName ?? "user",
206
- senderHandle,
207
- isGroup,
208
- messageId: params.messageId,
209
- });
210
- // First turn of a session is where flow violations cluster — the
211
- // model hasn't seen the contract in action yet. One line appended to
212
- // the turn-0 user message (never the system prompt, so the cached
213
- // prefix is untouched) pre-empts the 2x-token violation retry.
214
- // Skipped on flow retries: those already carry the full reminder.
215
- if (frontend && session.turns === 0 && !_internal.flowRetries) {
216
- prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
217
- }
524
+ const prompt = buildTurnPrompt(params, frontend, session.turns, _internal);
218
525
  log("agent", `[${chatId}] <- (${text.length} chars)`);
219
- traceMessage(chatId, "in", text, { senderName, isGroup });
526
+ traceMessage(chatId, "in", text, {
527
+ senderName: params.senderName,
528
+ isGroup: params.isGroup,
529
+ });
220
530
 
221
531
  yield { type: "run_started" };
222
532
 
223
533
  const qi = query({ prompt, options });
224
534
  activeQueries.set(chatId, qi);
225
535
 
226
- // Cold-start delivery-tool race. The Claude SDK spawns a fresh subprocess
227
- // per turn, and each reconnects to the hub's `${frontend}-tools` server.
228
- // `alwaysLoad` is meant to block startup until that server connects, but on
229
- // the FIRST turn of a freshly-opened chat the hub binding is created cold and
230
- // can still be `pending` when the model builds its turn-1 tool list — so
231
- // `end_turn`/`send` are absent and the reply silently fails (or the model
232
- // burns a `tool_search` round-trip to recover it). After a close+reconnect
233
- // the hub binding is warm, connects instantly, and the tools are present —
234
- // which is exactly why the bug "heals" on reconnect. Explicitly wait (bounded,
235
- // non-throwing) for the delivery server to leave `pending` before consuming
236
- // the stream. This mirrors the proven `refreshTools` gate and returns
237
- // immediately on warm turns, so it adds no steady-state latency.
536
+ // Cold-start delivery-tool race: on the FIRST turn of a freshly-opened
537
+ // chat the hub's `${frontend}-tools` binding can still be `pending` when
538
+ // the model builds its tool list, so `end_turn`/`send` are absent and the
539
+ // reply silently fails. Wait (bounded, non-throwing) for it to connect;
540
+ // returns immediately on warm turns. Mirrors the `refreshTools` gate.
238
541
  if (frontend) {
239
542
  await waitForMcpServersReady(qi, [`${frontend}-tools`], 5_000, 100);
240
543
  }
241
544
 
242
545
  const state = createStreamState();
243
- // tool_use id → tool name for calls we've announced this turn; lets
244
- // the tool_result emission name the tool without re-parsing blocks.
245
- const pendingTools = new Map<string, string>();
246
-
247
- // Per-API-call usage accumulator for live mid-turn stats. Each
248
- // assistant message carries its API call's usage as it lands; the
249
- // authoritative per-turn totals still come from the final result
250
- // message (processResultMessage) — this only feeds the live-turn
251
- // overlay so /status moves while a long agentic turn runs, and the
252
- // failure path below so an errored turn's burn isn't lost.
253
- const liveAcc = {
254
- input: 0,
255
- output: 0,
256
- cacheRead: 0,
257
- cacheWrite: 0,
258
- calls: 0,
259
- };
260
-
261
- let postResultTimer: ReturnType<typeof setTimeout> | null = null;
262
- let postResultForceClosed = false;
263
- const armPostResultWatchdog = (): void => {
264
- if (postResultTimer) return;
265
- const t = setTimeout(() => {
266
- postResultForceClosed = true;
267
- logWarn(
268
- "agent",
269
- `[${chatId}] SDK iterator stuck ${SDK_POST_RESULT_GRACE_MS}ms after result — aborting`,
270
- );
271
- incrementCounter("sdk.iterator_force_close_after_result");
272
- try {
273
- abortController.abort();
274
- } catch {
275
- /* abort() can throw if already aborted — ignore */
276
- }
277
- qi.return(undefined).catch(() => {
278
- /* the generator may already be in a terminal state — ignore */
279
- });
280
- }, SDK_POST_RESULT_GRACE_MS);
281
- t.unref();
282
- postResultTimer = t;
283
- };
284
-
285
- const captureIntoState = (
286
- toolName: string,
287
- input: Record<string, unknown>,
288
- ): void => {
289
- const norm = captureDeliveredText(toolName, input);
290
- if (norm) state.deliveredTextNorms.push(norm);
291
- };
546
+ const live = createLiveUsage();
547
+ const watchdog = createPostResultWatchdog(chatId, abortController, qi);
292
548
 
293
549
  let propagateError: AgentEvent | null = null;
294
550
  try {
295
- for await (const message of qi) {
296
- if (isSystemInit(message)) {
297
- state.newSessionId = message.session_id;
298
- continue;
299
- }
300
-
301
- if (isStreamEvent(message)) {
302
- const emit = processStreamDelta(message, state);
303
- if (emit) {
304
- if (emit.phase === "text") {
305
- yield { type: "text_delta", text: emit.text };
306
- } else {
307
- yield { type: "reasoning", text: emit.text };
308
- }
309
- }
310
- continue;
311
- }
312
-
313
- if (isAssistant(message)) {
314
- const result = processAssistantMessage(message, state);
315
- state.lastTrailingText = result.trailingText;
316
-
317
- const u = message.message.usage;
318
- if (u) {
319
- liveAcc.input += u.input_tokens ?? 0;
320
- liveAcc.output += u.output_tokens ?? 0;
321
- liveAcc.cacheRead += u.cache_read_input_tokens ?? 0;
322
- liveAcc.cacheWrite += u.cache_creation_input_tokens ?? 0;
323
- liveAcc.calls += 1;
324
- updateLiveTurn(chatId, {
325
- inputTokens: liveAcc.input,
326
- outputTokens: liveAcc.output,
327
- cacheRead: liveAcc.cacheRead,
328
- cacheWrite: liveAcc.cacheWrite,
329
- // This call's full prompt = current context fill.
330
- contextTokens:
331
- (u.input_tokens ?? 0) +
332
- (u.cache_read_input_tokens ?? 0) +
333
- (u.cache_creation_input_tokens ?? 0),
334
- contextWindow: state.contextWindow ?? 0,
335
- numApiCalls: liveAcc.calls,
336
- });
337
- }
338
-
339
- // Emit progress text segments BEFORE the tool calls they
340
- // precede, so a model that says "let me check…" then calls a
341
- // tool delivers the explanatory text first.
342
- for (const progress of result.progressTexts) {
343
- yield { type: "assistant_message", text: progress };
344
- }
345
-
346
- for (const tool of result.tools) {
347
- recordToolCall(chatId, tool.name, "claude");
348
- captureIntoState(tool.name, tool.input);
349
- if (isTurnTerminator(tool.name, tool.input)) {
350
- state.turnTerminated = true;
351
- }
352
- // Use the SDK's tool_use block id so the later tool_result
353
- // (same id) correlates — a UI spinner opened on this event
354
- // can only be closed by an event carrying the same id.
355
- const toolId =
356
- tool.id ||
357
- `${tool.name}-${Date.now()}-${Math.random()
358
- .toString(36)
359
- .slice(2, 8)}`;
360
- pendingTools.set(toolId, tool.name);
361
- yield {
362
- type: "tool_call",
363
- id: toolId,
364
- name: tool.name,
365
- input: tool.input,
366
- };
367
- }
368
- continue;
369
- }
370
-
371
- // Tool executions come back as tool_result blocks on synthetic
372
- // user messages. Emit the matching tool_result event — the other
373
- // half of the lifecycle a tool_call opens (frontends show a
374
- // spinner until it arrives).
375
- if (isUserMessage(message)) {
376
- for (const tr of extractToolResults(message)) {
377
- const name = pendingTools.get(tr.toolUseId);
378
- if (!name) continue; // not a tool we announced (subagent, replay)
379
- pendingTools.delete(tr.toolUseId);
380
- yield {
381
- type: "tool_result",
382
- id: tr.toolUseId,
383
- name,
384
- ...(tr.error ? { error: tr.error } : {}),
385
- };
386
- }
387
- continue;
388
- }
389
-
390
- // The turn just moved the plan's usage — drop the cached windows so
391
- // the next /status reads them again instead of showing pre-turn
392
- // figures.
393
- if (isRateLimitEvent(message)) {
394
- invalidatePlanUsage();
395
- continue;
396
- }
397
-
398
- if (isResult(message)) {
399
- processResultMessage(message, state, options.model ?? activeModel);
400
- armPostResultWatchdog();
401
- }
402
- }
403
-
551
+ yield* consumeSdkStream({
552
+ chatId,
553
+ qi,
554
+ state,
555
+ live,
556
+ watchdog,
557
+ model: options.model ?? activeModel,
558
+ pendingTools: new Map(),
559
+ });
404
560
  // The SDK doesn't throw on API errors — it converts them into a
405
561
  // synthetic assistant message and finishes the turn with an error-
406
562
  // flagged result (usage limits, 429s, auth failures all land here).
407
- // Rethrow the captured error text so this turn takes the SAME path
408
- // as a thrown SDK error: retry decision, failed-turn accounting, and
409
- // an `error` event carrying the real message — instead of being
410
- // treated as a normal reply (which, with no delivery tool call,
411
- // would trip the flow-violation re-prompt loop and burn more turns
412
- // against an already-exhausted limit).
563
+ // Rethrow so this turn takes the SAME path as a thrown SDK error
564
+ // instead of tripping the flow-violation re-prompt loop against an
565
+ // already-exhausted limit.
413
566
  if (state.resultErrorText) {
414
567
  throw new Error(state.resultErrorText);
415
568
  }
416
569
  } catch (err) {
417
- if (!postResultForceClosed) {
418
- const buildRetryStream = (
419
- fallbackModelId?: string,
420
- ): AsyncIterable<AgentEvent> =>
421
- runChatTurn(
422
- fallbackModelId
423
- ? {
424
- ...params,
425
- model: makeBareModelRef(
426
- params.model.backend,
427
- fallbackModelId,
428
- "fallback",
429
- ),
430
- }
431
- : params,
432
- { ..._internal, errorRetried: true },
433
- );
570
+ if (!watchdog.forceClosed) {
434
571
  const { retried, classified } = yield* applyRetryDecisionStream({
435
572
  err,
436
573
  chatId,
437
574
  activeModel,
438
575
  retried: _internal.errorRetried ?? false,
439
- buildRetryStream,
576
+ buildRetryStream: retryStreamBuilder(params, _internal),
440
577
  // No backendLabel — historical claude-sdk log shape was un-prefixed.
441
578
  });
442
- if (retried) {
443
- // The recursive stream already yielded its own usage + completed.
444
- return;
445
- }
579
+ // The recursive stream already yielded its own usage + completed.
580
+ if (retried) return;
446
581
  logError("agent", `[${chatId}] SDK error: ${classified.message}`);
447
- // Defer the actual yield until after the `finally` cleanup runs so the
448
- // watchdog timer and activeQueries entry are released first.
582
+ // Defer the yield until after `finally` releases the watchdog timer
583
+ // and the activeQueries entry.
449
584
  propagateError = {
450
585
  type: "error",
451
586
  error: classifiedToAgentError(classified),
452
587
  };
453
588
  }
454
589
  } finally {
455
- if (postResultTimer) {
456
- clearTimeout(postResultTimer);
457
- postResultTimer = null;
458
- }
590
+ watchdog.clear();
459
591
  if (activeQueries.get(chatId) === qi) {
460
592
  activeQueries.delete(chatId);
461
593
  }
462
594
  }
463
595
 
464
596
  if (propagateError) {
465
- // Terminal failure — account for whatever the turn consumed before
466
- // dying (failed turns burn real tokens). The result message never
467
- // arrived, so state.sdk* is usually empty; fall back to the per-call
468
- // accumulator. Retried turns return earlier and never reach here.
469
- const sawResultUsage =
470
- state.sdkInputTokens +
471
- state.sdkOutputTokens +
472
- state.sdkCacheRead +
473
- state.sdkCacheWrite >
474
- 0;
475
- recordFailedTurnAccounting({
476
- backend: "claude",
477
- chatId,
478
- durationMs: Date.now() - t0,
479
- toolCalls: state.toolCalls,
480
- apiCalls: state.numApiCalls || liveAcc.calls,
481
- model: activeModel,
482
- usage: sawResultUsage
483
- ? {
484
- inputTokens: state.sdkInputTokens,
485
- outputTokens: state.sdkOutputTokens,
486
- cacheRead: state.sdkCacheRead,
487
- cacheWrite: state.sdkCacheWrite,
488
- }
489
- : {
490
- inputTokens: liveAcc.input,
491
- outputTokens: liveAcc.output,
492
- cacheRead: liveAcc.cacheRead,
493
- cacheWrite: liveAcc.cacheWrite,
494
- },
495
- contextTokens: state.contextTokens,
496
- contextWindow: state.contextWindow,
497
- });
597
+ accountFailedClaudeTurn(chatId, state, live, activeModel, Date.now() - t0);
498
598
  yield propagateError;
499
599
  return;
500
600
  }
501
601
 
502
- // ── Persist session and usage ─────────────────────────────────────────────
503
-
504
602
  const durationMs = Date.now() - t0;
505
- recordTurnMetrics({
603
+ accountTurn({
506
604
  chatId,
507
605
  backend: "claude",
606
+ state,
508
607
  durationMs,
509
- toolCalls: state.toolCalls,
510
- apiCalls: state.numApiCalls,
511
- usage: {
512
- inputTokens: state.sdkInputTokens,
513
- outputTokens: state.sdkOutputTokens,
514
- cacheRead: state.sdkCacheRead,
515
- cacheWrite: state.sdkCacheWrite,
608
+ model: activeModel,
609
+ sessionId: state.newSessionId,
610
+ context: {
611
+ contextTokens: state.contextTokens,
612
+ contextWindow: state.contextWindow,
613
+ numApiCalls: state.numApiCalls,
614
+ costUsd: state.costUsd,
516
615
  },
517
616
  });
518
- if (state.newSessionId) setSessionId(chatId, state.newSessionId);
519
- recordUsage(chatId, {
520
- inputTokens: state.sdkInputTokens,
521
- outputTokens: state.sdkOutputTokens,
522
- cacheRead: state.sdkCacheRead,
523
- cacheWrite: state.sdkCacheWrite,
524
- durationMs,
525
- model: activeModel,
526
- contextTokens: state.contextTokens,
527
- contextWindow: state.contextWindow,
528
- numApiCalls: state.numApiCalls,
529
- costUsd: state.costUsd,
617
+ nameSessionFromFirstMessage({
618
+ chatId,
619
+ text,
620
+ previousTurns: session.turns,
621
+ isRetry: Boolean(_internal.flowRetries),
530
622
  });
531
623
 
532
- // Set a descriptive session name from the first message.
533
- // Guard against flow-violation retries, which pass the reminder text as
534
- // `text` — we only want the original user message, not the synthetic prompt.
535
- if (session.turns === 0 && text && !_internal.flowRetries) {
536
- const name = extractSessionName(text);
537
- if (name) setSessionName(chatId, name);
538
- }
539
-
540
- // ── Trailing-prose contract + flow-violation retry ──────────────────────
541
-
542
624
  // Only messaging frontends have a tool-only delivery contract to enforce.
543
625
  // In terminal mode (`frontend` undefined) there are no delivery tools and
544
- // trailing prose IS the reply — the terminal renderer surfaces it via
545
- // `result.text` when nothing came through the bridge (see
546
- // frontend/terminal/index.ts). Running the check there would flag the
547
- // model's natural prose as a violation and re-prompt it to call `end_turn`,
548
- // a tool that mode doesn't even register. Mirrors the `frontend`-gated
549
- // delivery-contract suffix above.
626
+ // trailing prose IS the reply (the terminal renderer surfaces it via
627
+ // `result.text`); running the check there would re-prompt the model to
628
+ // call `end_turn`, a tool that mode doesn't even register.
629
+ const flowRetries = _internal.flowRetries ?? 0;
550
630
  const violation = frontend
551
- ? detectFlowViolation({
552
- trailingText: state.lastTrailingText,
553
- turnTerminated: state.turnTerminated,
554
- deliveredTextNorms: state.deliveredTextNorms,
555
- toolCalls: state.toolCalls,
556
- retried: (_internal.flowRetries ?? 0) > 0,
557
- retryCount: _internal.flowRetries ?? 0,
558
- maxRetries: FLOW_VIOLATION_MAX_RETRIES,
631
+ ? enforceTrailingProse({
632
+ chatId,
633
+ state,
634
+ flowRetries,
559
635
  reminder: buildFlowViolationReminder(frontend),
560
636
  })
561
- : ({ violated: false } as const);
562
-
563
- if (violation.violated) {
564
- recordFlowViolation(
565
- chatId,
566
- violation.shouldRetry ? "retried" : "cap_exhausted",
637
+ : undefined;
638
+ if (violation?.violated && violation.shouldRetry) {
639
+ yield* runChatTurn(
640
+ { ...params, text: violation.reminder },
641
+ { ..._internal, flowRetries: flowRetries + 1 },
567
642
  );
568
- log(
569
- "agent",
570
- `[${chatId}] flow violation: ${violation.reason}. ${
571
- violation.shouldRetry
572
- ? "Re-prompting with reminder."
573
- : `Retry cap (${FLOW_VIOLATION_MAX_RETRIES}) exhausted — accepting silent drop.`
574
- }`,
575
- );
576
-
577
- if (violation.shouldRetry) {
578
- yield* runChatTurn(
579
- { ...params, text: violation.reminder },
580
- {
581
- ..._internal,
582
- flowRetries: (_internal.flowRetries ?? 0) + 1,
583
- },
584
- );
585
- return;
586
- }
643
+ return;
587
644
  }
588
645
 
589
646
  // Reached the non-retry path — this turn counts as one user-visible turn.
590
647
  incrementTurns(chatId);
591
648
 
592
- // ── Build result events ──────────────────────────────────────────────────
593
-
594
649
  state.allResponseText += state.currentBlockText;
650
+ reportCacheVerdict(chatId, state);
595
651
 
596
- // The aggregate `cache=NN%` can't distinguish a turn that reused the
597
- // previous turn's prefix from one that re-wrote it — see
598
- // shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
599
- // so warn when this turn's verdict proves the previous turn's overflow
600
- // cost a prefix re-write, then record this turn's overflow for the next.
601
- if (state.cacheStats) {
602
- const overflow = priorLookbackOverflow(chatId);
603
- if (
604
- overflow !== undefined &&
605
- crossTurnVerdict(state.cacheStats) === "miss"
606
- ) {
607
- logWarn(
608
- "agent",
609
- `[${chatId}] previous turn emitted ~${overflow} content blocks ` +
610
- `(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
611
- `missed — that turn's cache write was likely never read`,
612
- );
613
- }
614
- }
615
- noteLookbackRisk(chatId, state.toolCalls);
616
-
652
+ const usage = turnUsageSnapshot(state);
617
653
  log(
618
654
  "agent",
619
- `[${chatId}] -> (${summarizeUsage(
620
- {
621
- inputTokens: state.sdkInputTokens,
622
- outputTokens: state.sdkOutputTokens,
623
- cacheRead: state.sdkCacheRead,
624
- cacheWrite: state.sdkCacheWrite,
625
- },
626
- {
627
- durationMs,
628
- toolCalls: state.toolCalls,
629
- ...(state.cacheStats
630
- ? { suffix: formatTurnCache(state.cacheStats) }
631
- : {}),
632
- },
633
- )})`,
655
+ `[${chatId}] -> (${summarizeUsage(usage, {
656
+ durationMs,
657
+ toolCalls: state.toolCalls,
658
+ ...(state.cacheStats
659
+ ? { suffix: formatTurnCache(state.cacheStats) }
660
+ : {}),
661
+ })})`,
634
662
  );
635
663
  traceMessage(chatId, "out", state.allResponseText, {
636
664
  durationMs,
637
- inputTokens: state.sdkInputTokens,
638
- outputTokens: state.sdkOutputTokens,
639
- cacheRead: state.sdkCacheRead,
640
- cacheWrite: state.sdkCacheWrite,
665
+ ...usage,
641
666
  toolCalls: state.toolCalls,
642
667
  model: activeModel,
643
668
  });
644
-
645
- const usage = {
646
- inputTokens: state.sdkInputTokens,
647
- outputTokens: state.sdkOutputTokens,
648
- cacheRead: state.sdkCacheRead,
649
- cacheWrite: state.sdkCacheWrite,
669
+ yield* buildResultEvents({
670
+ text: state.allResponseText.trim(),
671
+ durationMs,
672
+ usage,
650
673
  modelId: activeModel,
651
- };
652
- yield { type: "usage", usage };
653
- yield {
654
- type: "completed",
655
- result: {
656
- text: state.allResponseText.trim(),
657
- durationMs,
658
- usage,
659
- modelId: activeModel,
660
- },
661
- };
674
+ });
662
675
  }