talon-agent 5.0.1 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.0.1",
3
+ "version": "5.1.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -99,7 +99,7 @@
99
99
  "build:fusefs": "node native/talon-fusefs/build.mjs"
100
100
  },
101
101
  "dependencies": {
102
- "@anthropic-ai/claude-agent-sdk": "^0.3.197",
102
+ "@anthropic-ai/claude-agent-sdk": "^0.3.277",
103
103
  "@anthropic-ai/sdk": "^0.104.1",
104
104
  "@brave/brave-search-mcp-server": "^2.0.75",
105
105
  "@clack/prompts": "^1.2.0",
@@ -156,23 +156,48 @@ function createPostResultWatchdog(
156
156
  }
157
157
 
158
158
  // ── Active query store ──────────────────────────────────────────────────────
159
- // Holds the Query reference for each in-flight chat so gateway actions
160
- // (e.g. reload_plugins) can call control methods like setMcpServers().
159
+ // Holds the in-flight turn for each chat so gateway actions (e.g.
160
+ // reload_plugins) can call control methods like setMcpServers(), and so a
161
+ // user-driven stop can mark the very turn it interrupts.
161
162
 
162
- const activeQueries = new Map<string, Query>();
163
+ type ActiveTurn = {
164
+ qi: Query;
165
+ /**
166
+ * Set by `interruptChatTurn` the moment a stop is requested. The turn's
167
+ * close-out reads it to tell a deliberate stop from a fault: whatever
168
+ * the SDK does after an interrupt is the stop.
169
+ *
170
+ * Since SDK 0.3.x that is emphatically not a clean `result`. The CLI
171
+ * emits an `is_error` result whose `errors[]` holds only an
172
+ * `[ede_diagnostic] …` line, `Query.readMessages` keeps it as
173
+ * `lastErrorResultText`, and when the underlying stream then errors the
174
+ * SDK replaces the error with `Error("Claude Code returned an error
175
+ * result: " + lastErrorResultText)`. Unmarked, that reads as a genuine
176
+ * SDK failure: an ERROR log per /stop, failed-turn accounting, and a
177
+ * fallback-model retry of the turn the user just stopped.
178
+ */
179
+ interrupted: boolean;
180
+ };
181
+
182
+ const activeQueries = new Map<string, ActiveTurn>();
163
183
 
164
184
  /**
165
185
  * Best-effort graceful interrupt of a chat's in-flight turn. Uses the SDK's
166
- * native `Query.interrupt()`, which stops the agent loop and closes the stream
167
- * with a `result` (subtype `interrupt`) — so the turn ends as a normal
168
- * completion (turn_end + usage), NOT an error, and never trips the
169
- * model-fallback retry path. No-op (returns false) when no turn is running.
186
+ * native `Query.interrupt()`, which stops the agent loop — so the turn ends
187
+ * as a normal completion (turn_end + usage), NOT an error, and never trips
188
+ * the model-fallback retry path.
189
+ *
190
+ * The turn is marked BEFORE the native interrupt is fired, exactly as the
191
+ * shared `runtime/turn/turn-interrupt.ts` marks `state.turnTerminated`
192
+ * first: the mark, not the SDK's own close-out shape, is what makes the
193
+ * contract above true. No-op (returns false) when no turn is running.
170
194
  */
171
195
  export async function interruptChatTurn(chatId: string): Promise<boolean> {
172
- const qi = activeQueries.get(chatId);
173
- if (!qi) return false;
196
+ const active = activeQueries.get(chatId);
197
+ if (!active) return false;
198
+ active.interrupted = true;
174
199
  try {
175
- await qi.interrupt();
200
+ await active.qi.interrupt();
176
201
  log("agent", `[${chatId}] turn interrupted by user`);
177
202
  incrementCounter("sdk.turn_interrupted");
178
203
  return true;
@@ -187,7 +212,7 @@ export async function interruptChatTurn(chatId: string): Promise<boolean> {
187
212
 
188
213
  /** Get the active Query for a chat, if one is in flight. */
189
214
  export function getActiveQuery(chatId: string): Query | undefined {
190
- return activeQueries.get(chatId);
215
+ return activeQueries.get(chatId)?.qi;
191
216
  }
192
217
 
193
218
  // ── Internal state passed across recursive retry calls ──────────────────────
@@ -428,12 +453,6 @@ function accountFailedClaudeTurn(
428
453
  model: string,
429
454
  durationMs: number,
430
455
  ): void {
431
- const sawResultUsage =
432
- state.sdkInputTokens +
433
- state.sdkOutputTokens +
434
- state.sdkCacheRead +
435
- state.sdkCacheWrite >
436
- 0;
437
456
  accountFailedTurn({
438
457
  backend: "claude",
439
458
  chatId,
@@ -441,7 +460,7 @@ function accountFailedClaudeTurn(
441
460
  durationMs,
442
461
  model,
443
462
  apiCalls: state.numApiCalls || live.calls,
444
- usage: sawResultUsage
463
+ usage: sawResultUsage(state)
445
464
  ? turnUsageSnapshot(state)
446
465
  : {
447
466
  inputTokens: live.input,
@@ -452,6 +471,37 @@ function accountFailedClaudeTurn(
452
471
  });
453
472
  }
454
473
 
474
+ /** Whether the turn's `result` message reported any tokens at all. */
475
+ function sawResultUsage(state: StreamState): boolean {
476
+ return (
477
+ state.sdkInputTokens +
478
+ state.sdkOutputTokens +
479
+ state.sdkCacheRead +
480
+ state.sdkCacheWrite >
481
+ 0
482
+ );
483
+ }
484
+
485
+ /**
486
+ * Close out a turn the user stopped. An interrupt is a completion, so it
487
+ * accounts like one (`accountTurn`, not `accountFailedTurn` — nothing
488
+ * failed) — but the `result` message may never have landed, leaving
489
+ * `state.sdk*` at zero while the per-API-call accumulator holds what the
490
+ * turn really burned. Fold the accumulator in so a stop doesn't lose the
491
+ * tokens, and mark the turn terminated so the flow-violation re-prompt
492
+ * can't resurrect what the user just stopped (the same guarantee
493
+ * `runtime/turn/turn-interrupt.ts` gives the callback backends).
494
+ */
495
+ function closeInterruptedTurn(state: StreamState, live: LiveUsage): void {
496
+ state.turnTerminated = true;
497
+ if (sawResultUsage(state)) return;
498
+ state.sdkInputTokens = live.input;
499
+ state.sdkOutputTokens = live.output;
500
+ state.sdkCacheRead = live.cacheRead;
501
+ state.sdkCacheWrite = live.cacheWrite;
502
+ if (!state.numApiCalls) state.numApiCalls = live.calls;
503
+ }
504
+
455
505
  /**
456
506
  * The aggregate `cache=NN%` can't distinguish a turn that reused the
457
507
  * previous turn's prefix from one that re-wrote it — see
@@ -488,6 +538,105 @@ function reportCacheVerdict(
488
538
  noteLookbackRisk(chatId, state.toolCalls);
489
539
  }
490
540
 
541
+ // ── Stream phase ────────────────────────────────────────────────────────────
542
+
543
+ /** What the stream phase left for the rest of the turn to do. */
544
+ type StreamOutcome =
545
+ /** Run the normal post-stream phases (this includes every user stop). */
546
+ | { kind: "ok" }
547
+ /** A retry already ran to completion and yielded its own events. */
548
+ | { kind: "retried" }
549
+ /** Terminal failure — account for it and yield this `error` event. */
550
+ | { kind: "failed"; event: AgentEvent };
551
+
552
+ /**
553
+ * Drive the SDK stream to exhaustion and decide what its ending means.
554
+ * Owns the error recovery (retry decision, model fallback) and the
555
+ * user-interrupt contract; releases the watchdog timer and the
556
+ * `activeQueries` entry on every exit.
557
+ */
558
+ async function* runTurnStream(inputs: {
559
+ chatId: string;
560
+ params: ChatRunParams;
561
+ internal: InternalState;
562
+ active: ActiveTurn;
563
+ state: StreamState;
564
+ live: LiveUsage;
565
+ watchdog: PostResultWatchdog;
566
+ /** Model the turn is accounted against. */
567
+ activeModel: string;
568
+ /** Model string the SDK attributes the result message's usage to. */
569
+ sdkModel: string;
570
+ }): AsyncGenerator<AgentEvent, StreamOutcome, void> {
571
+ const { chatId, active, state, live, watchdog } = inputs;
572
+ try {
573
+ yield* consumeSdkStream({
574
+ chatId,
575
+ qi: active.qi,
576
+ state,
577
+ live,
578
+ watchdog,
579
+ model: inputs.sdkModel,
580
+ pendingTools: new Map(),
581
+ });
582
+ // The SDK doesn't throw on API errors — it converts them into a
583
+ // synthetic assistant message and finishes the turn with an error-
584
+ // flagged result (usage limits, 429s, auth failures all land here).
585
+ // Rethrow so this turn takes the SAME path as a thrown SDK error
586
+ // instead of tripping the flow-violation re-prompt loop against an
587
+ // already-exhausted limit.
588
+ //
589
+ // …unless the user stopped this turn: an interrupted turn's result is
590
+ // flagged `is_error` with nothing but an `[ede_diagnostic]` line
591
+ // behind it, which `readResultError` then renders as bare trailing
592
+ // text or "Claude SDK turn failed (<subtype>)". That is the stop, not
593
+ // a failure — the turn closes as a completion.
594
+ if (state.resultErrorText && !active.interrupted) {
595
+ throw new Error(state.resultErrorText);
596
+ }
597
+ } catch (err) {
598
+ if (active.interrupted) {
599
+ // Same deal one layer down: after an interrupt the SDK re-labels the
600
+ // stream error with that error-result text. Quiet by design — no
601
+ // `logError`, no retry decision (which would classify the ede text,
602
+ // bump `errors.*`, and could fall back to another model for a turn
603
+ // the user just stopped), no `error` event. The shutdown drain takes
604
+ // this same path: its abort interrupts through here too.
605
+ log(
606
+ "agent",
607
+ `[${chatId}] turn ended by user interrupt: ${
608
+ err instanceof Error ? err.message : String(err)
609
+ }`,
610
+ );
611
+ incrementCounter("sdk.interrupt_closed_stream");
612
+ } else if (!watchdog.forceClosed) {
613
+ const { retried, classified } = yield* applyRetryDecisionStream({
614
+ err,
615
+ chatId,
616
+ activeModel: inputs.activeModel,
617
+ retried: inputs.internal.errorRetried ?? false,
618
+ buildRetryStream: retryStreamBuilder(inputs.params, inputs.internal),
619
+ // No backendLabel — historical claude-sdk log shape was un-prefixed.
620
+ });
621
+ if (retried) return { kind: "retried" };
622
+ logError("agent", `[${chatId}] SDK error: ${classified.message}`);
623
+ // Returning (rather than yielding) here defers the `error` event
624
+ // until after `finally` releases the watchdog timer and the
625
+ // activeQueries entry.
626
+ return {
627
+ kind: "failed",
628
+ event: { type: "error", error: classifiedToAgentError(classified) },
629
+ };
630
+ }
631
+ } finally {
632
+ watchdog.clear();
633
+ if (activeQueries.get(chatId) === active) {
634
+ activeQueries.delete(chatId);
635
+ }
636
+ }
637
+ return { kind: "ok" };
638
+ }
639
+
491
640
  // ── Main chat-turn generator ────────────────────────────────────────────────
492
641
 
493
642
  /**
@@ -498,7 +647,11 @@ function reportCacheVerdict(
498
647
  * recurse via `yield*` and produce the retry's event stream
499
648
  * transparently). On flow violation: `yield* runChatTurn(retry
500
649
  * params)` — the recursive call owns its `incrementTurns`, the
501
- * caller deliberately doesn't increment.
650
+ * caller deliberately doesn't increment. On a user interrupt
651
+ * (`interruptChatTurn`, which also backs the shutdown drain): the turn
652
+ * closes as a completion carrying the partial text and the real usage —
653
+ * never an `error` event, never a retry, whatever shape the SDK chose to
654
+ * end the stream in.
502
655
  */
503
656
  export async function* runChatTurn(
504
657
  params: ChatRunParams,
@@ -552,7 +705,8 @@ export async function* runChatTurn(
552
705
  yield { type: "run_started" };
553
706
 
554
707
  const qi = query({ prompt, options });
555
- activeQueries.set(chatId, qi);
708
+ const active: ActiveTurn = { qi, interrupted: false };
709
+ activeQueries.set(chatId, active);
556
710
 
557
711
  // Cold-start delivery-tool race: on the FIRST turn of a freshly-opened
558
712
  // chat the hub's `${frontend}-tools` binding can still be `pending` when
@@ -567,59 +721,27 @@ export async function* runChatTurn(
567
721
  const live = createLiveUsage();
568
722
  const watchdog = createPostResultWatchdog(chatId, abortController, qi);
569
723
 
570
- let propagateError: AgentEvent | null = null;
571
- try {
572
- yield* consumeSdkStream({
573
- chatId,
574
- qi,
575
- state,
576
- live,
577
- watchdog,
578
- model: options.model ?? activeModel,
579
- pendingTools: new Map(),
580
- });
581
- // The SDK doesn't throw on API errors — it converts them into a
582
- // synthetic assistant message and finishes the turn with an error-
583
- // flagged result (usage limits, 429s, auth failures all land here).
584
- // Rethrow so this turn takes the SAME path as a thrown SDK error
585
- // instead of tripping the flow-violation re-prompt loop against an
586
- // already-exhausted limit.
587
- if (state.resultErrorText) {
588
- throw new Error(state.resultErrorText);
589
- }
590
- } catch (err) {
591
- if (!watchdog.forceClosed) {
592
- const { retried, classified } = yield* applyRetryDecisionStream({
593
- err,
594
- chatId,
595
- activeModel,
596
- retried: _internal.errorRetried ?? false,
597
- buildRetryStream: retryStreamBuilder(params, _internal),
598
- // No backendLabel — historical claude-sdk log shape was un-prefixed.
599
- });
600
- // The recursive stream already yielded its own usage + completed.
601
- if (retried) return;
602
- logError("agent", `[${chatId}] SDK error: ${classified.message}`);
603
- // Defer the yield until after `finally` releases the watchdog timer
604
- // and the activeQueries entry.
605
- propagateError = {
606
- type: "error",
607
- error: classifiedToAgentError(classified),
608
- };
609
- }
610
- } finally {
611
- watchdog.clear();
612
- if (activeQueries.get(chatId) === qi) {
613
- activeQueries.delete(chatId);
614
- }
615
- }
616
-
617
- if (propagateError) {
724
+ const outcome = yield* runTurnStream({
725
+ chatId,
726
+ params,
727
+ internal: _internal,
728
+ active,
729
+ state,
730
+ live,
731
+ watchdog,
732
+ activeModel,
733
+ sdkModel: options.model ?? activeModel,
734
+ });
735
+ // The recursive retry stream already yielded its own usage + completed.
736
+ if (outcome.kind === "retried") return;
737
+ if (outcome.kind === "failed") {
618
738
  accountFailedClaudeTurn(chatId, state, live, activeModel, Date.now() - t0);
619
- yield propagateError;
739
+ yield outcome.event;
620
740
  return;
621
741
  }
622
742
 
743
+ if (active.interrupted) closeInterruptedTurn(state, live);
744
+
623
745
  const durationMs = Date.now() - t0;
624
746
  accountTurn({
625
747
  chatId,
@@ -20,6 +20,7 @@ import { ALLOWED_TOOLS_BACKGROUND } from "../../core/constants.js";
20
20
  import { EFFORT_MAP } from "./constants.js";
21
21
  import { buildMcpServers, buildPluginMcpServers } from "./options.js";
22
22
  import { warnIfBelowCacheMinimum } from "../runtime/cache/cache-telemetry.js";
23
+ import { emitAssistantText } from "../runtime/one-shot-hooks.js";
23
24
 
24
25
  const DEFAULT_SUBPROCESS_KILL_GRACE_MS = 5 * 1000;
25
26
 
@@ -65,6 +66,7 @@ export async function runOneShotAgent(
65
66
  contextLabel,
66
67
  abortController,
67
68
  appendLog,
69
+ onAssistantText,
68
70
  } = params;
69
71
 
70
72
  // Reasoning effort is opt-in for background runs (config `heartbeatEffort`
@@ -126,7 +128,7 @@ export async function runOneShotAgent(
126
128
  // settlement figure the task table records.
127
129
  let usage: OneShotUsage | undefined;
128
130
  for await (const msg of qi) {
129
- await formatAndAppendMessage(appendLog, msg);
131
+ await formatAndAppendMessage(appendLog, msg, onAssistantText);
130
132
  if (msg.type === "result") {
131
133
  const u = msg.usage;
132
134
  usage = {
@@ -183,6 +185,7 @@ function assembleMcpServers(contextLabel: string): Record<string, unknown> {
183
185
  async function formatAndAppendMessage(
184
186
  appendLog: (text: string) => Promise<void>,
185
187
  msg: SDKMessage,
188
+ onAssistantText?: OneShotAgentParams["onAssistantText"],
186
189
  ): Promise<void> {
187
190
  try {
188
191
  const ts = new Date().toISOString().slice(11, 19);
@@ -200,7 +203,11 @@ async function formatAndAppendMessage(
200
203
  });
201
204
 
202
205
  if (textBlocks.length > 0) {
203
- await appendLog(`\n## [${ts}] Assistant\n${textBlocks.join("\n")}\n`);
206
+ const assistantText = textBlocks.join("\n");
207
+ // Report before the log write: an append failure (full disk, closed
208
+ // handle) must not also swallow the run's result for a hook caller.
209
+ emitAssistantText(onAssistantText, assistantText);
210
+ await appendLog(`\n## [${ts}] Assistant\n${assistantText}\n`);
204
211
  }
205
212
  if (toolUseBlocks.length > 0) {
206
213
  await appendLog(`\n${toolUseBlocks.join("\n\n")}\n`);
@@ -208,6 +215,14 @@ async function formatAndAppendMessage(
208
215
  break;
209
216
  }
210
217
  case "result": {
218
+ // Deliberately NOT reported through `onAssistantText`. The SDK's
219
+ // terminal `result` message restates the last assistant turn's text
220
+ // (`subtype: "success"`) or carries an error string (the
221
+ // `error_*` subtypes) — never anything the `assistant` case above
222
+ // has not already emitted. Reporting it too would hand every hook
223
+ // caller a duplicate final segment, and the truncated copy at that
224
+ // (2000 chars, below). The log keeps it because a run log wants the
225
+ // settlement line; a sub-agent result does not.
211
226
  const result =
212
227
  "result" in msg
213
228
  ? (msg as { result: string }).result
@@ -422,6 +422,15 @@ function readResultError(msg: SDKResultMessage, state: StreamState): void {
422
422
  .filter((e) => typeof e === "string" && !e.startsWith("[ede_diagnostic]"))
423
423
  .join("; ")
424
424
  .slice(0, 500);
425
+ // A known startup failure (SDK ≥ 0.3.274) names its cause — the CLI
426
+ // never ran a turn, so nothing else in the result says why. Lead with
427
+ // it; the diagnostics text is the same line stderr carried.
428
+ if (msg.startup_failure_reason) {
429
+ state.resultErrorText =
430
+ `Claude Code failed to start (${msg.startup_failure_reason})` +
431
+ (diagnostics ? `: ${diagnostics}` : "");
432
+ return;
433
+ }
425
434
  state.resultErrorText =
426
435
  state.lastTrailingText.trim() ||
427
436
  diagnostics ||
@@ -20,6 +20,7 @@
20
20
  import type { OneShotAgentParams, OneShotUsage } from "../../core/types.js";
21
21
  import { log, logWarn } from "../../util/log.js";
22
22
  import { appendBackendSuffix } from "../runtime/index.js";
23
+ import { emitAssistantText } from "../runtime/one-shot-hooks.js";
23
24
  import { ensureCodex, getCodexAuthInfo } from "./init.js";
24
25
  import {
25
26
  CODEX_SYSTEM_PROMPT_SUFFIX,
@@ -79,6 +80,7 @@ export async function runOneShotAgent(
79
80
  contextLabel,
80
81
  abortController,
81
82
  appendLog,
83
+ onAssistantText,
82
84
  } = params;
83
85
 
84
86
  const codex = ensureCodex(contextLabel);
@@ -143,7 +145,7 @@ export async function runOneShotAgent(
143
145
  let usage: OneShotUsage | undefined;
144
146
  for await (const event of events) {
145
147
  if (abortController.signal.aborted) break;
146
- await appendCodexEvent(appendLog, event);
148
+ await appendCodexEvent(appendLog, event, onAssistantText);
147
149
  if (event.type === "turn.completed") {
148
150
  const u = (event as { usage?: Record<string, number> }).usage;
149
151
  if (u) {
@@ -217,6 +219,7 @@ export async function runOneShotAgent(
217
219
  async function appendCodexEvent(
218
220
  appendLog: (text: string) => Promise<void>,
219
221
  event: { type: string } & Record<string, unknown>,
222
+ onAssistantText?: OneShotAgentParams["onAssistantText"],
220
223
  ): Promise<void> {
221
224
  const ts = new Date().toISOString().slice(11, 19);
222
225
 
@@ -260,7 +263,7 @@ async function appendCodexEvent(
260
263
  case "item.completed": {
261
264
  const item = (event as unknown as { item?: Record<string, unknown> })
262
265
  .item;
263
- if (item) await appendCodexItem(appendLog, item, ts);
266
+ if (item) await appendCodexItem(appendLog, item, ts, onAssistantText);
264
267
  return;
265
268
  }
266
269
  default:
@@ -268,17 +271,28 @@ async function appendCodexEvent(
268
271
  }
269
272
  }
270
273
 
271
- /** Append one `ThreadItem` to the run log. */
274
+ /**
275
+ * Append one `ThreadItem` to the run log, and report the model's final
276
+ * answers to the run's optional `onAssistantText` consumer.
277
+ *
278
+ * Only `agent_message` items are reported: `reasoning` items are the model's
279
+ * thinking and the rest are tool/command/diff payloads, none of which is the
280
+ * run's answer.
281
+ */
272
282
  async function appendCodexItem(
273
283
  appendLog: (text: string) => Promise<void>,
274
284
  item: Record<string, unknown>,
275
285
  ts: string,
286
+ onAssistantText?: OneShotAgentParams["onAssistantText"],
276
287
  ): Promise<void> {
277
288
  const type = typeof item.type === "string" ? item.type : "unknown";
278
289
 
279
290
  if (type === "agent_message") {
280
291
  const text = typeof item.text === "string" ? item.text : "";
281
- if (text) await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
292
+ if (text) {
293
+ emitAssistantText(onAssistantText, text);
294
+ await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
295
+ }
282
296
  return;
283
297
  }
284
298
 
@@ -42,6 +42,7 @@ import {
42
42
  type RemoteAssistantInfo,
43
43
  } from "./session-helpers.js";
44
44
  import { appendBackendSuffix, sleep } from "../runtime/index.js";
45
+ import { emitAssistantText } from "../runtime/one-shot-hooks.js";
45
46
  import { buildPermissionRuleset } from "./sessions.js";
46
47
 
47
48
  // ── Client surface ──────────────────────────────────────────────────────────
@@ -108,6 +109,7 @@ export async function runRemoteOneShotAgent<
108
109
  contextLabel,
109
110
  abortController,
110
111
  appendLog,
112
+ onAssistantText,
111
113
  } = params;
112
114
  const { label, errMsg } = bindings;
113
115
 
@@ -210,7 +212,7 @@ export async function runRemoteOneShotAgent<
210
212
  : [];
211
213
 
212
214
  for (const part of parts) {
213
- await appendResponsePart(appendLog, part);
215
+ await appendResponsePart(appendLog, part, onAssistantText);
214
216
  }
215
217
 
216
218
  // The prompt response's assistant info carries the run's token usage —
@@ -250,17 +252,28 @@ export async function runRemoteOneShotAgent<
250
252
 
251
253
  // ── Run-log rendering ───────────────────────────────────────────────────────
252
254
 
253
- /** Render one response part into the Markdown run log. */
255
+ /**
256
+ * Render one response part into the Markdown run log, and report the
257
+ * assistant's final text to the run's optional `onAssistantText` consumer.
258
+ *
259
+ * Only the `text` parts are reported: `reasoning` parts are the model's
260
+ * thinking and tool parts are call payloads, neither of which is the run's
261
+ * answer.
262
+ */
254
263
  async function appendResponsePart(
255
264
  appendLog: (text: string) => Promise<void>,
256
265
  part: Record<string, unknown>,
266
+ onAssistantText?: OneShotAgentParams["onAssistantText"],
257
267
  ): Promise<void> {
258
268
  const ts = new Date().toISOString().slice(11, 19);
259
269
  const type = typeof part.type === "string" ? part.type : "unknown";
260
270
 
261
271
  if (type === "text") {
262
272
  const text = typeof part.text === "string" ? part.text : "";
263
- if (text) await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
273
+ if (text) {
274
+ emitAssistantText(onAssistantText, text);
275
+ await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
276
+ }
264
277
  return;
265
278
  }
266
279
 
@@ -0,0 +1,45 @@
1
+ /**
2
+ * One-shot run hooks — the shared, throw-proof call site for
3
+ * `OneShotAgentParams.onAssistantText`.
4
+ *
5
+ * Every backend's isolated runner (`BackgroundRunner.runOneShotAgent`) already
6
+ * renders the model's assistant text into the run log as markdown. The hook is
7
+ * the same text as data, so a caller (the sub-agent runner) can take a run's
8
+ * result without parsing the log back apart. It lives here rather than in each
9
+ * backend so all four report it identically: same "final answer only" meaning,
10
+ * same empty-string skip, same swallow-and-log on a throwing callback.
11
+ *
12
+ * Contract:
13
+ * - Final assistant text only. Reasoning/thinking blocks and tool-call
14
+ * payloads are log-only; they never reach the hook.
15
+ * - Synchronous. The runner does not await the consumer, so a hook that
16
+ * wants to do async work owns its own queueing.
17
+ * - Never throws into the run. A consumer bug must not abort a heartbeat.
18
+ */
19
+
20
+ import type { OneShotAgentParams } from "../../core/types.js";
21
+ import { logWarn } from "../../util/log.js";
22
+
23
+ /**
24
+ * Report one assistant text segment to the run's optional consumer.
25
+ *
26
+ * No-ops when there is no hook (the heartbeat/dream/cron path) or when the
27
+ * text is empty — the backends guard their log appends the same way, so the
28
+ * hook sees exactly the segments the log does.
29
+ */
30
+ export function emitAssistantText(
31
+ onAssistantText: OneShotAgentParams["onAssistantText"],
32
+ text: string,
33
+ ): void {
34
+ if (!onAssistantText || !text) return;
35
+ try {
36
+ onAssistantText(text);
37
+ } catch (err) {
38
+ logWarn(
39
+ "agent",
40
+ `one-shot onAssistantText hook threw (ignored): ${
41
+ err instanceof Error ? err.message : String(err)
42
+ }`,
43
+ );
44
+ }
45
+ }
@@ -31,9 +31,10 @@
31
31
  * They are deliberately not the same types. Two client arguments do not
32
32
  * survive a process boundary and the wire shapes say so:
33
33
  *
34
- * - `OneShotAgentParams` carries an `AbortController` and an
35
- * `appendLog` callback. `HostOneShotParams` is the serialisable
36
- * subset; Phase 2 maps `appendLog` onto `log` notices and the
34
+ * - `OneShotAgentParams` carries an `AbortController` and two
35
+ * callbacks (`appendLog`, `onAssistantText`). `HostOneShotParams`
36
+ * is the serialisable subset; Phase 2 maps `appendLog` onto `log`
37
+ * notices, `onAssistantText` onto a notice of its own, and the
37
38
  * abort onto an `interrupt`-shaped request.
38
39
  * - `hello.config` is the `claude-sdk` slice of `TalonConfig` as
39
40
  * JSON. The in-process client takes the real `TalonConfig` object.
@@ -59,9 +60,9 @@ export const AGENT_HOST_PROTOCOL_VERSION = 1;
59
60
  // ── Shared payload shapes ───────────────────────────────────────────────────
60
61
 
61
62
  /**
62
- * The serialisable half of `OneShotAgentParams`. `abortController` and
63
- * `appendLog` are host-local concerns (see the file header); everything
64
- * else is exactly what a background run needs.
63
+ * The serialisable half of `OneShotAgentParams`. `abortController`,
64
+ * `appendLog` and `onAssistantText` are host-local concerns (see the file
65
+ * header); everything else is exactly what a background run needs.
65
66
  *
66
67
  * Unexported on purpose — it is reachable as
67
68
  * `Extract<HostRequest, { type: "one_shot" }>["params"]`, and a second
@@ -103,6 +103,9 @@ export interface ChatBackend {
103
103
  * trigger log-file producers keep their direct write path.
104
104
  * Resolves with the run's token usage when the SDK reports it
105
105
  * (the task table records it at settlement); void otherwise.
106
+ * Implementations must also honour the optional `onAssistantText`
107
+ * hook — the run's final answers as data, for callers (the
108
+ * sub-agent runner) that need a result rather than a markdown log.
106
109
  * - `evictOrphanSubprocesses(label)` — backends that spawn
107
110
  * per-run subprocesses (Claude SDK) implement this so a hung
108
111
  * run can be force-cleaned after the abort grace window.
@@ -100,6 +100,16 @@ export function formatChildExit(
100
100
  return lines.length > 0 ? `${head}; stderr: ${lines.join(" | ")}` : head;
101
101
  }
102
102
 
103
+ /**
104
+ * Human-readable form of a child key for log lines. Keys join server name
105
+ * and chat id with a NUL byte (see `childKey` in index.ts) — unambiguous
106
+ * as a Map key, but it renders as `\u0000` in the JSON log.
107
+ */
108
+ function describeKey(key: string): string {
109
+ const nul = key.indexOf("\u0000");
110
+ return nul === -1 ? key : `${key.slice(0, nul)} chat=${key.slice(nul + 1)}`;
111
+ }
112
+
103
113
  /**
104
114
  * Bookkeeping for a child process that has gone away, asked or not.
105
115
  * Wired as the transport's `onclose` BEFORE `client.connect` so the SDK
@@ -125,7 +135,7 @@ function onChildClosed(key: string, transport: HubChildTransport): void {
125
135
  : "died before registration";
126
136
  logWarn(
127
137
  "gateway",
128
- `hub child ${key} ${phase} (pid ${transport.pid ?? "?"}): ${formatChildExit(record)}`,
138
+ `hub child ${describeKey(key)} ${phase} (pid ${transport.pid ?? "?"}): ${formatChildExit(record)}`,
129
139
  );
130
140
  }
131
141
 
@@ -199,7 +209,10 @@ async function spawnChild(key: string, spec: ChildSpec): Promise<ChildHandle> {
199
209
  try {
200
210
  await client.close();
201
211
  } catch (err) {
202
- logWarn("gateway", `hub child ${key} close failed: ${String(err)}`);
212
+ logWarn(
213
+ "gateway",
214
+ `hub child ${describeKey(key)} close failed: ${String(err)}`,
215
+ );
203
216
  }
204
217
  })());
205
218
  })(),
@@ -226,7 +239,10 @@ async function spawnChild(key: string, spec: ChildSpec): Promise<ChildHandle> {
226
239
  };
227
240
 
228
241
  children.set(key, entry);
229
- log("gateway", `hub child started: ${key} (pid ${transport.pid ?? "?"})`);
242
+ log(
243
+ "gateway",
244
+ `hub child started: ${describeKey(key)} (pid ${transport.pid ?? "?"})`,
245
+ );
230
246
  return entry.handle;
231
247
  }
232
248
 
@@ -327,9 +343,9 @@ function reapIdle(): void {
327
343
  if (entry.lastActivity >= cutoff) continue;
328
344
  children.delete(key);
329
345
  entry.close().catch((err) => {
330
- logError("gateway", `hub reap of ${key} failed`, err);
346
+ logError("gateway", `hub reap of ${describeKey(key)} failed`, err);
331
347
  });
332
- log("gateway", `hub child reaped (idle): ${key}`);
348
+ log("gateway", `hub child reaped (idle): ${describeKey(key)}`);
333
349
  }
334
350
  }
335
351
 
package/src/core/types.ts CHANGED
@@ -161,6 +161,17 @@ export type OneShotAgentParams = {
161
161
  abortController: AbortController;
162
162
  /** Append a string to the run log (markdown). */
163
163
  appendLog: (text: string) => Promise<void>;
164
+ /**
165
+ * Called with each assistant text segment as the run produces it (final
166
+ * answers, not reasoning/thinking, not tool-call payloads). Optional so
167
+ * heartbeat/dream/cron callers are untouched; a sub-agent runner uses it to
168
+ * capture the run's result without parsing the markdown log.
169
+ *
170
+ * Synchronous and best-effort: every backend invokes it through
171
+ * `emitAssistantText` (backend/runtime/one-shot-hooks.ts), which swallows
172
+ * and logs a throwing callback so a consumer bug can never fail the run.
173
+ */
174
+ onAssistantText?: (text: string) => void;
164
175
  };
165
176
 
166
177
  /** How much cache telemetry a backend can surface in /status. */