talon-agent 5.0.1 → 5.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/backend/claude-sdk/handler.ts +191 -69
- package/src/backend/claude-sdk/one-shot.ts +17 -2
- package/src/backend/claude-sdk/stream.ts +9 -0
- package/src/backend/codex/one-shot.ts +18 -4
- package/src/backend/remote-server/one-shot.ts +16 -3
- package/src/backend/runtime/one-shot-hooks.ts +45 -0
- package/src/core/agent-runtime/agent-host.ts +7 -6
- package/src/core/agent-runtime/capabilities.ts +3 -0
- package/src/core/mcp-hub/children.ts +21 -5
- package/src/core/types.ts +11 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "talon-agent",
|
|
3
|
-
"version": "5.0
|
|
3
|
+
"version": "5.1.0",
|
|
4
4
|
"description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
|
|
5
5
|
"author": "Dylan Neve",
|
|
6
6
|
"license": "MIT",
|
|
@@ -99,7 +99,7 @@
|
|
|
99
99
|
"build:fusefs": "node native/talon-fusefs/build.mjs"
|
|
100
100
|
},
|
|
101
101
|
"dependencies": {
|
|
102
|
-
"@anthropic-ai/claude-agent-sdk": "^0.3.
|
|
102
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.277",
|
|
103
103
|
"@anthropic-ai/sdk": "^0.104.1",
|
|
104
104
|
"@brave/brave-search-mcp-server": "^2.0.75",
|
|
105
105
|
"@clack/prompts": "^1.2.0",
|
|
@@ -156,23 +156,48 @@ function createPostResultWatchdog(
|
|
|
156
156
|
}
|
|
157
157
|
|
|
158
158
|
// ── Active query store ──────────────────────────────────────────────────────
|
|
159
|
-
// Holds the
|
|
160
|
-
//
|
|
159
|
+
// Holds the in-flight turn for each chat so gateway actions (e.g.
|
|
160
|
+
// reload_plugins) can call control methods like setMcpServers(), and so a
|
|
161
|
+
// user-driven stop can mark the very turn it interrupts.
|
|
161
162
|
|
|
162
|
-
|
|
163
|
+
type ActiveTurn = {
|
|
164
|
+
qi: Query;
|
|
165
|
+
/**
|
|
166
|
+
* Set by `interruptChatTurn` the moment a stop is requested. The turn's
|
|
167
|
+
* close-out reads it to tell a deliberate stop from a fault: whatever
|
|
168
|
+
* the SDK does after an interrupt is the stop.
|
|
169
|
+
*
|
|
170
|
+
* Since SDK 0.3.x that is emphatically not a clean `result`. The CLI
|
|
171
|
+
* emits an `is_error` result whose `errors[]` holds only an
|
|
172
|
+
* `[ede_diagnostic] …` line, `Query.readMessages` keeps it as
|
|
173
|
+
* `lastErrorResultText`, and when the underlying stream then errors the
|
|
174
|
+
* SDK replaces the error with `Error("Claude Code returned an error
|
|
175
|
+
* result: " + lastErrorResultText)`. Unmarked, that reads as a genuine
|
|
176
|
+
* SDK failure: an ERROR log per /stop, failed-turn accounting, and a
|
|
177
|
+
* fallback-model retry of the turn the user just stopped.
|
|
178
|
+
*/
|
|
179
|
+
interrupted: boolean;
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
const activeQueries = new Map<string, ActiveTurn>();
|
|
163
183
|
|
|
164
184
|
/**
|
|
165
185
|
* Best-effort graceful interrupt of a chat's in-flight turn. Uses the SDK's
|
|
166
|
-
* native `Query.interrupt()`, which stops the agent loop
|
|
167
|
-
*
|
|
168
|
-
*
|
|
169
|
-
*
|
|
186
|
+
* native `Query.interrupt()`, which stops the agent loop — so the turn ends
|
|
187
|
+
* as a normal completion (turn_end + usage), NOT an error, and never trips
|
|
188
|
+
* the model-fallback retry path.
|
|
189
|
+
*
|
|
190
|
+
* The turn is marked BEFORE the native interrupt is fired, exactly as the
|
|
191
|
+
* shared `runtime/turn/turn-interrupt.ts` marks `state.turnTerminated`
|
|
192
|
+
* first: the mark, not the SDK's own close-out shape, is what makes the
|
|
193
|
+
* contract above true. No-op (returns false) when no turn is running.
|
|
170
194
|
*/
|
|
171
195
|
export async function interruptChatTurn(chatId: string): Promise<boolean> {
|
|
172
|
-
const
|
|
173
|
-
if (!
|
|
196
|
+
const active = activeQueries.get(chatId);
|
|
197
|
+
if (!active) return false;
|
|
198
|
+
active.interrupted = true;
|
|
174
199
|
try {
|
|
175
|
-
await qi.interrupt();
|
|
200
|
+
await active.qi.interrupt();
|
|
176
201
|
log("agent", `[${chatId}] turn interrupted by user`);
|
|
177
202
|
incrementCounter("sdk.turn_interrupted");
|
|
178
203
|
return true;
|
|
@@ -187,7 +212,7 @@ export async function interruptChatTurn(chatId: string): Promise<boolean> {
|
|
|
187
212
|
|
|
188
213
|
/** Get the active Query for a chat, if one is in flight. */
|
|
189
214
|
export function getActiveQuery(chatId: string): Query | undefined {
|
|
190
|
-
return activeQueries.get(chatId);
|
|
215
|
+
return activeQueries.get(chatId)?.qi;
|
|
191
216
|
}
|
|
192
217
|
|
|
193
218
|
// ── Internal state passed across recursive retry calls ──────────────────────
|
|
@@ -428,12 +453,6 @@ function accountFailedClaudeTurn(
|
|
|
428
453
|
model: string,
|
|
429
454
|
durationMs: number,
|
|
430
455
|
): void {
|
|
431
|
-
const sawResultUsage =
|
|
432
|
-
state.sdkInputTokens +
|
|
433
|
-
state.sdkOutputTokens +
|
|
434
|
-
state.sdkCacheRead +
|
|
435
|
-
state.sdkCacheWrite >
|
|
436
|
-
0;
|
|
437
456
|
accountFailedTurn({
|
|
438
457
|
backend: "claude",
|
|
439
458
|
chatId,
|
|
@@ -441,7 +460,7 @@ function accountFailedClaudeTurn(
|
|
|
441
460
|
durationMs,
|
|
442
461
|
model,
|
|
443
462
|
apiCalls: state.numApiCalls || live.calls,
|
|
444
|
-
usage: sawResultUsage
|
|
463
|
+
usage: sawResultUsage(state)
|
|
445
464
|
? turnUsageSnapshot(state)
|
|
446
465
|
: {
|
|
447
466
|
inputTokens: live.input,
|
|
@@ -452,6 +471,37 @@ function accountFailedClaudeTurn(
|
|
|
452
471
|
});
|
|
453
472
|
}
|
|
454
473
|
|
|
474
|
+
/** Whether the turn's `result` message reported any tokens at all. */
|
|
475
|
+
function sawResultUsage(state: StreamState): boolean {
|
|
476
|
+
return (
|
|
477
|
+
state.sdkInputTokens +
|
|
478
|
+
state.sdkOutputTokens +
|
|
479
|
+
state.sdkCacheRead +
|
|
480
|
+
state.sdkCacheWrite >
|
|
481
|
+
0
|
|
482
|
+
);
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/**
|
|
486
|
+
* Close out a turn the user stopped. An interrupt is a completion, so it
|
|
487
|
+
* accounts like one (`accountTurn`, not `accountFailedTurn` — nothing
|
|
488
|
+
* failed) — but the `result` message may never have landed, leaving
|
|
489
|
+
* `state.sdk*` at zero while the per-API-call accumulator holds what the
|
|
490
|
+
* turn really burned. Fold the accumulator in so a stop doesn't lose the
|
|
491
|
+
* tokens, and mark the turn terminated so the flow-violation re-prompt
|
|
492
|
+
* can't resurrect what the user just stopped (the same guarantee
|
|
493
|
+
* `runtime/turn/turn-interrupt.ts` gives the callback backends).
|
|
494
|
+
*/
|
|
495
|
+
function closeInterruptedTurn(state: StreamState, live: LiveUsage): void {
|
|
496
|
+
state.turnTerminated = true;
|
|
497
|
+
if (sawResultUsage(state)) return;
|
|
498
|
+
state.sdkInputTokens = live.input;
|
|
499
|
+
state.sdkOutputTokens = live.output;
|
|
500
|
+
state.sdkCacheRead = live.cacheRead;
|
|
501
|
+
state.sdkCacheWrite = live.cacheWrite;
|
|
502
|
+
if (!state.numApiCalls) state.numApiCalls = live.calls;
|
|
503
|
+
}
|
|
504
|
+
|
|
455
505
|
/**
|
|
456
506
|
* The aggregate `cache=NN%` can't distinguish a turn that reused the
|
|
457
507
|
* previous turn's prefix from one that re-wrote it — see
|
|
@@ -488,6 +538,105 @@ function reportCacheVerdict(
|
|
|
488
538
|
noteLookbackRisk(chatId, state.toolCalls);
|
|
489
539
|
}
|
|
490
540
|
|
|
541
|
+
// ── Stream phase ────────────────────────────────────────────────────────────
|
|
542
|
+
|
|
543
|
+
/** What the stream phase left for the rest of the turn to do. */
|
|
544
|
+
type StreamOutcome =
|
|
545
|
+
/** Run the normal post-stream phases (this includes every user stop). */
|
|
546
|
+
| { kind: "ok" }
|
|
547
|
+
/** A retry already ran to completion and yielded its own events. */
|
|
548
|
+
| { kind: "retried" }
|
|
549
|
+
/** Terminal failure — account for it and yield this `error` event. */
|
|
550
|
+
| { kind: "failed"; event: AgentEvent };
|
|
551
|
+
|
|
552
|
+
/**
|
|
553
|
+
* Drive the SDK stream to exhaustion and decide what its ending means.
|
|
554
|
+
* Owns the error recovery (retry decision, model fallback) and the
|
|
555
|
+
* user-interrupt contract; releases the watchdog timer and the
|
|
556
|
+
* `activeQueries` entry on every exit.
|
|
557
|
+
*/
|
|
558
|
+
async function* runTurnStream(inputs: {
|
|
559
|
+
chatId: string;
|
|
560
|
+
params: ChatRunParams;
|
|
561
|
+
internal: InternalState;
|
|
562
|
+
active: ActiveTurn;
|
|
563
|
+
state: StreamState;
|
|
564
|
+
live: LiveUsage;
|
|
565
|
+
watchdog: PostResultWatchdog;
|
|
566
|
+
/** Model the turn is accounted against. */
|
|
567
|
+
activeModel: string;
|
|
568
|
+
/** Model string the SDK attributes the result message's usage to. */
|
|
569
|
+
sdkModel: string;
|
|
570
|
+
}): AsyncGenerator<AgentEvent, StreamOutcome, void> {
|
|
571
|
+
const { chatId, active, state, live, watchdog } = inputs;
|
|
572
|
+
try {
|
|
573
|
+
yield* consumeSdkStream({
|
|
574
|
+
chatId,
|
|
575
|
+
qi: active.qi,
|
|
576
|
+
state,
|
|
577
|
+
live,
|
|
578
|
+
watchdog,
|
|
579
|
+
model: inputs.sdkModel,
|
|
580
|
+
pendingTools: new Map(),
|
|
581
|
+
});
|
|
582
|
+
// The SDK doesn't throw on API errors — it converts them into a
|
|
583
|
+
// synthetic assistant message and finishes the turn with an error-
|
|
584
|
+
// flagged result (usage limits, 429s, auth failures all land here).
|
|
585
|
+
// Rethrow so this turn takes the SAME path as a thrown SDK error
|
|
586
|
+
// instead of tripping the flow-violation re-prompt loop against an
|
|
587
|
+
// already-exhausted limit.
|
|
588
|
+
//
|
|
589
|
+
// …unless the user stopped this turn: an interrupted turn's result is
|
|
590
|
+
// flagged `is_error` with nothing but an `[ede_diagnostic]` line
|
|
591
|
+
// behind it, which `readResultError` then renders as bare trailing
|
|
592
|
+
// text or "Claude SDK turn failed (<subtype>)". That is the stop, not
|
|
593
|
+
// a failure — the turn closes as a completion.
|
|
594
|
+
if (state.resultErrorText && !active.interrupted) {
|
|
595
|
+
throw new Error(state.resultErrorText);
|
|
596
|
+
}
|
|
597
|
+
} catch (err) {
|
|
598
|
+
if (active.interrupted) {
|
|
599
|
+
// Same deal one layer down: after an interrupt the SDK re-labels the
|
|
600
|
+
// stream error with that error-result text. Quiet by design — no
|
|
601
|
+
// `logError`, no retry decision (which would classify the ede text,
|
|
602
|
+
// bump `errors.*`, and could fall back to another model for a turn
|
|
603
|
+
// the user just stopped), no `error` event. The shutdown drain takes
|
|
604
|
+
// this same path: its abort interrupts through here too.
|
|
605
|
+
log(
|
|
606
|
+
"agent",
|
|
607
|
+
`[${chatId}] turn ended by user interrupt: ${
|
|
608
|
+
err instanceof Error ? err.message : String(err)
|
|
609
|
+
}`,
|
|
610
|
+
);
|
|
611
|
+
incrementCounter("sdk.interrupt_closed_stream");
|
|
612
|
+
} else if (!watchdog.forceClosed) {
|
|
613
|
+
const { retried, classified } = yield* applyRetryDecisionStream({
|
|
614
|
+
err,
|
|
615
|
+
chatId,
|
|
616
|
+
activeModel: inputs.activeModel,
|
|
617
|
+
retried: inputs.internal.errorRetried ?? false,
|
|
618
|
+
buildRetryStream: retryStreamBuilder(inputs.params, inputs.internal),
|
|
619
|
+
// No backendLabel — historical claude-sdk log shape was un-prefixed.
|
|
620
|
+
});
|
|
621
|
+
if (retried) return { kind: "retried" };
|
|
622
|
+
logError("agent", `[${chatId}] SDK error: ${classified.message}`);
|
|
623
|
+
// Returning (rather than yielding) here defers the `error` event
|
|
624
|
+
// until after `finally` releases the watchdog timer and the
|
|
625
|
+
// activeQueries entry.
|
|
626
|
+
return {
|
|
627
|
+
kind: "failed",
|
|
628
|
+
event: { type: "error", error: classifiedToAgentError(classified) },
|
|
629
|
+
};
|
|
630
|
+
}
|
|
631
|
+
} finally {
|
|
632
|
+
watchdog.clear();
|
|
633
|
+
if (activeQueries.get(chatId) === active) {
|
|
634
|
+
activeQueries.delete(chatId);
|
|
635
|
+
}
|
|
636
|
+
}
|
|
637
|
+
return { kind: "ok" };
|
|
638
|
+
}
|
|
639
|
+
|
|
491
640
|
// ── Main chat-turn generator ────────────────────────────────────────────────
|
|
492
641
|
|
|
493
642
|
/**
|
|
@@ -498,7 +647,11 @@ function reportCacheVerdict(
|
|
|
498
647
|
* recurse via `yield*` and produce the retry's event stream
|
|
499
648
|
* transparently). On flow violation: `yield* runChatTurn(retry
|
|
500
649
|
* params)` — the recursive call owns its `incrementTurns`, the
|
|
501
|
-
* caller deliberately doesn't increment.
|
|
650
|
+
* caller deliberately doesn't increment. On a user interrupt
|
|
651
|
+
* (`interruptChatTurn`, which also backs the shutdown drain): the turn
|
|
652
|
+
* closes as a completion carrying the partial text and the real usage —
|
|
653
|
+
* never an `error` event, never a retry, whatever shape the SDK chose to
|
|
654
|
+
* end the stream in.
|
|
502
655
|
*/
|
|
503
656
|
export async function* runChatTurn(
|
|
504
657
|
params: ChatRunParams,
|
|
@@ -552,7 +705,8 @@ export async function* runChatTurn(
|
|
|
552
705
|
yield { type: "run_started" };
|
|
553
706
|
|
|
554
707
|
const qi = query({ prompt, options });
|
|
555
|
-
|
|
708
|
+
const active: ActiveTurn = { qi, interrupted: false };
|
|
709
|
+
activeQueries.set(chatId, active);
|
|
556
710
|
|
|
557
711
|
// Cold-start delivery-tool race: on the FIRST turn of a freshly-opened
|
|
558
712
|
// chat the hub's `${frontend}-tools` binding can still be `pending` when
|
|
@@ -567,59 +721,27 @@ export async function* runChatTurn(
|
|
|
567
721
|
const live = createLiveUsage();
|
|
568
722
|
const watchdog = createPostResultWatchdog(chatId, abortController, qi);
|
|
569
723
|
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
// Rethrow so this turn takes the SAME path as a thrown SDK error
|
|
585
|
-
// instead of tripping the flow-violation re-prompt loop against an
|
|
586
|
-
// already-exhausted limit.
|
|
587
|
-
if (state.resultErrorText) {
|
|
588
|
-
throw new Error(state.resultErrorText);
|
|
589
|
-
}
|
|
590
|
-
} catch (err) {
|
|
591
|
-
if (!watchdog.forceClosed) {
|
|
592
|
-
const { retried, classified } = yield* applyRetryDecisionStream({
|
|
593
|
-
err,
|
|
594
|
-
chatId,
|
|
595
|
-
activeModel,
|
|
596
|
-
retried: _internal.errorRetried ?? false,
|
|
597
|
-
buildRetryStream: retryStreamBuilder(params, _internal),
|
|
598
|
-
// No backendLabel — historical claude-sdk log shape was un-prefixed.
|
|
599
|
-
});
|
|
600
|
-
// The recursive stream already yielded its own usage + completed.
|
|
601
|
-
if (retried) return;
|
|
602
|
-
logError("agent", `[${chatId}] SDK error: ${classified.message}`);
|
|
603
|
-
// Defer the yield until after `finally` releases the watchdog timer
|
|
604
|
-
// and the activeQueries entry.
|
|
605
|
-
propagateError = {
|
|
606
|
-
type: "error",
|
|
607
|
-
error: classifiedToAgentError(classified),
|
|
608
|
-
};
|
|
609
|
-
}
|
|
610
|
-
} finally {
|
|
611
|
-
watchdog.clear();
|
|
612
|
-
if (activeQueries.get(chatId) === qi) {
|
|
613
|
-
activeQueries.delete(chatId);
|
|
614
|
-
}
|
|
615
|
-
}
|
|
616
|
-
|
|
617
|
-
if (propagateError) {
|
|
724
|
+
const outcome = yield* runTurnStream({
|
|
725
|
+
chatId,
|
|
726
|
+
params,
|
|
727
|
+
internal: _internal,
|
|
728
|
+
active,
|
|
729
|
+
state,
|
|
730
|
+
live,
|
|
731
|
+
watchdog,
|
|
732
|
+
activeModel,
|
|
733
|
+
sdkModel: options.model ?? activeModel,
|
|
734
|
+
});
|
|
735
|
+
// The recursive retry stream already yielded its own usage + completed.
|
|
736
|
+
if (outcome.kind === "retried") return;
|
|
737
|
+
if (outcome.kind === "failed") {
|
|
618
738
|
accountFailedClaudeTurn(chatId, state, live, activeModel, Date.now() - t0);
|
|
619
|
-
yield
|
|
739
|
+
yield outcome.event;
|
|
620
740
|
return;
|
|
621
741
|
}
|
|
622
742
|
|
|
743
|
+
if (active.interrupted) closeInterruptedTurn(state, live);
|
|
744
|
+
|
|
623
745
|
const durationMs = Date.now() - t0;
|
|
624
746
|
accountTurn({
|
|
625
747
|
chatId,
|
|
@@ -20,6 +20,7 @@ import { ALLOWED_TOOLS_BACKGROUND } from "../../core/constants.js";
|
|
|
20
20
|
import { EFFORT_MAP } from "./constants.js";
|
|
21
21
|
import { buildMcpServers, buildPluginMcpServers } from "./options.js";
|
|
22
22
|
import { warnIfBelowCacheMinimum } from "../runtime/cache/cache-telemetry.js";
|
|
23
|
+
import { emitAssistantText } from "../runtime/one-shot-hooks.js";
|
|
23
24
|
|
|
24
25
|
const DEFAULT_SUBPROCESS_KILL_GRACE_MS = 5 * 1000;
|
|
25
26
|
|
|
@@ -65,6 +66,7 @@ export async function runOneShotAgent(
|
|
|
65
66
|
contextLabel,
|
|
66
67
|
abortController,
|
|
67
68
|
appendLog,
|
|
69
|
+
onAssistantText,
|
|
68
70
|
} = params;
|
|
69
71
|
|
|
70
72
|
// Reasoning effort is opt-in for background runs (config `heartbeatEffort`
|
|
@@ -126,7 +128,7 @@ export async function runOneShotAgent(
|
|
|
126
128
|
// settlement figure the task table records.
|
|
127
129
|
let usage: OneShotUsage | undefined;
|
|
128
130
|
for await (const msg of qi) {
|
|
129
|
-
await formatAndAppendMessage(appendLog, msg);
|
|
131
|
+
await formatAndAppendMessage(appendLog, msg, onAssistantText);
|
|
130
132
|
if (msg.type === "result") {
|
|
131
133
|
const u = msg.usage;
|
|
132
134
|
usage = {
|
|
@@ -183,6 +185,7 @@ function assembleMcpServers(contextLabel: string): Record<string, unknown> {
|
|
|
183
185
|
async function formatAndAppendMessage(
|
|
184
186
|
appendLog: (text: string) => Promise<void>,
|
|
185
187
|
msg: SDKMessage,
|
|
188
|
+
onAssistantText?: OneShotAgentParams["onAssistantText"],
|
|
186
189
|
): Promise<void> {
|
|
187
190
|
try {
|
|
188
191
|
const ts = new Date().toISOString().slice(11, 19);
|
|
@@ -200,7 +203,11 @@ async function formatAndAppendMessage(
|
|
|
200
203
|
});
|
|
201
204
|
|
|
202
205
|
if (textBlocks.length > 0) {
|
|
203
|
-
|
|
206
|
+
const assistantText = textBlocks.join("\n");
|
|
207
|
+
// Report before the log write: an append failure (full disk, closed
|
|
208
|
+
// handle) must not also swallow the run's result for a hook caller.
|
|
209
|
+
emitAssistantText(onAssistantText, assistantText);
|
|
210
|
+
await appendLog(`\n## [${ts}] Assistant\n${assistantText}\n`);
|
|
204
211
|
}
|
|
205
212
|
if (toolUseBlocks.length > 0) {
|
|
206
213
|
await appendLog(`\n${toolUseBlocks.join("\n\n")}\n`);
|
|
@@ -208,6 +215,14 @@ async function formatAndAppendMessage(
|
|
|
208
215
|
break;
|
|
209
216
|
}
|
|
210
217
|
case "result": {
|
|
218
|
+
// Deliberately NOT reported through `onAssistantText`. The SDK's
|
|
219
|
+
// terminal `result` message restates the last assistant turn's text
|
|
220
|
+
// (`subtype: "success"`) or carries an error string (the
|
|
221
|
+
// `error_*` subtypes) — never anything the `assistant` case above
|
|
222
|
+
// has not already emitted. Reporting it too would hand every hook
|
|
223
|
+
// caller a duplicate final segment, and the truncated copy at that
|
|
224
|
+
// (2000 chars, below). The log keeps it because a run log wants the
|
|
225
|
+
// settlement line; a sub-agent result does not.
|
|
211
226
|
const result =
|
|
212
227
|
"result" in msg
|
|
213
228
|
? (msg as { result: string }).result
|
|
@@ -422,6 +422,15 @@ function readResultError(msg: SDKResultMessage, state: StreamState): void {
|
|
|
422
422
|
.filter((e) => typeof e === "string" && !e.startsWith("[ede_diagnostic]"))
|
|
423
423
|
.join("; ")
|
|
424
424
|
.slice(0, 500);
|
|
425
|
+
// A known startup failure (SDK ≥ 0.3.274) names its cause — the CLI
|
|
426
|
+
// never ran a turn, so nothing else in the result says why. Lead with
|
|
427
|
+
// it; the diagnostics text is the same line stderr carried.
|
|
428
|
+
if (msg.startup_failure_reason) {
|
|
429
|
+
state.resultErrorText =
|
|
430
|
+
`Claude Code failed to start (${msg.startup_failure_reason})` +
|
|
431
|
+
(diagnostics ? `: ${diagnostics}` : "");
|
|
432
|
+
return;
|
|
433
|
+
}
|
|
425
434
|
state.resultErrorText =
|
|
426
435
|
state.lastTrailingText.trim() ||
|
|
427
436
|
diagnostics ||
|
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
import type { OneShotAgentParams, OneShotUsage } from "../../core/types.js";
|
|
21
21
|
import { log, logWarn } from "../../util/log.js";
|
|
22
22
|
import { appendBackendSuffix } from "../runtime/index.js";
|
|
23
|
+
import { emitAssistantText } from "../runtime/one-shot-hooks.js";
|
|
23
24
|
import { ensureCodex, getCodexAuthInfo } from "./init.js";
|
|
24
25
|
import {
|
|
25
26
|
CODEX_SYSTEM_PROMPT_SUFFIX,
|
|
@@ -79,6 +80,7 @@ export async function runOneShotAgent(
|
|
|
79
80
|
contextLabel,
|
|
80
81
|
abortController,
|
|
81
82
|
appendLog,
|
|
83
|
+
onAssistantText,
|
|
82
84
|
} = params;
|
|
83
85
|
|
|
84
86
|
const codex = ensureCodex(contextLabel);
|
|
@@ -143,7 +145,7 @@ export async function runOneShotAgent(
|
|
|
143
145
|
let usage: OneShotUsage | undefined;
|
|
144
146
|
for await (const event of events) {
|
|
145
147
|
if (abortController.signal.aborted) break;
|
|
146
|
-
await appendCodexEvent(appendLog, event);
|
|
148
|
+
await appendCodexEvent(appendLog, event, onAssistantText);
|
|
147
149
|
if (event.type === "turn.completed") {
|
|
148
150
|
const u = (event as { usage?: Record<string, number> }).usage;
|
|
149
151
|
if (u) {
|
|
@@ -217,6 +219,7 @@ export async function runOneShotAgent(
|
|
|
217
219
|
async function appendCodexEvent(
|
|
218
220
|
appendLog: (text: string) => Promise<void>,
|
|
219
221
|
event: { type: string } & Record<string, unknown>,
|
|
222
|
+
onAssistantText?: OneShotAgentParams["onAssistantText"],
|
|
220
223
|
): Promise<void> {
|
|
221
224
|
const ts = new Date().toISOString().slice(11, 19);
|
|
222
225
|
|
|
@@ -260,7 +263,7 @@ async function appendCodexEvent(
|
|
|
260
263
|
case "item.completed": {
|
|
261
264
|
const item = (event as unknown as { item?: Record<string, unknown> })
|
|
262
265
|
.item;
|
|
263
|
-
if (item) await appendCodexItem(appendLog, item, ts);
|
|
266
|
+
if (item) await appendCodexItem(appendLog, item, ts, onAssistantText);
|
|
264
267
|
return;
|
|
265
268
|
}
|
|
266
269
|
default:
|
|
@@ -268,17 +271,28 @@ async function appendCodexEvent(
|
|
|
268
271
|
}
|
|
269
272
|
}
|
|
270
273
|
|
|
271
|
-
/**
|
|
274
|
+
/**
|
|
275
|
+
* Append one `ThreadItem` to the run log, and report the model's final
|
|
276
|
+
* answers to the run's optional `onAssistantText` consumer.
|
|
277
|
+
*
|
|
278
|
+
* Only `agent_message` items are reported: `reasoning` items are the model's
|
|
279
|
+
* thinking and the rest are tool/command/diff payloads, none of which is the
|
|
280
|
+
* run's answer.
|
|
281
|
+
*/
|
|
272
282
|
async function appendCodexItem(
|
|
273
283
|
appendLog: (text: string) => Promise<void>,
|
|
274
284
|
item: Record<string, unknown>,
|
|
275
285
|
ts: string,
|
|
286
|
+
onAssistantText?: OneShotAgentParams["onAssistantText"],
|
|
276
287
|
): Promise<void> {
|
|
277
288
|
const type = typeof item.type === "string" ? item.type : "unknown";
|
|
278
289
|
|
|
279
290
|
if (type === "agent_message") {
|
|
280
291
|
const text = typeof item.text === "string" ? item.text : "";
|
|
281
|
-
if (text)
|
|
292
|
+
if (text) {
|
|
293
|
+
emitAssistantText(onAssistantText, text);
|
|
294
|
+
await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
|
|
295
|
+
}
|
|
282
296
|
return;
|
|
283
297
|
}
|
|
284
298
|
|
|
@@ -42,6 +42,7 @@ import {
|
|
|
42
42
|
type RemoteAssistantInfo,
|
|
43
43
|
} from "./session-helpers.js";
|
|
44
44
|
import { appendBackendSuffix, sleep } from "../runtime/index.js";
|
|
45
|
+
import { emitAssistantText } from "../runtime/one-shot-hooks.js";
|
|
45
46
|
import { buildPermissionRuleset } from "./sessions.js";
|
|
46
47
|
|
|
47
48
|
// ── Client surface ──────────────────────────────────────────────────────────
|
|
@@ -108,6 +109,7 @@ export async function runRemoteOneShotAgent<
|
|
|
108
109
|
contextLabel,
|
|
109
110
|
abortController,
|
|
110
111
|
appendLog,
|
|
112
|
+
onAssistantText,
|
|
111
113
|
} = params;
|
|
112
114
|
const { label, errMsg } = bindings;
|
|
113
115
|
|
|
@@ -210,7 +212,7 @@ export async function runRemoteOneShotAgent<
|
|
|
210
212
|
: [];
|
|
211
213
|
|
|
212
214
|
for (const part of parts) {
|
|
213
|
-
await appendResponsePart(appendLog, part);
|
|
215
|
+
await appendResponsePart(appendLog, part, onAssistantText);
|
|
214
216
|
}
|
|
215
217
|
|
|
216
218
|
// The prompt response's assistant info carries the run's token usage —
|
|
@@ -250,17 +252,28 @@ export async function runRemoteOneShotAgent<
|
|
|
250
252
|
|
|
251
253
|
// ── Run-log rendering ───────────────────────────────────────────────────────
|
|
252
254
|
|
|
253
|
-
/**
|
|
255
|
+
/**
|
|
256
|
+
* Render one response part into the Markdown run log, and report the
|
|
257
|
+
* assistant's final text to the run's optional `onAssistantText` consumer.
|
|
258
|
+
*
|
|
259
|
+
* Only the `text` parts are reported: `reasoning` parts are the model's
|
|
260
|
+
* thinking and tool parts are call payloads, neither of which is the run's
|
|
261
|
+
* answer.
|
|
262
|
+
*/
|
|
254
263
|
async function appendResponsePart(
|
|
255
264
|
appendLog: (text: string) => Promise<void>,
|
|
256
265
|
part: Record<string, unknown>,
|
|
266
|
+
onAssistantText?: OneShotAgentParams["onAssistantText"],
|
|
257
267
|
): Promise<void> {
|
|
258
268
|
const ts = new Date().toISOString().slice(11, 19);
|
|
259
269
|
const type = typeof part.type === "string" ? part.type : "unknown";
|
|
260
270
|
|
|
261
271
|
if (type === "text") {
|
|
262
272
|
const text = typeof part.text === "string" ? part.text : "";
|
|
263
|
-
if (text)
|
|
273
|
+
if (text) {
|
|
274
|
+
emitAssistantText(onAssistantText, text);
|
|
275
|
+
await appendLog(`\n## [${ts}] Assistant\n${text}\n`);
|
|
276
|
+
}
|
|
264
277
|
return;
|
|
265
278
|
}
|
|
266
279
|
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One-shot run hooks — the shared, throw-proof call site for
|
|
3
|
+
* `OneShotAgentParams.onAssistantText`.
|
|
4
|
+
*
|
|
5
|
+
* Every backend's isolated runner (`BackgroundRunner.runOneShotAgent`) already
|
|
6
|
+
* renders the model's assistant text into the run log as markdown. The hook is
|
|
7
|
+
* the same text as data, so a caller (the sub-agent runner) can take a run's
|
|
8
|
+
* result without parsing the log back apart. It lives here rather than in each
|
|
9
|
+
* backend so all four report it identically: same "final answer only" meaning,
|
|
10
|
+
* same empty-string skip, same swallow-and-log on a throwing callback.
|
|
11
|
+
*
|
|
12
|
+
* Contract:
|
|
13
|
+
* - Final assistant text only. Reasoning/thinking blocks and tool-call
|
|
14
|
+
* payloads are log-only; they never reach the hook.
|
|
15
|
+
* - Synchronous. The runner does not await the consumer, so a hook that
|
|
16
|
+
* wants to do async work owns its own queueing.
|
|
17
|
+
* - Never throws into the run. A consumer bug must not abort a heartbeat.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import type { OneShotAgentParams } from "../../core/types.js";
|
|
21
|
+
import { logWarn } from "../../util/log.js";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Report one assistant text segment to the run's optional consumer.
|
|
25
|
+
*
|
|
26
|
+
* No-ops when there is no hook (the heartbeat/dream/cron path) or when the
|
|
27
|
+
* text is empty — the backends guard their log appends the same way, so the
|
|
28
|
+
* hook sees exactly the segments the log does.
|
|
29
|
+
*/
|
|
30
|
+
export function emitAssistantText(
|
|
31
|
+
onAssistantText: OneShotAgentParams["onAssistantText"],
|
|
32
|
+
text: string,
|
|
33
|
+
): void {
|
|
34
|
+
if (!onAssistantText || !text) return;
|
|
35
|
+
try {
|
|
36
|
+
onAssistantText(text);
|
|
37
|
+
} catch (err) {
|
|
38
|
+
logWarn(
|
|
39
|
+
"agent",
|
|
40
|
+
`one-shot onAssistantText hook threw (ignored): ${
|
|
41
|
+
err instanceof Error ? err.message : String(err)
|
|
42
|
+
}`,
|
|
43
|
+
);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
@@ -31,9 +31,10 @@
|
|
|
31
31
|
* They are deliberately not the same types. Two client arguments do not
|
|
32
32
|
* survive a process boundary and the wire shapes say so:
|
|
33
33
|
*
|
|
34
|
-
* - `OneShotAgentParams` carries an `AbortController` and
|
|
35
|
-
* `appendLog`
|
|
36
|
-
* subset; Phase 2 maps `appendLog` onto `log`
|
|
34
|
+
* - `OneShotAgentParams` carries an `AbortController` and two
|
|
35
|
+
* callbacks (`appendLog`, `onAssistantText`). `HostOneShotParams`
|
|
36
|
+
* is the serialisable subset; Phase 2 maps `appendLog` onto `log`
|
|
37
|
+
* notices, `onAssistantText` onto a notice of its own, and the
|
|
37
38
|
* abort onto an `interrupt`-shaped request.
|
|
38
39
|
* - `hello.config` is the `claude-sdk` slice of `TalonConfig` as
|
|
39
40
|
* JSON. The in-process client takes the real `TalonConfig` object.
|
|
@@ -59,9 +60,9 @@ export const AGENT_HOST_PROTOCOL_VERSION = 1;
|
|
|
59
60
|
// ── Shared payload shapes ───────────────────────────────────────────────────
|
|
60
61
|
|
|
61
62
|
/**
|
|
62
|
-
* The serialisable half of `OneShotAgentParams`. `abortController
|
|
63
|
-
* `appendLog` are host-local concerns (see the file
|
|
64
|
-
* else is exactly what a background run needs.
|
|
63
|
+
* The serialisable half of `OneShotAgentParams`. `abortController`,
|
|
64
|
+
* `appendLog` and `onAssistantText` are host-local concerns (see the file
|
|
65
|
+
* header); everything else is exactly what a background run needs.
|
|
65
66
|
*
|
|
66
67
|
* Unexported on purpose — it is reachable as
|
|
67
68
|
* `Extract<HostRequest, { type: "one_shot" }>["params"]`, and a second
|
|
@@ -103,6 +103,9 @@ export interface ChatBackend {
|
|
|
103
103
|
* trigger log-file producers keep their direct write path.
|
|
104
104
|
* Resolves with the run's token usage when the SDK reports it
|
|
105
105
|
* (the task table records it at settlement); void otherwise.
|
|
106
|
+
* Implementations must also honour the optional `onAssistantText`
|
|
107
|
+
* hook — the run's final answers as data, for callers (the
|
|
108
|
+
* sub-agent runner) that need a result rather than a markdown log.
|
|
106
109
|
* - `evictOrphanSubprocesses(label)` — backends that spawn
|
|
107
110
|
* per-run subprocesses (Claude SDK) implement this so a hung
|
|
108
111
|
* run can be force-cleaned after the abort grace window.
|
|
@@ -100,6 +100,16 @@ export function formatChildExit(
|
|
|
100
100
|
return lines.length > 0 ? `${head}; stderr: ${lines.join(" | ")}` : head;
|
|
101
101
|
}
|
|
102
102
|
|
|
103
|
+
/**
|
|
104
|
+
* Human-readable form of a child key for log lines. Keys join server name
|
|
105
|
+
* and chat id with a NUL byte (see `childKey` in index.ts) — unambiguous
|
|
106
|
+
* as a Map key, but it renders as `\u0000` in the JSON log.
|
|
107
|
+
*/
|
|
108
|
+
function describeKey(key: string): string {
|
|
109
|
+
const nul = key.indexOf("\u0000");
|
|
110
|
+
return nul === -1 ? key : `${key.slice(0, nul)} chat=${key.slice(nul + 1)}`;
|
|
111
|
+
}
|
|
112
|
+
|
|
103
113
|
/**
|
|
104
114
|
* Bookkeeping for a child process that has gone away, asked or not.
|
|
105
115
|
* Wired as the transport's `onclose` BEFORE `client.connect` so the SDK
|
|
@@ -125,7 +135,7 @@ function onChildClosed(key: string, transport: HubChildTransport): void {
|
|
|
125
135
|
: "died before registration";
|
|
126
136
|
logWarn(
|
|
127
137
|
"gateway",
|
|
128
|
-
`hub child ${key} ${phase} (pid ${transport.pid ?? "?"}): ${formatChildExit(record)}`,
|
|
138
|
+
`hub child ${describeKey(key)} ${phase} (pid ${transport.pid ?? "?"}): ${formatChildExit(record)}`,
|
|
129
139
|
);
|
|
130
140
|
}
|
|
131
141
|
|
|
@@ -199,7 +209,10 @@ async function spawnChild(key: string, spec: ChildSpec): Promise<ChildHandle> {
|
|
|
199
209
|
try {
|
|
200
210
|
await client.close();
|
|
201
211
|
} catch (err) {
|
|
202
|
-
logWarn(
|
|
212
|
+
logWarn(
|
|
213
|
+
"gateway",
|
|
214
|
+
`hub child ${describeKey(key)} close failed: ${String(err)}`,
|
|
215
|
+
);
|
|
203
216
|
}
|
|
204
217
|
})());
|
|
205
218
|
})(),
|
|
@@ -226,7 +239,10 @@ async function spawnChild(key: string, spec: ChildSpec): Promise<ChildHandle> {
|
|
|
226
239
|
};
|
|
227
240
|
|
|
228
241
|
children.set(key, entry);
|
|
229
|
-
log(
|
|
242
|
+
log(
|
|
243
|
+
"gateway",
|
|
244
|
+
`hub child started: ${describeKey(key)} (pid ${transport.pid ?? "?"})`,
|
|
245
|
+
);
|
|
230
246
|
return entry.handle;
|
|
231
247
|
}
|
|
232
248
|
|
|
@@ -327,9 +343,9 @@ function reapIdle(): void {
|
|
|
327
343
|
if (entry.lastActivity >= cutoff) continue;
|
|
328
344
|
children.delete(key);
|
|
329
345
|
entry.close().catch((err) => {
|
|
330
|
-
logError("gateway", `hub reap of ${key} failed`, err);
|
|
346
|
+
logError("gateway", `hub reap of ${describeKey(key)} failed`, err);
|
|
331
347
|
});
|
|
332
|
-
log("gateway", `hub child reaped (idle): ${key}`);
|
|
348
|
+
log("gateway", `hub child reaped (idle): ${describeKey(key)}`);
|
|
333
349
|
}
|
|
334
350
|
}
|
|
335
351
|
|
package/src/core/types.ts
CHANGED
|
@@ -161,6 +161,17 @@ export type OneShotAgentParams = {
|
|
|
161
161
|
abortController: AbortController;
|
|
162
162
|
/** Append a string to the run log (markdown). */
|
|
163
163
|
appendLog: (text: string) => Promise<void>;
|
|
164
|
+
/**
|
|
165
|
+
* Called with each assistant text segment as the run produces it (final
|
|
166
|
+
* answers, not reasoning/thinking, not tool-call payloads). Optional so
|
|
167
|
+
* heartbeat/dream/cron callers are untouched; a sub-agent runner uses it to
|
|
168
|
+
* capture the run's result without parsing the markdown log.
|
|
169
|
+
*
|
|
170
|
+
* Synchronous and best-effort: every backend invokes it through
|
|
171
|
+
* `emitAssistantText` (backend/runtime/one-shot-hooks.ts), which swallows
|
|
172
|
+
* and logs a throwing callback so a consumer bug can never fail the run.
|
|
173
|
+
*/
|
|
174
|
+
onAssistantText?: (text: string) => void;
|
|
164
175
|
};
|
|
165
176
|
|
|
166
177
|
/** How much cache telemetry a backend can surface in /status. */
|