talon-agent 3.34.0 → 3.34.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -2
- package/src/backend/claude-sdk/handler.ts +428 -415
- package/src/backend/codex/handler/message.ts +275 -366
- package/src/backend/codex/handler/rollout-accounting.ts +137 -0
- package/src/backend/openai-agents/handler/message.ts +282 -356
- package/src/backend/remote-server/chat-turn.ts +66 -122
- package/src/backend/remote-server/index.ts +4 -0
- package/src/backend/remote-server/mcp.ts +73 -9
- package/src/backend/remote-server/model-catalog/presentation.ts +269 -228
- package/src/backend/remote-server/sessions.ts +2 -2
- package/src/backend/shared/cache-telemetry.ts +17 -1
- package/src/backend/shared/handler-to-events.ts +11 -16
- package/src/backend/shared/index.ts +14 -15
- package/src/backend/shared/result-events.ts +30 -0
- package/src/backend/shared/turn-phases.ts +277 -0
- package/src/core/background/triggers/exit.ts +19 -2
- package/src/core/background/triggers/resume.ts +14 -7
- package/src/core/background/triggers/state.ts +9 -0
- package/src/core/mcp-hub/child-transport.ts +215 -0
- package/src/core/mcp-hub/children.ts +72 -13
- package/src/core/mcp-hub/index.ts +32 -13
- package/src/core/models/active-model.ts +29 -6
- package/src/core/weaver/shuttle.ts +4 -0
- package/src/frontend/telegram/admin.ts +75 -52
- package/src/storage/trigger-store.ts +8 -0
- package/src/util/watchdog.ts +32 -7
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Claude SDK chat-turn handler — natively emits `AgentEvent`s.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
* / context overflow / model fallback via
|
|
7
|
-
*
|
|
8
|
-
*
|
|
4
|
+
* Owns the SDK-specific half of a turn: the options build, the `query()`
|
|
5
|
+
* subprocess, the per-message event translation, error recovery
|
|
6
|
+
* (session expired / context overflow / model fallback via
|
|
7
|
+
* `applyRetryDecisionStream`) and the post-result watchdog. The phases
|
|
8
|
+
* after the stream — accounting, the trailing-prose contract, the result
|
|
9
|
+
* events — are the shared ones in `backend/shared/turn-phases.ts`.
|
|
9
10
|
*
|
|
10
11
|
* The exported async generator `runChatTurn` is what the factory wires
|
|
11
12
|
* onto `ChatBackend.runChatTurn` — no wrapper, no callback shim.
|
|
@@ -15,9 +16,6 @@ import { query } from "@anthropic-ai/claude-agent-sdk";
|
|
|
15
16
|
import {
|
|
16
17
|
getSession,
|
|
17
18
|
incrementTurns,
|
|
18
|
-
recordUsage,
|
|
19
|
-
setSessionId,
|
|
20
|
-
setSessionName,
|
|
21
19
|
updateLiveTurn,
|
|
22
20
|
} from "../../storage/sessions.js";
|
|
23
21
|
import { log, logError, logWarn } from "../../util/log.js";
|
|
@@ -25,7 +23,11 @@ import { traceMessage } from "../../util/trace.js";
|
|
|
25
23
|
import { incrementCounter } from "../../storage/metrics.js";
|
|
26
24
|
import { isTurnTerminator } from "../../core/tools/index.js";
|
|
27
25
|
|
|
28
|
-
import type {
|
|
26
|
+
import type {
|
|
27
|
+
Query,
|
|
28
|
+
SDKAssistantMessage,
|
|
29
|
+
SDKUserMessage,
|
|
30
|
+
} from "@anthropic-ai/claude-agent-sdk";
|
|
29
31
|
import {
|
|
30
32
|
type AgentEvent,
|
|
31
33
|
classifiedToAgentError,
|
|
@@ -50,27 +52,28 @@ import {
|
|
|
50
52
|
processStreamDelta,
|
|
51
53
|
processAssistantMessage,
|
|
52
54
|
processResultMessage,
|
|
55
|
+
type StreamState,
|
|
53
56
|
} from "./stream.js";
|
|
54
57
|
import {
|
|
55
58
|
formatUserPrompt,
|
|
56
59
|
prepareSystemPrompt,
|
|
57
|
-
extractSessionName,
|
|
58
|
-
detectFlowViolation,
|
|
59
|
-
FLOW_VIOLATION_MAX_RETRIES,
|
|
60
60
|
captureDeliveredText,
|
|
61
61
|
summarizeUsage,
|
|
62
62
|
buildDeliveryContract,
|
|
63
63
|
buildFlowViolationReminder,
|
|
64
64
|
buildFirstTurnReminder,
|
|
65
65
|
recordToolCall,
|
|
66
|
-
recordTurnMetrics,
|
|
67
|
-
recordFailedTurnAccounting,
|
|
68
|
-
recordFlowViolation,
|
|
69
66
|
formatTurnCache,
|
|
70
67
|
crossTurnVerdict,
|
|
71
68
|
priorLookbackOverflow,
|
|
72
69
|
noteLookbackRisk,
|
|
73
70
|
CACHE_LOOKBACK_BLOCKS,
|
|
71
|
+
accountTurn,
|
|
72
|
+
accountFailedTurn,
|
|
73
|
+
nameSessionFromFirstMessage,
|
|
74
|
+
enforceTrailingProse,
|
|
75
|
+
buildResultEvents,
|
|
76
|
+
turnUsageSnapshot,
|
|
74
77
|
} from "../shared/index.js";
|
|
75
78
|
|
|
76
79
|
// ── Post-result watchdog ────────────────────────────────────────────────────
|
|
@@ -101,13 +104,60 @@ const SDK_POST_RESULT_GRACE_MS = envMs(
|
|
|
101
104
|
DEFAULT_SDK_POST_RESULT_GRACE_MS,
|
|
102
105
|
);
|
|
103
106
|
|
|
107
|
+
type PostResultWatchdog = {
|
|
108
|
+
/** Start the grace timer once `result` is processed (idempotent). */
|
|
109
|
+
arm(): void;
|
|
110
|
+
clear(): void;
|
|
111
|
+
/** True once the timer fired and force-closed the iterator. */
|
|
112
|
+
readonly forceClosed: boolean;
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
function createPostResultWatchdog(
|
|
116
|
+
chatId: string,
|
|
117
|
+
abortController: AbortController,
|
|
118
|
+
qi: Query,
|
|
119
|
+
): PostResultWatchdog {
|
|
120
|
+
let timer: ReturnType<typeof setTimeout> | null = null;
|
|
121
|
+
let forceClosed = false;
|
|
122
|
+
return {
|
|
123
|
+
get forceClosed() {
|
|
124
|
+
return forceClosed;
|
|
125
|
+
},
|
|
126
|
+
arm() {
|
|
127
|
+
if (timer) return;
|
|
128
|
+
const t = setTimeout(() => {
|
|
129
|
+
forceClosed = true;
|
|
130
|
+
logWarn(
|
|
131
|
+
"agent",
|
|
132
|
+
`[${chatId}] SDK iterator stuck ${SDK_POST_RESULT_GRACE_MS}ms after result — aborting`,
|
|
133
|
+
);
|
|
134
|
+
incrementCounter("sdk.iterator_force_close_after_result");
|
|
135
|
+
try {
|
|
136
|
+
abortController.abort();
|
|
137
|
+
} catch {
|
|
138
|
+
/* abort() can throw if already aborted — ignore */
|
|
139
|
+
}
|
|
140
|
+
qi.return(undefined).catch(() => {
|
|
141
|
+
/* the generator may already be in a terminal state — ignore */
|
|
142
|
+
});
|
|
143
|
+
}, SDK_POST_RESULT_GRACE_MS);
|
|
144
|
+
t.unref();
|
|
145
|
+
timer = t;
|
|
146
|
+
},
|
|
147
|
+
clear() {
|
|
148
|
+
if (!timer) return;
|
|
149
|
+
clearTimeout(timer);
|
|
150
|
+
timer = null;
|
|
151
|
+
},
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
|
|
104
155
|
// ── Active query store ──────────────────────────────────────────────────────
|
|
105
156
|
// Holds the Query reference for each in-flight chat so gateway actions
|
|
106
157
|
// (e.g. reload_plugins) can call control methods like setMcpServers().
|
|
107
158
|
|
|
108
159
|
const activeQueries = new Map<string, Query>();
|
|
109
160
|
|
|
110
|
-
/** Get the active Query for a chat, if one is in flight. */
|
|
111
161
|
/**
|
|
112
162
|
* Best-effort graceful interrupt of a chat's in-flight turn. Uses the SDK's
|
|
113
163
|
* native `Query.interrupt()`, which stops the agent loop and closes the stream
|
|
@@ -132,6 +182,7 @@ export async function interruptChatTurn(chatId: string): Promise<boolean> {
|
|
|
132
182
|
}
|
|
133
183
|
}
|
|
134
184
|
|
|
185
|
+
/** Get the active Query for a chat, if one is in flight. */
|
|
135
186
|
export function getActiveQuery(chatId: string): Query | undefined {
|
|
136
187
|
return activeQueries.get(chatId);
|
|
137
188
|
}
|
|
@@ -140,6 +191,282 @@ export function getActiveQuery(chatId: string): Query | undefined {
|
|
|
140
191
|
|
|
141
192
|
type InternalState = { flowRetries?: number; errorRetried?: boolean };
|
|
142
193
|
|
|
194
|
+
// ── Stream translation ──────────────────────────────────────────────────────
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* Per-API-call usage accumulator for live mid-turn stats. Each assistant
|
|
198
|
+
* message carries its API call's usage as it lands; the authoritative
|
|
199
|
+
* per-turn totals still come from the final result message
|
|
200
|
+
* (`processResultMessage`) — this only feeds the live-turn overlay so
|
|
201
|
+
* /status moves while a long agentic turn runs, and the failure path so
|
|
202
|
+
* an errored turn's burn isn't lost.
|
|
203
|
+
*/
|
|
204
|
+
type LiveUsage = {
|
|
205
|
+
input: number;
|
|
206
|
+
output: number;
|
|
207
|
+
cacheRead: number;
|
|
208
|
+
cacheWrite: number;
|
|
209
|
+
calls: number;
|
|
210
|
+
};
|
|
211
|
+
|
|
212
|
+
function noteAssistantUsage(
|
|
213
|
+
chatId: string,
|
|
214
|
+
state: StreamState,
|
|
215
|
+
live: LiveUsage,
|
|
216
|
+
message: SDKAssistantMessage,
|
|
217
|
+
): void {
|
|
218
|
+
const u = message.message.usage;
|
|
219
|
+
if (!u) return;
|
|
220
|
+
live.input += u.input_tokens ?? 0;
|
|
221
|
+
live.output += u.output_tokens ?? 0;
|
|
222
|
+
live.cacheRead += u.cache_read_input_tokens ?? 0;
|
|
223
|
+
live.cacheWrite += u.cache_creation_input_tokens ?? 0;
|
|
224
|
+
live.calls += 1;
|
|
225
|
+
updateLiveTurn(chatId, {
|
|
226
|
+
inputTokens: live.input,
|
|
227
|
+
outputTokens: live.output,
|
|
228
|
+
cacheRead: live.cacheRead,
|
|
229
|
+
cacheWrite: live.cacheWrite,
|
|
230
|
+
// This call's full prompt = current context fill.
|
|
231
|
+
contextTokens:
|
|
232
|
+
(u.input_tokens ?? 0) +
|
|
233
|
+
(u.cache_read_input_tokens ?? 0) +
|
|
234
|
+
(u.cache_creation_input_tokens ?? 0),
|
|
235
|
+
contextWindow: state.contextWindow ?? 0,
|
|
236
|
+
numApiCalls: live.calls,
|
|
237
|
+
});
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
type StreamContext = {
|
|
241
|
+
chatId: string;
|
|
242
|
+
qi: Query;
|
|
243
|
+
state: StreamState;
|
|
244
|
+
live: LiveUsage;
|
|
245
|
+
watchdog: PostResultWatchdog;
|
|
246
|
+
/** Model the result message's usage is attributed to. */
|
|
247
|
+
model: string;
|
|
248
|
+
/** tool_use id → tool name for calls announced this turn. */
|
|
249
|
+
pendingTools: Map<string, string>;
|
|
250
|
+
};
|
|
251
|
+
|
|
252
|
+
/**
|
|
253
|
+
* Translate one assistant message: progress text segments BEFORE the tool
|
|
254
|
+
* calls they precede, so a model that says "let me check…" then calls a
|
|
255
|
+
* tool delivers the explanatory text first.
|
|
256
|
+
*/
|
|
257
|
+
function* translateAssistantMessage(
|
|
258
|
+
ctx: StreamContext,
|
|
259
|
+
message: SDKAssistantMessage,
|
|
260
|
+
): Generator<AgentEvent, void, void> {
|
|
261
|
+
const { chatId, state } = ctx;
|
|
262
|
+
const result = processAssistantMessage(message, state);
|
|
263
|
+
state.lastTrailingText = result.trailingText;
|
|
264
|
+
noteAssistantUsage(chatId, state, ctx.live, message);
|
|
265
|
+
|
|
266
|
+
for (const progress of result.progressTexts) {
|
|
267
|
+
yield { type: "assistant_message", text: progress };
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
for (const tool of result.tools) {
|
|
271
|
+
recordToolCall(chatId, tool.name, "claude");
|
|
272
|
+
const norm = captureDeliveredText(tool.name, tool.input);
|
|
273
|
+
if (norm) state.deliveredTextNorms.push(norm);
|
|
274
|
+
if (isTurnTerminator(tool.name, tool.input)) {
|
|
275
|
+
state.turnTerminated = true;
|
|
276
|
+
}
|
|
277
|
+
// Use the SDK's tool_use block id so the later tool_result (same id)
|
|
278
|
+
// correlates — a UI spinner opened on this event can only be closed
|
|
279
|
+
// by an event carrying the same id.
|
|
280
|
+
const toolId =
|
|
281
|
+
tool.id ||
|
|
282
|
+
`${tool.name}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
|
|
283
|
+
ctx.pendingTools.set(toolId, tool.name);
|
|
284
|
+
yield { type: "tool_call", id: toolId, name: tool.name, input: tool.input };
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* Tool executions come back as tool_result blocks on synthetic user
|
|
290
|
+
* messages. Emit the matching tool_result event — the other half of the
|
|
291
|
+
* lifecycle a tool_call opens (frontends show a spinner until it arrives).
|
|
292
|
+
*/
|
|
293
|
+
function* translateToolResults(
|
|
294
|
+
ctx: StreamContext,
|
|
295
|
+
message: SDKUserMessage,
|
|
296
|
+
): Generator<AgentEvent, void, void> {
|
|
297
|
+
for (const tr of extractToolResults(message)) {
|
|
298
|
+
const name = ctx.pendingTools.get(tr.toolUseId);
|
|
299
|
+
if (!name) continue; // not a tool we announced (subagent, replay)
|
|
300
|
+
ctx.pendingTools.delete(tr.toolUseId);
|
|
301
|
+
yield {
|
|
302
|
+
type: "tool_result",
|
|
303
|
+
id: tr.toolUseId,
|
|
304
|
+
name,
|
|
305
|
+
...(tr.error ? { error: tr.error } : {}),
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/** The `for await` over the SDK's message stream, one event type at a time. */
|
|
311
|
+
async function* consumeSdkStream(
|
|
312
|
+
ctx: StreamContext,
|
|
313
|
+
): AsyncGenerator<AgentEvent, void, void> {
|
|
314
|
+
const { state } = ctx;
|
|
315
|
+
for await (const message of ctx.qi) {
|
|
316
|
+
if (isSystemInit(message)) {
|
|
317
|
+
state.newSessionId = message.session_id;
|
|
318
|
+
continue;
|
|
319
|
+
}
|
|
320
|
+
if (isStreamEvent(message)) {
|
|
321
|
+
const emit = processStreamDelta(message, state);
|
|
322
|
+
if (emit) {
|
|
323
|
+
yield emit.phase === "text"
|
|
324
|
+
? { type: "text_delta", text: emit.text }
|
|
325
|
+
: { type: "reasoning", text: emit.text };
|
|
326
|
+
}
|
|
327
|
+
continue;
|
|
328
|
+
}
|
|
329
|
+
if (isAssistant(message)) {
|
|
330
|
+
yield* translateAssistantMessage(ctx, message);
|
|
331
|
+
continue;
|
|
332
|
+
}
|
|
333
|
+
if (isUserMessage(message)) {
|
|
334
|
+
yield* translateToolResults(ctx, message);
|
|
335
|
+
continue;
|
|
336
|
+
}
|
|
337
|
+
// The turn just moved the plan's usage — drop the cached windows so
|
|
338
|
+
// the next /status reads them again instead of showing pre-turn
|
|
339
|
+
// figures.
|
|
340
|
+
if (isRateLimitEvent(message)) {
|
|
341
|
+
invalidatePlanUsage();
|
|
342
|
+
continue;
|
|
343
|
+
}
|
|
344
|
+
if (isResult(message)) {
|
|
345
|
+
processResultMessage(message, state, ctx.model);
|
|
346
|
+
ctx.watchdog.arm();
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
// ── Turn-local helpers ──────────────────────────────────────────────────────
|
|
352
|
+
|
|
353
|
+
function createLiveUsage(): LiveUsage {
|
|
354
|
+
return { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, calls: 0 };
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* The recursive retry stream for `applyRetryDecisionStream`. A fallback
|
|
359
|
+
* model id is threaded through the retry's params (`params.model`
|
|
360
|
+
* outranks chat settings, so a transient `setChatModel` flip would be a
|
|
361
|
+
* silent no-op).
|
|
362
|
+
*/
|
|
363
|
+
function retryStreamBuilder(
|
|
364
|
+
params: ChatRunParams,
|
|
365
|
+
internal: InternalState,
|
|
366
|
+
): (fallbackModelId?: string) => AsyncIterable<AgentEvent> {
|
|
367
|
+
return (fallbackModelId) =>
|
|
368
|
+
runChatTurn(
|
|
369
|
+
fallbackModelId
|
|
370
|
+
? {
|
|
371
|
+
...params,
|
|
372
|
+
model: makeBareModelRef(
|
|
373
|
+
params.model.backend,
|
|
374
|
+
fallbackModelId,
|
|
375
|
+
"fallback",
|
|
376
|
+
),
|
|
377
|
+
}
|
|
378
|
+
: params,
|
|
379
|
+
{ ...internal, errorRetried: true },
|
|
380
|
+
);
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* First turn of a session is where flow violations cluster — the model
|
|
385
|
+
* hasn't seen the contract in action yet. One line appended to the turn-0
|
|
386
|
+
* user message (never the system prompt, so the cached prefix is
|
|
387
|
+
* untouched) pre-empts the 2x-token violation retry. Skipped on flow
|
|
388
|
+
* retries: those already carry the full reminder.
|
|
389
|
+
*/
|
|
390
|
+
function buildTurnPrompt(
|
|
391
|
+
params: ChatRunParams,
|
|
392
|
+
frontend: string | undefined,
|
|
393
|
+
previousTurns: number,
|
|
394
|
+
internal: InternalState,
|
|
395
|
+
): string {
|
|
396
|
+
let prompt = formatUserPrompt({
|
|
397
|
+
text: params.text,
|
|
398
|
+
senderName: params.senderName ?? "user",
|
|
399
|
+
senderHandle: params.senderHandle,
|
|
400
|
+
isGroup: params.isGroup,
|
|
401
|
+
messageId: params.messageId,
|
|
402
|
+
});
|
|
403
|
+
if (frontend && previousTurns === 0 && !internal.flowRetries) {
|
|
404
|
+
prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
|
|
405
|
+
}
|
|
406
|
+
return prompt;
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
/**
|
|
410
|
+
* Terminal failure — account for whatever the turn consumed before dying
|
|
411
|
+
* (failed turns burn real tokens). The result message never arrived, so
|
|
412
|
+
* `state.sdk*` is usually empty; fall back to the per-call accumulator.
|
|
413
|
+
*/
|
|
414
|
+
function accountFailedClaudeTurn(
|
|
415
|
+
chatId: string,
|
|
416
|
+
state: StreamState,
|
|
417
|
+
live: LiveUsage,
|
|
418
|
+
model: string,
|
|
419
|
+
durationMs: number,
|
|
420
|
+
): void {
|
|
421
|
+
const sawResultUsage =
|
|
422
|
+
state.sdkInputTokens +
|
|
423
|
+
state.sdkOutputTokens +
|
|
424
|
+
state.sdkCacheRead +
|
|
425
|
+
state.sdkCacheWrite >
|
|
426
|
+
0;
|
|
427
|
+
accountFailedTurn({
|
|
428
|
+
backend: "claude",
|
|
429
|
+
chatId,
|
|
430
|
+
state,
|
|
431
|
+
durationMs,
|
|
432
|
+
model,
|
|
433
|
+
apiCalls: state.numApiCalls || live.calls,
|
|
434
|
+
usage: sawResultUsage
|
|
435
|
+
? turnUsageSnapshot(state)
|
|
436
|
+
: {
|
|
437
|
+
inputTokens: live.input,
|
|
438
|
+
outputTokens: live.output,
|
|
439
|
+
cacheRead: live.cacheRead,
|
|
440
|
+
cacheWrite: live.cacheWrite,
|
|
441
|
+
},
|
|
442
|
+
});
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
/**
|
|
446
|
+
* The aggregate `cache=NN%` can't distinguish a turn that reused the
|
|
447
|
+
* previous turn's prefix from one that re-wrote it — see
|
|
448
|
+
* shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
|
|
449
|
+
* so warn when this turn's verdict proves the previous turn's overflow
|
|
450
|
+
* cost a prefix re-write, then record this turn's overflow for the next.
|
|
451
|
+
*/
|
|
452
|
+
function reportCacheVerdict(chatId: string, state: StreamState): void {
|
|
453
|
+
if (state.cacheStats) {
|
|
454
|
+
const overflow = priorLookbackOverflow(chatId);
|
|
455
|
+
if (
|
|
456
|
+
overflow !== undefined &&
|
|
457
|
+
crossTurnVerdict(state.cacheStats) === "miss"
|
|
458
|
+
) {
|
|
459
|
+
logWarn(
|
|
460
|
+
"agent",
|
|
461
|
+
`[${chatId}] previous turn emitted ~${overflow} content blocks ` +
|
|
462
|
+
`(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
|
|
463
|
+
`missed — that turn's cache write was likely never read`,
|
|
464
|
+
);
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
noteLookbackRisk(chatId, state.toolCalls);
|
|
468
|
+
}
|
|
469
|
+
|
|
143
470
|
// ── Main chat-turn generator ────────────────────────────────────────────────
|
|
144
471
|
|
|
145
472
|
/**
|
|
@@ -157,19 +484,17 @@ export async function* runChatTurn(
|
|
|
157
484
|
_internal: InternalState = {},
|
|
158
485
|
): AsyncIterable<AgentEvent> {
|
|
159
486
|
const config = getConfig();
|
|
160
|
-
|
|
161
|
-
const { chatId, text, senderName, senderHandle, isGroup } = params;
|
|
487
|
+
const { chatId, text } = params;
|
|
162
488
|
const session = getSession(chatId);
|
|
163
489
|
const t0 = Date.now();
|
|
164
490
|
|
|
165
|
-
// The chat's OWNING messaging frontend (falling back to the primary
|
|
166
|
-
//
|
|
167
|
-
//
|
|
168
|
-
// frontend
|
|
169
|
-
//
|
|
170
|
-
//
|
|
171
|
-
//
|
|
172
|
-
// be asserted.
|
|
491
|
+
// The chat's OWNING messaging frontend (falling back to the primary for
|
|
492
|
+
// cross-surface chats). Drives the delivery-contract suffix and the
|
|
493
|
+
// frontend-aware flow-violation text — the tool NAMES differ per
|
|
494
|
+
// frontend, and with tool servers scoped per chat the wrong contract
|
|
495
|
+
// would instruct a tool that doesn't exist. Empty in terminal mode,
|
|
496
|
+
// where no delivery tools exist and the strict tool-only contract must
|
|
497
|
+
// not be asserted.
|
|
173
498
|
const frontend: string | undefined = frontendsForChat(
|
|
174
499
|
chatId,
|
|
175
500
|
getActiveFrontends(),
|
|
@@ -177,11 +502,8 @@ export async function* runChatTurn(
|
|
|
177
502
|
|
|
178
503
|
// Frozen per-session prompt (keyed by session epoch) — stable across
|
|
179
504
|
// turns so the provider's prompt-cache prefix survives other chats'
|
|
180
|
-
// session resets.
|
|
181
|
-
//
|
|
182
|
-
// prompt, the highest-salience spot — so models see the tool-only
|
|
183
|
-
// flow before their first turn instead of discovering it via a
|
|
184
|
-
// [FLOW VIOLATION] retry.
|
|
505
|
+
// session resets. The delivery contract joins as the backend suffix —
|
|
506
|
+
// the tail of the static prompt, the highest-salience spot.
|
|
185
507
|
const preparedPrompt = prepareSystemPrompt({
|
|
186
508
|
config,
|
|
187
509
|
previousTurns: session.turns,
|
|
@@ -199,464 +521,155 @@ export async function* runChatTurn(
|
|
|
199
521
|
params.model.id,
|
|
200
522
|
preparedPrompt,
|
|
201
523
|
);
|
|
202
|
-
|
|
203
|
-
let prompt = formatUserPrompt({
|
|
204
|
-
text,
|
|
205
|
-
senderName: senderName ?? "user",
|
|
206
|
-
senderHandle,
|
|
207
|
-
isGroup,
|
|
208
|
-
messageId: params.messageId,
|
|
209
|
-
});
|
|
210
|
-
// First turn of a session is where flow violations cluster — the
|
|
211
|
-
// model hasn't seen the contract in action yet. One line appended to
|
|
212
|
-
// the turn-0 user message (never the system prompt, so the cached
|
|
213
|
-
// prefix is untouched) pre-empts the 2x-token violation retry.
|
|
214
|
-
// Skipped on flow retries: those already carry the full reminder.
|
|
215
|
-
if (frontend && session.turns === 0 && !_internal.flowRetries) {
|
|
216
|
-
prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
|
|
217
|
-
}
|
|
524
|
+
const prompt = buildTurnPrompt(params, frontend, session.turns, _internal);
|
|
218
525
|
log("agent", `[${chatId}] <- (${text.length} chars)`);
|
|
219
|
-
traceMessage(chatId, "in", text, {
|
|
526
|
+
traceMessage(chatId, "in", text, {
|
|
527
|
+
senderName: params.senderName,
|
|
528
|
+
isGroup: params.isGroup,
|
|
529
|
+
});
|
|
220
530
|
|
|
221
531
|
yield { type: "run_started" };
|
|
222
532
|
|
|
223
533
|
const qi = query({ prompt, options });
|
|
224
534
|
activeQueries.set(chatId, qi);
|
|
225
535
|
|
|
226
|
-
// Cold-start delivery-tool race
|
|
227
|
-
//
|
|
228
|
-
//
|
|
229
|
-
//
|
|
230
|
-
//
|
|
231
|
-
// `end_turn`/`send` are absent and the reply silently fails (or the model
|
|
232
|
-
// burns a `tool_search` round-trip to recover it). After a close+reconnect
|
|
233
|
-
// the hub binding is warm, connects instantly, and the tools are present —
|
|
234
|
-
// which is exactly why the bug "heals" on reconnect. Explicitly wait (bounded,
|
|
235
|
-
// non-throwing) for the delivery server to leave `pending` before consuming
|
|
236
|
-
// the stream. This mirrors the proven `refreshTools` gate and returns
|
|
237
|
-
// immediately on warm turns, so it adds no steady-state latency.
|
|
536
|
+
// Cold-start delivery-tool race: on the FIRST turn of a freshly-opened
|
|
537
|
+
// chat the hub's `${frontend}-tools` binding can still be `pending` when
|
|
538
|
+
// the model builds its tool list, so `end_turn`/`send` are absent and the
|
|
539
|
+
// reply silently fails. Wait (bounded, non-throwing) for it to connect;
|
|
540
|
+
// returns immediately on warm turns. Mirrors the `refreshTools` gate.
|
|
238
541
|
if (frontend) {
|
|
239
542
|
await waitForMcpServersReady(qi, [`${frontend}-tools`], 5_000, 100);
|
|
240
543
|
}
|
|
241
544
|
|
|
242
545
|
const state = createStreamState();
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
const pendingTools = new Map<string, string>();
|
|
246
|
-
|
|
247
|
-
// Per-API-call usage accumulator for live mid-turn stats. Each
|
|
248
|
-
// assistant message carries its API call's usage as it lands; the
|
|
249
|
-
// authoritative per-turn totals still come from the final result
|
|
250
|
-
// message (processResultMessage) — this only feeds the live-turn
|
|
251
|
-
// overlay so /status moves while a long agentic turn runs, and the
|
|
252
|
-
// failure path below so an errored turn's burn isn't lost.
|
|
253
|
-
const liveAcc = {
|
|
254
|
-
input: 0,
|
|
255
|
-
output: 0,
|
|
256
|
-
cacheRead: 0,
|
|
257
|
-
cacheWrite: 0,
|
|
258
|
-
calls: 0,
|
|
259
|
-
};
|
|
260
|
-
|
|
261
|
-
let postResultTimer: ReturnType<typeof setTimeout> | null = null;
|
|
262
|
-
let postResultForceClosed = false;
|
|
263
|
-
const armPostResultWatchdog = (): void => {
|
|
264
|
-
if (postResultTimer) return;
|
|
265
|
-
const t = setTimeout(() => {
|
|
266
|
-
postResultForceClosed = true;
|
|
267
|
-
logWarn(
|
|
268
|
-
"agent",
|
|
269
|
-
`[${chatId}] SDK iterator stuck ${SDK_POST_RESULT_GRACE_MS}ms after result — aborting`,
|
|
270
|
-
);
|
|
271
|
-
incrementCounter("sdk.iterator_force_close_after_result");
|
|
272
|
-
try {
|
|
273
|
-
abortController.abort();
|
|
274
|
-
} catch {
|
|
275
|
-
/* abort() can throw if already aborted — ignore */
|
|
276
|
-
}
|
|
277
|
-
qi.return(undefined).catch(() => {
|
|
278
|
-
/* the generator may already be in a terminal state — ignore */
|
|
279
|
-
});
|
|
280
|
-
}, SDK_POST_RESULT_GRACE_MS);
|
|
281
|
-
t.unref();
|
|
282
|
-
postResultTimer = t;
|
|
283
|
-
};
|
|
284
|
-
|
|
285
|
-
const captureIntoState = (
|
|
286
|
-
toolName: string,
|
|
287
|
-
input: Record<string, unknown>,
|
|
288
|
-
): void => {
|
|
289
|
-
const norm = captureDeliveredText(toolName, input);
|
|
290
|
-
if (norm) state.deliveredTextNorms.push(norm);
|
|
291
|
-
};
|
|
546
|
+
const live = createLiveUsage();
|
|
547
|
+
const watchdog = createPostResultWatchdog(chatId, abortController, qi);
|
|
292
548
|
|
|
293
549
|
let propagateError: AgentEvent | null = null;
|
|
294
550
|
try {
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
if (emit.phase === "text") {
|
|
305
|
-
yield { type: "text_delta", text: emit.text };
|
|
306
|
-
} else {
|
|
307
|
-
yield { type: "reasoning", text: emit.text };
|
|
308
|
-
}
|
|
309
|
-
}
|
|
310
|
-
continue;
|
|
311
|
-
}
|
|
312
|
-
|
|
313
|
-
if (isAssistant(message)) {
|
|
314
|
-
const result = processAssistantMessage(message, state);
|
|
315
|
-
state.lastTrailingText = result.trailingText;
|
|
316
|
-
|
|
317
|
-
const u = message.message.usage;
|
|
318
|
-
if (u) {
|
|
319
|
-
liveAcc.input += u.input_tokens ?? 0;
|
|
320
|
-
liveAcc.output += u.output_tokens ?? 0;
|
|
321
|
-
liveAcc.cacheRead += u.cache_read_input_tokens ?? 0;
|
|
322
|
-
liveAcc.cacheWrite += u.cache_creation_input_tokens ?? 0;
|
|
323
|
-
liveAcc.calls += 1;
|
|
324
|
-
updateLiveTurn(chatId, {
|
|
325
|
-
inputTokens: liveAcc.input,
|
|
326
|
-
outputTokens: liveAcc.output,
|
|
327
|
-
cacheRead: liveAcc.cacheRead,
|
|
328
|
-
cacheWrite: liveAcc.cacheWrite,
|
|
329
|
-
// This call's full prompt = current context fill.
|
|
330
|
-
contextTokens:
|
|
331
|
-
(u.input_tokens ?? 0) +
|
|
332
|
-
(u.cache_read_input_tokens ?? 0) +
|
|
333
|
-
(u.cache_creation_input_tokens ?? 0),
|
|
334
|
-
contextWindow: state.contextWindow ?? 0,
|
|
335
|
-
numApiCalls: liveAcc.calls,
|
|
336
|
-
});
|
|
337
|
-
}
|
|
338
|
-
|
|
339
|
-
// Emit progress text segments BEFORE the tool calls they
|
|
340
|
-
// precede, so a model that says "let me check…" then calls a
|
|
341
|
-
// tool delivers the explanatory text first.
|
|
342
|
-
for (const progress of result.progressTexts) {
|
|
343
|
-
yield { type: "assistant_message", text: progress };
|
|
344
|
-
}
|
|
345
|
-
|
|
346
|
-
for (const tool of result.tools) {
|
|
347
|
-
recordToolCall(chatId, tool.name, "claude");
|
|
348
|
-
captureIntoState(tool.name, tool.input);
|
|
349
|
-
if (isTurnTerminator(tool.name, tool.input)) {
|
|
350
|
-
state.turnTerminated = true;
|
|
351
|
-
}
|
|
352
|
-
// Use the SDK's tool_use block id so the later tool_result
|
|
353
|
-
// (same id) correlates — a UI spinner opened on this event
|
|
354
|
-
// can only be closed by an event carrying the same id.
|
|
355
|
-
const toolId =
|
|
356
|
-
tool.id ||
|
|
357
|
-
`${tool.name}-${Date.now()}-${Math.random()
|
|
358
|
-
.toString(36)
|
|
359
|
-
.slice(2, 8)}`;
|
|
360
|
-
pendingTools.set(toolId, tool.name);
|
|
361
|
-
yield {
|
|
362
|
-
type: "tool_call",
|
|
363
|
-
id: toolId,
|
|
364
|
-
name: tool.name,
|
|
365
|
-
input: tool.input,
|
|
366
|
-
};
|
|
367
|
-
}
|
|
368
|
-
continue;
|
|
369
|
-
}
|
|
370
|
-
|
|
371
|
-
// Tool executions come back as tool_result blocks on synthetic
|
|
372
|
-
// user messages. Emit the matching tool_result event — the other
|
|
373
|
-
// half of the lifecycle a tool_call opens (frontends show a
|
|
374
|
-
// spinner until it arrives).
|
|
375
|
-
if (isUserMessage(message)) {
|
|
376
|
-
for (const tr of extractToolResults(message)) {
|
|
377
|
-
const name = pendingTools.get(tr.toolUseId);
|
|
378
|
-
if (!name) continue; // not a tool we announced (subagent, replay)
|
|
379
|
-
pendingTools.delete(tr.toolUseId);
|
|
380
|
-
yield {
|
|
381
|
-
type: "tool_result",
|
|
382
|
-
id: tr.toolUseId,
|
|
383
|
-
name,
|
|
384
|
-
...(tr.error ? { error: tr.error } : {}),
|
|
385
|
-
};
|
|
386
|
-
}
|
|
387
|
-
continue;
|
|
388
|
-
}
|
|
389
|
-
|
|
390
|
-
// The turn just moved the plan's usage — drop the cached windows so
|
|
391
|
-
// the next /status reads them again instead of showing pre-turn
|
|
392
|
-
// figures.
|
|
393
|
-
if (isRateLimitEvent(message)) {
|
|
394
|
-
invalidatePlanUsage();
|
|
395
|
-
continue;
|
|
396
|
-
}
|
|
397
|
-
|
|
398
|
-
if (isResult(message)) {
|
|
399
|
-
processResultMessage(message, state, options.model ?? activeModel);
|
|
400
|
-
armPostResultWatchdog();
|
|
401
|
-
}
|
|
402
|
-
}
|
|
403
|
-
|
|
551
|
+
yield* consumeSdkStream({
|
|
552
|
+
chatId,
|
|
553
|
+
qi,
|
|
554
|
+
state,
|
|
555
|
+
live,
|
|
556
|
+
watchdog,
|
|
557
|
+
model: options.model ?? activeModel,
|
|
558
|
+
pendingTools: new Map(),
|
|
559
|
+
});
|
|
404
560
|
// The SDK doesn't throw on API errors — it converts them into a
|
|
405
561
|
// synthetic assistant message and finishes the turn with an error-
|
|
406
562
|
// flagged result (usage limits, 429s, auth failures all land here).
|
|
407
|
-
// Rethrow
|
|
408
|
-
//
|
|
409
|
-
//
|
|
410
|
-
// treated as a normal reply (which, with no delivery tool call,
|
|
411
|
-
// would trip the flow-violation re-prompt loop and burn more turns
|
|
412
|
-
// against an already-exhausted limit).
|
|
563
|
+
// Rethrow so this turn takes the SAME path as a thrown SDK error
|
|
564
|
+
// instead of tripping the flow-violation re-prompt loop against an
|
|
565
|
+
// already-exhausted limit.
|
|
413
566
|
if (state.resultErrorText) {
|
|
414
567
|
throw new Error(state.resultErrorText);
|
|
415
568
|
}
|
|
416
569
|
} catch (err) {
|
|
417
|
-
if (!
|
|
418
|
-
const buildRetryStream = (
|
|
419
|
-
fallbackModelId?: string,
|
|
420
|
-
): AsyncIterable<AgentEvent> =>
|
|
421
|
-
runChatTurn(
|
|
422
|
-
fallbackModelId
|
|
423
|
-
? {
|
|
424
|
-
...params,
|
|
425
|
-
model: makeBareModelRef(
|
|
426
|
-
params.model.backend,
|
|
427
|
-
fallbackModelId,
|
|
428
|
-
"fallback",
|
|
429
|
-
),
|
|
430
|
-
}
|
|
431
|
-
: params,
|
|
432
|
-
{ ..._internal, errorRetried: true },
|
|
433
|
-
);
|
|
570
|
+
if (!watchdog.forceClosed) {
|
|
434
571
|
const { retried, classified } = yield* applyRetryDecisionStream({
|
|
435
572
|
err,
|
|
436
573
|
chatId,
|
|
437
574
|
activeModel,
|
|
438
575
|
retried: _internal.errorRetried ?? false,
|
|
439
|
-
buildRetryStream,
|
|
576
|
+
buildRetryStream: retryStreamBuilder(params, _internal),
|
|
440
577
|
// No backendLabel — historical claude-sdk log shape was un-prefixed.
|
|
441
578
|
});
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
return;
|
|
445
|
-
}
|
|
579
|
+
// The recursive stream already yielded its own usage + completed.
|
|
580
|
+
if (retried) return;
|
|
446
581
|
logError("agent", `[${chatId}] SDK error: ${classified.message}`);
|
|
447
|
-
// Defer the
|
|
448
|
-
//
|
|
582
|
+
// Defer the yield until after `finally` releases the watchdog timer
|
|
583
|
+
// and the activeQueries entry.
|
|
449
584
|
propagateError = {
|
|
450
585
|
type: "error",
|
|
451
586
|
error: classifiedToAgentError(classified),
|
|
452
587
|
};
|
|
453
588
|
}
|
|
454
589
|
} finally {
|
|
455
|
-
|
|
456
|
-
clearTimeout(postResultTimer);
|
|
457
|
-
postResultTimer = null;
|
|
458
|
-
}
|
|
590
|
+
watchdog.clear();
|
|
459
591
|
if (activeQueries.get(chatId) === qi) {
|
|
460
592
|
activeQueries.delete(chatId);
|
|
461
593
|
}
|
|
462
594
|
}
|
|
463
595
|
|
|
464
596
|
if (propagateError) {
|
|
465
|
-
|
|
466
|
-
// dying (failed turns burn real tokens). The result message never
|
|
467
|
-
// arrived, so state.sdk* is usually empty; fall back to the per-call
|
|
468
|
-
// accumulator. Retried turns return earlier and never reach here.
|
|
469
|
-
const sawResultUsage =
|
|
470
|
-
state.sdkInputTokens +
|
|
471
|
-
state.sdkOutputTokens +
|
|
472
|
-
state.sdkCacheRead +
|
|
473
|
-
state.sdkCacheWrite >
|
|
474
|
-
0;
|
|
475
|
-
recordFailedTurnAccounting({
|
|
476
|
-
backend: "claude",
|
|
477
|
-
chatId,
|
|
478
|
-
durationMs: Date.now() - t0,
|
|
479
|
-
toolCalls: state.toolCalls,
|
|
480
|
-
apiCalls: state.numApiCalls || liveAcc.calls,
|
|
481
|
-
model: activeModel,
|
|
482
|
-
usage: sawResultUsage
|
|
483
|
-
? {
|
|
484
|
-
inputTokens: state.sdkInputTokens,
|
|
485
|
-
outputTokens: state.sdkOutputTokens,
|
|
486
|
-
cacheRead: state.sdkCacheRead,
|
|
487
|
-
cacheWrite: state.sdkCacheWrite,
|
|
488
|
-
}
|
|
489
|
-
: {
|
|
490
|
-
inputTokens: liveAcc.input,
|
|
491
|
-
outputTokens: liveAcc.output,
|
|
492
|
-
cacheRead: liveAcc.cacheRead,
|
|
493
|
-
cacheWrite: liveAcc.cacheWrite,
|
|
494
|
-
},
|
|
495
|
-
contextTokens: state.contextTokens,
|
|
496
|
-
contextWindow: state.contextWindow,
|
|
497
|
-
});
|
|
597
|
+
accountFailedClaudeTurn(chatId, state, live, activeModel, Date.now() - t0);
|
|
498
598
|
yield propagateError;
|
|
499
599
|
return;
|
|
500
600
|
}
|
|
501
601
|
|
|
502
|
-
// ── Persist session and usage ─────────────────────────────────────────────
|
|
503
|
-
|
|
504
602
|
const durationMs = Date.now() - t0;
|
|
505
|
-
|
|
603
|
+
accountTurn({
|
|
506
604
|
chatId,
|
|
507
605
|
backend: "claude",
|
|
606
|
+
state,
|
|
508
607
|
durationMs,
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
608
|
+
model: activeModel,
|
|
609
|
+
sessionId: state.newSessionId,
|
|
610
|
+
context: {
|
|
611
|
+
contextTokens: state.contextTokens,
|
|
612
|
+
contextWindow: state.contextWindow,
|
|
613
|
+
numApiCalls: state.numApiCalls,
|
|
614
|
+
costUsd: state.costUsd,
|
|
516
615
|
},
|
|
517
616
|
});
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
cacheWrite: state.sdkCacheWrite,
|
|
524
|
-
durationMs,
|
|
525
|
-
model: activeModel,
|
|
526
|
-
contextTokens: state.contextTokens,
|
|
527
|
-
contextWindow: state.contextWindow,
|
|
528
|
-
numApiCalls: state.numApiCalls,
|
|
529
|
-
costUsd: state.costUsd,
|
|
617
|
+
nameSessionFromFirstMessage({
|
|
618
|
+
chatId,
|
|
619
|
+
text,
|
|
620
|
+
previousTurns: session.turns,
|
|
621
|
+
isRetry: Boolean(_internal.flowRetries),
|
|
530
622
|
});
|
|
531
623
|
|
|
532
|
-
// Set a descriptive session name from the first message.
|
|
533
|
-
// Guard against flow-violation retries, which pass the reminder text as
|
|
534
|
-
// `text` — we only want the original user message, not the synthetic prompt.
|
|
535
|
-
if (session.turns === 0 && text && !_internal.flowRetries) {
|
|
536
|
-
const name = extractSessionName(text);
|
|
537
|
-
if (name) setSessionName(chatId, name);
|
|
538
|
-
}
|
|
539
|
-
|
|
540
|
-
// ── Trailing-prose contract + flow-violation retry ──────────────────────
|
|
541
|
-
|
|
542
624
|
// Only messaging frontends have a tool-only delivery contract to enforce.
|
|
543
625
|
// In terminal mode (`frontend` undefined) there are no delivery tools and
|
|
544
|
-
// trailing prose IS the reply
|
|
545
|
-
// `result.text`
|
|
546
|
-
//
|
|
547
|
-
|
|
548
|
-
// a tool that mode doesn't even register. Mirrors the `frontend`-gated
|
|
549
|
-
// delivery-contract suffix above.
|
|
626
|
+
// trailing prose IS the reply (the terminal renderer surfaces it via
|
|
627
|
+
// `result.text`); running the check there would re-prompt the model to
|
|
628
|
+
// call `end_turn`, a tool that mode doesn't even register.
|
|
629
|
+
const flowRetries = _internal.flowRetries ?? 0;
|
|
550
630
|
const violation = frontend
|
|
551
|
-
?
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
toolCalls: state.toolCalls,
|
|
556
|
-
retried: (_internal.flowRetries ?? 0) > 0,
|
|
557
|
-
retryCount: _internal.flowRetries ?? 0,
|
|
558
|
-
maxRetries: FLOW_VIOLATION_MAX_RETRIES,
|
|
631
|
+
? enforceTrailingProse({
|
|
632
|
+
chatId,
|
|
633
|
+
state,
|
|
634
|
+
flowRetries,
|
|
559
635
|
reminder: buildFlowViolationReminder(frontend),
|
|
560
636
|
})
|
|
561
|
-
:
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
violation.shouldRetry ? "retried" : "cap_exhausted",
|
|
637
|
+
: undefined;
|
|
638
|
+
if (violation?.violated && violation.shouldRetry) {
|
|
639
|
+
yield* runChatTurn(
|
|
640
|
+
{ ...params, text: violation.reminder },
|
|
641
|
+
{ ..._internal, flowRetries: flowRetries + 1 },
|
|
567
642
|
);
|
|
568
|
-
|
|
569
|
-
"agent",
|
|
570
|
-
`[${chatId}] flow violation: ${violation.reason}. ${
|
|
571
|
-
violation.shouldRetry
|
|
572
|
-
? "Re-prompting with reminder."
|
|
573
|
-
: `Retry cap (${FLOW_VIOLATION_MAX_RETRIES}) exhausted — accepting silent drop.`
|
|
574
|
-
}`,
|
|
575
|
-
);
|
|
576
|
-
|
|
577
|
-
if (violation.shouldRetry) {
|
|
578
|
-
yield* runChatTurn(
|
|
579
|
-
{ ...params, text: violation.reminder },
|
|
580
|
-
{
|
|
581
|
-
..._internal,
|
|
582
|
-
flowRetries: (_internal.flowRetries ?? 0) + 1,
|
|
583
|
-
},
|
|
584
|
-
);
|
|
585
|
-
return;
|
|
586
|
-
}
|
|
643
|
+
return;
|
|
587
644
|
}
|
|
588
645
|
|
|
589
646
|
// Reached the non-retry path — this turn counts as one user-visible turn.
|
|
590
647
|
incrementTurns(chatId);
|
|
591
648
|
|
|
592
|
-
// ── Build result events ──────────────────────────────────────────────────
|
|
593
|
-
|
|
594
649
|
state.allResponseText += state.currentBlockText;
|
|
650
|
+
reportCacheVerdict(chatId, state);
|
|
595
651
|
|
|
596
|
-
|
|
597
|
-
// previous turn's prefix from one that re-wrote it — see
|
|
598
|
-
// shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
|
|
599
|
-
// so warn when this turn's verdict proves the previous turn's overflow
|
|
600
|
-
// cost a prefix re-write, then record this turn's overflow for the next.
|
|
601
|
-
if (state.cacheStats) {
|
|
602
|
-
const overflow = priorLookbackOverflow(chatId);
|
|
603
|
-
if (
|
|
604
|
-
overflow !== undefined &&
|
|
605
|
-
crossTurnVerdict(state.cacheStats) === "miss"
|
|
606
|
-
) {
|
|
607
|
-
logWarn(
|
|
608
|
-
"agent",
|
|
609
|
-
`[${chatId}] previous turn emitted ~${overflow} content blocks ` +
|
|
610
|
-
`(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
|
|
611
|
-
`missed — that turn's cache write was likely never read`,
|
|
612
|
-
);
|
|
613
|
-
}
|
|
614
|
-
}
|
|
615
|
-
noteLookbackRisk(chatId, state.toolCalls);
|
|
616
|
-
|
|
652
|
+
const usage = turnUsageSnapshot(state);
|
|
617
653
|
log(
|
|
618
654
|
"agent",
|
|
619
|
-
`[${chatId}] -> (${summarizeUsage(
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
{
|
|
627
|
-
durationMs,
|
|
628
|
-
toolCalls: state.toolCalls,
|
|
629
|
-
...(state.cacheStats
|
|
630
|
-
? { suffix: formatTurnCache(state.cacheStats) }
|
|
631
|
-
: {}),
|
|
632
|
-
},
|
|
633
|
-
)})`,
|
|
655
|
+
`[${chatId}] -> (${summarizeUsage(usage, {
|
|
656
|
+
durationMs,
|
|
657
|
+
toolCalls: state.toolCalls,
|
|
658
|
+
...(state.cacheStats
|
|
659
|
+
? { suffix: formatTurnCache(state.cacheStats) }
|
|
660
|
+
: {}),
|
|
661
|
+
})})`,
|
|
634
662
|
);
|
|
635
663
|
traceMessage(chatId, "out", state.allResponseText, {
|
|
636
664
|
durationMs,
|
|
637
|
-
|
|
638
|
-
outputTokens: state.sdkOutputTokens,
|
|
639
|
-
cacheRead: state.sdkCacheRead,
|
|
640
|
-
cacheWrite: state.sdkCacheWrite,
|
|
665
|
+
...usage,
|
|
641
666
|
toolCalls: state.toolCalls,
|
|
642
667
|
model: activeModel,
|
|
643
668
|
});
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
cacheRead: state.sdkCacheRead,
|
|
649
|
-
cacheWrite: state.sdkCacheWrite,
|
|
669
|
+
yield* buildResultEvents({
|
|
670
|
+
text: state.allResponseText.trim(),
|
|
671
|
+
durationMs,
|
|
672
|
+
usage,
|
|
650
673
|
modelId: activeModel,
|
|
651
|
-
};
|
|
652
|
-
yield { type: "usage", usage };
|
|
653
|
-
yield {
|
|
654
|
-
type: "completed",
|
|
655
|
-
result: {
|
|
656
|
-
text: state.allResponseText.trim(),
|
|
657
|
-
durationMs,
|
|
658
|
-
usage,
|
|
659
|
-
modelId: activeModel,
|
|
660
|
-
},
|
|
661
|
-
};
|
|
674
|
+
});
|
|
662
675
|
}
|