talon-agent 3.34.0 → 3.35.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/LICENSE +1 -1
  2. package/README.md +0 -10
  3. package/package.json +6 -4
  4. package/src/backend/claude-sdk/handler.ts +428 -415
  5. package/src/backend/codex/auth.ts +40 -1
  6. package/src/backend/codex/handler/message.ts +290 -366
  7. package/src/backend/codex/handler/rollout-accounting.ts +137 -0
  8. package/src/backend/codex/plan-usage.ts +27 -2
  9. package/src/backend/kilo/handler/message.ts +2 -0
  10. package/src/backend/kilo/server.ts +1 -0
  11. package/src/backend/openai-agents/handler/message.ts +282 -356
  12. package/src/backend/opencode/handler/message.ts +2 -0
  13. package/src/backend/opencode/server.ts +1 -0
  14. package/src/backend/remote-server/chat-turn.ts +76 -122
  15. package/src/backend/remote-server/index.ts +4 -0
  16. package/src/backend/remote-server/lifecycle.ts +37 -0
  17. package/src/backend/remote-server/mcp.ts +73 -9
  18. package/src/backend/remote-server/model-catalog/presentation.ts +269 -228
  19. package/src/backend/remote-server/server-bindings.ts +12 -0
  20. package/src/backend/remote-server/sessions.ts +2 -2
  21. package/src/backend/remote-server/state.ts +8 -0
  22. package/src/backend/remote-server/turn.ts +93 -23
  23. package/src/backend/shared/cache-telemetry.ts +17 -1
  24. package/src/backend/shared/handler-to-events.ts +11 -16
  25. package/src/backend/shared/index.ts +14 -15
  26. package/src/backend/shared/result-events.ts +30 -0
  27. package/src/backend/shared/turn-phases.ts +277 -0
  28. package/src/bootstrap.ts +24 -10
  29. package/src/core/auth/expiry-monitor.ts +89 -0
  30. package/src/core/auth/login-flow.ts +247 -0
  31. package/src/core/auth/status.ts +193 -0
  32. package/src/core/background/triggers/exit.ts +19 -2
  33. package/src/core/background/triggers/resume.ts +14 -7
  34. package/src/core/background/triggers/state.ts +9 -0
  35. package/src/core/engine/gateway-actions/history.ts +12 -1
  36. package/src/core/errors.ts +1 -1
  37. package/src/core/mcp-hub/child-transport.ts +215 -0
  38. package/src/core/mcp-hub/children.ts +72 -13
  39. package/src/core/mcp-hub/index.ts +32 -13
  40. package/src/core/models/active-model.ts +29 -6
  41. package/src/core/tools/history.ts +10 -1
  42. package/src/core/weaver/shuttle.ts +4 -0
  43. package/src/frontend/shared/model-commands.ts +400 -0
  44. package/src/frontend/telegram/admin.ts +75 -52
  45. package/src/frontend/telegram/auth-panel.ts +205 -0
  46. package/src/frontend/telegram/callbacks/auth.ts +70 -0
  47. package/src/frontend/telegram/callbacks/index.ts +16 -0
  48. package/src/frontend/telegram/callbacks/whatsapp.ts +50 -0
  49. package/src/frontend/telegram/commands/auth.ts +58 -0
  50. package/src/frontend/telegram/commands/definitions.ts +4 -0
  51. package/src/frontend/telegram/commands/index.ts +10 -4
  52. package/src/frontend/telegram/commands/whatsapp-pairing.ts +121 -79
  53. package/src/frontend/whatsapp/actions/chat-info.ts +2 -36
  54. package/src/frontend/whatsapp/actions/history.ts +100 -0
  55. package/src/frontend/whatsapp/actions/index.ts +4 -1
  56. package/src/frontend/whatsapp/commands.ts +371 -0
  57. package/src/frontend/whatsapp/inbound.ts +24 -46
  58. package/src/frontend/whatsapp/index.ts +12 -1
  59. package/src/frontend/whatsapp/media-store.ts +20 -1
  60. package/src/frontend/whatsapp/message-store.ts +101 -24
  61. package/src/storage/history.ts +22 -1
  62. package/src/storage/repositories/history-repo.ts +14 -0
  63. package/src/storage/repositories/whatsapp-messages-repo.ts +101 -0
  64. package/src/storage/sql/history.sql +12 -0
  65. package/src/storage/sql/schema.sql +21 -0
  66. package/src/storage/sql/statements.generated.ts +50 -1
  67. package/src/storage/sql/whatsapp-messages.sql +29 -0
  68. package/src/storage/trigger-store.ts +8 -0
  69. package/src/storage/whatsapp-messages.ts +51 -0
  70. package/src/util/watchdog.ts +32 -7
@@ -1,12 +1,13 @@
1
1
  /**
2
2
  * Codex main message handler.
3
3
  *
4
- * Orchestrates the full turn lifecycle on top of `@openai/codex-sdk`'s
5
- * `Thread.runStreamed`. Shares the non-SDK-specific primitives with the other
6
- * backends via `../../shared/`. Codex-specific bits: reading the `runStreamed`
7
- * event stream, translating items into shared stream state (see `events.ts`),
8
- * resuming via `codex.resumeThread(id)`, the rollout-JSONL live/settle usage
9
- * accounting, and the ChatGPT-OAuth model-mismatch recovery ladder.
4
+ * Orchestrates the turn on top of `@openai/codex-sdk`'s `Thread.runStreamed`.
5
+ * Codex-specific bits: reading the `runStreamed` event stream, translating
6
+ * items into shared stream state (see `events.ts`), resuming via
7
+ * `codex.resumeThread(id)`, the rollout-JSONL live/settle usage accounting
8
+ * (`rollout-accounting.ts`), and the ChatGPT-OAuth model-mismatch recovery
9
+ * ladder. The post-stream phases are the shared ones in
10
+ * `backend/shared/turn-phases.ts`.
10
11
  */
11
12
 
12
13
  import type { Thread, Usage } from "@openai/codex-sdk";
@@ -14,9 +15,6 @@ import type { QueryParams, QueryResult } from "../../shared/handler-types.js";
14
15
  import {
15
16
  getSession,
16
17
  incrementTurns,
17
- recordUsage,
18
- setSessionName,
19
- setSessionId,
20
18
  resetSession,
21
19
  } from "../../../storage/sessions.js";
22
20
  import { getChatSettings } from "../../../storage/chat-settings.js";
@@ -26,20 +24,19 @@ import { incrementCounter } from "../../../storage/metrics.js";
26
24
 
27
25
  import {
28
26
  createStreamState,
29
- recordTokens,
30
27
  finalizeResponseText,
31
28
  formatUserPrompt,
32
29
  prepareSystemPrompt,
33
- extractSessionName,
34
- summarizeUsage,
35
30
  routeDelivery,
36
31
  buildDeliveryFailureReminder,
37
32
  TextBlockDeliveryError,
38
33
  applyRetryDecision,
39
- recordTurnMetrics,
40
- recordFailedTurnAccounting,
41
- pushLiveUsage,
42
34
  registerTurnInterrupt,
35
+ accountTurn,
36
+ accountFailedTurn,
37
+ nameSessionFromFirstMessage,
38
+ finishCallbackTurn,
39
+ type StreamState,
43
40
  } from "../../shared/index.js";
44
41
 
45
42
  import {
@@ -47,7 +44,6 @@ import {
47
44
  CODEX_DEFAULT_MODEL,
48
45
  CODEX_CHATGPT_DEFAULT_MODEL,
49
46
  CODEX_THREAD_PERMISSIONS,
50
- CODEX_LIVE_POLL_INTERVAL_MS,
51
47
  } from "../constants.js";
52
48
  import {
53
49
  frontendsForChat,
@@ -56,7 +52,9 @@ import {
56
52
  import { getState } from "../state.js";
57
53
  import { ensureCodex, getCodexAuthInfo } from "../init.js";
58
54
  import {
55
+ codexLoginExpiredError,
59
56
  isChatGptModelMismatchError,
57
+ isCodexRefreshTokenError,
60
58
  isSilentOAuthExitError,
61
59
  } from "../auth.js";
62
60
  import {
@@ -67,16 +65,37 @@ import {
67
65
  import { supportsReasoningLevel } from "../../../core/models/reasoning-levels.js";
68
66
  import { toCodexReasoningEffort } from "../effort.js";
69
67
  import { markOAuthIncompat } from "../oauth-incompat.js";
70
- import { readLastRolloutSnapshot } from "../token-usage.js";
71
68
  import { activeAborts } from "./state.js";
72
69
  import { CodexUsageExhaustedError, probeUsageExhausted } from "./usage.js";
73
- import { handleEvent } from "./events.js";
70
+ import { handleEvent, type HandleEventContext } from "./events.js";
71
+ import {
72
+ createRolloutAccounting,
73
+ type RolloutAccounting,
74
+ } from "./rollout-accounting.js";
74
75
 
75
76
  // ── Local utility ───────────────────────────────────────────────────────────
76
77
 
77
78
  const errMsg = (e: unknown): string =>
78
79
  e instanceof Error ? e.message : String(e);
79
80
 
81
+ /** The expected close on `end_turn` / a user interrupt: abort after the terminator. */
82
+ const isTerminatorAbort = (state: StreamState, err: unknown): boolean =>
83
+ state.turnTerminated &&
84
+ (errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)));
85
+
86
+ /**
87
+ * Swap an expired-login exit for the user-facing auth error before the
88
+ * shared retry ladder classifies it. The raw SDK text is the CLI banner
89
+ * plus a stderr dump (see `isCodexRefreshTokenError`); left alone it
90
+ * reads as an opaque exit-1 and the user never learns that `codex
91
+ * login` is the fix. Any other error passes through unchanged.
92
+ */
93
+ function surfaceLoginExpiry(err: unknown, turnFailedError?: string): unknown {
94
+ return isCodexRefreshTokenError(`${turnFailedError ?? ""} ${errMsg(err)}`)
95
+ ? codexLoginExpiredError(err)
96
+ : err;
97
+ }
98
+
80
99
  /**
81
100
  * One-shot ChatGPT-OAuth model-mismatch recovery.
82
101
  *
@@ -183,52 +202,24 @@ async function maybeFallbackForChatGptMismatch(
183
202
  return await handleMessage({ ...params, model: fallbackModel }, true);
184
203
  }
185
204
 
186
- // ── Main handler ────────────────────────────────────────────────────────────
205
+ // ── Model resolution ────────────────────────────────────────────────────────
187
206
 
188
- export async function handleMessage(
189
- params: QueryParams,
190
- _retried = false,
191
- ): Promise<QueryResult> {
192
- const state = getState();
193
- const config = state.config;
194
- if (!config) {
195
- throw new Error("Codex agent not initialized");
196
- }
197
- const codex = ensureCodex(params.chatId);
198
-
199
- const {
200
- chatId,
201
- text,
202
- senderName,
203
- senderHandle,
204
- isGroup,
205
- messageId,
206
- onTextBlock,
207
- onToolUse,
208
- onToolStart,
209
- onToolEnd,
210
- } = params;
211
- const t0 = Date.now();
212
- const session = getSession(chatId);
213
- const previousTurns = session.turns;
214
-
215
- // Resolve active model. Codex accepts arbitrary model strings; we
216
- // pass through whatever the chat settings hold. The fallback chain
217
- // is: chat-settings → config → auth-aware default. The auth-aware
218
- // default is `gpt-5-codex` when an API key is present, `gpt-5.5`
219
- // when only ChatGPT OAuth is configured (because `gpt-5-codex` is
220
- // rejected with a 400 on ChatGPT-mode accounts).
221
- const chatSettings = getChatSettings(chatId);
207
+ /**
208
+ * Codex accepts arbitrary model strings; we pass through whatever the
209
+ * caller resolved (chat-settings → config) and fall back to the auth-aware
210
+ * default: `gpt-5-codex` when an API key is present, `gpt-5.5` when only
211
+ * ChatGPT OAuth is configured (because `gpt-5-codex` is rejected with a
212
+ * 400 on ChatGPT-mode accounts). A model known to be OAuth-incompat on a
213
+ * ChatGPT-OAuth account is swapped pre-emptively rather than letting the
214
+ * first turn fail.
215
+ */
216
+ function resolveCodexModel(chatId: string, requested: string | undefined) {
222
217
  const authInfo = getCodexAuthInfo();
223
218
  const authAwareDefault =
224
219
  authInfo?.mode === "chatgpt"
225
220
  ? CODEX_CHATGPT_DEFAULT_MODEL
226
221
  : CODEX_DEFAULT_MODEL;
227
- const requestedModel =
228
- params.model ?? chatSettings.model ?? config.model ?? authAwareDefault;
229
- // If the resolved model is known OAuth-incompat AND we're on
230
- // ChatGPT OAuth, pre-emptively swap to the chatgpt-compatible
231
- // fallback rather than letting the first turn fail.
222
+ const requestedModel = requested ?? authAwareDefault;
232
223
  let activeModel = requestedModel;
233
224
  if (authInfo?.mode === "chatgpt" && isCodexOAuthIncompat(requestedModel)) {
234
225
  const fallback =
@@ -246,6 +237,188 @@ export async function handleMessage(
246
237
  }
247
238
  }
248
239
  log("agent", `[${chatId}] Codex model resolved: ${activeModel}`);
240
+ return activeModel;
241
+ }
242
+
243
+ /**
244
+ * Availability check (does this model offer the level?) then vocabulary
245
+ * translation (can Codex express it?) — the latter is shared with the
246
+ * one-shot path via `toCodexReasoningEffort` so the two can't drift.
247
+ */
248
+ async function buildThreadOptions(
249
+ activeModel: string,
250
+ requestedEffort: ReturnType<typeof getChatSettings>["effort"],
251
+ ) {
252
+ const activeModelInfo = await getModelInfo(activeModel).catch(
253
+ () => undefined,
254
+ );
255
+ const supportedReasoningLevels =
256
+ activeModelInfo?.supportedReasoningLevels ?? [];
257
+ const modelReasoningEffort =
258
+ requestedEffort &&
259
+ supportsReasoningLevel(requestedEffort, supportedReasoningLevels)
260
+ ? toCodexReasoningEffort(requestedEffort)
261
+ : undefined;
262
+ return {
263
+ activeModelInfo,
264
+ threadOptions: {
265
+ model: activeModel,
266
+ skipGitRepoCheck: true,
267
+ ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
268
+ ...CODEX_THREAD_PERMISSIONS,
269
+ },
270
+ };
271
+ }
272
+
273
+ // ── Stream loop ─────────────────────────────────────────────────────────────
274
+
275
+ function createEventContext(
276
+ params: QueryParams,
277
+ state: StreamState,
278
+ ): HandleEventContext {
279
+ return {
280
+ state,
281
+ seenToolCallIds: new Set<string>(),
282
+ startedToolIds: new Set<string>(),
283
+ codexToolMetrics: { count: 0 },
284
+ onTextBlock: params.onTextBlock,
285
+ onToolUse: params.onToolUse,
286
+ onToolStart: params.onToolStart,
287
+ onToolEnd: params.onToolEnd,
288
+ chatId: params.chatId,
289
+ };
290
+ }
291
+
292
+ /** What the stream reported, readable mid-loop by the failure path too. */
293
+ type CodexStreamOutcome = {
294
+ usage: Usage | null;
295
+ turnFailedError: string | undefined;
296
+ };
297
+
298
+ async function driveCodexStream(inputs: {
299
+ thread: Thread;
300
+ inputText: string;
301
+ abortController: AbortController;
302
+ eventContext: HandleEventContext;
303
+ rollout: RolloutAccounting;
304
+ outcome: CodexStreamOutcome;
305
+ }): Promise<void> {
306
+ const { abortController, eventContext, rollout, outcome } = inputs;
307
+ const { state, chatId } = eventContext;
308
+ const { events } = await inputs.thread.runStreamed(inputs.inputText, {
309
+ signal: abortController.signal,
310
+ });
311
+
312
+ for await (const event of events) {
313
+ if (abortController.signal.aborted && !state.turnTerminated) break;
314
+ handleEvent(event, eventContext);
315
+
316
+ if (event.type === "thread.started") {
317
+ rollout.threadId = event.thread_id;
318
+ } else if (event.type === "turn.completed") {
319
+ outcome.usage = event.usage;
320
+ } else if (event.type === "turn.failed") {
321
+ outcome.turnFailedError = event.error.message;
322
+ } else if (event.type === "error") {
323
+ outcome.turnFailedError = event.message;
324
+ }
325
+
326
+ rollout.pollLive();
327
+
328
+ // Terminator-driven abort: a delivery tool already shipped the
329
+ // reply via the bridge. Cancel further model generation to skip
330
+ // the wrap-up round-trip Codex would otherwise burn.
331
+ if (state.turnTerminated && !abortController.signal.aborted) {
332
+ log("agent", `[${chatId}] terminator fired — aborting Codex turn`);
333
+ try {
334
+ abortController.abort();
335
+ } catch (err) {
336
+ logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
337
+ }
338
+ }
339
+ }
340
+ }
341
+
342
+ /**
343
+ * The failure ladder: ChatGPT-OAuth mismatch recovery, then the shared
344
+ * retry decision, then terminal-failure accounting and the throw.
345
+ */
346
+ async function recoverCodexFailure(inputs: {
347
+ err: unknown;
348
+ params: QueryParams;
349
+ retried: boolean;
350
+ activeModel: string;
351
+ state: StreamState;
352
+ rollout: RolloutAccounting;
353
+ outcome: CodexStreamOutcome;
354
+ toolCalls: number;
355
+ t0: number;
356
+ }): Promise<QueryResult> {
357
+ const { err, params, retried, activeModel, state, rollout, outcome } = inputs;
358
+ const { chatId } = params;
359
+
360
+ // Check both the captured event-stream message and the thrown error —
361
+ // Codex SDK surfaces it via both channels. Only use the thread ID from
362
+ // this run.
363
+ const fallback = await maybeFallbackForChatGptMismatch(
364
+ `${outcome.turnFailedError ?? ""} ${errMsg(err)}`,
365
+ activeModel,
366
+ params,
367
+ retried,
368
+ chatId,
369
+ rollout.threadId,
370
+ );
371
+ if (fallback) return fallback;
372
+
373
+ const decision = await applyRetryDecision({
374
+ err: surfaceLoginExpiry(err, outcome.turnFailedError),
375
+ chatId,
376
+ activeModel,
377
+ retried,
378
+ params,
379
+ recurseWithRetried: (p) => handleMessage(p, true),
380
+ backendLabel: "Codex",
381
+ resetNoun: "thread",
382
+ });
383
+ if (decision.retry) return decision.retry;
384
+
385
+ // Terminal failure — recover whatever usage the rollout recorded
386
+ // before the turn died, then account for it.
387
+ await rollout.settle(outcome.usage).catch(() => {});
388
+ accountFailedTurn({
389
+ backend: "codex",
390
+ chatId,
391
+ state,
392
+ durationMs: Date.now() - inputs.t0,
393
+ model: activeModel,
394
+ toolCalls: inputs.toolCalls,
395
+ });
396
+ logError("agent", `[${chatId}] Codex error: ${decision.classified.message}`);
397
+ throw decision.classified;
398
+ }
399
+
400
+ // ── Main handler ────────────────────────────────────────────────────────────
401
+
402
+ export async function handleMessage(
403
+ params: QueryParams,
404
+ _retried = false,
405
+ ): Promise<QueryResult> {
406
+ const config = getState().config;
407
+ if (!config) {
408
+ throw new Error("Codex agent not initialized");
409
+ }
410
+ const codex = ensureCodex(params.chatId);
411
+
412
+ const { chatId, text, senderName, senderHandle, isGroup, messageId } = params;
413
+ const t0 = Date.now();
414
+ const session = getSession(chatId);
415
+ const previousTurns = session.turns;
416
+
417
+ const chatSettings = getChatSettings(chatId);
418
+ const activeModel = resolveCodexModel(
419
+ chatId,
420
+ params.model ?? chatSettings.model ?? config.model,
421
+ );
249
422
 
250
423
  // Per-session frozen prompt + Codex-specific delivery suffix.
251
424
  const { text: systemPrompt } = prepareSystemPrompt({
@@ -270,49 +443,22 @@ export async function handleMessage(
270
443
  log("agent", `[${chatId}] <- (${text.length} chars)`);
271
444
  traceMessage(chatId, "in", text, { senderName, isGroup });
272
445
 
273
- // Resume an existing Codex thread or start a fresh one. Codex persists
274
- // threads under `~/.codex/sessions/`; we store the thread id in Talon's
275
- // session storage so `resumeThread()` keeps the conversation continuous.
276
- const activeModelInfo = await getModelInfo(activeModel).catch(
277
- () => undefined,
446
+ // Resume the stored Codex thread (persisted under `~/.codex/sessions/`).
447
+ const { activeModelInfo, threadOptions } = await buildThreadOptions(
448
+ activeModel,
449
+ chatSettings.effort,
278
450
  );
279
- const supportedReasoningLevels =
280
- activeModelInfo?.supportedReasoningLevels ?? [];
281
- const requestedEffort = chatSettings.effort;
282
- // Availability check (does this model offer the level?) then vocabulary
283
- // translation (can Codex express it?) — the latter is shared with the
284
- // one-shot path via `toCodexReasoningEffort` so the two can't drift.
285
- const modelReasoningEffort =
286
- requestedEffort &&
287
- supportsReasoningLevel(requestedEffort, supportedReasoningLevels)
288
- ? toCodexReasoningEffort(requestedEffort)
289
- : undefined;
290
- const threadOptions = {
291
- model: activeModel,
292
- skipGitRepoCheck: true,
293
- ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
294
- ...CODEX_THREAD_PERMISSIONS,
295
- };
296
451
  const thread: Thread = session.sessionId
297
452
  ? codex.resumeThread(session.sessionId, threadOptions)
298
453
  : codex.startThread(threadOptions);
299
454
 
300
- // Baseline cumulative token totals from the rollout JSONL, captured
301
- // BEFORE the turn runs. `total_token_usage` accumulates across the
302
- // whole session file, so this turn's usage = post-turn totals minus
303
- // this baseline. Fresh threads have no rollout yet → zero baseline.
304
- // `null` = resumed thread whose baseline couldn't be read.
305
- const baselineTotals = session.sessionId
306
- ? ((await readLastRolloutSnapshot(session.sessionId).catch(() => null))
307
- ?.totals ?? null)
308
- : { inputTokens: 0, cachedInputTokens: 0, outputTokens: 0 };
309
-
310
- // Bind the stream state to the chat so token mutators mirror counts
311
- // into the live-turn overlay — /status updates while the turn runs.
455
+ // Chat-bound state mirrors counts into the live-turn overlay.
312
456
  const streamState = createStreamState(chatId);
313
- const seenToolCallIds = new Set<string>();
314
- const startedToolIds = new Set<string>();
315
- const codexToolMetrics = { count: 0 };
457
+ const rollout = await createRolloutAccounting({
458
+ state: streamState,
459
+ sessionId: session.sessionId,
460
+ });
461
+ const eventContext = createEventContext(params, streamState);
316
462
  const abortController = new AbortController();
317
463
  activeAborts.set(chatId, abortController);
318
464
  // A user interrupt is a synthetic turn terminator: marking the flag
@@ -324,217 +470,42 @@ export async function handleMessage(
324
470
  abortController.abort();
325
471
  });
326
472
 
327
- let usage: Usage | null = null;
328
- let turnFailedError: string | undefined;
329
- let resolvedThreadId: string | undefined;
330
-
331
- // Throttled mid-turn rollout poll. The Codex CLI appends a
332
- // `token_count` event to the rollout JSONL after every API call, so
333
- // tailing it during the turn gives live context-fill / token / API-call
334
- // stats long before `turn.completed`. Fire-and-forget with an in-flight
335
- // guard — never blocks the event loop, never throws.
336
- let rolloutPollInFlight = false;
337
- let lastRolloutPollAt = 0;
338
- const pollRolloutForLiveStats = () => {
339
- if (!resolvedThreadId || rolloutPollInFlight) return;
340
- const now = Date.now();
341
- if (now - lastRolloutPollAt < CODEX_LIVE_POLL_INTERVAL_MS) return;
342
- rolloutPollInFlight = true;
343
- lastRolloutPollAt = now;
344
- readLastRolloutSnapshot(resolvedThreadId)
345
- .then((snap) => {
346
- if (!snap) return;
347
- if (snap.usage) {
348
- streamState.contextTokens = snap.usage.contextTokens;
349
- if (snap.usage.contextWindow) {
350
- streamState.contextWindow = snap.usage.contextWindow;
351
- }
352
- }
353
- if (typeof snap.numApiCalls === "number") {
354
- streamState.numApiCalls = snap.numApiCalls;
355
- }
356
- // Same delta-vs-baseline math as the post-loop accounting; the
357
- // final pass recomputes and overwrites, so a torn mid-turn read
358
- // can't corrupt the committed numbers.
359
- if (snap.totals && baselineTotals) {
360
- streamState.sdkInputTokens = Math.max(
361
- 0,
362
- snap.totals.inputTokens - baselineTotals.inputTokens,
363
- );
364
- streamState.sdkOutputTokens = Math.max(
365
- 0,
366
- snap.totals.outputTokens - baselineTotals.outputTokens,
367
- );
368
- streamState.sdkCacheRead = Math.max(
369
- 0,
370
- snap.totals.cachedInputTokens - baselineTotals.cachedInputTokens,
371
- );
372
- }
373
- pushLiveUsage(streamState);
374
- })
375
- .catch(() => {})
376
- .finally(() => {
377
- rolloutPollInFlight = false;
378
- });
379
- };
380
-
381
- // Final authoritative usage settlement — shared by the success post-loop
382
- // and the terminal-failure path so failed turns account for the tokens
383
- // they burned too. Codex's `turn.completed.usage` is CUMULATIVE across
384
- // every API call in the turn — never in the per-turn units the shared
385
- // stream state (and everything downstream: /status, the companion's
386
- // per-message counts) speaks. The rollout JSONL's totals diffed against
387
- // the pre-turn baseline are this turn's real usage; that is the ONLY
388
- // authoritative source. The SDK figure is a last-resort fallback when
389
- // the rollout can't be read, and it overstates multi-call turns.
390
- const settleUsageAccounting = async (): Promise<void> => {
391
- const last = resolvedThreadId
392
- ? await readLastRolloutSnapshot(resolvedThreadId).catch(() => null)
393
- : null;
394
- if (last?.usage) {
395
- streamState.contextTokens = last.usage.contextTokens;
396
- if (last.usage.contextWindow) {
397
- streamState.contextWindow = last.usage.contextWindow;
398
- }
399
- }
400
- if (typeof last?.numApiCalls === "number") {
401
- streamState.numApiCalls = last.numApiCalls;
402
- }
403
- if (last?.totals && baselineTotals) {
404
- recordTokens(streamState, {
405
- inputTokens: last.totals.inputTokens - baselineTotals.inputTokens,
406
- outputTokens: last.totals.outputTokens - baselineTotals.outputTokens,
407
- cacheRead:
408
- last.totals.cachedInputTokens - baselineTotals.cachedInputTokens,
409
- cacheWrite: 0, // Codex doesn't report cache writes
410
- });
411
- } else if (usage) {
412
- recordTokens(streamState, {
413
- inputTokens: usage.input_tokens,
414
- outputTokens: usage.output_tokens,
415
- cacheRead: usage.cached_input_tokens,
416
- cacheWrite: 0, // Codex doesn't report cache writes
417
- });
418
- }
473
+ const outcome: CodexStreamOutcome = {
474
+ usage: null,
475
+ turnFailedError: undefined,
419
476
  };
420
-
421
477
  const setupMs = Date.now() - t0;
422
478
  let turnMs = 0;
423
479
 
424
480
  try {
425
481
  const turnStart = Date.now();
426
-
427
- // Codex's SDK does not expose `system` directly on `runStreamed`;
428
- // system prompts are baked at thread creation via the CLI's config.
429
- // Talon-side workaround: prepend the system prompt to the user prompt
430
- // as a fenced block on the first turn only. Subsequent turns inherit
431
- // instructions from the resumed thread.
482
+ // `runStreamed` has no `system` slot: prepend the system prompt as a
483
+ // fenced block on the first turn only; resumed threads inherit it.
432
484
  const inputText =
433
485
  previousTurns === 0 ? `${systemPrompt}\n\n---\n\n${prompt}` : prompt;
434
-
435
- const { events } = await thread.runStreamed(inputText, {
436
- signal: abortController.signal,
486
+ await driveCodexStream({
487
+ thread,
488
+ inputText,
489
+ abortController,
490
+ eventContext,
491
+ rollout,
492
+ outcome,
437
493
  });
438
-
439
- for await (const event of events) {
440
- if (abortController.signal.aborted && !streamState.turnTerminated) break;
441
- handleEvent(event, {
442
- state: streamState,
443
- seenToolCallIds,
444
- startedToolIds,
445
- codexToolMetrics,
446
- onTextBlock,
447
- onToolUse,
448
- onToolStart,
449
- onToolEnd,
450
- chatId,
451
- });
452
-
453
- if (event.type === "thread.started") {
454
- resolvedThreadId = event.thread_id;
455
- } else if (event.type === "turn.completed") {
456
- usage = event.usage;
457
- } else if (event.type === "turn.failed") {
458
- turnFailedError = event.error.message;
459
- } else if (event.type === "error") {
460
- turnFailedError = event.message;
461
- }
462
-
463
- pollRolloutForLiveStats();
464
-
465
- // Terminator-driven abort: a delivery tool already shipped the
466
- // reply via the bridge. Cancel further model generation to skip
467
- // the wrap-up round-trip Codex would otherwise burn.
468
- if (streamState.turnTerminated && !abortController.signal.aborted) {
469
- log("agent", `[${chatId}] terminator fired — aborting Codex turn`);
470
- try {
471
- abortController.abort();
472
- } catch (err) {
473
- logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
474
- }
475
- }
476
- }
477
-
478
494
  turnMs = Date.now() - turnStart;
479
495
  } catch (err) {
480
496
  // Aborted-by-terminator path is the expected close on `end_turn`.
481
- if (
482
- streamState.turnTerminated &&
483
- (errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)))
484
- ) {
485
- // Swallow — turn completed via terminator tool.
486
- } else {
487
- // ChatGPT-OAuth model-mismatch path. Check both the captured
488
- // event-stream message and the thrown error — Codex SDK surfaces
489
- // it via both channels. Only use the thread ID from this run.
490
- const fallback = await maybeFallbackForChatGptMismatch(
491
- `${turnFailedError ?? ""} ${errMsg(err)}`,
492
- activeModel,
493
- params,
494
- _retried,
495
- chatId,
496
- resolvedThreadId,
497
- );
498
- if (fallback) return fallback;
499
-
500
- const outcome = await applyRetryDecision({
497
+ if (!isTerminatorAbort(streamState, err)) {
498
+ return await recoverCodexFailure({
501
499
  err,
502
- chatId,
503
- activeModel,
504
- retried: _retried,
505
500
  params,
506
- recurseWithRetried: (p) => handleMessage(p, true),
507
- backendLabel: "Codex",
508
- resetNoun: "thread",
509
- });
510
- if (outcome.retry) return outcome.retry;
511
-
512
- // Terminal failure — recover whatever usage the rollout recorded
513
- // before the turn died, then account for it (failed turns burn
514
- // real tokens; they must not vanish from /status and /metrics).
515
- await settleUsageAccounting().catch(() => {});
516
- recordFailedTurnAccounting({
517
- backend: "codex",
518
- chatId,
519
- durationMs: Date.now() - t0,
520
- toolCalls: codexToolMetrics.count,
521
- apiCalls: streamState.numApiCalls,
522
- model: activeModel,
523
- usage: {
524
- inputTokens: streamState.sdkInputTokens,
525
- outputTokens: streamState.sdkOutputTokens,
526
- cacheRead: streamState.sdkCacheRead,
527
- cacheWrite: streamState.sdkCacheWrite,
528
- },
529
- contextTokens: streamState.contextTokens,
530
- contextWindow: streamState.contextWindow,
501
+ retried: _retried,
502
+ activeModel,
503
+ state: streamState,
504
+ rollout,
505
+ outcome,
506
+ toolCalls: eventContext.codexToolMetrics.count,
507
+ t0,
531
508
  });
532
-
533
- logError(
534
- "agent",
535
- `[${chatId}] Codex error: ${outcome.classified.message}`,
536
- );
537
- throw outcome.classified;
538
509
  }
539
510
  } finally {
540
511
  unregisterInterrupt();
@@ -548,9 +519,9 @@ export async function handleMessage(
548
519
  // Event-only ChatGPT-mismatch recovery: if the SDK emitted a
549
520
  // `turn.failed` carrying the mismatch text but DIDN'T rethrow, the
550
521
  // catch block above never fired. Catch it here too.
551
- if (turnFailedError && !_retried) {
522
+ if (outcome.turnFailedError && !_retried) {
552
523
  const fallback = await maybeFallbackForChatGptMismatch(
553
- turnFailedError,
524
+ outcome.turnFailedError,
554
525
  activeModel,
555
526
  params,
556
527
  _retried,
@@ -559,57 +530,35 @@ export async function handleMessage(
559
530
  if (fallback) return fallback;
560
531
  }
561
532
 
562
- if (resolvedThreadId) {
563
- const stored = getSession(chatId).sessionId;
564
- if (stored !== resolvedThreadId) {
565
- setSessionId(chatId, resolvedThreadId);
566
- }
567
- }
568
-
569
- await settleUsageAccounting();
533
+ await rollout.settle(outcome.usage);
570
534
 
571
535
  // Surface a synthetic error if Codex failed the turn upstream.
572
- if (turnFailedError) {
573
- streamState.syntheticError = turnFailedError;
536
+ if (outcome.turnFailedError) {
537
+ streamState.syntheticError = outcome.turnFailedError;
574
538
  }
575
539
 
576
540
  const responseText = finalizeResponseText(streamState);
577
541
  const durationMs = Date.now() - t0;
578
- recordTurnMetrics({
542
+ accountTurn({
579
543
  chatId,
580
544
  backend: "codex",
581
- durationMs,
582
- toolCalls: codexToolMetrics.count,
583
- apiCalls: streamState.numApiCalls,
584
- failed: Boolean(turnFailedError),
585
- usage: {
586
- inputTokens: streamState.sdkInputTokens,
587
- outputTokens: streamState.sdkOutputTokens,
588
- cacheRead: streamState.sdkCacheRead,
589
- cacheWrite: streamState.sdkCacheWrite,
590
- },
591
- });
592
-
593
- recordUsage(chatId, {
594
- inputTokens: streamState.sdkInputTokens,
595
- outputTokens: streamState.sdkOutputTokens,
596
- cacheRead: streamState.sdkCacheRead,
597
- cacheWrite: streamState.sdkCacheWrite,
545
+ state: streamState,
598
546
  durationMs,
599
547
  model: activeModel,
600
- // contextTokens comes from the rollout JSONL when available. Falls
601
- // back to 0 → /status shows "unknown", correct under-promise behaviour.
602
- contextTokens: streamState.contextTokens || undefined,
603
- // Prefer the rollout's reported context window over the static catalog.
604
- contextWindow: streamState.contextWindow ?? activeModelInfo?.contextWindow,
605
- numApiCalls: streamState.numApiCalls || undefined,
548
+ sessionId: rollout.threadId,
549
+ failed: Boolean(outcome.turnFailedError),
550
+ toolCalls: eventContext.codexToolMetrics.count,
551
+ context: {
552
+ // contextTokens comes from the rollout JSONL when available. Falls
553
+ // back to 0 → /status shows "unknown", correct under-promise behaviour.
554
+ contextTokens: streamState.contextTokens || undefined,
555
+ // Prefer the rollout's reported context window over the static catalog.
556
+ contextWindow:
557
+ streamState.contextWindow ?? activeModelInfo?.contextWindow,
558
+ numApiCalls: streamState.numApiCalls || undefined,
559
+ },
606
560
  });
607
-
608
- // Set a descriptive session name from the user's first message.
609
- if (previousTurns === 0) {
610
- const name = extractSessionName(text);
611
- if (name) setSessionName(chatId, name);
612
- }
561
+ nameSessionFromFirstMessage({ chatId, text, previousTurns });
613
562
 
614
563
  // ── Delivery — decision tree shared with the other backends ────────────────
615
564
  let delivery;
@@ -619,7 +568,7 @@ export async function handleMessage(
619
568
  chatId,
620
569
  state: streamState,
621
570
  responseText,
622
- onTextBlock,
571
+ onTextBlock: params.onTextBlock,
623
572
  propagateDeliveryFailure: true,
624
573
  });
625
574
  } catch (err) {
@@ -638,38 +587,13 @@ export async function handleMessage(
638
587
  }
639
588
 
640
589
  incrementTurns(chatId);
641
-
642
- log(
643
- "agent",
644
- `[${chatId}] delivery: ${delivery.route} (${delivery.chars} chars)`,
645
- );
646
-
647
- log(
648
- "agent",
649
- `[${chatId}] -> (${summarizeUsage(
650
- {
651
- inputTokens: streamState.sdkInputTokens,
652
- outputTokens: streamState.sdkOutputTokens,
653
- cacheRead: streamState.sdkCacheRead,
654
- cacheWrite: streamState.sdkCacheWrite,
655
- },
656
- { durationMs, toolCalls: streamState.toolCalls },
657
- )} terminator=${streamState.turnTerminated ? "yes" : "no"} ` +
658
- `delivered=${streamState.deliveredTextNorms.length} ` +
659
- `respLen=${responseText.length} ` +
660
- `setup=${setupMs}ms turn=${turnMs}ms)`,
661
- );
662
- traceMessage(chatId, "out", responseText, {
590
+ return finishCallbackTurn({
591
+ chatId,
592
+ state: streamState,
593
+ responseText,
663
594
  durationMs,
664
- toolCalls: streamState.toolCalls,
595
+ setupMs,
596
+ turnMs,
597
+ delivery,
665
598
  });
666
-
667
- return {
668
- text: responseText,
669
- durationMs,
670
- inputTokens: streamState.sdkInputTokens,
671
- outputTokens: streamState.sdkOutputTokens,
672
- cacheRead: streamState.sdkCacheRead,
673
- cacheWrite: streamState.sdkCacheWrite,
674
- };
675
599
  }