talon-agent 3.34.0 → 3.34.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,26 +2,19 @@
2
2
  * OpenAI Agents backend message handler.
3
3
  *
4
4
  * Drives a single-agent run on top of `@openai/agents`'s `run()` in streaming
5
- * mode. Shares the non-SDK-specific primitives with the other backends via
6
- * `../../shared/`. OpenAI-Agents-specific bits: building the `Agent` with the
7
- * per-chat MCP bundle, iterating the `StreamedRunResult` (see `events.ts`),
8
- * `.cancel()` on terminator, and the trailing-prose flow-violation retry.
5
+ * mode. OpenAI-Agents-specific bits: building the `Agent` with the per-chat
6
+ * MCP bundle, iterating the `StreamedRunResult` (see `events.ts`), and
7
+ * `.cancel()` on terminator. The post-stream phases — accounting, the
8
+ * trailing-prose flow-violation retry, the result — are the shared ones in
9
+ * `backend/shared/turn-phases.ts`.
9
10
  */
10
11
 
11
12
  import { Agent, run } from "@openai/agents";
12
13
  import type { QueryParams, QueryResult } from "../../shared/handler-types.js";
13
- import {
14
- getSession,
15
- incrementTurns,
16
- recordUsage,
17
- setSessionName,
18
- resetSession,
19
- } from "../../../storage/sessions.js";
14
+ import { getSession, incrementTurns } from "../../../storage/sessions.js";
20
15
  import { getChatSettings } from "../../../storage/chat-settings.js";
21
- import { classify } from "../../../core/errors.js";
22
16
  import { log, logError, logWarn } from "../../../util/log.js";
23
17
  import { traceMessage } from "../../../util/trace.js";
24
- import { incrementCounter } from "../../../storage/metrics.js";
25
18
 
26
19
  import {
27
20
  createStreamState,
@@ -29,21 +22,18 @@ import {
29
22
  finalizeResponseText,
30
23
  formatUserPrompt,
31
24
  prepareSystemPrompt,
32
- extractSessionName,
33
- classifyRetry,
34
- summarizeUsage,
35
25
  routeDelivery,
36
26
  buildFirstTurnReminder,
37
27
  buildFlowViolationReminder,
38
- recordTurnMetrics,
39
- recordFailedTurnAccounting,
40
- recordFlowViolation,
28
+ applyRetryDecision,
41
29
  registerTurnInterrupt,
30
+ accountTurn,
31
+ accountFailedTurn,
32
+ nameSessionFromFirstMessage,
33
+ enforceTrailingProse,
34
+ finishCallbackTurn,
35
+ type StreamState,
42
36
  } from "../../shared/index.js";
43
- import {
44
- detectFlowViolation,
45
- FLOW_VIOLATION_MAX_RETRIES,
46
- } from "../../shared/flow-violation.js";
47
37
 
48
38
  import {
49
39
  buildOpenAiAgentsSuffix,
@@ -64,32 +54,232 @@ import { handleRunItem } from "./events.js";
64
54
  const errMsg = (e: unknown): string =>
65
55
  e instanceof Error ? e.message : String(e);
66
56
 
57
+ /** The expected close on `end_turn` / a user interrupt: abort after the terminator. */
58
+ const isTerminatorAbort = (state: StreamState, err: unknown): boolean =>
59
+ state.turnTerminated &&
60
+ (errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)));
61
+
62
+ type McpBundle = Awaited<ReturnType<typeof getOrCreateBundle>>;
63
+
64
+ type RunUsage = {
65
+ inputTokens?: number;
66
+ outputTokens?: number;
67
+ inputTokensDetails?: { cachedTokens?: number };
68
+ };
69
+
67
70
  /**
68
71
  * Read the aggregated usage off a run's state. `_context` is named with
69
72
  * an underscore in the SDK type but is structurally public; the SDK
70
73
  * updates it as each model call in the agentic loop completes, so this
71
74
  * is valid both mid-stream (live stats) and after `stream.completed`.
72
75
  */
73
- function readRunUsage(runState: unknown):
74
- | {
75
- inputTokens?: number;
76
- outputTokens?: number;
77
- inputTokensDetails?: { cachedTokens?: number };
76
+ function readRunUsage(runState: unknown): RunUsage | undefined {
77
+ return (runState as { _context?: { usage?: RunUsage } })._context?.usage;
78
+ }
79
+
80
+ function recordRunUsage(state: StreamState, usage: RunUsage): void {
81
+ recordTokens(state, {
82
+ inputTokens: usage.inputTokens ?? 0,
83
+ outputTokens: usage.outputTokens ?? 0,
84
+ cacheRead: usage.inputTokensDetails?.cachedTokens ?? 0,
85
+ cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
86
+ });
87
+ }
88
+
89
+ // ── Setup ───────────────────────────────────────────────────────────────────
90
+
91
+ /**
92
+ * First-turn nudge — turn 0 is where flow violations cluster, and one
93
+ * line in the user message costs nothing on later turns and never
94
+ * touches the cached prefix.
95
+ */
96
+ function buildTurnPrompt(
97
+ params: QueryParams,
98
+ frontend: string | undefined,
99
+ previousTurns: number,
100
+ isRetry: boolean,
101
+ ): string {
102
+ let prompt = formatUserPrompt({
103
+ text: params.text,
104
+ senderName: params.senderName ?? "user",
105
+ senderHandle: params.senderHandle,
106
+ isGroup: params.isGroup,
107
+ messageId: params.messageId,
108
+ });
109
+ if (frontend && previousTurns === 0 && !isRetry) {
110
+ prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
111
+ }
112
+ return prompt;
113
+ }
114
+
115
+ /**
116
+ * Acquire the per-chat MCP bundle. Persistent across turns — built on
117
+ * first use, kept alive until `releaseBundle(chatId)`. Avoids the
118
+ * ~15-subprocess re-spawn the original per-turn build caused.
119
+ */
120
+ async function acquireMcpBundle(
121
+ chatId: string,
122
+ frontends: readonly string[],
123
+ ): Promise<McpBundle> {
124
+ const state = getState();
125
+ const config = state.config;
126
+ if (!config) throw new Error("OpenAI Agents backend not initialized");
127
+ try {
128
+ return await getOrCreateBundle({
129
+ chatId,
130
+ bridgeUrl: `http://127.0.0.1:${state.gatewayPortFn()}`,
131
+ frontends,
132
+ braveApiKey: config.braveApiKey,
133
+ toolExclusions: config,
134
+ });
135
+ } catch (err) {
136
+ logError(
137
+ "agent",
138
+ `[${chatId}] OpenAI Agents: MCP setup failed: ${errMsg(err)}`,
139
+ );
140
+ throw err;
141
+ }
142
+ }
143
+
144
+ /**
145
+ * Diagnostic — enumerate every tool the model will see this turn.
146
+ * Critical for tracking down "model never calls end_turn": if it
147
+ * isn't in this list, the problem is MCP registration, not the model.
148
+ */
149
+ async function logRegisteredTools(
150
+ chatId: string,
151
+ mcpBundle: McpBundle,
152
+ ): Promise<void> {
153
+ try {
154
+ const builtinNames = OPENAI_AGENTS_BUILTIN_TOOLS.map((t) => t.name);
155
+ const mcpToolLists = await Promise.all(
156
+ mcpBundle.servers.map((s) =>
157
+ s
158
+ .listTools()
159
+ .then((ts: Array<{ name?: string }>) => ts.map((t) => t.name ?? "?"))
160
+ .catch(() => [] as string[]),
161
+ ),
162
+ );
163
+ const mcpNames = mcpToolLists.flat();
164
+ log(
165
+ "agent",
166
+ `[${chatId}] tools registered: builtins=[${builtinNames.join(", ")}] mcp=[${mcpNames.join(", ")}]`,
167
+ );
168
+ } catch {
169
+ /* best-effort diagnostic */
170
+ }
171
+ }
172
+
173
+ // ── Stream loop ─────────────────────────────────────────────────────────────
174
+
175
+ async function driveAgentRun(inputs: {
176
+ chatId: string;
177
+ prompt: string;
178
+ systemPrompt: string;
179
+ activeModel: string;
180
+ mcpBundle: McpBundle;
181
+ abortController: AbortController;
182
+ state: StreamState;
183
+ onToolUse: QueryParams["onToolUse"];
184
+ }): Promise<void> {
185
+ const { chatId, mcpBundle, abortController, state } = inputs;
186
+ await logRegisteredTools(chatId, mcpBundle);
187
+
188
+ // Build the agent. `tools` carries the filesystem + shell built-ins
189
+ // for parity with the Claude SDK backend; `mcpServers` carries the
190
+ // Talon frontend + plugin MCP servers. Single agent, no handoffs.
191
+ // `mcpConfig.includeServerInToolNames` namespaces MCP tools as
192
+ // `mcp_<serverName>__<toolName>` so colliding names across plugins
193
+ // both stay available. Built-in tools stay unprefixed.
194
+ const agent = new Agent({
195
+ name: OPENAI_AGENTS_AGENT_NAME,
196
+ instructions: inputs.systemPrompt,
197
+ model: inputs.activeModel,
198
+ tools: [...OPENAI_AGENTS_BUILTIN_TOOLS],
199
+ mcpServers: mcpBundle.servers,
200
+ mcpConfig: { includeServerInToolNames: true },
201
+ });
202
+
203
+ // Per-chat MemorySession so the SDK preserves the full multi-turn
204
+ // record (model outputs, tool calls + results, reasoning items).
205
+ // Without this, every turn starts blind to prior context.
206
+ const stream = await run(agent, inputs.prompt, {
207
+ stream: true,
208
+ maxTurns: OPENAI_AGENTS_MAX_TURNS,
209
+ signal: abortController.signal,
210
+ session: getOrCreateSession(chatId),
211
+ });
212
+
213
+ // The Agents SDK aggregates usage on the run context as each model
214
+ // call completes — sample it (throttled) so the live-turn overlay
215
+ // tracks the agentic loop instead of jumping from 0 to final.
216
+ let lastLiveUsagePushAt = 0;
217
+ const pushRunUsageLive = (): void => {
218
+ const now = Date.now();
219
+ if (now - lastLiveUsagePushAt < 1000) return;
220
+ lastLiveUsagePushAt = now;
221
+ const u = readRunUsage(stream.state);
222
+ if (u) recordRunUsage(state, u);
223
+ };
224
+
225
+ const seenToolCallIds = new Set<string>();
226
+ for await (const event of stream) {
227
+ if (abortController.signal.aborted && !state.turnTerminated) break;
228
+
229
+ if (event.type === "run_item_stream_event") {
230
+ handleRunItem(event, {
231
+ state,
232
+ seenToolCallIds,
233
+ onToolUse: inputs.onToolUse,
234
+ chatId,
235
+ });
236
+ pushRunUsageLive();
78
237
  }
79
- | undefined {
80
- return (
81
- runState as {
82
- _context?: {
83
- usage?: {
84
- inputTokens?: number;
85
- outputTokens?: number;
86
- inputTokensDetails?: { cachedTokens?: number };
87
- };
88
- };
238
+ // `raw_model_stream_event` and `agent_updated_stream_event`
239
+ // events are intentionally not surfaced — token-by-token streaming
240
+ // would expose private chain-of-thought; the final-message event
241
+ // is enough.
242
+
243
+ // Terminator-driven abort. The SDK emits TWO events for each tool
244
+ // call: `tool_called` (RPC about to run) and `tool_output` (RPC
245
+ // completed; message reached the frontend). We must NOT abort on
246
+ // `tool_called` — that cancels the in-flight RPC and the message
247
+ // never ships. Aborting on `tool_output` after we've flagged the
248
+ // turn terminated means delivery happened AND we skip the SDK's
249
+ // wrap-up round-trip (otherwise 5–10s of lingering typing).
250
+ if (
251
+ state.turnTerminated &&
252
+ event.type === "run_item_stream_event" &&
253
+ (event as { name?: string }).name === "tool_output" &&
254
+ !abortController.signal.aborted
255
+ ) {
256
+ log(
257
+ "agent",
258
+ `[${chatId}] terminator tool result received — aborting wrap-up`,
259
+ );
260
+ try {
261
+ abortController.abort();
262
+ } catch (err) {
263
+ logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
264
+ }
89
265
  }
90
- )._context?.usage;
266
+ }
267
+
268
+ // Await the final completion so usage + final state are populated.
269
+ // Safe to call even when we aborted via the terminator (resolves to
270
+ // the partial state).
271
+ await stream.completed.catch(() => {
272
+ /* swallow — aborted-by-terminator path */
273
+ });
274
+
275
+ // Token usage from the underlying RunResult. The SDK aggregates
276
+ // `usage` across all turns in the loop.
277
+ const usage = readRunUsage(stream.state);
278
+ if (usage) recordRunUsage(state, usage);
91
279
  }
92
280
 
281
+ // ── Main handler ────────────────────────────────────────────────────────────
282
+
93
283
  export async function handleMessage(
94
284
  params: QueryParams,
95
285
  _retried = false,
@@ -101,19 +291,11 @@ export async function handleMessage(
101
291
  throw new Error("OpenAI Agents backend not initialized");
102
292
  }
103
293
 
104
- const {
105
- chatId,
106
- text,
107
- senderName,
108
- senderHandle,
109
- isGroup,
110
- messageId,
111
- onTextBlock,
112
- onToolUse,
113
- } = params;
294
+ const { chatId, text, senderName, isGroup, onTextBlock } = params;
114
295
  const t0 = Date.now();
115
296
  const session = getSession(chatId);
116
297
  const previousTurns = session.turns;
298
+ const isRetry = _retried || _flowRetries > 0;
117
299
 
118
300
  // Resolve the active model — chat-settings → config → default.
119
301
  const chatSettings = getChatSettings(chatId);
@@ -140,49 +322,16 @@ export async function handleMessage(
140
322
  chatId,
141
323
  sessionEpoch: session.createdAt,
142
324
  });
143
-
144
- let prompt = formatUserPrompt({
145
- text,
146
- senderName: senderName ?? "user",
147
- senderHandle,
148
- isGroup,
149
- messageId,
150
- });
151
- // First-turn nudge — turn 0 is where flow violations cluster, and one
152
- // line in the user message costs nothing on later turns and never
153
- // touches the cached prefix.
154
- if (frontend && previousTurns === 0 && !_retried && _flowRetries === 0) {
155
- prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
156
- }
325
+ const prompt = buildTurnPrompt(params, frontend, previousTurns, isRetry);
157
326
 
158
327
  log("agent", `[${chatId}] <- (${text.length} chars)`);
159
328
  traceMessage(chatId, "in", text, { senderName, isGroup });
160
329
 
161
- // Acquire the per-chat MCP bundle. Persistent across turns — built on
162
- // first use, kept alive until `releaseBundle(chatId)`. Avoids the
163
- // ~15-subprocess re-spawn the original per-turn build caused.
164
- const bridgeUrl = `http://127.0.0.1:${state.gatewayPortFn()}`;
165
- let mcpBundle: Awaited<ReturnType<typeof getOrCreateBundle>>;
166
- try {
167
- mcpBundle = await getOrCreateBundle({
168
- chatId,
169
- bridgeUrl,
170
- frontends,
171
- braveApiKey: config.braveApiKey,
172
- toolExclusions: config,
173
- });
174
- } catch (err) {
175
- logError(
176
- "agent",
177
- `[${chatId}] OpenAI Agents: MCP setup failed: ${errMsg(err)}`,
178
- );
179
- throw err;
180
- }
330
+ const mcpBundle = await acquireMcpBundle(chatId, frontends);
181
331
 
182
332
  // Bind the stream state to the chat so token mutators mirror counts
183
333
  // into the live-turn overlay — /status updates while the turn runs.
184
334
  const streamState = createStreamState(chatId);
185
- const seenToolCallIds = new Set<string>();
186
335
  const abortController = new AbortController();
187
336
  activeAborts.set(chatId, abortController);
188
337
  // A user interrupt is a synthetic turn terminator: marking the flag
@@ -199,201 +348,47 @@ export async function handleMessage(
199
348
 
200
349
  try {
201
350
  const turnStart = Date.now();
202
-
203
- // Build the agent. `tools` carries the filesystem + shell built-ins
204
- // for parity with the Claude SDK backend; `mcpServers` carries the
205
- // Talon frontend + plugin MCP servers. Single agent, no handoffs.
206
- // `mcpConfig.includeServerInToolNames` namespaces MCP tools as
207
- // `mcp_<serverName>__<toolName>` so colliding names across plugins
208
- // both stay available. Built-in tools stay unprefixed.
209
-
210
- // Diagnostic — enumerate every tool the model will see this turn.
211
- // Critical for tracking down "model never calls end_turn": if it
212
- // isn't in this list, the problem is MCP registration, not the model.
213
- try {
214
- const builtinNames = OPENAI_AGENTS_BUILTIN_TOOLS.map((t) => t.name);
215
- const mcpToolLists = await Promise.all(
216
- mcpBundle.servers.map((s) =>
217
- s
218
- .listTools()
219
- .then((ts: Array<{ name?: string }>) =>
220
- ts.map((t) => t.name ?? "?"),
221
- )
222
- .catch(() => [] as string[]),
223
- ),
224
- );
225
- const mcpNames = mcpToolLists.flat();
226
- log(
227
- "agent",
228
- `[${chatId}] tools registered: builtins=[${builtinNames.join(", ")}] mcp=[${mcpNames.join(", ")}]`,
229
- );
230
- } catch {
231
- /* best-effort diagnostic */
232
- }
233
-
234
- const agent = new Agent({
235
- name: OPENAI_AGENTS_AGENT_NAME,
236
- instructions: systemPrompt,
237
- model: activeModel,
238
- tools: [...OPENAI_AGENTS_BUILTIN_TOOLS],
239
- mcpServers: mcpBundle.servers,
240
- mcpConfig: { includeServerInToolNames: true },
241
- });
242
-
243
- // Per-chat MemorySession so the SDK preserves the full multi-turn
244
- // record (model outputs, tool calls + results, reasoning items).
245
- // Without this, every turn starts blind to prior context.
246
- const stream = await run(agent, prompt, {
247
- stream: true,
248
- maxTurns: OPENAI_AGENTS_MAX_TURNS,
249
- signal: abortController.signal,
250
- session: getOrCreateSession(chatId),
251
- });
252
-
253
- // The Agents SDK aggregates usage on the run context as each model
254
- // call completes — sample it (throttled) so the live-turn overlay
255
- // tracks the agentic loop instead of jumping from 0 to final.
256
- let lastLiveUsagePushAt = 0;
257
- const pushRunUsageLive = (): void => {
258
- const now = Date.now();
259
- if (now - lastLiveUsagePushAt < 1000) return;
260
- lastLiveUsagePushAt = now;
261
- const u = readRunUsage(stream.state);
262
- if (!u) return;
263
- recordTokens(streamState, {
264
- inputTokens: u.inputTokens ?? 0,
265
- outputTokens: u.outputTokens ?? 0,
266
- cacheRead: u.inputTokensDetails?.cachedTokens ?? 0,
267
- cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
268
- });
269
- };
270
-
271
- for await (const event of stream) {
272
- if (abortController.signal.aborted && !streamState.turnTerminated) break;
273
-
274
- if (event.type === "run_item_stream_event") {
275
- handleRunItem(event, {
276
- state: streamState,
277
- seenToolCallIds,
278
- onToolUse,
279
- chatId,
280
- });
281
- pushRunUsageLive();
282
- }
283
- // `raw_model_stream_event` and `agent_updated_stream_event`
284
- // events are intentionally not surfaced — token-by-token streaming
285
- // would expose private chain-of-thought; the final-message event
286
- // is enough.
287
-
288
- // Terminator-driven abort. The SDK emits TWO events for each tool
289
- // call: `tool_called` (RPC about to run) and `tool_output` (RPC
290
- // completed; message reached the frontend). We must NOT abort on
291
- // `tool_called` — that cancels the in-flight RPC and the message
292
- // never ships. Aborting on `tool_output` after we've flagged the
293
- // turn terminated means delivery happened AND we skip the SDK's
294
- // wrap-up round-trip (otherwise 5–10s of lingering typing).
295
- if (
296
- streamState.turnTerminated &&
297
- event.type === "run_item_stream_event" &&
298
- (event as { name?: string }).name === "tool_output" &&
299
- !abortController.signal.aborted
300
- ) {
301
- log(
302
- "agent",
303
- `[${chatId}] terminator tool result received — aborting wrap-up`,
304
- );
305
- try {
306
- abortController.abort();
307
- } catch (err) {
308
- logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
309
- }
310
- }
311
- }
312
-
313
- // Await the final completion so usage + final state are populated.
314
- // Safe to call even when we aborted via the terminator (resolves to
315
- // the partial state).
316
- await stream.completed.catch(() => {
317
- /* swallow — aborted-by-terminator path */
351
+ await driveAgentRun({
352
+ chatId,
353
+ prompt,
354
+ systemPrompt,
355
+ activeModel,
356
+ mcpBundle,
357
+ abortController,
358
+ state: streamState,
359
+ onToolUse: params.onToolUse,
318
360
  });
319
-
320
- // Token usage from the underlying RunResult. The SDK aggregates
321
- // `usage` across all turns in the loop.
322
- const usage = readRunUsage(stream.state);
323
- if (usage) {
324
- recordTokens(streamState, {
325
- inputTokens: usage.inputTokens ?? 0,
326
- outputTokens: usage.outputTokens ?? 0,
327
- cacheRead: usage.inputTokensDetails?.cachedTokens ?? 0,
328
- cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
329
- });
330
- }
331
-
332
361
  turnMs = Date.now() - turnStart;
333
362
  } catch (err) {
334
- if (
335
- streamState.turnTerminated &&
336
- (errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)))
337
- ) {
338
- // Swallow — turn completed via terminator tool.
339
- } else {
340
- const classified = classify(err);
341
- incrementCounter(`errors.${classified.reason ?? "unknown"}`);
342
-
343
- const decision = classifyRetry({
344
- error: classified,
363
+ // Swallow the terminator abort — the turn completed via a delivery tool.
364
+ if (!isTerminatorAbort(streamState, err)) {
365
+ // MCP bundle is retained across a retry — subprocesses are
366
+ // stateless wrt the model conversation. See `mcp-pool.ts`.
367
+ const outcome = await applyRetryDecision({
368
+ err,
369
+ chatId,
345
370
  activeModel,
346
371
  retried: _retried,
372
+ params,
373
+ recurseWithRetried: (p) => handleMessage(p, true),
374
+ backendLabel: "OpenAI Agents",
347
375
  });
348
-
349
- if (decision.kind === "reset_and_retry") {
350
- logWarn(
351
- "agent",
352
- `[${chatId}] OpenAI Agents ${decision.reason}, resetting session and retrying`,
353
- );
354
- resetSession(chatId);
355
- // MCP bundle is retained across the retry — subprocesses are
356
- // stateless wrt the model conversation. See `mcp-pool.ts`.
357
- return handleMessage(params, true);
358
- }
359
-
360
- if (decision.kind === "fallback_model") {
361
- logWarn(
362
- "agent",
363
- `[${chatId}] ${classified.reason}, falling back to ${decision.fallbackModelId}`,
364
- );
365
- resetSession(chatId);
366
- return await handleMessage(
367
- { ...params, model: decision.fallbackModelId },
368
- true,
369
- );
370
- }
376
+ if (outcome.retry) return outcome.retry;
371
377
 
372
378
  // Terminal failure — account for whatever the turn consumed before
373
- // dying and drop the live overlay (the retry branches above re-enter
374
- // handleMessage, which does its own accounting).
375
- recordFailedTurnAccounting({
379
+ // dying (the retry above did its own accounting).
380
+ accountFailedTurn({
376
381
  backend: "openai-agents",
377
382
  chatId,
383
+ state: streamState,
378
384
  durationMs: Date.now() - t0,
379
- toolCalls: streamState.toolCalls,
380
- apiCalls: streamState.numApiCalls,
381
385
  model: activeModel,
382
- usage: {
383
- inputTokens: streamState.sdkInputTokens,
384
- outputTokens: streamState.sdkOutputTokens,
385
- cacheRead: streamState.sdkCacheRead,
386
- cacheWrite: streamState.sdkCacheWrite,
387
- },
388
- contextTokens: streamState.contextTokens,
389
- contextWindow: streamState.contextWindow,
390
386
  });
391
-
392
387
  logError(
393
388
  "agent",
394
- `[${chatId}] OpenAI Agents error: ${classified.message}`,
389
+ `[${chatId}] OpenAI Agents error: ${outcome.classified.message}`,
395
390
  );
396
- throw classified;
391
+ throw outcome.classified;
397
392
  }
398
393
  } finally {
399
394
  unregisterInterrupt();
@@ -409,86 +404,41 @@ export async function handleMessage(
409
404
 
410
405
  const responseText = finalizeResponseText(streamState);
411
406
  const durationMs = Date.now() - t0;
412
- recordTurnMetrics({
407
+ accountTurn({
413
408
  chatId,
414
409
  backend: "openai-agents",
415
- durationMs,
416
- toolCalls: streamState.toolCalls,
417
- apiCalls: streamState.numApiCalls,
418
- usage: {
419
- inputTokens: streamState.sdkInputTokens,
420
- outputTokens: streamState.sdkOutputTokens,
421
- cacheRead: streamState.sdkCacheRead,
422
- cacheWrite: streamState.sdkCacheWrite,
423
- },
424
- });
425
-
426
- recordUsage(chatId, {
427
- inputTokens: streamState.sdkInputTokens,
428
- outputTokens: streamState.sdkOutputTokens,
429
- cacheRead: streamState.sdkCacheRead,
430
- cacheWrite: streamState.sdkCacheWrite,
410
+ state: streamState,
431
411
  durationMs,
432
412
  model: activeModel,
433
413
  });
434
414
 
435
- // ── Trailing-prose contract + flow-violation retry ──────────────────────
436
415
  // Replies MUST go through `end_turn` (canonical) or `send` (mid-turn).
437
- // If the model wrote prose without calling either, the user would see
438
- // nothing — re-prompt once with a synthetic reminder. A second violation
439
- // accepts a silent drop. Only enforced when delivery tools are registered
440
- // (non-empty mcpBundle.servers). `incrementTurns` is deferred until AFTER
441
- // the check so the retry path doesn't double-count.
416
+ // Only enforced when delivery tools are registered (non-empty
417
+ // mcpBundle.servers). `incrementTurns` is deferred until AFTER the
418
+ // check so the retry path doesn't double-count.
442
419
  const violation =
443
420
  mcpBundle.servers.length > 0
444
- ? detectFlowViolation({
445
- trailingText: streamState.lastTrailingText,
446
- turnTerminated: streamState.turnTerminated,
447
- deliveredTextNorms: streamState.deliveredTextNorms,
448
- toolCalls: streamState.toolCalls,
449
- retried: _flowRetries > 0,
450
- retryCount: _flowRetries,
451
- maxRetries: FLOW_VIOLATION_MAX_RETRIES,
421
+ ? enforceTrailingProse({
422
+ chatId,
423
+ state: streamState,
424
+ flowRetries: _flowRetries,
452
425
  ...(frontend
453
426
  ? { reminder: buildFlowViolationReminder(frontend) }
454
427
  : {}),
455
428
  })
456
- : ({ violated: false } as const);
457
-
458
- if (violation.violated) {
459
- recordFlowViolation(
460
- chatId,
461
- violation.shouldRetry ? "retried" : "cap_exhausted",
462
- );
463
- log(
464
- "agent",
465
- `[${chatId}] flow violation: trailing prose (${violation.trailing.length} chars) without end_turn/send. ${
466
- violation.shouldRetry
467
- ? "Re-prompting with reminder."
468
- : "Already retried — accepting silent drop."
469
- }`,
429
+ : undefined;
430
+ if (violation?.violated && violation.shouldRetry) {
431
+ // Recursive call owns the `incrementTurns` for this user message.
432
+ return handleMessage(
433
+ { ...params, text: violation.reminder },
434
+ _retried,
435
+ _flowRetries + 1,
470
436
  );
471
-
472
- if (violation.shouldRetry) {
473
- // Recursive call owns the `incrementTurns` for this user message.
474
- return handleMessage(
475
- { ...params, text: violation.reminder },
476
- _retried,
477
- _flowRetries + 1,
478
- );
479
- }
480
437
  }
481
438
 
482
439
  // Reached the non-retry path — this turn counts as one user-visible turn.
483
440
  incrementTurns(chatId);
484
-
485
- // Set a descriptive session name from the user's *first* message.
486
- // Guarded by `!_retried` so the reminder doesn't get captured as the
487
- // session name when the retry recurses with `params.text = reminder`.
488
- if (previousTurns === 0 && !_retried && _flowRetries === 0) {
489
- const name = extractSessionName(text);
490
- if (name) setSessionName(chatId, name);
491
- }
441
+ nameSessionFromFirstMessage({ chatId, text, previousTurns, isRetry });
492
442
 
493
443
  // ── Delivery — strict tool-only ──────────────────────────────────────────
494
444
  // Replies must reach the user via a delivery tool. Trailing prose is
@@ -507,37 +457,13 @@ export async function handleMessage(
507
457
  })
508
458
  : { route: "silent" as const, chars: 0 };
509
459
 
510
- log(
511
- "agent",
512
- `[${chatId}] delivery: ${delivery.route} (${delivery.chars} chars)`,
513
- );
514
-
515
- log(
516
- "agent",
517
- `[${chatId}] -> (${summarizeUsage(
518
- {
519
- inputTokens: streamState.sdkInputTokens,
520
- outputTokens: streamState.sdkOutputTokens,
521
- cacheRead: streamState.sdkCacheRead,
522
- cacheWrite: streamState.sdkCacheWrite,
523
- },
524
- { durationMs, toolCalls: streamState.toolCalls },
525
- )} terminator=${streamState.turnTerminated ? "yes" : "no"} ` +
526
- `delivered=${streamState.deliveredTextNorms.length} ` +
527
- `respLen=${responseText.length} ` +
528
- `setup=${setupMs}ms turn=${turnMs}ms)`,
529
- );
530
- traceMessage(chatId, "out", responseText, {
460
+ return finishCallbackTurn({
461
+ chatId,
462
+ state: streamState,
463
+ responseText,
531
464
  durationMs,
532
- toolCalls: streamState.toolCalls,
465
+ setupMs,
466
+ turnMs,
467
+ delivery,
533
468
  });
534
-
535
- return {
536
- text: responseText,
537
- durationMs,
538
- inputTokens: streamState.sdkInputTokens,
539
- outputTokens: streamState.sdkOutputTokens,
540
- cacheRead: streamState.sdkCacheRead,
541
- cacheWrite: streamState.sdkCacheWrite,
542
- };
543
469
  }