talon-agent 3.34.0 → 3.34.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -2
- package/src/backend/claude-sdk/handler.ts +428 -415
- package/src/backend/codex/handler/message.ts +275 -366
- package/src/backend/codex/handler/rollout-accounting.ts +137 -0
- package/src/backend/openai-agents/handler/message.ts +282 -356
- package/src/backend/remote-server/chat-turn.ts +66 -122
- package/src/backend/remote-server/index.ts +4 -0
- package/src/backend/remote-server/mcp.ts +73 -9
- package/src/backend/remote-server/model-catalog/presentation.ts +269 -228
- package/src/backend/remote-server/sessions.ts +2 -2
- package/src/backend/shared/cache-telemetry.ts +17 -1
- package/src/backend/shared/handler-to-events.ts +11 -16
- package/src/backend/shared/index.ts +14 -15
- package/src/backend/shared/result-events.ts +30 -0
- package/src/backend/shared/turn-phases.ts +277 -0
- package/src/core/background/triggers/exit.ts +19 -2
- package/src/core/background/triggers/resume.ts +14 -7
- package/src/core/background/triggers/state.ts +9 -0
- package/src/core/mcp-hub/child-transport.ts +215 -0
- package/src/core/mcp-hub/children.ts +72 -13
- package/src/core/mcp-hub/index.ts +32 -13
- package/src/core/models/active-model.ts +29 -6
- package/src/core/weaver/shuttle.ts +4 -0
- package/src/frontend/telegram/admin.ts +75 -52
- package/src/storage/trigger-store.ts +8 -0
- package/src/util/watchdog.ts +32 -7
|
@@ -2,26 +2,19 @@
|
|
|
2
2
|
* OpenAI Agents backend message handler.
|
|
3
3
|
*
|
|
4
4
|
* Drives a single-agent run on top of `@openai/agents`'s `run()` in streaming
|
|
5
|
-
* mode.
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
5
|
+
* mode. OpenAI-Agents-specific bits: building the `Agent` with the per-chat
|
|
6
|
+
* MCP bundle, iterating the `StreamedRunResult` (see `events.ts`), and
|
|
7
|
+
* `.cancel()` on terminator. The post-stream phases — accounting, the
|
|
8
|
+
* trailing-prose flow-violation retry, the result — are the shared ones in
|
|
9
|
+
* `backend/shared/turn-phases.ts`.
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
import { Agent, run } from "@openai/agents";
|
|
12
13
|
import type { QueryParams, QueryResult } from "../../shared/handler-types.js";
|
|
13
|
-
import {
|
|
14
|
-
getSession,
|
|
15
|
-
incrementTurns,
|
|
16
|
-
recordUsage,
|
|
17
|
-
setSessionName,
|
|
18
|
-
resetSession,
|
|
19
|
-
} from "../../../storage/sessions.js";
|
|
14
|
+
import { getSession, incrementTurns } from "../../../storage/sessions.js";
|
|
20
15
|
import { getChatSettings } from "../../../storage/chat-settings.js";
|
|
21
|
-
import { classify } from "../../../core/errors.js";
|
|
22
16
|
import { log, logError, logWarn } from "../../../util/log.js";
|
|
23
17
|
import { traceMessage } from "../../../util/trace.js";
|
|
24
|
-
import { incrementCounter } from "../../../storage/metrics.js";
|
|
25
18
|
|
|
26
19
|
import {
|
|
27
20
|
createStreamState,
|
|
@@ -29,21 +22,18 @@ import {
|
|
|
29
22
|
finalizeResponseText,
|
|
30
23
|
formatUserPrompt,
|
|
31
24
|
prepareSystemPrompt,
|
|
32
|
-
extractSessionName,
|
|
33
|
-
classifyRetry,
|
|
34
|
-
summarizeUsage,
|
|
35
25
|
routeDelivery,
|
|
36
26
|
buildFirstTurnReminder,
|
|
37
27
|
buildFlowViolationReminder,
|
|
38
|
-
|
|
39
|
-
recordFailedTurnAccounting,
|
|
40
|
-
recordFlowViolation,
|
|
28
|
+
applyRetryDecision,
|
|
41
29
|
registerTurnInterrupt,
|
|
30
|
+
accountTurn,
|
|
31
|
+
accountFailedTurn,
|
|
32
|
+
nameSessionFromFirstMessage,
|
|
33
|
+
enforceTrailingProse,
|
|
34
|
+
finishCallbackTurn,
|
|
35
|
+
type StreamState,
|
|
42
36
|
} from "../../shared/index.js";
|
|
43
|
-
import {
|
|
44
|
-
detectFlowViolation,
|
|
45
|
-
FLOW_VIOLATION_MAX_RETRIES,
|
|
46
|
-
} from "../../shared/flow-violation.js";
|
|
47
37
|
|
|
48
38
|
import {
|
|
49
39
|
buildOpenAiAgentsSuffix,
|
|
@@ -64,32 +54,232 @@ import { handleRunItem } from "./events.js";
|
|
|
64
54
|
const errMsg = (e: unknown): string =>
|
|
65
55
|
e instanceof Error ? e.message : String(e);
|
|
66
56
|
|
|
57
|
+
/** The expected close on `end_turn` / a user interrupt: abort after the terminator. */
|
|
58
|
+
const isTerminatorAbort = (state: StreamState, err: unknown): boolean =>
|
|
59
|
+
state.turnTerminated &&
|
|
60
|
+
(errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)));
|
|
61
|
+
|
|
62
|
+
type McpBundle = Awaited<ReturnType<typeof getOrCreateBundle>>;
|
|
63
|
+
|
|
64
|
+
type RunUsage = {
|
|
65
|
+
inputTokens?: number;
|
|
66
|
+
outputTokens?: number;
|
|
67
|
+
inputTokensDetails?: { cachedTokens?: number };
|
|
68
|
+
};
|
|
69
|
+
|
|
67
70
|
/**
|
|
68
71
|
* Read the aggregated usage off a run's state. `_context` is named with
|
|
69
72
|
* an underscore in the SDK type but is structurally public; the SDK
|
|
70
73
|
* updates it as each model call in the agentic loop completes, so this
|
|
71
74
|
* is valid both mid-stream (live stats) and after `stream.completed`.
|
|
72
75
|
*/
|
|
73
|
-
function readRunUsage(runState: unknown):
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
76
|
+
function readRunUsage(runState: unknown): RunUsage | undefined {
|
|
77
|
+
return (runState as { _context?: { usage?: RunUsage } })._context?.usage;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function recordRunUsage(state: StreamState, usage: RunUsage): void {
|
|
81
|
+
recordTokens(state, {
|
|
82
|
+
inputTokens: usage.inputTokens ?? 0,
|
|
83
|
+
outputTokens: usage.outputTokens ?? 0,
|
|
84
|
+
cacheRead: usage.inputTokensDetails?.cachedTokens ?? 0,
|
|
85
|
+
cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// ── Setup ───────────────────────────────────────────────────────────────────
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* First-turn nudge — turn 0 is where flow violations cluster, and one
|
|
93
|
+
* line in the user message costs nothing on later turns and never
|
|
94
|
+
* touches the cached prefix.
|
|
95
|
+
*/
|
|
96
|
+
function buildTurnPrompt(
|
|
97
|
+
params: QueryParams,
|
|
98
|
+
frontend: string | undefined,
|
|
99
|
+
previousTurns: number,
|
|
100
|
+
isRetry: boolean,
|
|
101
|
+
): string {
|
|
102
|
+
let prompt = formatUserPrompt({
|
|
103
|
+
text: params.text,
|
|
104
|
+
senderName: params.senderName ?? "user",
|
|
105
|
+
senderHandle: params.senderHandle,
|
|
106
|
+
isGroup: params.isGroup,
|
|
107
|
+
messageId: params.messageId,
|
|
108
|
+
});
|
|
109
|
+
if (frontend && previousTurns === 0 && !isRetry) {
|
|
110
|
+
prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
|
|
111
|
+
}
|
|
112
|
+
return prompt;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Acquire the per-chat MCP bundle. Persistent across turns — built on
|
|
117
|
+
* first use, kept alive until `releaseBundle(chatId)`. Avoids the
|
|
118
|
+
* ~15-subprocess re-spawn the original per-turn build caused.
|
|
119
|
+
*/
|
|
120
|
+
async function acquireMcpBundle(
|
|
121
|
+
chatId: string,
|
|
122
|
+
frontends: readonly string[],
|
|
123
|
+
): Promise<McpBundle> {
|
|
124
|
+
const state = getState();
|
|
125
|
+
const config = state.config;
|
|
126
|
+
if (!config) throw new Error("OpenAI Agents backend not initialized");
|
|
127
|
+
try {
|
|
128
|
+
return await getOrCreateBundle({
|
|
129
|
+
chatId,
|
|
130
|
+
bridgeUrl: `http://127.0.0.1:${state.gatewayPortFn()}`,
|
|
131
|
+
frontends,
|
|
132
|
+
braveApiKey: config.braveApiKey,
|
|
133
|
+
toolExclusions: config,
|
|
134
|
+
});
|
|
135
|
+
} catch (err) {
|
|
136
|
+
logError(
|
|
137
|
+
"agent",
|
|
138
|
+
`[${chatId}] OpenAI Agents: MCP setup failed: ${errMsg(err)}`,
|
|
139
|
+
);
|
|
140
|
+
throw err;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Diagnostic — enumerate every tool the model will see this turn.
|
|
146
|
+
* Critical for tracking down "model never calls end_turn": if it
|
|
147
|
+
* isn't in this list, the problem is MCP registration, not the model.
|
|
148
|
+
*/
|
|
149
|
+
async function logRegisteredTools(
|
|
150
|
+
chatId: string,
|
|
151
|
+
mcpBundle: McpBundle,
|
|
152
|
+
): Promise<void> {
|
|
153
|
+
try {
|
|
154
|
+
const builtinNames = OPENAI_AGENTS_BUILTIN_TOOLS.map((t) => t.name);
|
|
155
|
+
const mcpToolLists = await Promise.all(
|
|
156
|
+
mcpBundle.servers.map((s) =>
|
|
157
|
+
s
|
|
158
|
+
.listTools()
|
|
159
|
+
.then((ts: Array<{ name?: string }>) => ts.map((t) => t.name ?? "?"))
|
|
160
|
+
.catch(() => [] as string[]),
|
|
161
|
+
),
|
|
162
|
+
);
|
|
163
|
+
const mcpNames = mcpToolLists.flat();
|
|
164
|
+
log(
|
|
165
|
+
"agent",
|
|
166
|
+
`[${chatId}] tools registered: builtins=[${builtinNames.join(", ")}] mcp=[${mcpNames.join(", ")}]`,
|
|
167
|
+
);
|
|
168
|
+
} catch {
|
|
169
|
+
/* best-effort diagnostic */
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// ── Stream loop ─────────────────────────────────────────────────────────────
|
|
174
|
+
|
|
175
|
+
async function driveAgentRun(inputs: {
|
|
176
|
+
chatId: string;
|
|
177
|
+
prompt: string;
|
|
178
|
+
systemPrompt: string;
|
|
179
|
+
activeModel: string;
|
|
180
|
+
mcpBundle: McpBundle;
|
|
181
|
+
abortController: AbortController;
|
|
182
|
+
state: StreamState;
|
|
183
|
+
onToolUse: QueryParams["onToolUse"];
|
|
184
|
+
}): Promise<void> {
|
|
185
|
+
const { chatId, mcpBundle, abortController, state } = inputs;
|
|
186
|
+
await logRegisteredTools(chatId, mcpBundle);
|
|
187
|
+
|
|
188
|
+
// Build the agent. `tools` carries the filesystem + shell built-ins
|
|
189
|
+
// for parity with the Claude SDK backend; `mcpServers` carries the
|
|
190
|
+
// Talon frontend + plugin MCP servers. Single agent, no handoffs.
|
|
191
|
+
// `mcpConfig.includeServerInToolNames` namespaces MCP tools as
|
|
192
|
+
// `mcp_<serverName>__<toolName>` so colliding names across plugins
|
|
193
|
+
// both stay available. Built-in tools stay unprefixed.
|
|
194
|
+
const agent = new Agent({
|
|
195
|
+
name: OPENAI_AGENTS_AGENT_NAME,
|
|
196
|
+
instructions: inputs.systemPrompt,
|
|
197
|
+
model: inputs.activeModel,
|
|
198
|
+
tools: [...OPENAI_AGENTS_BUILTIN_TOOLS],
|
|
199
|
+
mcpServers: mcpBundle.servers,
|
|
200
|
+
mcpConfig: { includeServerInToolNames: true },
|
|
201
|
+
});
|
|
202
|
+
|
|
203
|
+
// Per-chat MemorySession so the SDK preserves the full multi-turn
|
|
204
|
+
// record (model outputs, tool calls + results, reasoning items).
|
|
205
|
+
// Without this, every turn starts blind to prior context.
|
|
206
|
+
const stream = await run(agent, inputs.prompt, {
|
|
207
|
+
stream: true,
|
|
208
|
+
maxTurns: OPENAI_AGENTS_MAX_TURNS,
|
|
209
|
+
signal: abortController.signal,
|
|
210
|
+
session: getOrCreateSession(chatId),
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
// The Agents SDK aggregates usage on the run context as each model
|
|
214
|
+
// call completes — sample it (throttled) so the live-turn overlay
|
|
215
|
+
// tracks the agentic loop instead of jumping from 0 to final.
|
|
216
|
+
let lastLiveUsagePushAt = 0;
|
|
217
|
+
const pushRunUsageLive = (): void => {
|
|
218
|
+
const now = Date.now();
|
|
219
|
+
if (now - lastLiveUsagePushAt < 1000) return;
|
|
220
|
+
lastLiveUsagePushAt = now;
|
|
221
|
+
const u = readRunUsage(stream.state);
|
|
222
|
+
if (u) recordRunUsage(state, u);
|
|
223
|
+
};
|
|
224
|
+
|
|
225
|
+
const seenToolCallIds = new Set<string>();
|
|
226
|
+
for await (const event of stream) {
|
|
227
|
+
if (abortController.signal.aborted && !state.turnTerminated) break;
|
|
228
|
+
|
|
229
|
+
if (event.type === "run_item_stream_event") {
|
|
230
|
+
handleRunItem(event, {
|
|
231
|
+
state,
|
|
232
|
+
seenToolCallIds,
|
|
233
|
+
onToolUse: inputs.onToolUse,
|
|
234
|
+
chatId,
|
|
235
|
+
});
|
|
236
|
+
pushRunUsageLive();
|
|
78
237
|
}
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
238
|
+
// `raw_model_stream_event` and `agent_updated_stream_event`
|
|
239
|
+
// events are intentionally not surfaced — token-by-token streaming
|
|
240
|
+
// would expose private chain-of-thought; the final-message event
|
|
241
|
+
// is enough.
|
|
242
|
+
|
|
243
|
+
// Terminator-driven abort. The SDK emits TWO events for each tool
|
|
244
|
+
// call: `tool_called` (RPC about to run) and `tool_output` (RPC
|
|
245
|
+
// completed; message reached the frontend). We must NOT abort on
|
|
246
|
+
// `tool_called` — that cancels the in-flight RPC and the message
|
|
247
|
+
// never ships. Aborting on `tool_output` after we've flagged the
|
|
248
|
+
// turn terminated means delivery happened AND we skip the SDK's
|
|
249
|
+
// wrap-up round-trip (otherwise 5–10s of lingering typing).
|
|
250
|
+
if (
|
|
251
|
+
state.turnTerminated &&
|
|
252
|
+
event.type === "run_item_stream_event" &&
|
|
253
|
+
(event as { name?: string }).name === "tool_output" &&
|
|
254
|
+
!abortController.signal.aborted
|
|
255
|
+
) {
|
|
256
|
+
log(
|
|
257
|
+
"agent",
|
|
258
|
+
`[${chatId}] terminator tool result received — aborting wrap-up`,
|
|
259
|
+
);
|
|
260
|
+
try {
|
|
261
|
+
abortController.abort();
|
|
262
|
+
} catch (err) {
|
|
263
|
+
logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
|
|
264
|
+
}
|
|
89
265
|
}
|
|
90
|
-
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// Await the final completion so usage + final state are populated.
|
|
269
|
+
// Safe to call even when we aborted via the terminator (resolves to
|
|
270
|
+
// the partial state).
|
|
271
|
+
await stream.completed.catch(() => {
|
|
272
|
+
/* swallow — aborted-by-terminator path */
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
// Token usage from the underlying RunResult. The SDK aggregates
|
|
276
|
+
// `usage` across all turns in the loop.
|
|
277
|
+
const usage = readRunUsage(stream.state);
|
|
278
|
+
if (usage) recordRunUsage(state, usage);
|
|
91
279
|
}
|
|
92
280
|
|
|
281
|
+
// ── Main handler ────────────────────────────────────────────────────────────
|
|
282
|
+
|
|
93
283
|
export async function handleMessage(
|
|
94
284
|
params: QueryParams,
|
|
95
285
|
_retried = false,
|
|
@@ -101,19 +291,11 @@ export async function handleMessage(
|
|
|
101
291
|
throw new Error("OpenAI Agents backend not initialized");
|
|
102
292
|
}
|
|
103
293
|
|
|
104
|
-
const {
|
|
105
|
-
chatId,
|
|
106
|
-
text,
|
|
107
|
-
senderName,
|
|
108
|
-
senderHandle,
|
|
109
|
-
isGroup,
|
|
110
|
-
messageId,
|
|
111
|
-
onTextBlock,
|
|
112
|
-
onToolUse,
|
|
113
|
-
} = params;
|
|
294
|
+
const { chatId, text, senderName, isGroup, onTextBlock } = params;
|
|
114
295
|
const t0 = Date.now();
|
|
115
296
|
const session = getSession(chatId);
|
|
116
297
|
const previousTurns = session.turns;
|
|
298
|
+
const isRetry = _retried || _flowRetries > 0;
|
|
117
299
|
|
|
118
300
|
// Resolve the active model — chat-settings → config → default.
|
|
119
301
|
const chatSettings = getChatSettings(chatId);
|
|
@@ -140,49 +322,16 @@ export async function handleMessage(
|
|
|
140
322
|
chatId,
|
|
141
323
|
sessionEpoch: session.createdAt,
|
|
142
324
|
});
|
|
143
|
-
|
|
144
|
-
let prompt = formatUserPrompt({
|
|
145
|
-
text,
|
|
146
|
-
senderName: senderName ?? "user",
|
|
147
|
-
senderHandle,
|
|
148
|
-
isGroup,
|
|
149
|
-
messageId,
|
|
150
|
-
});
|
|
151
|
-
// First-turn nudge — turn 0 is where flow violations cluster, and one
|
|
152
|
-
// line in the user message costs nothing on later turns and never
|
|
153
|
-
// touches the cached prefix.
|
|
154
|
-
if (frontend && previousTurns === 0 && !_retried && _flowRetries === 0) {
|
|
155
|
-
prompt += `\n\n${buildFirstTurnReminder(frontend)}`;
|
|
156
|
-
}
|
|
325
|
+
const prompt = buildTurnPrompt(params, frontend, previousTurns, isRetry);
|
|
157
326
|
|
|
158
327
|
log("agent", `[${chatId}] <- (${text.length} chars)`);
|
|
159
328
|
traceMessage(chatId, "in", text, { senderName, isGroup });
|
|
160
329
|
|
|
161
|
-
|
|
162
|
-
// first use, kept alive until `releaseBundle(chatId)`. Avoids the
|
|
163
|
-
// ~15-subprocess re-spawn the original per-turn build caused.
|
|
164
|
-
const bridgeUrl = `http://127.0.0.1:${state.gatewayPortFn()}`;
|
|
165
|
-
let mcpBundle: Awaited<ReturnType<typeof getOrCreateBundle>>;
|
|
166
|
-
try {
|
|
167
|
-
mcpBundle = await getOrCreateBundle({
|
|
168
|
-
chatId,
|
|
169
|
-
bridgeUrl,
|
|
170
|
-
frontends,
|
|
171
|
-
braveApiKey: config.braveApiKey,
|
|
172
|
-
toolExclusions: config,
|
|
173
|
-
});
|
|
174
|
-
} catch (err) {
|
|
175
|
-
logError(
|
|
176
|
-
"agent",
|
|
177
|
-
`[${chatId}] OpenAI Agents: MCP setup failed: ${errMsg(err)}`,
|
|
178
|
-
);
|
|
179
|
-
throw err;
|
|
180
|
-
}
|
|
330
|
+
const mcpBundle = await acquireMcpBundle(chatId, frontends);
|
|
181
331
|
|
|
182
332
|
// Bind the stream state to the chat so token mutators mirror counts
|
|
183
333
|
// into the live-turn overlay — /status updates while the turn runs.
|
|
184
334
|
const streamState = createStreamState(chatId);
|
|
185
|
-
const seenToolCallIds = new Set<string>();
|
|
186
335
|
const abortController = new AbortController();
|
|
187
336
|
activeAborts.set(chatId, abortController);
|
|
188
337
|
// A user interrupt is a synthetic turn terminator: marking the flag
|
|
@@ -199,201 +348,47 @@ export async function handleMessage(
|
|
|
199
348
|
|
|
200
349
|
try {
|
|
201
350
|
const turnStart = Date.now();
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
// Critical for tracking down "model never calls end_turn": if it
|
|
212
|
-
// isn't in this list, the problem is MCP registration, not the model.
|
|
213
|
-
try {
|
|
214
|
-
const builtinNames = OPENAI_AGENTS_BUILTIN_TOOLS.map((t) => t.name);
|
|
215
|
-
const mcpToolLists = await Promise.all(
|
|
216
|
-
mcpBundle.servers.map((s) =>
|
|
217
|
-
s
|
|
218
|
-
.listTools()
|
|
219
|
-
.then((ts: Array<{ name?: string }>) =>
|
|
220
|
-
ts.map((t) => t.name ?? "?"),
|
|
221
|
-
)
|
|
222
|
-
.catch(() => [] as string[]),
|
|
223
|
-
),
|
|
224
|
-
);
|
|
225
|
-
const mcpNames = mcpToolLists.flat();
|
|
226
|
-
log(
|
|
227
|
-
"agent",
|
|
228
|
-
`[${chatId}] tools registered: builtins=[${builtinNames.join(", ")}] mcp=[${mcpNames.join(", ")}]`,
|
|
229
|
-
);
|
|
230
|
-
} catch {
|
|
231
|
-
/* best-effort diagnostic */
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
const agent = new Agent({
|
|
235
|
-
name: OPENAI_AGENTS_AGENT_NAME,
|
|
236
|
-
instructions: systemPrompt,
|
|
237
|
-
model: activeModel,
|
|
238
|
-
tools: [...OPENAI_AGENTS_BUILTIN_TOOLS],
|
|
239
|
-
mcpServers: mcpBundle.servers,
|
|
240
|
-
mcpConfig: { includeServerInToolNames: true },
|
|
241
|
-
});
|
|
242
|
-
|
|
243
|
-
// Per-chat MemorySession so the SDK preserves the full multi-turn
|
|
244
|
-
// record (model outputs, tool calls + results, reasoning items).
|
|
245
|
-
// Without this, every turn starts blind to prior context.
|
|
246
|
-
const stream = await run(agent, prompt, {
|
|
247
|
-
stream: true,
|
|
248
|
-
maxTurns: OPENAI_AGENTS_MAX_TURNS,
|
|
249
|
-
signal: abortController.signal,
|
|
250
|
-
session: getOrCreateSession(chatId),
|
|
251
|
-
});
|
|
252
|
-
|
|
253
|
-
// The Agents SDK aggregates usage on the run context as each model
|
|
254
|
-
// call completes — sample it (throttled) so the live-turn overlay
|
|
255
|
-
// tracks the agentic loop instead of jumping from 0 to final.
|
|
256
|
-
let lastLiveUsagePushAt = 0;
|
|
257
|
-
const pushRunUsageLive = (): void => {
|
|
258
|
-
const now = Date.now();
|
|
259
|
-
if (now - lastLiveUsagePushAt < 1000) return;
|
|
260
|
-
lastLiveUsagePushAt = now;
|
|
261
|
-
const u = readRunUsage(stream.state);
|
|
262
|
-
if (!u) return;
|
|
263
|
-
recordTokens(streamState, {
|
|
264
|
-
inputTokens: u.inputTokens ?? 0,
|
|
265
|
-
outputTokens: u.outputTokens ?? 0,
|
|
266
|
-
cacheRead: u.inputTokensDetails?.cachedTokens ?? 0,
|
|
267
|
-
cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
|
|
268
|
-
});
|
|
269
|
-
};
|
|
270
|
-
|
|
271
|
-
for await (const event of stream) {
|
|
272
|
-
if (abortController.signal.aborted && !streamState.turnTerminated) break;
|
|
273
|
-
|
|
274
|
-
if (event.type === "run_item_stream_event") {
|
|
275
|
-
handleRunItem(event, {
|
|
276
|
-
state: streamState,
|
|
277
|
-
seenToolCallIds,
|
|
278
|
-
onToolUse,
|
|
279
|
-
chatId,
|
|
280
|
-
});
|
|
281
|
-
pushRunUsageLive();
|
|
282
|
-
}
|
|
283
|
-
// `raw_model_stream_event` and `agent_updated_stream_event`
|
|
284
|
-
// events are intentionally not surfaced — token-by-token streaming
|
|
285
|
-
// would expose private chain-of-thought; the final-message event
|
|
286
|
-
// is enough.
|
|
287
|
-
|
|
288
|
-
// Terminator-driven abort. The SDK emits TWO events for each tool
|
|
289
|
-
// call: `tool_called` (RPC about to run) and `tool_output` (RPC
|
|
290
|
-
// completed; message reached the frontend). We must NOT abort on
|
|
291
|
-
// `tool_called` — that cancels the in-flight RPC and the message
|
|
292
|
-
// never ships. Aborting on `tool_output` after we've flagged the
|
|
293
|
-
// turn terminated means delivery happened AND we skip the SDK's
|
|
294
|
-
// wrap-up round-trip (otherwise 5–10s of lingering typing).
|
|
295
|
-
if (
|
|
296
|
-
streamState.turnTerminated &&
|
|
297
|
-
event.type === "run_item_stream_event" &&
|
|
298
|
-
(event as { name?: string }).name === "tool_output" &&
|
|
299
|
-
!abortController.signal.aborted
|
|
300
|
-
) {
|
|
301
|
-
log(
|
|
302
|
-
"agent",
|
|
303
|
-
`[${chatId}] terminator tool result received — aborting wrap-up`,
|
|
304
|
-
);
|
|
305
|
-
try {
|
|
306
|
-
abortController.abort();
|
|
307
|
-
} catch (err) {
|
|
308
|
-
logWarn("agent", `[${chatId}] abort failed: ${errMsg(err)}`);
|
|
309
|
-
}
|
|
310
|
-
}
|
|
311
|
-
}
|
|
312
|
-
|
|
313
|
-
// Await the final completion so usage + final state are populated.
|
|
314
|
-
// Safe to call even when we aborted via the terminator (resolves to
|
|
315
|
-
// the partial state).
|
|
316
|
-
await stream.completed.catch(() => {
|
|
317
|
-
/* swallow — aborted-by-terminator path */
|
|
351
|
+
await driveAgentRun({
|
|
352
|
+
chatId,
|
|
353
|
+
prompt,
|
|
354
|
+
systemPrompt,
|
|
355
|
+
activeModel,
|
|
356
|
+
mcpBundle,
|
|
357
|
+
abortController,
|
|
358
|
+
state: streamState,
|
|
359
|
+
onToolUse: params.onToolUse,
|
|
318
360
|
});
|
|
319
|
-
|
|
320
|
-
// Token usage from the underlying RunResult. The SDK aggregates
|
|
321
|
-
// `usage` across all turns in the loop.
|
|
322
|
-
const usage = readRunUsage(stream.state);
|
|
323
|
-
if (usage) {
|
|
324
|
-
recordTokens(streamState, {
|
|
325
|
-
inputTokens: usage.inputTokens ?? 0,
|
|
326
|
-
outputTokens: usage.outputTokens ?? 0,
|
|
327
|
-
cacheRead: usage.inputTokensDetails?.cachedTokens ?? 0,
|
|
328
|
-
cacheWrite: 0, // OpenAI Responses API doesn't report cache writes.
|
|
329
|
-
});
|
|
330
|
-
}
|
|
331
|
-
|
|
332
361
|
turnMs = Date.now() - turnStart;
|
|
333
362
|
} catch (err) {
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
incrementCounter(`errors.${classified.reason ?? "unknown"}`);
|
|
342
|
-
|
|
343
|
-
const decision = classifyRetry({
|
|
344
|
-
error: classified,
|
|
363
|
+
// Swallow the terminator abort — the turn completed via a delivery tool.
|
|
364
|
+
if (!isTerminatorAbort(streamState, err)) {
|
|
365
|
+
// MCP bundle is retained across a retry — subprocesses are
|
|
366
|
+
// stateless wrt the model conversation. See `mcp-pool.ts`.
|
|
367
|
+
const outcome = await applyRetryDecision({
|
|
368
|
+
err,
|
|
369
|
+
chatId,
|
|
345
370
|
activeModel,
|
|
346
371
|
retried: _retried,
|
|
372
|
+
params,
|
|
373
|
+
recurseWithRetried: (p) => handleMessage(p, true),
|
|
374
|
+
backendLabel: "OpenAI Agents",
|
|
347
375
|
});
|
|
348
|
-
|
|
349
|
-
if (decision.kind === "reset_and_retry") {
|
|
350
|
-
logWarn(
|
|
351
|
-
"agent",
|
|
352
|
-
`[${chatId}] OpenAI Agents ${decision.reason}, resetting session and retrying`,
|
|
353
|
-
);
|
|
354
|
-
resetSession(chatId);
|
|
355
|
-
// MCP bundle is retained across the retry — subprocesses are
|
|
356
|
-
// stateless wrt the model conversation. See `mcp-pool.ts`.
|
|
357
|
-
return handleMessage(params, true);
|
|
358
|
-
}
|
|
359
|
-
|
|
360
|
-
if (decision.kind === "fallback_model") {
|
|
361
|
-
logWarn(
|
|
362
|
-
"agent",
|
|
363
|
-
`[${chatId}] ${classified.reason}, falling back to ${decision.fallbackModelId}`,
|
|
364
|
-
);
|
|
365
|
-
resetSession(chatId);
|
|
366
|
-
return await handleMessage(
|
|
367
|
-
{ ...params, model: decision.fallbackModelId },
|
|
368
|
-
true,
|
|
369
|
-
);
|
|
370
|
-
}
|
|
376
|
+
if (outcome.retry) return outcome.retry;
|
|
371
377
|
|
|
372
378
|
// Terminal failure — account for whatever the turn consumed before
|
|
373
|
-
// dying
|
|
374
|
-
|
|
375
|
-
recordFailedTurnAccounting({
|
|
379
|
+
// dying (the retry above did its own accounting).
|
|
380
|
+
accountFailedTurn({
|
|
376
381
|
backend: "openai-agents",
|
|
377
382
|
chatId,
|
|
383
|
+
state: streamState,
|
|
378
384
|
durationMs: Date.now() - t0,
|
|
379
|
-
toolCalls: streamState.toolCalls,
|
|
380
|
-
apiCalls: streamState.numApiCalls,
|
|
381
385
|
model: activeModel,
|
|
382
|
-
usage: {
|
|
383
|
-
inputTokens: streamState.sdkInputTokens,
|
|
384
|
-
outputTokens: streamState.sdkOutputTokens,
|
|
385
|
-
cacheRead: streamState.sdkCacheRead,
|
|
386
|
-
cacheWrite: streamState.sdkCacheWrite,
|
|
387
|
-
},
|
|
388
|
-
contextTokens: streamState.contextTokens,
|
|
389
|
-
contextWindow: streamState.contextWindow,
|
|
390
386
|
});
|
|
391
|
-
|
|
392
387
|
logError(
|
|
393
388
|
"agent",
|
|
394
|
-
`[${chatId}] OpenAI Agents error: ${classified.message}`,
|
|
389
|
+
`[${chatId}] OpenAI Agents error: ${outcome.classified.message}`,
|
|
395
390
|
);
|
|
396
|
-
throw classified;
|
|
391
|
+
throw outcome.classified;
|
|
397
392
|
}
|
|
398
393
|
} finally {
|
|
399
394
|
unregisterInterrupt();
|
|
@@ -409,86 +404,41 @@ export async function handleMessage(
|
|
|
409
404
|
|
|
410
405
|
const responseText = finalizeResponseText(streamState);
|
|
411
406
|
const durationMs = Date.now() - t0;
|
|
412
|
-
|
|
407
|
+
accountTurn({
|
|
413
408
|
chatId,
|
|
414
409
|
backend: "openai-agents",
|
|
415
|
-
|
|
416
|
-
toolCalls: streamState.toolCalls,
|
|
417
|
-
apiCalls: streamState.numApiCalls,
|
|
418
|
-
usage: {
|
|
419
|
-
inputTokens: streamState.sdkInputTokens,
|
|
420
|
-
outputTokens: streamState.sdkOutputTokens,
|
|
421
|
-
cacheRead: streamState.sdkCacheRead,
|
|
422
|
-
cacheWrite: streamState.sdkCacheWrite,
|
|
423
|
-
},
|
|
424
|
-
});
|
|
425
|
-
|
|
426
|
-
recordUsage(chatId, {
|
|
427
|
-
inputTokens: streamState.sdkInputTokens,
|
|
428
|
-
outputTokens: streamState.sdkOutputTokens,
|
|
429
|
-
cacheRead: streamState.sdkCacheRead,
|
|
430
|
-
cacheWrite: streamState.sdkCacheWrite,
|
|
410
|
+
state: streamState,
|
|
431
411
|
durationMs,
|
|
432
412
|
model: activeModel,
|
|
433
413
|
});
|
|
434
414
|
|
|
435
|
-
// ── Trailing-prose contract + flow-violation retry ──────────────────────
|
|
436
415
|
// Replies MUST go through `end_turn` (canonical) or `send` (mid-turn).
|
|
437
|
-
//
|
|
438
|
-
//
|
|
439
|
-
//
|
|
440
|
-
// (non-empty mcpBundle.servers). `incrementTurns` is deferred until AFTER
|
|
441
|
-
// the check so the retry path doesn't double-count.
|
|
416
|
+
// Only enforced when delivery tools are registered (non-empty
|
|
417
|
+
// mcpBundle.servers). `incrementTurns` is deferred until AFTER the
|
|
418
|
+
// check so the retry path doesn't double-count.
|
|
442
419
|
const violation =
|
|
443
420
|
mcpBundle.servers.length > 0
|
|
444
|
-
?
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
toolCalls: streamState.toolCalls,
|
|
449
|
-
retried: _flowRetries > 0,
|
|
450
|
-
retryCount: _flowRetries,
|
|
451
|
-
maxRetries: FLOW_VIOLATION_MAX_RETRIES,
|
|
421
|
+
? enforceTrailingProse({
|
|
422
|
+
chatId,
|
|
423
|
+
state: streamState,
|
|
424
|
+
flowRetries: _flowRetries,
|
|
452
425
|
...(frontend
|
|
453
426
|
? { reminder: buildFlowViolationReminder(frontend) }
|
|
454
427
|
: {}),
|
|
455
428
|
})
|
|
456
|
-
:
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
log(
|
|
464
|
-
"agent",
|
|
465
|
-
`[${chatId}] flow violation: trailing prose (${violation.trailing.length} chars) without end_turn/send. ${
|
|
466
|
-
violation.shouldRetry
|
|
467
|
-
? "Re-prompting with reminder."
|
|
468
|
-
: "Already retried — accepting silent drop."
|
|
469
|
-
}`,
|
|
429
|
+
: undefined;
|
|
430
|
+
if (violation?.violated && violation.shouldRetry) {
|
|
431
|
+
// Recursive call owns the `incrementTurns` for this user message.
|
|
432
|
+
return handleMessage(
|
|
433
|
+
{ ...params, text: violation.reminder },
|
|
434
|
+
_retried,
|
|
435
|
+
_flowRetries + 1,
|
|
470
436
|
);
|
|
471
|
-
|
|
472
|
-
if (violation.shouldRetry) {
|
|
473
|
-
// Recursive call owns the `incrementTurns` for this user message.
|
|
474
|
-
return handleMessage(
|
|
475
|
-
{ ...params, text: violation.reminder },
|
|
476
|
-
_retried,
|
|
477
|
-
_flowRetries + 1,
|
|
478
|
-
);
|
|
479
|
-
}
|
|
480
437
|
}
|
|
481
438
|
|
|
482
439
|
// Reached the non-retry path — this turn counts as one user-visible turn.
|
|
483
440
|
incrementTurns(chatId);
|
|
484
|
-
|
|
485
|
-
// Set a descriptive session name from the user's *first* message.
|
|
486
|
-
// Guarded by `!_retried` so the reminder doesn't get captured as the
|
|
487
|
-
// session name when the retry recurses with `params.text = reminder`.
|
|
488
|
-
if (previousTurns === 0 && !_retried && _flowRetries === 0) {
|
|
489
|
-
const name = extractSessionName(text);
|
|
490
|
-
if (name) setSessionName(chatId, name);
|
|
491
|
-
}
|
|
441
|
+
nameSessionFromFirstMessage({ chatId, text, previousTurns, isRetry });
|
|
492
442
|
|
|
493
443
|
// ── Delivery — strict tool-only ──────────────────────────────────────────
|
|
494
444
|
// Replies must reach the user via a delivery tool. Trailing prose is
|
|
@@ -507,37 +457,13 @@ export async function handleMessage(
|
|
|
507
457
|
})
|
|
508
458
|
: { route: "silent" as const, chars: 0 };
|
|
509
459
|
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
log(
|
|
516
|
-
"agent",
|
|
517
|
-
`[${chatId}] -> (${summarizeUsage(
|
|
518
|
-
{
|
|
519
|
-
inputTokens: streamState.sdkInputTokens,
|
|
520
|
-
outputTokens: streamState.sdkOutputTokens,
|
|
521
|
-
cacheRead: streamState.sdkCacheRead,
|
|
522
|
-
cacheWrite: streamState.sdkCacheWrite,
|
|
523
|
-
},
|
|
524
|
-
{ durationMs, toolCalls: streamState.toolCalls },
|
|
525
|
-
)} terminator=${streamState.turnTerminated ? "yes" : "no"} ` +
|
|
526
|
-
`delivered=${streamState.deliveredTextNorms.length} ` +
|
|
527
|
-
`respLen=${responseText.length} ` +
|
|
528
|
-
`setup=${setupMs}ms turn=${turnMs}ms)`,
|
|
529
|
-
);
|
|
530
|
-
traceMessage(chatId, "out", responseText, {
|
|
460
|
+
return finishCallbackTurn({
|
|
461
|
+
chatId,
|
|
462
|
+
state: streamState,
|
|
463
|
+
responseText,
|
|
531
464
|
durationMs,
|
|
532
|
-
|
|
465
|
+
setupMs,
|
|
466
|
+
turnMs,
|
|
467
|
+
delivery,
|
|
533
468
|
});
|
|
534
|
-
|
|
535
|
-
return {
|
|
536
|
-
text: responseText,
|
|
537
|
-
durationMs,
|
|
538
|
-
inputTokens: streamState.sdkInputTokens,
|
|
539
|
-
outputTokens: streamState.sdkOutputTokens,
|
|
540
|
-
cacheRead: streamState.sdkCacheRead,
|
|
541
|
-
cacheWrite: streamState.sdkCacheWrite,
|
|
542
|
-
};
|
|
543
469
|
}
|