@prestyj/agent 5.10.0 → 5.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +140 -26
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +73 -2
- package/dist/index.d.ts +73 -2
- package/dist/index.js +143 -27
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/index.d.cts
CHANGED
|
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
|
|
|
20
20
|
* batch runs in source order so stateful mutations cannot race each other.
|
|
21
21
|
*/
|
|
22
22
|
executionMode?: ToolExecutionMode;
|
|
23
|
+
/**
|
|
24
|
+
* Overrides the loop's default per-tool timeout. A tool that owns a longer
|
|
25
|
+
* internal budget than the default must declare it here, or the loop cancels
|
|
26
|
+
* it first and the tool's own timeout — with its specific, actionable error
|
|
27
|
+
* message — becomes unreachable.
|
|
28
|
+
*/
|
|
29
|
+
timeoutMs?: number;
|
|
23
30
|
execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
|
|
24
31
|
}
|
|
25
32
|
interface AgentTextDeltaEvent {
|
|
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
|
|
|
70
77
|
usage: Usage;
|
|
71
78
|
timing: AgentTurnTiming;
|
|
72
79
|
}
|
|
80
|
+
/**
|
|
81
|
+
* A safe point between steps: the assistant message and every tool result for
|
|
82
|
+
* this turn are now in the message array, and no provider call is in flight.
|
|
83
|
+
*
|
|
84
|
+
* Hosts that persist a transcript flush here. Without it a crash mid-run loses
|
|
85
|
+
* the WHOLE turn — including tool results whose side effects already landed on
|
|
86
|
+
* disk — because the only flush happens after the loop returns.
|
|
87
|
+
*
|
|
88
|
+
* Yielded immediately after tool results are appended, so it pairs with
|
|
89
|
+
* `turn_end` (which covers the assistant half) to cover every message.
|
|
90
|
+
*/
|
|
91
|
+
interface AgentCheckpointEvent {
|
|
92
|
+
type: "checkpoint";
|
|
93
|
+
turn: number;
|
|
94
|
+
}
|
|
73
95
|
interface AgentDoneEvent {
|
|
74
96
|
type: "agent_done";
|
|
75
97
|
totalTurns: number;
|
|
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
|
|
|
87
109
|
totalTurns: number;
|
|
88
110
|
maxTurns: number;
|
|
89
111
|
}
|
|
112
|
+
/**
|
|
113
|
+
* Emitted when the loop was about to stop on an exhausted turn budget but the
|
|
114
|
+
* host granted an extension instead. The effective budget is raised and the
|
|
115
|
+
* loop continues with a continuation prompt, so this is NOT terminal — unlike
|
|
116
|
+
* `max_turns`, which still fires if the extended budget is also spent.
|
|
117
|
+
*/
|
|
118
|
+
interface AgentTurnBudgetExtendedEvent {
|
|
119
|
+
type: "turn_budget_extended";
|
|
120
|
+
/** Turn number at which the budget was exhausted. */
|
|
121
|
+
turn: number;
|
|
122
|
+
/** New effective `maxTurns` after the extension. */
|
|
123
|
+
grantedTurns: number;
|
|
124
|
+
/** 1-based extension count for this run. */
|
|
125
|
+
extension: number;
|
|
126
|
+
}
|
|
90
127
|
/**
|
|
91
128
|
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
129
|
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
|
|
|
148
185
|
type: "follow_up_message";
|
|
149
186
|
content: Message["content"];
|
|
150
187
|
}
|
|
151
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
188
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
189
|
interface TransformContextOptions {
|
|
153
190
|
/** Force a transform after the provider reports context overflow. */
|
|
154
191
|
force?: boolean;
|
|
@@ -168,10 +205,32 @@ interface AgentOptions {
|
|
|
168
205
|
/** Control whether tools may/must be called, or select a named tool when supported. */
|
|
169
206
|
toolChoice?: StreamOptions["toolChoice"];
|
|
170
207
|
maxTurns?: number;
|
|
208
|
+
/**
|
|
209
|
+
* How many times `onTurnBudgetExhausted` may grant extra turns in one run.
|
|
210
|
+
* Each grant raises the effective budget by the original `maxTurns`.
|
|
211
|
+
* Default: 2. Set 0 to disable extensions entirely.
|
|
212
|
+
*/
|
|
213
|
+
maxTurnExtensions?: number;
|
|
171
214
|
maxTokens?: number;
|
|
172
215
|
temperature?: number;
|
|
173
216
|
thinking?: StreamOptions["thinking"];
|
|
174
217
|
apiKey?: string;
|
|
218
|
+
/**
|
|
219
|
+
* Re-resolve the credential at the start of every turn. A run can span many
|
|
220
|
+
* minutes, and an OAuth grant refreshed by any process (another app window, a
|
|
221
|
+
* CLI session, the usage poller) invalidates the access token captured when
|
|
222
|
+
* the run began — so a pinned `apiKey` goes dead mid-run and every remaining
|
|
223
|
+
* turn fails with an authentication error. Returning the current credential
|
|
224
|
+
* here keeps a long run alive across rotations.
|
|
225
|
+
*
|
|
226
|
+
* Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
|
|
227
|
+
* resolver throws (the provider call then surfaces the real auth error).
|
|
228
|
+
*/
|
|
229
|
+
resolveCredentials?: () => Promise<{
|
|
230
|
+
apiKey: string;
|
|
231
|
+
accountId?: string;
|
|
232
|
+
projectId?: string;
|
|
233
|
+
}>;
|
|
175
234
|
baseUrl?: string;
|
|
176
235
|
signal?: AbortSignal;
|
|
177
236
|
accountId?: string;
|
|
@@ -234,6 +293,18 @@ interface AgentOptions {
|
|
|
234
293
|
* on read.
|
|
235
294
|
*/
|
|
236
295
|
getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
|
|
296
|
+
/**
|
|
297
|
+
* Consulted when a tool-running turn exhausts the turn budget mid-task,
|
|
298
|
+
* before the loop emits the terminal `max_turns` event. Return true to grant
|
|
299
|
+
* another `maxTurns` worth of turns; false (the default when unset) keeps
|
|
300
|
+
* today's hard cut-off. Hosts should only grant on evidence of progress —
|
|
301
|
+
* extending a spinning agent just buys it more tokens to spin with.
|
|
302
|
+
*/
|
|
303
|
+
onTurnBudgetExhausted?: (ctx: {
|
|
304
|
+
turn: number;
|
|
305
|
+
maxTurns: number;
|
|
306
|
+
extension: number;
|
|
307
|
+
}) => Promise<boolean> | boolean;
|
|
237
308
|
}
|
|
238
309
|
interface AgentResult {
|
|
239
310
|
message: AssistantMessage;
|
|
@@ -319,7 +390,7 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
319
390
|
* plan running out of usage). Unlike a transient per-minute 429, this does NOT
|
|
320
391
|
* clear with a quick retry — the user must wait for the window to reset — so the
|
|
321
392
|
* loop surfaces it immediately instead of retrying for minutes. Matches the
|
|
322
|
-
* canonical message
|
|
393
|
+
* canonical message @prestyj/ai stamps onto the provider error.
|
|
323
394
|
*/
|
|
324
395
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
325
396
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
package/dist/index.d.ts
CHANGED
|
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
|
|
|
20
20
|
* batch runs in source order so stateful mutations cannot race each other.
|
|
21
21
|
*/
|
|
22
22
|
executionMode?: ToolExecutionMode;
|
|
23
|
+
/**
|
|
24
|
+
* Overrides the loop's default per-tool timeout. A tool that owns a longer
|
|
25
|
+
* internal budget than the default must declare it here, or the loop cancels
|
|
26
|
+
* it first and the tool's own timeout — with its specific, actionable error
|
|
27
|
+
* message — becomes unreachable.
|
|
28
|
+
*/
|
|
29
|
+
timeoutMs?: number;
|
|
23
30
|
execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
|
|
24
31
|
}
|
|
25
32
|
interface AgentTextDeltaEvent {
|
|
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
|
|
|
70
77
|
usage: Usage;
|
|
71
78
|
timing: AgentTurnTiming;
|
|
72
79
|
}
|
|
80
|
+
/**
|
|
81
|
+
* A safe point between steps: the assistant message and every tool result for
|
|
82
|
+
* this turn are now in the message array, and no provider call is in flight.
|
|
83
|
+
*
|
|
84
|
+
* Hosts that persist a transcript flush here. Without it a crash mid-run loses
|
|
85
|
+
* the WHOLE turn — including tool results whose side effects already landed on
|
|
86
|
+
* disk — because the only flush happens after the loop returns.
|
|
87
|
+
*
|
|
88
|
+
* Yielded immediately after tool results are appended, so it pairs with
|
|
89
|
+
* `turn_end` (which covers the assistant half) to cover every message.
|
|
90
|
+
*/
|
|
91
|
+
interface AgentCheckpointEvent {
|
|
92
|
+
type: "checkpoint";
|
|
93
|
+
turn: number;
|
|
94
|
+
}
|
|
73
95
|
interface AgentDoneEvent {
|
|
74
96
|
type: "agent_done";
|
|
75
97
|
totalTurns: number;
|
|
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
|
|
|
87
109
|
totalTurns: number;
|
|
88
110
|
maxTurns: number;
|
|
89
111
|
}
|
|
112
|
+
/**
|
|
113
|
+
* Emitted when the loop was about to stop on an exhausted turn budget but the
|
|
114
|
+
* host granted an extension instead. The effective budget is raised and the
|
|
115
|
+
* loop continues with a continuation prompt, so this is NOT terminal — unlike
|
|
116
|
+
* `max_turns`, which still fires if the extended budget is also spent.
|
|
117
|
+
*/
|
|
118
|
+
interface AgentTurnBudgetExtendedEvent {
|
|
119
|
+
type: "turn_budget_extended";
|
|
120
|
+
/** Turn number at which the budget was exhausted. */
|
|
121
|
+
turn: number;
|
|
122
|
+
/** New effective `maxTurns` after the extension. */
|
|
123
|
+
grantedTurns: number;
|
|
124
|
+
/** 1-based extension count for this run. */
|
|
125
|
+
extension: number;
|
|
126
|
+
}
|
|
90
127
|
/**
|
|
91
128
|
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
129
|
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
|
|
|
148
185
|
type: "follow_up_message";
|
|
149
186
|
content: Message["content"];
|
|
150
187
|
}
|
|
151
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
188
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
189
|
interface TransformContextOptions {
|
|
153
190
|
/** Force a transform after the provider reports context overflow. */
|
|
154
191
|
force?: boolean;
|
|
@@ -168,10 +205,32 @@ interface AgentOptions {
|
|
|
168
205
|
/** Control whether tools may/must be called, or select a named tool when supported. */
|
|
169
206
|
toolChoice?: StreamOptions["toolChoice"];
|
|
170
207
|
maxTurns?: number;
|
|
208
|
+
/**
|
|
209
|
+
* How many times `onTurnBudgetExhausted` may grant extra turns in one run.
|
|
210
|
+
* Each grant raises the effective budget by the original `maxTurns`.
|
|
211
|
+
* Default: 2. Set 0 to disable extensions entirely.
|
|
212
|
+
*/
|
|
213
|
+
maxTurnExtensions?: number;
|
|
171
214
|
maxTokens?: number;
|
|
172
215
|
temperature?: number;
|
|
173
216
|
thinking?: StreamOptions["thinking"];
|
|
174
217
|
apiKey?: string;
|
|
218
|
+
/**
|
|
219
|
+
* Re-resolve the credential at the start of every turn. A run can span many
|
|
220
|
+
* minutes, and an OAuth grant refreshed by any process (another app window, a
|
|
221
|
+
* CLI session, the usage poller) invalidates the access token captured when
|
|
222
|
+
* the run began — so a pinned `apiKey` goes dead mid-run and every remaining
|
|
223
|
+
* turn fails with an authentication error. Returning the current credential
|
|
224
|
+
* here keeps a long run alive across rotations.
|
|
225
|
+
*
|
|
226
|
+
* Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
|
|
227
|
+
* resolver throws (the provider call then surfaces the real auth error).
|
|
228
|
+
*/
|
|
229
|
+
resolveCredentials?: () => Promise<{
|
|
230
|
+
apiKey: string;
|
|
231
|
+
accountId?: string;
|
|
232
|
+
projectId?: string;
|
|
233
|
+
}>;
|
|
175
234
|
baseUrl?: string;
|
|
176
235
|
signal?: AbortSignal;
|
|
177
236
|
accountId?: string;
|
|
@@ -234,6 +293,18 @@ interface AgentOptions {
|
|
|
234
293
|
* on read.
|
|
235
294
|
*/
|
|
236
295
|
getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
|
|
296
|
+
/**
|
|
297
|
+
* Consulted when a tool-running turn exhausts the turn budget mid-task,
|
|
298
|
+
* before the loop emits the terminal `max_turns` event. Return true to grant
|
|
299
|
+
* another `maxTurns` worth of turns; false (the default when unset) keeps
|
|
300
|
+
* today's hard cut-off. Hosts should only grant on evidence of progress —
|
|
301
|
+
* extending a spinning agent just buys it more tokens to spin with.
|
|
302
|
+
*/
|
|
303
|
+
onTurnBudgetExhausted?: (ctx: {
|
|
304
|
+
turn: number;
|
|
305
|
+
maxTurns: number;
|
|
306
|
+
extension: number;
|
|
307
|
+
}) => Promise<boolean> | boolean;
|
|
237
308
|
}
|
|
238
309
|
interface AgentResult {
|
|
239
310
|
message: AssistantMessage;
|
|
@@ -319,7 +390,7 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
319
390
|
* plan running out of usage). Unlike a transient per-minute 429, this does NOT
|
|
320
391
|
* clear with a quick retry — the user must wait for the window to reset — so the
|
|
321
392
|
* loop surfaces it immediately instead of retrying for minutes. Matches the
|
|
322
|
-
* canonical message
|
|
393
|
+
* canonical message @prestyj/ai stamps onto the provider error.
|
|
323
394
|
*/
|
|
324
395
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
325
396
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
package/dist/index.js
CHANGED
|
@@ -8,7 +8,9 @@ import {
|
|
|
8
8
|
EventStream,
|
|
9
9
|
EZCoderAIError,
|
|
10
10
|
isHardBillingMessage,
|
|
11
|
-
redactValue
|
|
11
|
+
redactValue,
|
|
12
|
+
sliceHead,
|
|
13
|
+
sliceTail
|
|
12
14
|
} from "@prestyj/ai";
|
|
13
15
|
|
|
14
16
|
// src/local-backend.ts
|
|
@@ -35,6 +37,7 @@ function isLocalBackendUrl(baseUrl) {
|
|
|
35
37
|
|
|
36
38
|
// src/agent-loop.ts
|
|
37
39
|
var DEFAULT_MAX_TURNS = 300;
|
|
40
|
+
var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
|
|
38
41
|
var _diagFn = null;
|
|
39
42
|
function setStreamDiagnostic(fn) {
|
|
40
43
|
_diagFn = fn;
|
|
@@ -54,7 +57,13 @@ function isContextOverflow(err) {
|
|
|
54
57
|
if (overflowStatus === 402) return false;
|
|
55
58
|
if (isBillingError(err)) return false;
|
|
56
59
|
const msg = err.message.toLowerCase();
|
|
57
|
-
|
|
60
|
+
if (msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten")) {
|
|
61
|
+
return true;
|
|
62
|
+
}
|
|
63
|
+
const rateLimited = overflowStatus === 429 || msg.includes("rate limit") || msg.includes("rate_limit") || msg.includes("too many requests");
|
|
64
|
+
const perUnitTime = msg.includes("per min") || msg.includes("/min") || msg.includes("per minute") || msg.includes("per hour") || msg.includes("per day") || msg.includes("tpm") || msg.includes("rpm");
|
|
65
|
+
if (rateLimited && perUnitTime) return false;
|
|
66
|
+
return msg.includes("token") && msg.includes("exceed");
|
|
58
67
|
}
|
|
59
68
|
function parseOverflowNumber(value) {
|
|
60
69
|
return Number(value.replace(/[,_\s]/g, ""));
|
|
@@ -167,6 +176,17 @@ function isMalformedStream(err) {
|
|
|
167
176
|
const msg = err.message;
|
|
168
177
|
return /\bin JSON at position \d+/i.test(msg);
|
|
169
178
|
}
|
|
179
|
+
var TIMEOUT_NAMES = /* @__PURE__ */ new Set(["TimeoutError", "ConnectTimeoutError", "HeadersTimeoutError"]);
|
|
180
|
+
var TIMEOUT_MESSAGES = [
|
|
181
|
+
/^request timed out\.?$/i,
|
|
182
|
+
/\brequest to [\w .-]+ timed out\b/i,
|
|
183
|
+
/\b(?:connection|socket|headers|stream) timed out\b/i
|
|
184
|
+
];
|
|
185
|
+
function isBareTimeout(e) {
|
|
186
|
+
if (typeof e.name === "string" && TIMEOUT_NAMES.has(e.name)) return true;
|
|
187
|
+
if (typeof e.message !== "string") return false;
|
|
188
|
+
return TIMEOUT_MESSAGES.some((re) => re.test(e.message));
|
|
189
|
+
}
|
|
170
190
|
function isTransportFailure(err) {
|
|
171
191
|
const codes = /* @__PURE__ */ new Set([
|
|
172
192
|
"ECONNRESET",
|
|
@@ -203,6 +223,8 @@ function isTransportFailure(err) {
|
|
|
203
223
|
if (typeof e.message === "string") {
|
|
204
224
|
for (const re of messages) if (re.test(e.message)) return true;
|
|
205
225
|
}
|
|
226
|
+
const clientError = typeof e.status === "number" && e.status >= 400 && e.status < 500;
|
|
227
|
+
if (!clientError && isBareTimeout(e)) return true;
|
|
206
228
|
cur = e.cause;
|
|
207
229
|
}
|
|
208
230
|
return false;
|
|
@@ -237,6 +259,9 @@ function abortablePromise(promise, signal) {
|
|
|
237
259
|
promise.then(resolveOnce, rejectOnce);
|
|
238
260
|
});
|
|
239
261
|
}
|
|
262
|
+
function turnBudgetContinuationPrompt() {
|
|
263
|
+
return "[You reached the turn limit for this segment but the work is not finished, so you have been granted more turns. Before continuing, state in one or two sentences what is already done and what remains, then keep going from there \u2014 do not restart work that is already complete.]";
|
|
264
|
+
}
|
|
240
265
|
function abortableSleep(ms, signal) {
|
|
241
266
|
if (signal?.aborted) return Promise.reject(createAbortError());
|
|
242
267
|
return new Promise((resolve, reject) => {
|
|
@@ -258,6 +283,9 @@ function closeIterator(iterator) {
|
|
|
258
283
|
}
|
|
259
284
|
async function* agentLoop(messages, options) {
|
|
260
285
|
const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
|
|
286
|
+
let effectiveMaxTurns = maxTurns;
|
|
287
|
+
const maxTurnExtensions = Math.max(0, options.maxTurnExtensions ?? 2);
|
|
288
|
+
let turnExtensions = 0;
|
|
261
289
|
const maxContinuations = options.maxContinuations ?? 5;
|
|
262
290
|
let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
|
|
263
291
|
const totalUsage = { inputTokens: 0, outputTokens: 0 };
|
|
@@ -305,12 +333,12 @@ async function* agentLoop(messages, options) {
|
|
|
305
333
|
const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
|
|
306
334
|
const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
|
|
307
335
|
const MAX_TOOLCALL_DELTA_CHARS = 1e6;
|
|
308
|
-
const
|
|
336
|
+
const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
|
|
309
337
|
let logicalTurnStartedAt = 0;
|
|
310
338
|
let firstProviderEventAt;
|
|
311
339
|
let providerDurationMs = 0;
|
|
312
340
|
try {
|
|
313
|
-
while (turn <
|
|
341
|
+
while (turn < effectiveMaxTurns) {
|
|
314
342
|
options.signal?.throwIfAborted();
|
|
315
343
|
turn++;
|
|
316
344
|
if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
|
|
@@ -381,6 +409,7 @@ async function* agentLoop(messages, options) {
|
|
|
381
409
|
let lastEventType = "";
|
|
382
410
|
let toolcallDeltaChars = 0;
|
|
383
411
|
let toolcallDeltaCount = 0;
|
|
412
|
+
let toolcallNoProgressCount = 0;
|
|
384
413
|
let runawayDetected = null;
|
|
385
414
|
let attemptText = "";
|
|
386
415
|
let lastYieldEndTime = Date.now();
|
|
@@ -421,6 +450,21 @@ async function* agentLoop(messages, options) {
|
|
|
421
450
|
diag("stream_call", { nonStreaming: useNonStreamingFallback });
|
|
422
451
|
streamCallStart = Date.now();
|
|
423
452
|
providerAttemptStartedAt = streamCallStart;
|
|
453
|
+
let liveApiKey = options.apiKey;
|
|
454
|
+
let liveAccountId = options.accountId;
|
|
455
|
+
let liveProjectId = options.projectId;
|
|
456
|
+
if (options.resolveCredentials) {
|
|
457
|
+
try {
|
|
458
|
+
const fresh = await options.resolveCredentials();
|
|
459
|
+
liveApiKey = fresh.apiKey;
|
|
460
|
+
if (fresh.accountId !== void 0) liveAccountId = fresh.accountId;
|
|
461
|
+
if (fresh.projectId !== void 0) liveProjectId = fresh.projectId;
|
|
462
|
+
} catch (credErr) {
|
|
463
|
+
diag("credential_refresh_failed", {
|
|
464
|
+
error: (credErr instanceof Error ? credErr.message : String(credErr)).slice(0, 200)
|
|
465
|
+
});
|
|
466
|
+
}
|
|
467
|
+
}
|
|
424
468
|
const result = stream({
|
|
425
469
|
provider: options.provider,
|
|
426
470
|
model: options.model,
|
|
@@ -432,12 +476,12 @@ async function* agentLoop(messages, options) {
|
|
|
432
476
|
maxTokens: options.maxTokens,
|
|
433
477
|
temperature: options.temperature,
|
|
434
478
|
thinking: options.thinking,
|
|
435
|
-
apiKey:
|
|
479
|
+
apiKey: liveApiKey,
|
|
436
480
|
baseUrl: options.baseUrl,
|
|
437
481
|
signal: streamController.signal,
|
|
438
|
-
accountId:
|
|
482
|
+
accountId: liveAccountId,
|
|
439
483
|
transportSessionId: options.transportSessionId,
|
|
440
|
-
projectId:
|
|
484
|
+
projectId: liveProjectId,
|
|
441
485
|
cacheRetention: options.cacheRetention,
|
|
442
486
|
promptCacheKey: options.promptCacheKey,
|
|
443
487
|
serviceTier: options.serviceTier,
|
|
@@ -535,11 +579,13 @@ async function* agentLoop(messages, options) {
|
|
|
535
579
|
const chunkChars = event.argsJson?.length ?? 0;
|
|
536
580
|
toolcallDeltaChars += chunkChars;
|
|
537
581
|
toolcallDeltaCount++;
|
|
538
|
-
|
|
582
|
+
toolcallNoProgressCount = chunkChars > 0 ? 0 : toolcallNoProgressCount + 1;
|
|
583
|
+
if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallNoProgressCount > MAX_TOOLCALL_NO_PROGRESS_EVENTS)) {
|
|
539
584
|
runawayDetected = {
|
|
540
585
|
kind: toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS ? "chars" : "events",
|
|
541
586
|
chars: toolcallDeltaChars,
|
|
542
|
-
events: toolcallDeltaCount
|
|
587
|
+
events: toolcallDeltaCount,
|
|
588
|
+
noProgressEvents: toolcallNoProgressCount
|
|
543
589
|
};
|
|
544
590
|
diag("runaway_toolcall_detected", {
|
|
545
591
|
...runawayDetected,
|
|
@@ -721,11 +767,15 @@ async function* agentLoop(messages, options) {
|
|
|
721
767
|
turn--;
|
|
722
768
|
continue;
|
|
723
769
|
}
|
|
724
|
-
const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.
|
|
770
|
+
const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.noProgressEvents} consecutive tool-call delta events without argument progress`;
|
|
725
771
|
yield {
|
|
726
772
|
type: "error",
|
|
727
|
-
error: new
|
|
728
|
-
`The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}).
|
|
773
|
+
error: new EZCoderAIError(
|
|
774
|
+
`The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Your conversation is preserved.`,
|
|
775
|
+
{
|
|
776
|
+
source: "provider",
|
|
777
|
+
hint: "Retry once. If it keeps happening, switch models and continue the preserved conversation."
|
|
778
|
+
}
|
|
729
779
|
)
|
|
730
780
|
};
|
|
731
781
|
break;
|
|
@@ -752,7 +802,15 @@ async function* agentLoop(messages, options) {
|
|
|
752
802
|
role: "assistant",
|
|
753
803
|
content: [{ type: "text", text: attemptText }]
|
|
754
804
|
});
|
|
755
|
-
messages.push({
|
|
805
|
+
messages.push({
|
|
806
|
+
role: "user",
|
|
807
|
+
content: PARTIAL_CONTINUATION_PROMPT,
|
|
808
|
+
provenance: {
|
|
809
|
+
source: "runtime",
|
|
810
|
+
kind: "continuation",
|
|
811
|
+
visibility: "hidden"
|
|
812
|
+
}
|
|
813
|
+
});
|
|
756
814
|
preservedChars = attemptText.length;
|
|
757
815
|
}
|
|
758
816
|
diag("retry", {
|
|
@@ -871,6 +929,9 @@ async function* agentLoop(messages, options) {
|
|
|
871
929
|
useNonStreamingFallback = false;
|
|
872
930
|
totalUsage.inputTokens += response.usage.inputTokens;
|
|
873
931
|
totalUsage.outputTokens += response.usage.outputTokens;
|
|
932
|
+
if (response.usage.reasoningTokens) {
|
|
933
|
+
totalUsage.reasoningTokens = (totalUsage.reasoningTokens ?? 0) + response.usage.reasoningTokens;
|
|
934
|
+
}
|
|
874
935
|
if (response.usage.cacheRead) {
|
|
875
936
|
totalUsage.cacheRead = (totalUsage.cacheRead ?? 0) + response.usage.cacheRead;
|
|
876
937
|
}
|
|
@@ -922,7 +983,15 @@ async function* agentLoop(messages, options) {
|
|
|
922
983
|
model: options.model
|
|
923
984
|
});
|
|
924
985
|
yield { type: "truncated", reason: "max_tokens", continued: true };
|
|
925
|
-
messages.push({
|
|
986
|
+
messages.push({
|
|
987
|
+
role: "user",
|
|
988
|
+
content: MAX_TOKENS_CONTINUATION_PROMPT,
|
|
989
|
+
provenance: {
|
|
990
|
+
source: "runtime",
|
|
991
|
+
kind: "continuation",
|
|
992
|
+
visibility: "hidden"
|
|
993
|
+
}
|
|
994
|
+
});
|
|
926
995
|
continue;
|
|
927
996
|
}
|
|
928
997
|
yield { type: "truncated", reason: "max_tokens", continued: false };
|
|
@@ -998,6 +1067,7 @@ async function* agentLoop(messages, options) {
|
|
|
998
1067
|
);
|
|
999
1068
|
const executionResult = hasSequentialToolCall ? yield* executeToolCallsMixed(toolCalls, toolResults, executionOptions) : yield* executeToolCallsParallel(toolCalls, toolResults, executionOptions);
|
|
1000
1069
|
messages.push({ role: "tool", content: executionResult.toolResults });
|
|
1070
|
+
yield { type: "checkpoint", turn };
|
|
1001
1071
|
const toolsAborted = executionResult.aborted;
|
|
1002
1072
|
if (fatalToolArgumentError) {
|
|
1003
1073
|
if (fatalToolArgumentRecoverable && !toolArgumentAutoContinueUsed) {
|
|
@@ -1029,8 +1099,51 @@ async function* agentLoop(messages, options) {
|
|
|
1029
1099
|
}
|
|
1030
1100
|
}
|
|
1031
1101
|
}
|
|
1032
|
-
if (turn >=
|
|
1033
|
-
|
|
1102
|
+
if (turn >= effectiveMaxTurns) {
|
|
1103
|
+
let extended = false;
|
|
1104
|
+
if (options.onTurnBudgetExhausted && turnExtensions < maxTurnExtensions) {
|
|
1105
|
+
const extension = turnExtensions + 1;
|
|
1106
|
+
let granted = false;
|
|
1107
|
+
try {
|
|
1108
|
+
granted = await options.onTurnBudgetExhausted({
|
|
1109
|
+
turn,
|
|
1110
|
+
maxTurns: effectiveMaxTurns,
|
|
1111
|
+
extension
|
|
1112
|
+
});
|
|
1113
|
+
} catch {
|
|
1114
|
+
granted = false;
|
|
1115
|
+
}
|
|
1116
|
+
if (granted) {
|
|
1117
|
+
turnExtensions = extension;
|
|
1118
|
+
effectiveMaxTurns += maxTurns;
|
|
1119
|
+
extended = true;
|
|
1120
|
+
diag("turn_budget_extended", {
|
|
1121
|
+
turn,
|
|
1122
|
+
grantedTurns: effectiveMaxTurns,
|
|
1123
|
+
extension,
|
|
1124
|
+
provider: options.provider,
|
|
1125
|
+
model: options.model
|
|
1126
|
+
});
|
|
1127
|
+
yield {
|
|
1128
|
+
type: "turn_budget_extended",
|
|
1129
|
+
turn,
|
|
1130
|
+
grantedTurns: effectiveMaxTurns,
|
|
1131
|
+
extension
|
|
1132
|
+
};
|
|
1133
|
+
messages.push({
|
|
1134
|
+
role: "user",
|
|
1135
|
+
content: turnBudgetContinuationPrompt(),
|
|
1136
|
+
provenance: {
|
|
1137
|
+
source: "runtime",
|
|
1138
|
+
kind: "continuation",
|
|
1139
|
+
visibility: "hidden"
|
|
1140
|
+
}
|
|
1141
|
+
});
|
|
1142
|
+
}
|
|
1143
|
+
}
|
|
1144
|
+
if (!extended) {
|
|
1145
|
+
hitMaxTurns = true;
|
|
1146
|
+
}
|
|
1034
1147
|
}
|
|
1035
1148
|
}
|
|
1036
1149
|
} finally {
|
|
@@ -1046,14 +1159,15 @@ async function* agentLoop(messages, options) {
|
|
|
1046
1159
|
if (hitMaxTurns) {
|
|
1047
1160
|
diag("max_turns_reached", {
|
|
1048
1161
|
turn,
|
|
1049
|
-
maxTurns,
|
|
1162
|
+
maxTurns: effectiveMaxTurns,
|
|
1163
|
+
extensions: turnExtensions,
|
|
1050
1164
|
provider: options.provider,
|
|
1051
1165
|
model: options.model
|
|
1052
1166
|
});
|
|
1053
1167
|
yield {
|
|
1054
1168
|
type: "max_turns",
|
|
1055
1169
|
totalTurns: turn,
|
|
1056
|
-
maxTurns
|
|
1170
|
+
maxTurns: effectiveMaxTurns
|
|
1057
1171
|
};
|
|
1058
1172
|
}
|
|
1059
1173
|
yield {
|
|
@@ -1089,7 +1203,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
|
|
|
1089
1203
|
try {
|
|
1090
1204
|
const parsed = tool.parameters.parse(toolCall.args);
|
|
1091
1205
|
const callerSignal = options.signal;
|
|
1092
|
-
const toolTimeout = AbortSignal.timeout(
|
|
1206
|
+
const toolTimeout = AbortSignal.timeout(tool.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS);
|
|
1093
1207
|
const ctx = {
|
|
1094
1208
|
signal: callerSignal ? AbortSignal.any([callerSignal, toolTimeout]) : toolTimeout,
|
|
1095
1209
|
toolCallId: toolCall.id,
|
|
@@ -1302,9 +1416,9 @@ function capToolResults(toolResults, maxToolResultChars) {
|
|
|
1302
1416
|
const originalChars = toolResult.content.length;
|
|
1303
1417
|
const headChars = Math.floor(max * 0.7);
|
|
1304
1418
|
const tailChars = max - headChars;
|
|
1305
|
-
const head = toolResult.content
|
|
1306
|
-
const tail = toolResult.content
|
|
1307
|
-
const omitted = originalChars -
|
|
1419
|
+
const head = sliceHead(toolResult.content, headChars);
|
|
1420
|
+
const tail = sliceTail(toolResult.content, tailChars);
|
|
1421
|
+
const omitted = originalChars - head.length - tail.length;
|
|
1308
1422
|
toolResult.content = head + `
|
|
1309
1423
|
|
|
1310
1424
|
[... ${omitted} characters omitted ...]
|
|
@@ -1339,11 +1453,11 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
|
|
|
1339
1453
|
const headChars = Math.floor(fairShare * 0.7);
|
|
1340
1454
|
const tailChars = fairShare - headChars;
|
|
1341
1455
|
const omitted = originalChars - fairShare;
|
|
1342
|
-
toolResult.content = toolResult.content
|
|
1456
|
+
toolResult.content = sliceHead(toolResult.content, headChars) + `
|
|
1343
1457
|
|
|
1344
1458
|
[... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
|
|
1345
1459
|
|
|
1346
|
-
` + (
|
|
1460
|
+
` + sliceTail(toolResult.content, tailChars);
|
|
1347
1461
|
toolResult.capped = {
|
|
1348
1462
|
originalChars: toolResult.capped?.originalChars ?? originalChars,
|
|
1349
1463
|
keptChars: toolResult.content.length,
|
|
@@ -1362,12 +1476,14 @@ function truncateToolResultText(text, maxChars) {
|
|
|
1362
1476
|
if (text.length <= maxChars) return text;
|
|
1363
1477
|
const tailChars = Math.min(Math.floor(maxChars * 0.3), 2e4);
|
|
1364
1478
|
const headChars = Math.max(maxChars - tailChars, 0);
|
|
1365
|
-
const
|
|
1366
|
-
|
|
1479
|
+
const head = sliceHead(text, headChars);
|
|
1480
|
+
const tail = sliceTail(text, tailChars);
|
|
1481
|
+
const omitted = text.length - head.length - tail.length;
|
|
1482
|
+
return `${head}
|
|
1367
1483
|
|
|
1368
1484
|
[... ${omitted} characters omitted after context overflow ...]
|
|
1369
1485
|
|
|
1370
|
-
${
|
|
1486
|
+
${tail}`;
|
|
1371
1487
|
}
|
|
1372
1488
|
function truncateOversizedToolResults(messages, maxChars) {
|
|
1373
1489
|
if (maxChars <= 0) return false;
|