@prestyj/agent 5.9.0 → 5.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +155 -29
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +66 -2
- package/dist/index.d.ts +66 -2
- package/dist/index.js +157 -30
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/index.d.cts
CHANGED
|
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
|
|
|
20
20
|
* batch runs in source order so stateful mutations cannot race each other.
|
|
21
21
|
*/
|
|
22
22
|
executionMode?: ToolExecutionMode;
|
|
23
|
+
/**
|
|
24
|
+
* Overrides the loop's default per-tool timeout. A tool that owns a longer
|
|
25
|
+
* internal budget than the default must declare it here, or the loop cancels
|
|
26
|
+
* it first and the tool's own timeout — with its specific, actionable error
|
|
27
|
+
* message — becomes unreachable.
|
|
28
|
+
*/
|
|
29
|
+
timeoutMs?: number;
|
|
23
30
|
execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
|
|
24
31
|
}
|
|
25
32
|
interface AgentTextDeltaEvent {
|
|
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
|
|
|
70
77
|
usage: Usage;
|
|
71
78
|
timing: AgentTurnTiming;
|
|
72
79
|
}
|
|
80
|
+
/**
|
|
81
|
+
* A safe point between steps: the assistant message and every tool result for
|
|
82
|
+
* this turn are now in the message array, and no provider call is in flight.
|
|
83
|
+
*
|
|
84
|
+
* Hosts that persist a transcript flush here. Without it a crash mid-run loses
|
|
85
|
+
* the WHOLE turn — including tool results whose side effects already landed on
|
|
86
|
+
* disk — because the only flush happens after the loop returns.
|
|
87
|
+
*
|
|
88
|
+
* Yielded immediately after tool results are appended, so it pairs with
|
|
89
|
+
* `turn_end` (which covers the assistant half) to cover every message.
|
|
90
|
+
*/
|
|
91
|
+
interface AgentCheckpointEvent {
|
|
92
|
+
type: "checkpoint";
|
|
93
|
+
turn: number;
|
|
94
|
+
}
|
|
73
95
|
interface AgentDoneEvent {
|
|
74
96
|
type: "agent_done";
|
|
75
97
|
totalTurns: number;
|
|
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
|
|
|
87
109
|
totalTurns: number;
|
|
88
110
|
maxTurns: number;
|
|
89
111
|
}
|
|
112
|
+
/**
|
|
113
|
+
* Emitted when the loop was about to stop on an exhausted turn budget but the
|
|
114
|
+
* host granted an extension instead. The effective budget is raised and the
|
|
115
|
+
* loop continues with a continuation prompt, so this is NOT terminal — unlike
|
|
116
|
+
* `max_turns`, which still fires if the extended budget is also spent.
|
|
117
|
+
*/
|
|
118
|
+
interface AgentTurnBudgetExtendedEvent {
|
|
119
|
+
type: "turn_budget_extended";
|
|
120
|
+
/** Turn number at which the budget was exhausted. */
|
|
121
|
+
turn: number;
|
|
122
|
+
/** New effective `maxTurns` after the extension. */
|
|
123
|
+
grantedTurns: number;
|
|
124
|
+
/** 1-based extension count for this run. */
|
|
125
|
+
extension: number;
|
|
126
|
+
}
|
|
90
127
|
/**
|
|
91
128
|
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
129
|
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
|
|
|
148
185
|
type: "follow_up_message";
|
|
149
186
|
content: Message["content"];
|
|
150
187
|
}
|
|
151
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
188
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
189
|
interface TransformContextOptions {
|
|
153
190
|
/** Force a transform after the provider reports context overflow. */
|
|
154
191
|
force?: boolean;
|
|
@@ -168,6 +205,12 @@ interface AgentOptions {
|
|
|
168
205
|
/** Control whether tools may/must be called, or select a named tool when supported. */
|
|
169
206
|
toolChoice?: StreamOptions["toolChoice"];
|
|
170
207
|
maxTurns?: number;
|
|
208
|
+
/**
|
|
209
|
+
* How many times `onTurnBudgetExhausted` may grant extra turns in one run.
|
|
210
|
+
* Each grant raises the effective budget by the original `maxTurns`.
|
|
211
|
+
* Default: 2. Set 0 to disable extensions entirely.
|
|
212
|
+
*/
|
|
213
|
+
maxTurnExtensions?: number;
|
|
171
214
|
maxTokens?: number;
|
|
172
215
|
temperature?: number;
|
|
173
216
|
thinking?: StreamOptions["thinking"];
|
|
@@ -234,6 +277,18 @@ interface AgentOptions {
|
|
|
234
277
|
* on read.
|
|
235
278
|
*/
|
|
236
279
|
getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
|
|
280
|
+
/**
|
|
281
|
+
* Consulted when a tool-running turn exhausts the turn budget mid-task,
|
|
282
|
+
* before the loop emits the terminal `max_turns` event. Return true to grant
|
|
283
|
+
* another `maxTurns` worth of turns; false (the default when unset) keeps
|
|
284
|
+
* today's hard cut-off. Hosts should only grant on evidence of progress —
|
|
285
|
+
* extending a spinning agent just buys it more tokens to spin with.
|
|
286
|
+
*/
|
|
287
|
+
onTurnBudgetExhausted?: (ctx: {
|
|
288
|
+
turn: number;
|
|
289
|
+
maxTurns: number;
|
|
290
|
+
extension: number;
|
|
291
|
+
}) => Promise<boolean> | boolean;
|
|
237
292
|
}
|
|
238
293
|
interface AgentResult {
|
|
239
294
|
message: AssistantMessage;
|
|
@@ -324,4 +379,13 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
324
379
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
325
380
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
|
326
381
|
|
|
327
|
-
|
|
382
|
+
/**
|
|
383
|
+
* A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
|
|
384
|
+
* minutes prefilling a large prompt before it emits its first token. The
|
|
385
|
+
* first-event watchdog that protects hosted streams turns that into an abort
|
|
386
|
+
* → retry → cold-prefill loop that never converges, so it is disabled for
|
|
387
|
+
* loopback backends.
|
|
388
|
+
*/
|
|
389
|
+
declare function isLocalBackendUrl(baseUrl?: string): boolean;
|
|
390
|
+
|
|
391
|
+
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
|
package/dist/index.d.ts
CHANGED
|
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
|
|
|
20
20
|
* batch runs in source order so stateful mutations cannot race each other.
|
|
21
21
|
*/
|
|
22
22
|
executionMode?: ToolExecutionMode;
|
|
23
|
+
/**
|
|
24
|
+
* Overrides the loop's default per-tool timeout. A tool that owns a longer
|
|
25
|
+
* internal budget than the default must declare it here, or the loop cancels
|
|
26
|
+
* it first and the tool's own timeout — with its specific, actionable error
|
|
27
|
+
* message — becomes unreachable.
|
|
28
|
+
*/
|
|
29
|
+
timeoutMs?: number;
|
|
23
30
|
execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
|
|
24
31
|
}
|
|
25
32
|
interface AgentTextDeltaEvent {
|
|
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
|
|
|
70
77
|
usage: Usage;
|
|
71
78
|
timing: AgentTurnTiming;
|
|
72
79
|
}
|
|
80
|
+
/**
|
|
81
|
+
* A safe point between steps: the assistant message and every tool result for
|
|
82
|
+
* this turn are now in the message array, and no provider call is in flight.
|
|
83
|
+
*
|
|
84
|
+
* Hosts that persist a transcript flush here. Without it a crash mid-run loses
|
|
85
|
+
* the WHOLE turn — including tool results whose side effects already landed on
|
|
86
|
+
* disk — because the only flush happens after the loop returns.
|
|
87
|
+
*
|
|
88
|
+
* Yielded immediately after tool results are appended, so it pairs with
|
|
89
|
+
* `turn_end` (which covers the assistant half) to cover every message.
|
|
90
|
+
*/
|
|
91
|
+
interface AgentCheckpointEvent {
|
|
92
|
+
type: "checkpoint";
|
|
93
|
+
turn: number;
|
|
94
|
+
}
|
|
73
95
|
interface AgentDoneEvent {
|
|
74
96
|
type: "agent_done";
|
|
75
97
|
totalTurns: number;
|
|
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
|
|
|
87
109
|
totalTurns: number;
|
|
88
110
|
maxTurns: number;
|
|
89
111
|
}
|
|
112
|
+
/**
|
|
113
|
+
* Emitted when the loop was about to stop on an exhausted turn budget but the
|
|
114
|
+
* host granted an extension instead. The effective budget is raised and the
|
|
115
|
+
* loop continues with a continuation prompt, so this is NOT terminal — unlike
|
|
116
|
+
* `max_turns`, which still fires if the extended budget is also spent.
|
|
117
|
+
*/
|
|
118
|
+
interface AgentTurnBudgetExtendedEvent {
|
|
119
|
+
type: "turn_budget_extended";
|
|
120
|
+
/** Turn number at which the budget was exhausted. */
|
|
121
|
+
turn: number;
|
|
122
|
+
/** New effective `maxTurns` after the extension. */
|
|
123
|
+
grantedTurns: number;
|
|
124
|
+
/** 1-based extension count for this run. */
|
|
125
|
+
extension: number;
|
|
126
|
+
}
|
|
90
127
|
/**
|
|
91
128
|
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
129
|
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
|
|
|
148
185
|
type: "follow_up_message";
|
|
149
186
|
content: Message["content"];
|
|
150
187
|
}
|
|
151
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
188
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
189
|
interface TransformContextOptions {
|
|
153
190
|
/** Force a transform after the provider reports context overflow. */
|
|
154
191
|
force?: boolean;
|
|
@@ -168,6 +205,12 @@ interface AgentOptions {
|
|
|
168
205
|
/** Control whether tools may/must be called, or select a named tool when supported. */
|
|
169
206
|
toolChoice?: StreamOptions["toolChoice"];
|
|
170
207
|
maxTurns?: number;
|
|
208
|
+
/**
|
|
209
|
+
* How many times `onTurnBudgetExhausted` may grant extra turns in one run.
|
|
210
|
+
* Each grant raises the effective budget by the original `maxTurns`.
|
|
211
|
+
* Default: 2. Set 0 to disable extensions entirely.
|
|
212
|
+
*/
|
|
213
|
+
maxTurnExtensions?: number;
|
|
171
214
|
maxTokens?: number;
|
|
172
215
|
temperature?: number;
|
|
173
216
|
thinking?: StreamOptions["thinking"];
|
|
@@ -234,6 +277,18 @@ interface AgentOptions {
|
|
|
234
277
|
* on read.
|
|
235
278
|
*/
|
|
236
279
|
getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
|
|
280
|
+
/**
|
|
281
|
+
* Consulted when a tool-running turn exhausts the turn budget mid-task,
|
|
282
|
+
* before the loop emits the terminal `max_turns` event. Return true to grant
|
|
283
|
+
* another `maxTurns` worth of turns; false (the default when unset) keeps
|
|
284
|
+
* today's hard cut-off. Hosts should only grant on evidence of progress —
|
|
285
|
+
* extending a spinning agent just buys it more tokens to spin with.
|
|
286
|
+
*/
|
|
287
|
+
onTurnBudgetExhausted?: (ctx: {
|
|
288
|
+
turn: number;
|
|
289
|
+
maxTurns: number;
|
|
290
|
+
extension: number;
|
|
291
|
+
}) => Promise<boolean> | boolean;
|
|
237
292
|
}
|
|
238
293
|
interface AgentResult {
|
|
239
294
|
message: AssistantMessage;
|
|
@@ -324,4 +379,13 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
324
379
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
325
380
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
|
326
381
|
|
|
327
|
-
|
|
382
|
+
/**
|
|
383
|
+
* A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
|
|
384
|
+
* minutes prefilling a large prompt before it emits its first token. The
|
|
385
|
+
* first-event watchdog that protects hosted streams turns that into an abort
|
|
386
|
+
* → retry → cold-prefill loop that never converges, so it is disabled for
|
|
387
|
+
* loopback backends.
|
|
388
|
+
*/
|
|
389
|
+
declare function isLocalBackendUrl(baseUrl?: string): boolean;
|
|
390
|
+
|
|
391
|
+
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
|
package/dist/index.js
CHANGED
|
@@ -8,9 +8,36 @@ import {
|
|
|
8
8
|
EventStream,
|
|
9
9
|
EZCoderAIError,
|
|
10
10
|
isHardBillingMessage,
|
|
11
|
-
redactValue
|
|
11
|
+
redactValue,
|
|
12
|
+
sliceHead,
|
|
13
|
+
sliceTail
|
|
12
14
|
} from "@prestyj/ai";
|
|
15
|
+
|
|
16
|
+
// src/local-backend.ts
|
|
17
|
+
function isLocalBackendUrl(baseUrl) {
|
|
18
|
+
if (!baseUrl) return false;
|
|
19
|
+
let host;
|
|
20
|
+
try {
|
|
21
|
+
host = new URL(baseUrl).hostname.toLowerCase();
|
|
22
|
+
} catch {
|
|
23
|
+
return false;
|
|
24
|
+
}
|
|
25
|
+
if (host === "[::1]" || host === "::1") return true;
|
|
26
|
+
if (host === "localhost" || host.endsWith(".localhost")) return true;
|
|
27
|
+
if (host === "0.0.0.0") return true;
|
|
28
|
+
if (host.endsWith(".local")) return true;
|
|
29
|
+
const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
|
|
30
|
+
if (ipv4) {
|
|
31
|
+
const octets = ipv4.slice(1).map(Number);
|
|
32
|
+
if (octets.some((n) => n > 255)) return false;
|
|
33
|
+
return octets[0] === 127;
|
|
34
|
+
}
|
|
35
|
+
return false;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// src/agent-loop.ts
|
|
13
39
|
var DEFAULT_MAX_TURNS = 300;
|
|
40
|
+
var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
|
|
14
41
|
var _diagFn = null;
|
|
15
42
|
function setStreamDiagnostic(fn) {
|
|
16
43
|
_diagFn = fn;
|
|
@@ -213,6 +240,9 @@ function abortablePromise(promise, signal) {
|
|
|
213
240
|
promise.then(resolveOnce, rejectOnce);
|
|
214
241
|
});
|
|
215
242
|
}
|
|
243
|
+
function turnBudgetContinuationPrompt() {
|
|
244
|
+
return "[You reached the turn limit for this segment but the work is not finished, so you have been granted more turns. Before continuing, state in one or two sentences what is already done and what remains, then keep going from there \u2014 do not restart work that is already complete.]";
|
|
245
|
+
}
|
|
216
246
|
function abortableSleep(ms, signal) {
|
|
217
247
|
if (signal?.aborted) return Promise.reject(createAbortError());
|
|
218
248
|
return new Promise((resolve, reject) => {
|
|
@@ -234,6 +264,9 @@ function closeIterator(iterator) {
|
|
|
234
264
|
}
|
|
235
265
|
async function* agentLoop(messages, options) {
|
|
236
266
|
const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
|
|
267
|
+
let effectiveMaxTurns = maxTurns;
|
|
268
|
+
const maxTurnExtensions = Math.max(0, options.maxTurnExtensions ?? 2);
|
|
269
|
+
let turnExtensions = 0;
|
|
237
270
|
const maxContinuations = options.maxContinuations ?? 5;
|
|
238
271
|
let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
|
|
239
272
|
const totalUsage = { inputTokens: 0, outputTokens: 0 };
|
|
@@ -276,16 +309,17 @@ async function* agentLoop(messages, options) {
|
|
|
276
309
|
const STREAM_THINKING_IDLE_TIMEOUT_MS = 3e5;
|
|
277
310
|
const STREAM_THINKING_HARD_TIMEOUT_MS = 6e5;
|
|
278
311
|
const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
|
|
279
|
-
const
|
|
280
|
-
const
|
|
281
|
-
const
|
|
312
|
+
const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
|
|
313
|
+
const localBackend = isLocalBackendUrl(options.baseUrl);
|
|
314
|
+
const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
|
|
315
|
+
const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
|
|
282
316
|
const MAX_TOOLCALL_DELTA_CHARS = 1e6;
|
|
283
|
-
const
|
|
317
|
+
const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
|
|
284
318
|
let logicalTurnStartedAt = 0;
|
|
285
319
|
let firstProviderEventAt;
|
|
286
320
|
let providerDurationMs = 0;
|
|
287
321
|
try {
|
|
288
|
-
while (turn <
|
|
322
|
+
while (turn < effectiveMaxTurns) {
|
|
289
323
|
options.signal?.throwIfAborted();
|
|
290
324
|
turn++;
|
|
291
325
|
if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
|
|
@@ -306,7 +340,11 @@ async function* agentLoop(messages, options) {
|
|
|
306
340
|
messages: messages.length,
|
|
307
341
|
chars: msgChars,
|
|
308
342
|
provider: options.provider,
|
|
309
|
-
model: options.model
|
|
343
|
+
model: options.model,
|
|
344
|
+
thinking: options.thinking ?? "off",
|
|
345
|
+
firstEventTimeoutMs,
|
|
346
|
+
initialHardTimeoutMs,
|
|
347
|
+
localBackend
|
|
310
348
|
});
|
|
311
349
|
}
|
|
312
350
|
if (firstTurn && options.getSteeringMessages) {
|
|
@@ -352,6 +390,7 @@ async function* agentLoop(messages, options) {
|
|
|
352
390
|
let lastEventType = "";
|
|
353
391
|
let toolcallDeltaChars = 0;
|
|
354
392
|
let toolcallDeltaCount = 0;
|
|
393
|
+
let toolcallNoProgressCount = 0;
|
|
355
394
|
let runawayDetected = null;
|
|
356
395
|
let attemptText = "";
|
|
357
396
|
let lastYieldEndTime = Date.now();
|
|
@@ -364,6 +403,7 @@ async function* agentLoop(messages, options) {
|
|
|
364
403
|
if (useNonStreamingFallback) return;
|
|
365
404
|
if (idleTimer) clearTimeout(idleTimer);
|
|
366
405
|
const timeoutMs = hasReceivedEvent ? STREAM_IDLE_TIMEOUT_MS : hasReceivedThinking ? STREAM_THINKING_IDLE_TIMEOUT_MS : firstEventTimeoutMs;
|
|
406
|
+
if (!Number.isFinite(timeoutMs)) return;
|
|
367
407
|
idleTimer = setTimeout(() => {
|
|
368
408
|
diag("idle_timeout_fired", {
|
|
369
409
|
events: streamEventCount,
|
|
@@ -505,11 +545,13 @@ async function* agentLoop(messages, options) {
|
|
|
505
545
|
const chunkChars = event.argsJson?.length ?? 0;
|
|
506
546
|
toolcallDeltaChars += chunkChars;
|
|
507
547
|
toolcallDeltaCount++;
|
|
508
|
-
|
|
548
|
+
toolcallNoProgressCount = chunkChars > 0 ? 0 : toolcallNoProgressCount + 1;
|
|
549
|
+
if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallNoProgressCount > MAX_TOOLCALL_NO_PROGRESS_EVENTS)) {
|
|
509
550
|
runawayDetected = {
|
|
510
551
|
kind: toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS ? "chars" : "events",
|
|
511
552
|
chars: toolcallDeltaChars,
|
|
512
|
-
events: toolcallDeltaCount
|
|
553
|
+
events: toolcallDeltaCount,
|
|
554
|
+
noProgressEvents: toolcallNoProgressCount
|
|
513
555
|
};
|
|
514
556
|
diag("runaway_toolcall_detected", {
|
|
515
557
|
...runawayDetected,
|
|
@@ -691,11 +733,15 @@ async function* agentLoop(messages, options) {
|
|
|
691
733
|
turn--;
|
|
692
734
|
continue;
|
|
693
735
|
}
|
|
694
|
-
const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.
|
|
736
|
+
const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.noProgressEvents} consecutive tool-call delta events without argument progress`;
|
|
695
737
|
yield {
|
|
696
738
|
type: "error",
|
|
697
|
-
error: new
|
|
698
|
-
`The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}).
|
|
739
|
+
error: new EZCoderAIError(
|
|
740
|
+
`The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Your conversation is preserved.`,
|
|
741
|
+
{
|
|
742
|
+
source: "provider",
|
|
743
|
+
hint: "Retry once. If it keeps happening, switch models and continue the preserved conversation."
|
|
744
|
+
}
|
|
699
745
|
)
|
|
700
746
|
};
|
|
701
747
|
break;
|
|
@@ -722,7 +768,15 @@ async function* agentLoop(messages, options) {
|
|
|
722
768
|
role: "assistant",
|
|
723
769
|
content: [{ type: "text", text: attemptText }]
|
|
724
770
|
});
|
|
725
|
-
messages.push({
|
|
771
|
+
messages.push({
|
|
772
|
+
role: "user",
|
|
773
|
+
content: PARTIAL_CONTINUATION_PROMPT,
|
|
774
|
+
provenance: {
|
|
775
|
+
source: "runtime",
|
|
776
|
+
kind: "continuation",
|
|
777
|
+
visibility: "hidden"
|
|
778
|
+
}
|
|
779
|
+
});
|
|
726
780
|
preservedChars = attemptText.length;
|
|
727
781
|
}
|
|
728
782
|
diag("retry", {
|
|
@@ -748,15 +802,29 @@ async function* agentLoop(messages, options) {
|
|
|
748
802
|
continue;
|
|
749
803
|
}
|
|
750
804
|
if (transportFailure) {
|
|
805
|
+
const cause = malformed ? "malformed_stream" : socketDrop ? "socket_drop" : "stream_stall";
|
|
751
806
|
diag("stall_exhausted", {
|
|
752
807
|
stallRetries: MAX_STALL_RETRIES,
|
|
753
808
|
provider: options.provider,
|
|
754
|
-
model: options.model
|
|
809
|
+
model: options.model,
|
|
810
|
+
cause,
|
|
811
|
+
nonStreaming: useNonStreamingFallback,
|
|
812
|
+
events: streamEventCount,
|
|
813
|
+
eventTypes: eventTypeCounts,
|
|
814
|
+
lastEventType,
|
|
815
|
+
sinceLastEventMs: Date.now() - lastEventTime,
|
|
816
|
+
attemptDurationMs: Date.now() - streamCallStart,
|
|
817
|
+
maxConsumerLagMs
|
|
755
818
|
});
|
|
756
819
|
yield {
|
|
757
820
|
type: "error",
|
|
758
|
-
error: new
|
|
759
|
-
`The API provider
|
|
821
|
+
error: new EZCoderAIError(
|
|
822
|
+
`The connection to the API provider stopped responding after ${MAX_STALL_RETRIES} automatic retries. Your conversation is preserved.`,
|
|
823
|
+
{
|
|
824
|
+
source: "network",
|
|
825
|
+
hint: "Retry once. If it keeps happening on this device, disable any VPN or proxy and allow EZ Coder through firewall or antivirus web protection.",
|
|
826
|
+
cause: err
|
|
827
|
+
}
|
|
760
828
|
)
|
|
761
829
|
};
|
|
762
830
|
break;
|
|
@@ -827,6 +895,9 @@ async function* agentLoop(messages, options) {
|
|
|
827
895
|
useNonStreamingFallback = false;
|
|
828
896
|
totalUsage.inputTokens += response.usage.inputTokens;
|
|
829
897
|
totalUsage.outputTokens += response.usage.outputTokens;
|
|
898
|
+
if (response.usage.reasoningTokens) {
|
|
899
|
+
totalUsage.reasoningTokens = (totalUsage.reasoningTokens ?? 0) + response.usage.reasoningTokens;
|
|
900
|
+
}
|
|
830
901
|
if (response.usage.cacheRead) {
|
|
831
902
|
totalUsage.cacheRead = (totalUsage.cacheRead ?? 0) + response.usage.cacheRead;
|
|
832
903
|
}
|
|
@@ -878,7 +949,15 @@ async function* agentLoop(messages, options) {
|
|
|
878
949
|
model: options.model
|
|
879
950
|
});
|
|
880
951
|
yield { type: "truncated", reason: "max_tokens", continued: true };
|
|
881
|
-
messages.push({
|
|
952
|
+
messages.push({
|
|
953
|
+
role: "user",
|
|
954
|
+
content: MAX_TOKENS_CONTINUATION_PROMPT,
|
|
955
|
+
provenance: {
|
|
956
|
+
source: "runtime",
|
|
957
|
+
kind: "continuation",
|
|
958
|
+
visibility: "hidden"
|
|
959
|
+
}
|
|
960
|
+
});
|
|
882
961
|
continue;
|
|
883
962
|
}
|
|
884
963
|
yield { type: "truncated", reason: "max_tokens", continued: false };
|
|
@@ -954,6 +1033,7 @@ async function* agentLoop(messages, options) {
|
|
|
954
1033
|
);
|
|
955
1034
|
const executionResult = hasSequentialToolCall ? yield* executeToolCallsMixed(toolCalls, toolResults, executionOptions) : yield* executeToolCallsParallel(toolCalls, toolResults, executionOptions);
|
|
956
1035
|
messages.push({ role: "tool", content: executionResult.toolResults });
|
|
1036
|
+
yield { type: "checkpoint", turn };
|
|
957
1037
|
const toolsAborted = executionResult.aborted;
|
|
958
1038
|
if (fatalToolArgumentError) {
|
|
959
1039
|
if (fatalToolArgumentRecoverable && !toolArgumentAutoContinueUsed) {
|
|
@@ -985,8 +1065,51 @@ async function* agentLoop(messages, options) {
|
|
|
985
1065
|
}
|
|
986
1066
|
}
|
|
987
1067
|
}
|
|
988
|
-
if (turn >=
|
|
989
|
-
|
|
1068
|
+
if (turn >= effectiveMaxTurns) {
|
|
1069
|
+
let extended = false;
|
|
1070
|
+
if (options.onTurnBudgetExhausted && turnExtensions < maxTurnExtensions) {
|
|
1071
|
+
const extension = turnExtensions + 1;
|
|
1072
|
+
let granted = false;
|
|
1073
|
+
try {
|
|
1074
|
+
granted = await options.onTurnBudgetExhausted({
|
|
1075
|
+
turn,
|
|
1076
|
+
maxTurns: effectiveMaxTurns,
|
|
1077
|
+
extension
|
|
1078
|
+
});
|
|
1079
|
+
} catch {
|
|
1080
|
+
granted = false;
|
|
1081
|
+
}
|
|
1082
|
+
if (granted) {
|
|
1083
|
+
turnExtensions = extension;
|
|
1084
|
+
effectiveMaxTurns += maxTurns;
|
|
1085
|
+
extended = true;
|
|
1086
|
+
diag("turn_budget_extended", {
|
|
1087
|
+
turn,
|
|
1088
|
+
grantedTurns: effectiveMaxTurns,
|
|
1089
|
+
extension,
|
|
1090
|
+
provider: options.provider,
|
|
1091
|
+
model: options.model
|
|
1092
|
+
});
|
|
1093
|
+
yield {
|
|
1094
|
+
type: "turn_budget_extended",
|
|
1095
|
+
turn,
|
|
1096
|
+
grantedTurns: effectiveMaxTurns,
|
|
1097
|
+
extension
|
|
1098
|
+
};
|
|
1099
|
+
messages.push({
|
|
1100
|
+
role: "user",
|
|
1101
|
+
content: turnBudgetContinuationPrompt(),
|
|
1102
|
+
provenance: {
|
|
1103
|
+
source: "runtime",
|
|
1104
|
+
kind: "continuation",
|
|
1105
|
+
visibility: "hidden"
|
|
1106
|
+
}
|
|
1107
|
+
});
|
|
1108
|
+
}
|
|
1109
|
+
}
|
|
1110
|
+
if (!extended) {
|
|
1111
|
+
hitMaxTurns = true;
|
|
1112
|
+
}
|
|
990
1113
|
}
|
|
991
1114
|
}
|
|
992
1115
|
} finally {
|
|
@@ -1002,14 +1125,15 @@ async function* agentLoop(messages, options) {
|
|
|
1002
1125
|
if (hitMaxTurns) {
|
|
1003
1126
|
diag("max_turns_reached", {
|
|
1004
1127
|
turn,
|
|
1005
|
-
maxTurns,
|
|
1128
|
+
maxTurns: effectiveMaxTurns,
|
|
1129
|
+
extensions: turnExtensions,
|
|
1006
1130
|
provider: options.provider,
|
|
1007
1131
|
model: options.model
|
|
1008
1132
|
});
|
|
1009
1133
|
yield {
|
|
1010
1134
|
type: "max_turns",
|
|
1011
1135
|
totalTurns: turn,
|
|
1012
|
-
maxTurns
|
|
1136
|
+
maxTurns: effectiveMaxTurns
|
|
1013
1137
|
};
|
|
1014
1138
|
}
|
|
1015
1139
|
yield {
|
|
@@ -1045,7 +1169,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
|
|
|
1045
1169
|
try {
|
|
1046
1170
|
const parsed = tool.parameters.parse(toolCall.args);
|
|
1047
1171
|
const callerSignal = options.signal;
|
|
1048
|
-
const toolTimeout = AbortSignal.timeout(
|
|
1172
|
+
const toolTimeout = AbortSignal.timeout(tool.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS);
|
|
1049
1173
|
const ctx = {
|
|
1050
1174
|
signal: callerSignal ? AbortSignal.any([callerSignal, toolTimeout]) : toolTimeout,
|
|
1051
1175
|
toolCallId: toolCall.id,
|
|
@@ -1258,9 +1382,9 @@ function capToolResults(toolResults, maxToolResultChars) {
|
|
|
1258
1382
|
const originalChars = toolResult.content.length;
|
|
1259
1383
|
const headChars = Math.floor(max * 0.7);
|
|
1260
1384
|
const tailChars = max - headChars;
|
|
1261
|
-
const head = toolResult.content
|
|
1262
|
-
const tail = toolResult.content
|
|
1263
|
-
const omitted = originalChars -
|
|
1385
|
+
const head = sliceHead(toolResult.content, headChars);
|
|
1386
|
+
const tail = sliceTail(toolResult.content, tailChars);
|
|
1387
|
+
const omitted = originalChars - head.length - tail.length;
|
|
1264
1388
|
toolResult.content = head + `
|
|
1265
1389
|
|
|
1266
1390
|
[... ${omitted} characters omitted ...]
|
|
@@ -1295,11 +1419,11 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
|
|
|
1295
1419
|
const headChars = Math.floor(fairShare * 0.7);
|
|
1296
1420
|
const tailChars = fairShare - headChars;
|
|
1297
1421
|
const omitted = originalChars - fairShare;
|
|
1298
|
-
toolResult.content = toolResult.content
|
|
1422
|
+
toolResult.content = sliceHead(toolResult.content, headChars) + `
|
|
1299
1423
|
|
|
1300
1424
|
[... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
|
|
1301
1425
|
|
|
1302
|
-
` + (
|
|
1426
|
+
` + sliceTail(toolResult.content, tailChars);
|
|
1303
1427
|
toolResult.capped = {
|
|
1304
1428
|
originalChars: toolResult.capped?.originalChars ?? originalChars,
|
|
1305
1429
|
keptChars: toolResult.content.length,
|
|
@@ -1318,12 +1442,14 @@ function truncateToolResultText(text, maxChars) {
|
|
|
1318
1442
|
if (text.length <= maxChars) return text;
|
|
1319
1443
|
const tailChars = Math.min(Math.floor(maxChars * 0.3), 2e4);
|
|
1320
1444
|
const headChars = Math.max(maxChars - tailChars, 0);
|
|
1321
|
-
const
|
|
1322
|
-
|
|
1445
|
+
const head = sliceHead(text, headChars);
|
|
1446
|
+
const tail = sliceTail(text, tailChars);
|
|
1447
|
+
const omitted = text.length - head.length - tail.length;
|
|
1448
|
+
return `${head}
|
|
1323
1449
|
|
|
1324
1450
|
[... ${omitted} characters omitted after context overflow ...]
|
|
1325
1451
|
|
|
1326
|
-
${
|
|
1452
|
+
${tail}`;
|
|
1327
1453
|
}
|
|
1328
1454
|
function truncateOversizedToolResults(messages, maxChars) {
|
|
1329
1455
|
if (maxChars <= 0) return false;
|
|
@@ -1575,6 +1701,7 @@ export {
|
|
|
1575
1701
|
isAbortError,
|
|
1576
1702
|
isBillingError,
|
|
1577
1703
|
isContextOverflow,
|
|
1704
|
+
isLocalBackendUrl,
|
|
1578
1705
|
isUsageLimitError,
|
|
1579
1706
|
setStreamDiagnostic
|
|
1580
1707
|
};
|