@prestyj/agent 5.7.0 → 5.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +120 -6
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +32 -7
- package/dist/index.d.ts +32 -7
- package/dist/index.js +120 -6
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/index.d.cts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition,
|
|
1
|
+
import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, Usage, StopReason, AssistantMessage } from '@prestyj/ai';
|
|
2
2
|
import { z } from 'zod';
|
|
3
3
|
|
|
4
4
|
interface StructuredToolResult {
|
|
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
|
|
|
87
87
|
totalTurns: number;
|
|
88
88
|
maxTurns: number;
|
|
89
89
|
}
|
|
90
|
+
/**
|
|
91
|
+
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
|
+
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
93
|
+
* or a provider-reported `error` stop. Distinguishes a truncated/degraded
|
|
94
|
+
* completion from a clean one so hosts can warn the user instead of silently
|
|
95
|
+
* presenting incomplete output as done.
|
|
96
|
+
*/
|
|
97
|
+
interface AgentTruncatedEvent {
|
|
98
|
+
type: "truncated";
|
|
99
|
+
reason: "max_tokens" | "refusal" | "provider_error";
|
|
100
|
+
/** True when the loop injected a continuation and will keep going. */
|
|
101
|
+
continued: boolean;
|
|
102
|
+
}
|
|
90
103
|
interface AgentRetryEvent {
|
|
91
104
|
type: "retry";
|
|
92
|
-
reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
|
|
105
|
+
reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
|
|
93
106
|
attempt: number;
|
|
94
107
|
maxAttempts: number;
|
|
95
108
|
delayMs: number;
|
|
@@ -135,7 +148,15 @@ interface AgentFollowUpMessageEvent {
|
|
|
135
148
|
type: "follow_up_message";
|
|
136
149
|
content: Message["content"];
|
|
137
150
|
}
|
|
138
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
|
|
151
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
|
+
interface TransformContextOptions {
|
|
153
|
+
/** Force a transform after the provider reports context overflow. */
|
|
154
|
+
force?: boolean;
|
|
155
|
+
/** Latest successful provider usage, anchored at its assistant message. */
|
|
156
|
+
usage?: Usage;
|
|
157
|
+
/** Messages appended after that usage sample and not yet seen by the provider. */
|
|
158
|
+
pendingMessages: Message[];
|
|
159
|
+
}
|
|
139
160
|
interface AgentOptions {
|
|
140
161
|
provider: StreamOptions["provider"];
|
|
141
162
|
model: string;
|
|
@@ -180,6 +201,10 @@ interface AgentOptions {
|
|
|
180
201
|
clearToolUses?: boolean;
|
|
181
202
|
/** Max characters for a single tool result. Results exceeding this are truncated with a notice. */
|
|
182
203
|
maxToolResultChars?: number;
|
|
204
|
+
/** Aggregate budget for ALL tool results in one assistant turn. Protects
|
|
205
|
+
* against parallel fan-outs injecting huge uncached context in one turn;
|
|
206
|
+
* the largest results are trimmed (water-filling) with a re-run notice. */
|
|
207
|
+
maxTurnToolResultChars?: number;
|
|
183
208
|
/** Max consecutive pause_turn continuations before stopping (default: 5).
|
|
184
209
|
* Prevents infinite loops when server-side tools keep pausing. */
|
|
185
210
|
maxContinuations?: number;
|
|
@@ -188,12 +213,12 @@ interface AgentOptions {
|
|
|
188
213
|
* the messages array (e.g. compaction, truncation). Return the same array
|
|
189
214
|
* for no-op, or a new array to replace the conversation context.
|
|
190
215
|
*
|
|
216
|
+
* The latest provider usage is authoritative for the history through its
|
|
217
|
+
* assistant response. `pendingMessages` contains context appended afterward.
|
|
191
218
|
* When `options.force` is true, the caller should compact unconditionally
|
|
192
219
|
* (e.g. after a context overflow error from the API).
|
|
193
220
|
*/
|
|
194
|
-
transformContext?: (messages: Message[], options
|
|
195
|
-
force?: boolean;
|
|
196
|
-
}) => Message[] | Promise<Message[]>;
|
|
221
|
+
transformContext?: (messages: Message[], options: TransformContextOptions) => Message[] | Promise<Message[]>;
|
|
197
222
|
/**
|
|
198
223
|
* Polled after tool execution completes each turn. Returns user messages
|
|
199
224
|
* to inject into the conversation before the next LLM call (steering).
|
|
@@ -299,4 +324,4 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
299
324
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
300
325
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
|
301
326
|
|
|
302
|
-
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
|
|
327
|
+
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition,
|
|
1
|
+
import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, Usage, StopReason, AssistantMessage } from '@prestyj/ai';
|
|
2
2
|
import { z } from 'zod';
|
|
3
3
|
|
|
4
4
|
interface StructuredToolResult {
|
|
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
|
|
|
87
87
|
totalTurns: number;
|
|
88
88
|
maxTurns: number;
|
|
89
89
|
}
|
|
90
|
+
/**
|
|
91
|
+
* Warning signal emitted when a turn ended on a non-clean stop reason —
|
|
92
|
+
* `max_tokens` (output clipped at the model's output-token limit), `refusal`,
|
|
93
|
+
* or a provider-reported `error` stop. Distinguishes a truncated/degraded
|
|
94
|
+
* completion from a clean one so hosts can warn the user instead of silently
|
|
95
|
+
* presenting incomplete output as done.
|
|
96
|
+
*/
|
|
97
|
+
interface AgentTruncatedEvent {
|
|
98
|
+
type: "truncated";
|
|
99
|
+
reason: "max_tokens" | "refusal" | "provider_error";
|
|
100
|
+
/** True when the loop injected a continuation and will keep going. */
|
|
101
|
+
continued: boolean;
|
|
102
|
+
}
|
|
90
103
|
interface AgentRetryEvent {
|
|
91
104
|
type: "retry";
|
|
92
|
-
reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
|
|
105
|
+
reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
|
|
93
106
|
attempt: number;
|
|
94
107
|
maxAttempts: number;
|
|
95
108
|
delayMs: number;
|
|
@@ -135,7 +148,15 @@ interface AgentFollowUpMessageEvent {
|
|
|
135
148
|
type: "follow_up_message";
|
|
136
149
|
content: Message["content"];
|
|
137
150
|
}
|
|
138
|
-
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
|
|
151
|
+
type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
|
|
152
|
+
interface TransformContextOptions {
|
|
153
|
+
/** Force a transform after the provider reports context overflow. */
|
|
154
|
+
force?: boolean;
|
|
155
|
+
/** Latest successful provider usage, anchored at its assistant message. */
|
|
156
|
+
usage?: Usage;
|
|
157
|
+
/** Messages appended after that usage sample and not yet seen by the provider. */
|
|
158
|
+
pendingMessages: Message[];
|
|
159
|
+
}
|
|
139
160
|
interface AgentOptions {
|
|
140
161
|
provider: StreamOptions["provider"];
|
|
141
162
|
model: string;
|
|
@@ -180,6 +201,10 @@ interface AgentOptions {
|
|
|
180
201
|
clearToolUses?: boolean;
|
|
181
202
|
/** Max characters for a single tool result. Results exceeding this are truncated with a notice. */
|
|
182
203
|
maxToolResultChars?: number;
|
|
204
|
+
/** Aggregate budget for ALL tool results in one assistant turn. Protects
|
|
205
|
+
* against parallel fan-outs injecting huge uncached context in one turn;
|
|
206
|
+
* the largest results are trimmed (water-filling) with a re-run notice. */
|
|
207
|
+
maxTurnToolResultChars?: number;
|
|
183
208
|
/** Max consecutive pause_turn continuations before stopping (default: 5).
|
|
184
209
|
* Prevents infinite loops when server-side tools keep pausing. */
|
|
185
210
|
maxContinuations?: number;
|
|
@@ -188,12 +213,12 @@ interface AgentOptions {
|
|
|
188
213
|
* the messages array (e.g. compaction, truncation). Return the same array
|
|
189
214
|
* for no-op, or a new array to replace the conversation context.
|
|
190
215
|
*
|
|
216
|
+
* The latest provider usage is authoritative for the history through its
|
|
217
|
+
* assistant response. `pendingMessages` contains context appended afterward.
|
|
191
218
|
* When `options.force` is true, the caller should compact unconditionally
|
|
192
219
|
* (e.g. after a context overflow error from the API).
|
|
193
220
|
*/
|
|
194
|
-
transformContext?: (messages: Message[], options
|
|
195
|
-
force?: boolean;
|
|
196
|
-
}) => Message[] | Promise<Message[]>;
|
|
221
|
+
transformContext?: (messages: Message[], options: TransformContextOptions) => Message[] | Promise<Message[]>;
|
|
197
222
|
/**
|
|
198
223
|
* Polled after tool execution completes each turn. Returns user messages
|
|
199
224
|
* to inject into the conversation before the next LLM call (steering).
|
|
@@ -299,4 +324,4 @@ declare function isBillingError(err: unknown): boolean;
|
|
|
299
324
|
declare function isUsageLimitError(err: unknown): boolean;
|
|
300
325
|
declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
|
|
301
326
|
|
|
302
|
-
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
|
|
327
|
+
export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
|
package/dist/index.js
CHANGED
|
@@ -127,7 +127,7 @@ function classifyOverload(err) {
|
|
|
127
127
|
if (statusCode === 529 || msg.includes("overloaded") || msg.includes("529")) {
|
|
128
128
|
return "overloaded";
|
|
129
129
|
}
|
|
130
|
-
if (statusCode === 500 || statusCode === 502 || statusCode === 503 || statusCode === 504 || msg.includes("api_error") || msg.includes("server_error") || msg.includes("internal server error") || msg.includes("bad gateway") || msg.includes("service unavailable") || msg.includes("gateway timeout")) {
|
|
130
|
+
if (statusCode === 500 || statusCode === 502 || statusCode === 503 || statusCode === 504 || statusCode === 507 || msg.includes("api_error") || msg.includes("server_error") || msg.includes("internal server error") || msg.includes("bad gateway") || msg.includes("service unavailable") || msg.includes("gateway timeout") || msg.includes("exceeded request buffer limit while retrying upstream")) {
|
|
131
131
|
return "provider_error";
|
|
132
132
|
}
|
|
133
133
|
if (isOpaqueProviderMessage(err.message)) {
|
|
@@ -188,7 +188,11 @@ function createAbortError() {
|
|
|
188
188
|
}
|
|
189
189
|
function abortablePromise(promise, signal) {
|
|
190
190
|
if (!signal) return promise;
|
|
191
|
-
if (signal.aborted)
|
|
191
|
+
if (signal.aborted) {
|
|
192
|
+
promise.catch(() => {
|
|
193
|
+
});
|
|
194
|
+
return Promise.reject(createAbortError());
|
|
195
|
+
}
|
|
192
196
|
return new Promise((resolve, reject) => {
|
|
193
197
|
let settled = false;
|
|
194
198
|
const cleanup = () => signal.removeEventListener("abort", onAbort);
|
|
@@ -233,6 +237,8 @@ async function* agentLoop(messages, options) {
|
|
|
233
237
|
const maxContinuations = options.maxContinuations ?? 5;
|
|
234
238
|
let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
|
|
235
239
|
const totalUsage = { inputTokens: 0, outputTokens: 0 };
|
|
240
|
+
let latestProviderUsage;
|
|
241
|
+
let usageAnchorIndex;
|
|
236
242
|
let turn = 0;
|
|
237
243
|
let hitMaxTurns = false;
|
|
238
244
|
let firstTurn = true;
|
|
@@ -242,6 +248,7 @@ async function* agentLoop(messages, options) {
|
|
|
242
248
|
let overloadRetries = 0;
|
|
243
249
|
let emptyResponseRetries = 0;
|
|
244
250
|
let stallRetries = 0;
|
|
251
|
+
let runawayToolcallRetries = 0;
|
|
245
252
|
let overflowCompactionAttempts = 0;
|
|
246
253
|
let toolResultTruncationAttempted = false;
|
|
247
254
|
const invalidToolArgumentCounts = /* @__PURE__ */ new Map();
|
|
@@ -250,11 +257,16 @@ async function* agentLoop(messages, options) {
|
|
|
250
257
|
const MAX_OVERLOAD_RETRIES = 10;
|
|
251
258
|
const MAX_EMPTY_RESPONSE_RETRIES = 2;
|
|
252
259
|
const MAX_STALL_RETRIES = 10;
|
|
260
|
+
const MAX_RUNAWAY_TOOLCALL_RETRIES = 2;
|
|
261
|
+
const RUNAWAY_TOOLCALL_RETRY_DELAY_MS = 1e3;
|
|
253
262
|
const MAX_OVERFLOW_COMPACTIONS = 2;
|
|
254
263
|
const STALL_RETRIES_BEFORE_NON_STREAMING = 2;
|
|
255
264
|
const STALL_DELAY_MS = 1e3;
|
|
256
265
|
const MIN_PARTIAL_PRESERVE_CHARS = 200;
|
|
257
266
|
const PARTIAL_CONTINUATION_PROMPT = "[Your previous response was cut off by a connection failure. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
|
|
267
|
+
const MAX_OUTPUT_CONTINUATIONS = 2;
|
|
268
|
+
const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
|
|
269
|
+
let maxTokensContinuations = 0;
|
|
258
270
|
const OVERLOAD_BASE_DELAY_MS = 2e3;
|
|
259
271
|
const OVERLOAD_MAX_DELAY_MS = 3e4;
|
|
260
272
|
const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
|
|
@@ -309,7 +321,11 @@ async function* agentLoop(messages, options) {
|
|
|
309
321
|
firstTurn = false;
|
|
310
322
|
if (options.transformContext) {
|
|
311
323
|
diag("transform_start");
|
|
312
|
-
const
|
|
324
|
+
const pendingMessages = usageAnchorIndex === void 0 ? [] : messages.slice(usageAnchorIndex + 1);
|
|
325
|
+
const transformed = await options.transformContext(messages, {
|
|
326
|
+
usage: latestProviderUsage,
|
|
327
|
+
pendingMessages
|
|
328
|
+
});
|
|
313
329
|
if (transformed !== messages) {
|
|
314
330
|
diag("transform_compacted", {
|
|
315
331
|
before: messages.length,
|
|
@@ -317,6 +333,8 @@ async function* agentLoop(messages, options) {
|
|
|
317
333
|
});
|
|
318
334
|
messages.length = 0;
|
|
319
335
|
messages.push(...transformed);
|
|
336
|
+
latestProviderUsage = void 0;
|
|
337
|
+
usageAnchorIndex = void 0;
|
|
320
338
|
}
|
|
321
339
|
diag("transform_end");
|
|
322
340
|
}
|
|
@@ -577,10 +595,17 @@ async function* agentLoop(messages, options) {
|
|
|
577
595
|
...overflowDetails
|
|
578
596
|
});
|
|
579
597
|
try {
|
|
580
|
-
const
|
|
598
|
+
const pendingMessages = usageAnchorIndex === void 0 ? [] : messages.slice(usageAnchorIndex + 1);
|
|
599
|
+
const compacted = await options.transformContext(messages, {
|
|
600
|
+
force: true,
|
|
601
|
+
usage: latestProviderUsage,
|
|
602
|
+
pendingMessages
|
|
603
|
+
});
|
|
581
604
|
if (compacted !== messages && compacted.length < messages.length) {
|
|
582
605
|
messages.length = 0;
|
|
583
606
|
messages.push(...compacted);
|
|
607
|
+
latestProviderUsage = void 0;
|
|
608
|
+
usageAnchorIndex = void 0;
|
|
584
609
|
diag("overflow_compact_success", {
|
|
585
610
|
attempt: overflowCompactionAttempts,
|
|
586
611
|
messages: messages.length,
|
|
@@ -644,11 +669,33 @@ async function* agentLoop(messages, options) {
|
|
|
644
669
|
provider: options.provider,
|
|
645
670
|
model: options.model
|
|
646
671
|
});
|
|
672
|
+
if (runawayToolcallRetries < MAX_RUNAWAY_TOOLCALL_RETRIES) {
|
|
673
|
+
runawayToolcallRetries++;
|
|
674
|
+
const delayMs = RUNAWAY_TOOLCALL_RETRY_DELAY_MS * runawayToolcallRetries;
|
|
675
|
+
diag("retry", {
|
|
676
|
+
reason: "runaway_toolcall",
|
|
677
|
+
attempt: runawayToolcallRetries,
|
|
678
|
+
maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
|
|
679
|
+
delayMs,
|
|
680
|
+
...runawayDetected
|
|
681
|
+
});
|
|
682
|
+
yield {
|
|
683
|
+
type: "retry",
|
|
684
|
+
reason: "runaway_toolcall",
|
|
685
|
+
attempt: runawayToolcallRetries,
|
|
686
|
+
maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
|
|
687
|
+
delayMs,
|
|
688
|
+
silent: true
|
|
689
|
+
};
|
|
690
|
+
await abortableSleep(delayMs, options.signal);
|
|
691
|
+
turn--;
|
|
692
|
+
continue;
|
|
693
|
+
}
|
|
647
694
|
const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
|
|
648
695
|
yield {
|
|
649
696
|
type: "error",
|
|
650
697
|
error: new Error(
|
|
651
|
-
`The model
|
|
698
|
+
`The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
|
|
652
699
|
)
|
|
653
700
|
};
|
|
654
701
|
break;
|
|
@@ -749,6 +796,7 @@ async function* agentLoop(messages, options) {
|
|
|
749
796
|
}
|
|
750
797
|
overloadRetries = 0;
|
|
751
798
|
stallRetries = 0;
|
|
799
|
+
runawayToolcallRetries = 0;
|
|
752
800
|
const contentArr = Array.isArray(response.message.content) ? response.message.content : null;
|
|
753
801
|
const hasActionableContent = response.message.content !== "" && contentArr !== null && contentArr.some(
|
|
754
802
|
(p) => p.type === "text" || p.type === "tool_call" || p.type === "server_tool_call"
|
|
@@ -786,6 +834,8 @@ async function* agentLoop(messages, options) {
|
|
|
786
834
|
totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
|
|
787
835
|
}
|
|
788
836
|
messages.push(response.message);
|
|
837
|
+
latestProviderUsage = response.usage;
|
|
838
|
+
usageAnchorIndex = messages.length - 1;
|
|
789
839
|
const completedAt = Date.now();
|
|
790
840
|
const outputTokensPerSecond = providerDurationMs > 0 && response.usage.outputTokens > 0 ? response.usage.outputTokens / (providerDurationMs / 1e3) : void 0;
|
|
791
841
|
const timing = {
|
|
@@ -818,6 +868,27 @@ async function* agentLoop(messages, options) {
|
|
|
818
868
|
consecutivePauses = 0;
|
|
819
869
|
const allToolCalls = extractToolCalls(response.message.content);
|
|
820
870
|
if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
|
|
871
|
+
if (response.stopReason === "max_tokens") {
|
|
872
|
+
if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
|
|
873
|
+
maxTokensContinuations++;
|
|
874
|
+
diag("max_tokens_continuation", {
|
|
875
|
+
attempt: maxTokensContinuations,
|
|
876
|
+
maxAttempts: MAX_OUTPUT_CONTINUATIONS,
|
|
877
|
+
provider: options.provider,
|
|
878
|
+
model: options.model
|
|
879
|
+
});
|
|
880
|
+
yield { type: "truncated", reason: "max_tokens", continued: true };
|
|
881
|
+
messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
|
|
882
|
+
continue;
|
|
883
|
+
}
|
|
884
|
+
yield { type: "truncated", reason: "max_tokens", continued: false };
|
|
885
|
+
} else if (response.stopReason === "refusal" || response.stopReason === "error") {
|
|
886
|
+
yield {
|
|
887
|
+
type: "truncated",
|
|
888
|
+
reason: response.stopReason === "refusal" ? "refusal" : "provider_error",
|
|
889
|
+
continued: false
|
|
890
|
+
};
|
|
891
|
+
}
|
|
821
892
|
if (options.getSteeringMessages) {
|
|
822
893
|
const steering = await options.getSteeringMessages();
|
|
823
894
|
if (steering && steering.length > 0) {
|
|
@@ -873,6 +944,7 @@ async function* agentLoop(messages, options) {
|
|
|
873
944
|
const executionOptions = {
|
|
874
945
|
signal: options.signal,
|
|
875
946
|
maxToolResultChars: options.maxToolResultChars,
|
|
947
|
+
maxTurnToolResultChars: options.maxTurnToolResultChars,
|
|
876
948
|
toolMap,
|
|
877
949
|
invalidToolArgumentCounts,
|
|
878
950
|
markFatalToolArgumentError
|
|
@@ -1112,6 +1184,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
|
|
|
1112
1184
|
}
|
|
1113
1185
|
const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
|
|
1114
1186
|
capToolResults(toolResults, options.maxToolResultChars);
|
|
1187
|
+
capTurnToolResults(toolResults, options.maxTurnToolResultChars);
|
|
1115
1188
|
return { toolResults, aborted };
|
|
1116
1189
|
}
|
|
1117
1190
|
async function* executeToolCallsParallel(toolCalls, initialToolResults, options) {
|
|
@@ -1151,6 +1224,7 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
|
|
|
1151
1224
|
}
|
|
1152
1225
|
const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
|
|
1153
1226
|
capToolResults(toolResults, options.maxToolResultChars);
|
|
1227
|
+
capTurnToolResults(toolResults, options.maxTurnToolResultChars);
|
|
1154
1228
|
return { toolResults, aborted };
|
|
1155
1229
|
}
|
|
1156
1230
|
function buildToolResults(initialToolResults, toolCalls, resultsById) {
|
|
@@ -1181,16 +1255,56 @@ function capToolResults(toolResults, maxToolResultChars) {
|
|
|
1181
1255
|
const max = Math.min(maxToolResultChars, hardMax);
|
|
1182
1256
|
for (const toolResult of toolResults) {
|
|
1183
1257
|
if (typeof toolResult.content !== "string" || toolResult.content.length <= max) continue;
|
|
1258
|
+
const originalChars = toolResult.content.length;
|
|
1184
1259
|
const headChars = Math.floor(max * 0.7);
|
|
1185
1260
|
const tailChars = max - headChars;
|
|
1186
1261
|
const head = toolResult.content.slice(0, headChars);
|
|
1187
1262
|
const tail = toolResult.content.slice(-tailChars);
|
|
1188
|
-
const omitted =
|
|
1263
|
+
const omitted = originalChars - headChars - tailChars;
|
|
1189
1264
|
toolResult.content = head + `
|
|
1190
1265
|
|
|
1191
1266
|
[... ${omitted} characters omitted ...]
|
|
1192
1267
|
|
|
1193
1268
|
` + tail;
|
|
1269
|
+
toolResult.capped = {
|
|
1270
|
+
originalChars,
|
|
1271
|
+
keptChars: toolResult.content.length,
|
|
1272
|
+
scope: "per-result"
|
|
1273
|
+
};
|
|
1274
|
+
}
|
|
1275
|
+
}
|
|
1276
|
+
function capTurnToolResults(toolResults, maxTurnToolResultChars) {
|
|
1277
|
+
if (!maxTurnToolResultChars) return;
|
|
1278
|
+
const textResults = toolResults.filter(
|
|
1279
|
+
(toolResult) => typeof toolResult.content === "string"
|
|
1280
|
+
);
|
|
1281
|
+
const total = textResults.reduce((sum, toolResult) => sum + toolResult.content.length, 0);
|
|
1282
|
+
if (total <= maxTurnToolResultChars) return;
|
|
1283
|
+
const bySize = [...textResults].sort((a, b) => a.content.length - b.content.length);
|
|
1284
|
+
let remaining = maxTurnToolResultChars;
|
|
1285
|
+
let left = bySize.length;
|
|
1286
|
+
for (const toolResult of bySize) {
|
|
1287
|
+
const fairShare = Math.floor(remaining / left);
|
|
1288
|
+
left--;
|
|
1289
|
+
if (toolResult.content.length <= fairShare) {
|
|
1290
|
+
remaining -= toolResult.content.length;
|
|
1291
|
+
continue;
|
|
1292
|
+
}
|
|
1293
|
+
remaining -= fairShare;
|
|
1294
|
+
const originalChars = toolResult.content.length;
|
|
1295
|
+
const headChars = Math.floor(fairShare * 0.7);
|
|
1296
|
+
const tailChars = fairShare - headChars;
|
|
1297
|
+
const omitted = originalChars - fairShare;
|
|
1298
|
+
toolResult.content = toolResult.content.slice(0, headChars) + `
|
|
1299
|
+
|
|
1300
|
+
[... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
|
|
1301
|
+
|
|
1302
|
+
` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
|
|
1303
|
+
toolResult.capped = {
|
|
1304
|
+
originalChars: toolResult.capped?.originalChars ?? originalChars,
|
|
1305
|
+
keptChars: toolResult.content.length,
|
|
1306
|
+
scope: "per-turn"
|
|
1307
|
+
};
|
|
1194
1308
|
}
|
|
1195
1309
|
}
|
|
1196
1310
|
function normalizeToolResult(raw) {
|