@prestyj/agent 5.7.0 → 5.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,4 +1,4 @@
1
- import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, StopReason, Usage, AssistantMessage } from '@prestyj/ai';
1
+ import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, Usage, StopReason, AssistantMessage } from '@prestyj/ai';
2
2
  import { z } from 'zod';
3
3
 
4
4
  interface StructuredToolResult {
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
87
87
  totalTurns: number;
88
88
  maxTurns: number;
89
89
  }
90
+ /**
91
+ * Warning signal emitted when a turn ended on a non-clean stop reason —
92
+ * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
93
+ * or a provider-reported `error` stop. Distinguishes a truncated/degraded
94
+ * completion from a clean one so hosts can warn the user instead of silently
95
+ * presenting incomplete output as done.
96
+ */
97
+ interface AgentTruncatedEvent {
98
+ type: "truncated";
99
+ reason: "max_tokens" | "refusal" | "provider_error";
100
+ /** True when the loop injected a continuation and will keep going. */
101
+ continued: boolean;
102
+ }
90
103
  interface AgentRetryEvent {
91
104
  type: "retry";
92
- reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
105
+ reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
93
106
  attempt: number;
94
107
  maxAttempts: number;
95
108
  delayMs: number;
@@ -135,7 +148,15 @@ interface AgentFollowUpMessageEvent {
135
148
  type: "follow_up_message";
136
149
  content: Message["content"];
137
150
  }
138
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
151
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
152
+ interface TransformContextOptions {
153
+ /** Force a transform after the provider reports context overflow. */
154
+ force?: boolean;
155
+ /** Latest successful provider usage, anchored at its assistant message. */
156
+ usage?: Usage;
157
+ /** Messages appended after that usage sample and not yet seen by the provider. */
158
+ pendingMessages: Message[];
159
+ }
139
160
  interface AgentOptions {
140
161
  provider: StreamOptions["provider"];
141
162
  model: string;
@@ -180,6 +201,10 @@ interface AgentOptions {
180
201
  clearToolUses?: boolean;
181
202
  /** Max characters for a single tool result. Results exceeding this are truncated with a notice. */
182
203
  maxToolResultChars?: number;
204
+ /** Aggregate budget for ALL tool results in one assistant turn. Protects
205
+ * against parallel fan-outs injecting huge uncached context in one turn;
206
+ * the largest results are trimmed (water-filling) with a re-run notice. */
207
+ maxTurnToolResultChars?: number;
183
208
  /** Max consecutive pause_turn continuations before stopping (default: 5).
184
209
  * Prevents infinite loops when server-side tools keep pausing. */
185
210
  maxContinuations?: number;
@@ -188,12 +213,12 @@ interface AgentOptions {
188
213
  * the messages array (e.g. compaction, truncation). Return the same array
189
214
  * for no-op, or a new array to replace the conversation context.
190
215
  *
216
+ * The latest provider usage is authoritative for the history through its
217
+ * assistant response. `pendingMessages` contains context appended afterward.
191
218
  * When `options.force` is true, the caller should compact unconditionally
192
219
  * (e.g. after a context overflow error from the API).
193
220
  */
194
- transformContext?: (messages: Message[], options?: {
195
- force?: boolean;
196
- }) => Message[] | Promise<Message[]>;
221
+ transformContext?: (messages: Message[], options: TransformContextOptions) => Message[] | Promise<Message[]>;
197
222
  /**
198
223
  * Polled after tool execution completes each turn. Returns user messages
199
224
  * to inject into the conversation before the next LLM call (steering).
@@ -299,4 +324,4 @@ declare function isBillingError(err: unknown): boolean;
299
324
  declare function isUsageLimitError(err: unknown): boolean;
300
325
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
301
326
 
302
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
327
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
package/dist/index.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, StopReason, Usage, AssistantMessage } from '@prestyj/ai';
1
+ import { StreamOptions, Message, Tool, ToolResultContent, ServerToolDefinition, Usage, StopReason, AssistantMessage } from '@prestyj/ai';
2
2
  import { z } from 'zod';
3
3
 
4
4
  interface StructuredToolResult {
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
87
87
  totalTurns: number;
88
88
  maxTurns: number;
89
89
  }
90
+ /**
91
+ * Warning signal emitted when a turn ended on a non-clean stop reason —
92
+ * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
93
+ * or a provider-reported `error` stop. Distinguishes a truncated/degraded
94
+ * completion from a clean one so hosts can warn the user instead of silently
95
+ * presenting incomplete output as done.
96
+ */
97
+ interface AgentTruncatedEvent {
98
+ type: "truncated";
99
+ reason: "max_tokens" | "refusal" | "provider_error";
100
+ /** True when the loop injected a continuation and will keep going. */
101
+ continued: boolean;
102
+ }
90
103
  interface AgentRetryEvent {
91
104
  type: "retry";
92
- reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
105
+ reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
93
106
  attempt: number;
94
107
  maxAttempts: number;
95
108
  delayMs: number;
@@ -135,7 +148,15 @@ interface AgentFollowUpMessageEvent {
135
148
  type: "follow_up_message";
136
149
  content: Message["content"];
137
150
  }
138
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
151
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
152
+ interface TransformContextOptions {
153
+ /** Force a transform after the provider reports context overflow. */
154
+ force?: boolean;
155
+ /** Latest successful provider usage, anchored at its assistant message. */
156
+ usage?: Usage;
157
+ /** Messages appended after that usage sample and not yet seen by the provider. */
158
+ pendingMessages: Message[];
159
+ }
139
160
  interface AgentOptions {
140
161
  provider: StreamOptions["provider"];
141
162
  model: string;
@@ -180,6 +201,10 @@ interface AgentOptions {
180
201
  clearToolUses?: boolean;
181
202
  /** Max characters for a single tool result. Results exceeding this are truncated with a notice. */
182
203
  maxToolResultChars?: number;
204
+ /** Aggregate budget for ALL tool results in one assistant turn. Protects
205
+ * against parallel fan-outs injecting huge uncached context in one turn;
206
+ * the largest results are trimmed (water-filling) with a re-run notice. */
207
+ maxTurnToolResultChars?: number;
183
208
  /** Max consecutive pause_turn continuations before stopping (default: 5).
184
209
  * Prevents infinite loops when server-side tools keep pausing. */
185
210
  maxContinuations?: number;
@@ -188,12 +213,12 @@ interface AgentOptions {
188
213
  * the messages array (e.g. compaction, truncation). Return the same array
189
214
  * for no-op, or a new array to replace the conversation context.
190
215
  *
216
+ * The latest provider usage is authoritative for the history through its
217
+ * assistant response. `pendingMessages` contains context appended afterward.
191
218
  * When `options.force` is true, the caller should compact unconditionally
192
219
  * (e.g. after a context overflow error from the API).
193
220
  */
194
- transformContext?: (messages: Message[], options?: {
195
- force?: boolean;
196
- }) => Message[] | Promise<Message[]>;
221
+ transformContext?: (messages: Message[], options: TransformContextOptions) => Message[] | Promise<Message[]>;
197
222
  /**
198
223
  * Polled after tool execution completes each turn. Returns user messages
199
224
  * to inject into the conversation before the next LLM call (steering).
@@ -299,4 +324,4 @@ declare function isBillingError(err: unknown): boolean;
299
324
  declare function isUsageLimitError(err: unknown): boolean;
300
325
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
301
326
 
302
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
327
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
package/dist/index.js CHANGED
@@ -127,7 +127,7 @@ function classifyOverload(err) {
127
127
  if (statusCode === 529 || msg.includes("overloaded") || msg.includes("529")) {
128
128
  return "overloaded";
129
129
  }
130
- if (statusCode === 500 || statusCode === 502 || statusCode === 503 || statusCode === 504 || msg.includes("api_error") || msg.includes("server_error") || msg.includes("internal server error") || msg.includes("bad gateway") || msg.includes("service unavailable") || msg.includes("gateway timeout")) {
130
+ if (statusCode === 500 || statusCode === 502 || statusCode === 503 || statusCode === 504 || statusCode === 507 || msg.includes("api_error") || msg.includes("server_error") || msg.includes("internal server error") || msg.includes("bad gateway") || msg.includes("service unavailable") || msg.includes("gateway timeout") || msg.includes("exceeded request buffer limit while retrying upstream")) {
131
131
  return "provider_error";
132
132
  }
133
133
  if (isOpaqueProviderMessage(err.message)) {
@@ -188,7 +188,11 @@ function createAbortError() {
188
188
  }
189
189
  function abortablePromise(promise, signal) {
190
190
  if (!signal) return promise;
191
- if (signal.aborted) return Promise.reject(createAbortError());
191
+ if (signal.aborted) {
192
+ promise.catch(() => {
193
+ });
194
+ return Promise.reject(createAbortError());
195
+ }
192
196
  return new Promise((resolve, reject) => {
193
197
  let settled = false;
194
198
  const cleanup = () => signal.removeEventListener("abort", onAbort);
@@ -233,6 +237,8 @@ async function* agentLoop(messages, options) {
233
237
  const maxContinuations = options.maxContinuations ?? 5;
234
238
  let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
235
239
  const totalUsage = { inputTokens: 0, outputTokens: 0 };
240
+ let latestProviderUsage;
241
+ let usageAnchorIndex;
236
242
  let turn = 0;
237
243
  let hitMaxTurns = false;
238
244
  let firstTurn = true;
@@ -242,6 +248,7 @@ async function* agentLoop(messages, options) {
242
248
  let overloadRetries = 0;
243
249
  let emptyResponseRetries = 0;
244
250
  let stallRetries = 0;
251
+ let runawayToolcallRetries = 0;
245
252
  let overflowCompactionAttempts = 0;
246
253
  let toolResultTruncationAttempted = false;
247
254
  const invalidToolArgumentCounts = /* @__PURE__ */ new Map();
@@ -250,11 +257,16 @@ async function* agentLoop(messages, options) {
250
257
  const MAX_OVERLOAD_RETRIES = 10;
251
258
  const MAX_EMPTY_RESPONSE_RETRIES = 2;
252
259
  const MAX_STALL_RETRIES = 10;
260
+ const MAX_RUNAWAY_TOOLCALL_RETRIES = 2;
261
+ const RUNAWAY_TOOLCALL_RETRY_DELAY_MS = 1e3;
253
262
  const MAX_OVERFLOW_COMPACTIONS = 2;
254
263
  const STALL_RETRIES_BEFORE_NON_STREAMING = 2;
255
264
  const STALL_DELAY_MS = 1e3;
256
265
  const MIN_PARTIAL_PRESERVE_CHARS = 200;
257
266
  const PARTIAL_CONTINUATION_PROMPT = "[Your previous response was cut off by a connection failure. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
267
+ const MAX_OUTPUT_CONTINUATIONS = 2;
268
+ const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
269
+ let maxTokensContinuations = 0;
258
270
  const OVERLOAD_BASE_DELAY_MS = 2e3;
259
271
  const OVERLOAD_MAX_DELAY_MS = 3e4;
260
272
  const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
@@ -309,7 +321,11 @@ async function* agentLoop(messages, options) {
309
321
  firstTurn = false;
310
322
  if (options.transformContext) {
311
323
  diag("transform_start");
312
- const transformed = await options.transformContext(messages);
324
+ const pendingMessages = usageAnchorIndex === void 0 ? [] : messages.slice(usageAnchorIndex + 1);
325
+ const transformed = await options.transformContext(messages, {
326
+ usage: latestProviderUsage,
327
+ pendingMessages
328
+ });
313
329
  if (transformed !== messages) {
314
330
  diag("transform_compacted", {
315
331
  before: messages.length,
@@ -317,6 +333,8 @@ async function* agentLoop(messages, options) {
317
333
  });
318
334
  messages.length = 0;
319
335
  messages.push(...transformed);
336
+ latestProviderUsage = void 0;
337
+ usageAnchorIndex = void 0;
320
338
  }
321
339
  diag("transform_end");
322
340
  }
@@ -577,10 +595,17 @@ async function* agentLoop(messages, options) {
577
595
  ...overflowDetails
578
596
  });
579
597
  try {
580
- const compacted = await options.transformContext(messages, { force: true });
598
+ const pendingMessages = usageAnchorIndex === void 0 ? [] : messages.slice(usageAnchorIndex + 1);
599
+ const compacted = await options.transformContext(messages, {
600
+ force: true,
601
+ usage: latestProviderUsage,
602
+ pendingMessages
603
+ });
581
604
  if (compacted !== messages && compacted.length < messages.length) {
582
605
  messages.length = 0;
583
606
  messages.push(...compacted);
607
+ latestProviderUsage = void 0;
608
+ usageAnchorIndex = void 0;
584
609
  diag("overflow_compact_success", {
585
610
  attempt: overflowCompactionAttempts,
586
611
  messages: messages.length,
@@ -644,11 +669,33 @@ async function* agentLoop(messages, options) {
644
669
  provider: options.provider,
645
670
  model: options.model
646
671
  });
672
+ if (runawayToolcallRetries < MAX_RUNAWAY_TOOLCALL_RETRIES) {
673
+ runawayToolcallRetries++;
674
+ const delayMs = RUNAWAY_TOOLCALL_RETRY_DELAY_MS * runawayToolcallRetries;
675
+ diag("retry", {
676
+ reason: "runaway_toolcall",
677
+ attempt: runawayToolcallRetries,
678
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
679
+ delayMs,
680
+ ...runawayDetected
681
+ });
682
+ yield {
683
+ type: "retry",
684
+ reason: "runaway_toolcall",
685
+ attempt: runawayToolcallRetries,
686
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
687
+ delayMs,
688
+ silent: true
689
+ };
690
+ await abortableSleep(delayMs, options.signal);
691
+ turn--;
692
+ continue;
693
+ }
647
694
  const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
648
695
  yield {
649
696
  type: "error",
650
697
  error: new Error(
651
- `The model glitched mid-tool-call and produced ${detail} without closing the call. This is usually an upstream model bug \u2014 try the same request again or switch models. Your conversation is preserved.`
698
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
652
699
  )
653
700
  };
654
701
  break;
@@ -749,6 +796,7 @@ async function* agentLoop(messages, options) {
749
796
  }
750
797
  overloadRetries = 0;
751
798
  stallRetries = 0;
799
+ runawayToolcallRetries = 0;
752
800
  const contentArr = Array.isArray(response.message.content) ? response.message.content : null;
753
801
  const hasActionableContent = response.message.content !== "" && contentArr !== null && contentArr.some(
754
802
  (p) => p.type === "text" || p.type === "tool_call" || p.type === "server_tool_call"
@@ -786,6 +834,8 @@ async function* agentLoop(messages, options) {
786
834
  totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
787
835
  }
788
836
  messages.push(response.message);
837
+ latestProviderUsage = response.usage;
838
+ usageAnchorIndex = messages.length - 1;
789
839
  const completedAt = Date.now();
790
840
  const outputTokensPerSecond = providerDurationMs > 0 && response.usage.outputTokens > 0 ? response.usage.outputTokens / (providerDurationMs / 1e3) : void 0;
791
841
  const timing = {
@@ -818,6 +868,27 @@ async function* agentLoop(messages, options) {
818
868
  consecutivePauses = 0;
819
869
  const allToolCalls = extractToolCalls(response.message.content);
820
870
  if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
871
+ if (response.stopReason === "max_tokens") {
872
+ if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
873
+ maxTokensContinuations++;
874
+ diag("max_tokens_continuation", {
875
+ attempt: maxTokensContinuations,
876
+ maxAttempts: MAX_OUTPUT_CONTINUATIONS,
877
+ provider: options.provider,
878
+ model: options.model
879
+ });
880
+ yield { type: "truncated", reason: "max_tokens", continued: true };
881
+ messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
882
+ continue;
883
+ }
884
+ yield { type: "truncated", reason: "max_tokens", continued: false };
885
+ } else if (response.stopReason === "refusal" || response.stopReason === "error") {
886
+ yield {
887
+ type: "truncated",
888
+ reason: response.stopReason === "refusal" ? "refusal" : "provider_error",
889
+ continued: false
890
+ };
891
+ }
821
892
  if (options.getSteeringMessages) {
822
893
  const steering = await options.getSteeringMessages();
823
894
  if (steering && steering.length > 0) {
@@ -873,6 +944,7 @@ async function* agentLoop(messages, options) {
873
944
  const executionOptions = {
874
945
  signal: options.signal,
875
946
  maxToolResultChars: options.maxToolResultChars,
947
+ maxTurnToolResultChars: options.maxTurnToolResultChars,
876
948
  toolMap,
877
949
  invalidToolArgumentCounts,
878
950
  markFatalToolArgumentError
@@ -1112,6 +1184,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1112
1184
  }
1113
1185
  const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1114
1186
  capToolResults(toolResults, options.maxToolResultChars);
1187
+ capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1115
1188
  return { toolResults, aborted };
1116
1189
  }
1117
1190
  async function* executeToolCallsParallel(toolCalls, initialToolResults, options) {
@@ -1151,6 +1224,7 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
1151
1224
  }
1152
1225
  const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1153
1226
  capToolResults(toolResults, options.maxToolResultChars);
1227
+ capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1154
1228
  return { toolResults, aborted };
1155
1229
  }
1156
1230
  function buildToolResults(initialToolResults, toolCalls, resultsById) {
@@ -1181,16 +1255,56 @@ function capToolResults(toolResults, maxToolResultChars) {
1181
1255
  const max = Math.min(maxToolResultChars, hardMax);
1182
1256
  for (const toolResult of toolResults) {
1183
1257
  if (typeof toolResult.content !== "string" || toolResult.content.length <= max) continue;
1258
+ const originalChars = toolResult.content.length;
1184
1259
  const headChars = Math.floor(max * 0.7);
1185
1260
  const tailChars = max - headChars;
1186
1261
  const head = toolResult.content.slice(0, headChars);
1187
1262
  const tail = toolResult.content.slice(-tailChars);
1188
- const omitted = toolResult.content.length - headChars - tailChars;
1263
+ const omitted = originalChars - headChars - tailChars;
1189
1264
  toolResult.content = head + `
1190
1265
 
1191
1266
  [... ${omitted} characters omitted ...]
1192
1267
 
1193
1268
  ` + tail;
1269
+ toolResult.capped = {
1270
+ originalChars,
1271
+ keptChars: toolResult.content.length,
1272
+ scope: "per-result"
1273
+ };
1274
+ }
1275
+ }
1276
+ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1277
+ if (!maxTurnToolResultChars) return;
1278
+ const textResults = toolResults.filter(
1279
+ (toolResult) => typeof toolResult.content === "string"
1280
+ );
1281
+ const total = textResults.reduce((sum, toolResult) => sum + toolResult.content.length, 0);
1282
+ if (total <= maxTurnToolResultChars) return;
1283
+ const bySize = [...textResults].sort((a, b) => a.content.length - b.content.length);
1284
+ let remaining = maxTurnToolResultChars;
1285
+ let left = bySize.length;
1286
+ for (const toolResult of bySize) {
1287
+ const fairShare = Math.floor(remaining / left);
1288
+ left--;
1289
+ if (toolResult.content.length <= fairShare) {
1290
+ remaining -= toolResult.content.length;
1291
+ continue;
1292
+ }
1293
+ remaining -= fairShare;
1294
+ const originalChars = toolResult.content.length;
1295
+ const headChars = Math.floor(fairShare * 0.7);
1296
+ const tailChars = fairShare - headChars;
1297
+ const omitted = originalChars - fairShare;
1298
+ toolResult.content = toolResult.content.slice(0, headChars) + `
1299
+
1300
+ [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1301
+
1302
+ ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1303
+ toolResult.capped = {
1304
+ originalChars: toolResult.capped?.originalChars ?? originalChars,
1305
+ keptChars: toolResult.content.length,
1306
+ scope: "per-turn"
1307
+ };
1194
1308
  }
1195
1309
  }
1196
1310
  function normalizeToolResult(raw) {