@prestyj/agent 5.10.0 → 5.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
20
20
  * batch runs in source order so stateful mutations cannot race each other.
21
21
  */
22
22
  executionMode?: ToolExecutionMode;
23
+ /**
24
+ * Overrides the loop's default per-tool timeout. A tool that owns a longer
25
+ * internal budget than the default must declare it here, or the loop cancels
26
+ * it first and the tool's own timeout — with its specific, actionable error
27
+ * message — becomes unreachable.
28
+ */
29
+ timeoutMs?: number;
23
30
  execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
24
31
  }
25
32
  interface AgentTextDeltaEvent {
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
70
77
  usage: Usage;
71
78
  timing: AgentTurnTiming;
72
79
  }
80
+ /**
81
+ * A safe point between steps: the assistant message and every tool result for
82
+ * this turn are now in the message array, and no provider call is in flight.
83
+ *
84
+ * Hosts that persist a transcript flush here. Without it a crash mid-run loses
85
+ * the WHOLE turn — including tool results whose side effects already landed on
86
+ * disk — because the only flush happens after the loop returns.
87
+ *
88
+ * Yielded immediately after tool results are appended, so it pairs with
89
+ * `turn_end` (which covers the assistant half) to cover every message.
90
+ */
91
+ interface AgentCheckpointEvent {
92
+ type: "checkpoint";
93
+ turn: number;
94
+ }
73
95
  interface AgentDoneEvent {
74
96
  type: "agent_done";
75
97
  totalTurns: number;
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
87
109
  totalTurns: number;
88
110
  maxTurns: number;
89
111
  }
112
+ /**
113
+ * Emitted when the loop was about to stop on an exhausted turn budget but the
114
+ * host granted an extension instead. The effective budget is raised and the
115
+ * loop continues with a continuation prompt, so this is NOT terminal — unlike
116
+ * `max_turns`, which still fires if the extended budget is also spent.
117
+ */
118
+ interface AgentTurnBudgetExtendedEvent {
119
+ type: "turn_budget_extended";
120
+ /** Turn number at which the budget was exhausted. */
121
+ turn: number;
122
+ /** New effective `maxTurns` after the extension. */
123
+ grantedTurns: number;
124
+ /** 1-based extension count for this run. */
125
+ extension: number;
126
+ }
90
127
  /**
91
128
  * Warning signal emitted when a turn ended on a non-clean stop reason —
92
129
  * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
148
185
  type: "follow_up_message";
149
186
  content: Message["content"];
150
187
  }
151
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
188
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
152
189
  interface TransformContextOptions {
153
190
  /** Force a transform after the provider reports context overflow. */
154
191
  force?: boolean;
@@ -168,10 +205,32 @@ interface AgentOptions {
168
205
  /** Control whether tools may/must be called, or select a named tool when supported. */
169
206
  toolChoice?: StreamOptions["toolChoice"];
170
207
  maxTurns?: number;
208
+ /**
209
+ * How many times `onTurnBudgetExhausted` may grant extra turns in one run.
210
+ * Each grant raises the effective budget by the original `maxTurns`.
211
+ * Default: 2. Set 0 to disable extensions entirely.
212
+ */
213
+ maxTurnExtensions?: number;
171
214
  maxTokens?: number;
172
215
  temperature?: number;
173
216
  thinking?: StreamOptions["thinking"];
174
217
  apiKey?: string;
218
+ /**
219
+ * Re-resolve the credential at the start of every turn. A run can span many
220
+ * minutes, and an OAuth grant refreshed by any process (another app window, a
221
+ * CLI session, the usage poller) invalidates the access token captured when
222
+ * the run began — so a pinned `apiKey` goes dead mid-run and every remaining
223
+ * turn fails with an authentication error. Returning the current credential
224
+ * here keeps a long run alive across rotations.
225
+ *
226
+ * Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
227
+ * resolver throws (the provider call then surfaces the real auth error).
228
+ */
229
+ resolveCredentials?: () => Promise<{
230
+ apiKey: string;
231
+ accountId?: string;
232
+ projectId?: string;
233
+ }>;
175
234
  baseUrl?: string;
176
235
  signal?: AbortSignal;
177
236
  accountId?: string;
@@ -234,6 +293,18 @@ interface AgentOptions {
234
293
  * on read.
235
294
  */
236
295
  getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
296
+ /**
297
+ * Consulted when a tool-running turn exhausts the turn budget mid-task,
298
+ * before the loop emits the terminal `max_turns` event. Return true to grant
299
+ * another `maxTurns` worth of turns; false (the default when unset) keeps
300
+ * today's hard cut-off. Hosts should only grant on evidence of progress —
301
+ * extending a spinning agent just buys it more tokens to spin with.
302
+ */
303
+ onTurnBudgetExhausted?: (ctx: {
304
+ turn: number;
305
+ maxTurns: number;
306
+ extension: number;
307
+ }) => Promise<boolean> | boolean;
237
308
  }
238
309
  interface AgentResult {
239
310
  message: AssistantMessage;
@@ -319,7 +390,7 @@ declare function isBillingError(err: unknown): boolean;
319
390
  * plan running out of usage). Unlike a transient per-minute 429, this does NOT
320
391
  * clear with a quick retry — the user must wait for the window to reset — so the
321
392
  * loop surfaces it immediately instead of retrying for minutes. Matches the
322
- * canonical message gg-ai stamps onto the provider error.
393
+ * canonical message @prestyj/ai stamps onto the provider error.
323
394
  */
324
395
  declare function isUsageLimitError(err: unknown): boolean;
325
396
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
package/dist/index.d.ts CHANGED
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
20
20
  * batch runs in source order so stateful mutations cannot race each other.
21
21
  */
22
22
  executionMode?: ToolExecutionMode;
23
+ /**
24
+ * Overrides the loop's default per-tool timeout. A tool that owns a longer
25
+ * internal budget than the default must declare it here, or the loop cancels
26
+ * it first and the tool's own timeout — with its specific, actionable error
27
+ * message — becomes unreachable.
28
+ */
29
+ timeoutMs?: number;
23
30
  execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
24
31
  }
25
32
  interface AgentTextDeltaEvent {
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
70
77
  usage: Usage;
71
78
  timing: AgentTurnTiming;
72
79
  }
80
+ /**
81
+ * A safe point between steps: the assistant message and every tool result for
82
+ * this turn are now in the message array, and no provider call is in flight.
83
+ *
84
+ * Hosts that persist a transcript flush here. Without it a crash mid-run loses
85
+ * the WHOLE turn — including tool results whose side effects already landed on
86
+ * disk — because the only flush happens after the loop returns.
87
+ *
88
+ * Yielded immediately after tool results are appended, so it pairs with
89
+ * `turn_end` (which covers the assistant half) to cover every message.
90
+ */
91
+ interface AgentCheckpointEvent {
92
+ type: "checkpoint";
93
+ turn: number;
94
+ }
73
95
  interface AgentDoneEvent {
74
96
  type: "agent_done";
75
97
  totalTurns: number;
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
87
109
  totalTurns: number;
88
110
  maxTurns: number;
89
111
  }
112
+ /**
113
+ * Emitted when the loop was about to stop on an exhausted turn budget but the
114
+ * host granted an extension instead. The effective budget is raised and the
115
+ * loop continues with a continuation prompt, so this is NOT terminal — unlike
116
+ * `max_turns`, which still fires if the extended budget is also spent.
117
+ */
118
+ interface AgentTurnBudgetExtendedEvent {
119
+ type: "turn_budget_extended";
120
+ /** Turn number at which the budget was exhausted. */
121
+ turn: number;
122
+ /** New effective `maxTurns` after the extension. */
123
+ grantedTurns: number;
124
+ /** 1-based extension count for this run. */
125
+ extension: number;
126
+ }
90
127
  /**
91
128
  * Warning signal emitted when a turn ended on a non-clean stop reason —
92
129
  * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
148
185
  type: "follow_up_message";
149
186
  content: Message["content"];
150
187
  }
151
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
188
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
152
189
  interface TransformContextOptions {
153
190
  /** Force a transform after the provider reports context overflow. */
154
191
  force?: boolean;
@@ -168,10 +205,32 @@ interface AgentOptions {
168
205
  /** Control whether tools may/must be called, or select a named tool when supported. */
169
206
  toolChoice?: StreamOptions["toolChoice"];
170
207
  maxTurns?: number;
208
+ /**
209
+ * How many times `onTurnBudgetExhausted` may grant extra turns in one run.
210
+ * Each grant raises the effective budget by the original `maxTurns`.
211
+ * Default: 2. Set 0 to disable extensions entirely.
212
+ */
213
+ maxTurnExtensions?: number;
171
214
  maxTokens?: number;
172
215
  temperature?: number;
173
216
  thinking?: StreamOptions["thinking"];
174
217
  apiKey?: string;
218
+ /**
219
+ * Re-resolve the credential at the start of every turn. A run can span many
220
+ * minutes, and an OAuth grant refreshed by any process (another app window, a
221
+ * CLI session, the usage poller) invalidates the access token captured when
222
+ * the run began — so a pinned `apiKey` goes dead mid-run and every remaining
223
+ * turn fails with an authentication error. Returning the current credential
224
+ * here keeps a long run alive across rotations.
225
+ *
226
+ * Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
227
+ * resolver throws (the provider call then surfaces the real auth error).
228
+ */
229
+ resolveCredentials?: () => Promise<{
230
+ apiKey: string;
231
+ accountId?: string;
232
+ projectId?: string;
233
+ }>;
175
234
  baseUrl?: string;
176
235
  signal?: AbortSignal;
177
236
  accountId?: string;
@@ -234,6 +293,18 @@ interface AgentOptions {
234
293
  * on read.
235
294
  */
236
295
  getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
296
+ /**
297
+ * Consulted when a tool-running turn exhausts the turn budget mid-task,
298
+ * before the loop emits the terminal `max_turns` event. Return true to grant
299
+ * another `maxTurns` worth of turns; false (the default when unset) keeps
300
+ * today's hard cut-off. Hosts should only grant on evidence of progress —
301
+ * extending a spinning agent just buys it more tokens to spin with.
302
+ */
303
+ onTurnBudgetExhausted?: (ctx: {
304
+ turn: number;
305
+ maxTurns: number;
306
+ extension: number;
307
+ }) => Promise<boolean> | boolean;
237
308
  }
238
309
  interface AgentResult {
239
310
  message: AssistantMessage;
@@ -319,7 +390,7 @@ declare function isBillingError(err: unknown): boolean;
319
390
  * plan running out of usage). Unlike a transient per-minute 429, this does NOT
320
391
  * clear with a quick retry — the user must wait for the window to reset — so the
321
392
  * loop surfaces it immediately instead of retrying for minutes. Matches the
322
- * canonical message gg-ai stamps onto the provider error.
393
+ * canonical message @prestyj/ai stamps onto the provider error.
323
394
  */
324
395
  declare function isUsageLimitError(err: unknown): boolean;
325
396
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
package/dist/index.js CHANGED
@@ -8,7 +8,9 @@ import {
8
8
  EventStream,
9
9
  EZCoderAIError,
10
10
  isHardBillingMessage,
11
- redactValue
11
+ redactValue,
12
+ sliceHead,
13
+ sliceTail
12
14
  } from "@prestyj/ai";
13
15
 
14
16
  // src/local-backend.ts
@@ -35,6 +37,7 @@ function isLocalBackendUrl(baseUrl) {
35
37
 
36
38
  // src/agent-loop.ts
37
39
  var DEFAULT_MAX_TURNS = 300;
40
+ var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
38
41
  var _diagFn = null;
39
42
  function setStreamDiagnostic(fn) {
40
43
  _diagFn = fn;
@@ -54,7 +57,13 @@ function isContextOverflow(err) {
54
57
  if (overflowStatus === 402) return false;
55
58
  if (isBillingError(err)) return false;
56
59
  const msg = err.message.toLowerCase();
57
- return msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten") || msg.includes("token") && msg.includes("exceed");
60
+ if (msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten")) {
61
+ return true;
62
+ }
63
+ const rateLimited = overflowStatus === 429 || msg.includes("rate limit") || msg.includes("rate_limit") || msg.includes("too many requests");
64
+ const perUnitTime = msg.includes("per min") || msg.includes("/min") || msg.includes("per minute") || msg.includes("per hour") || msg.includes("per day") || msg.includes("tpm") || msg.includes("rpm");
65
+ if (rateLimited && perUnitTime) return false;
66
+ return msg.includes("token") && msg.includes("exceed");
58
67
  }
59
68
  function parseOverflowNumber(value) {
60
69
  return Number(value.replace(/[,_\s]/g, ""));
@@ -167,6 +176,17 @@ function isMalformedStream(err) {
167
176
  const msg = err.message;
168
177
  return /\bin JSON at position \d+/i.test(msg);
169
178
  }
179
+ var TIMEOUT_NAMES = /* @__PURE__ */ new Set(["TimeoutError", "ConnectTimeoutError", "HeadersTimeoutError"]);
180
+ var TIMEOUT_MESSAGES = [
181
+ /^request timed out\.?$/i,
182
+ /\brequest to [\w .-]+ timed out\b/i,
183
+ /\b(?:connection|socket|headers|stream) timed out\b/i
184
+ ];
185
+ function isBareTimeout(e) {
186
+ if (typeof e.name === "string" && TIMEOUT_NAMES.has(e.name)) return true;
187
+ if (typeof e.message !== "string") return false;
188
+ return TIMEOUT_MESSAGES.some((re) => re.test(e.message));
189
+ }
170
190
  function isTransportFailure(err) {
171
191
  const codes = /* @__PURE__ */ new Set([
172
192
  "ECONNRESET",
@@ -203,6 +223,8 @@ function isTransportFailure(err) {
203
223
  if (typeof e.message === "string") {
204
224
  for (const re of messages) if (re.test(e.message)) return true;
205
225
  }
226
+ const clientError = typeof e.status === "number" && e.status >= 400 && e.status < 500;
227
+ if (!clientError && isBareTimeout(e)) return true;
206
228
  cur = e.cause;
207
229
  }
208
230
  return false;
@@ -237,6 +259,9 @@ function abortablePromise(promise, signal) {
237
259
  promise.then(resolveOnce, rejectOnce);
238
260
  });
239
261
  }
262
+ function turnBudgetContinuationPrompt() {
263
+ return "[You reached the turn limit for this segment but the work is not finished, so you have been granted more turns. Before continuing, state in one or two sentences what is already done and what remains, then keep going from there \u2014 do not restart work that is already complete.]";
264
+ }
240
265
  function abortableSleep(ms, signal) {
241
266
  if (signal?.aborted) return Promise.reject(createAbortError());
242
267
  return new Promise((resolve, reject) => {
@@ -258,6 +283,9 @@ function closeIterator(iterator) {
258
283
  }
259
284
  async function* agentLoop(messages, options) {
260
285
  const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
286
+ let effectiveMaxTurns = maxTurns;
287
+ const maxTurnExtensions = Math.max(0, options.maxTurnExtensions ?? 2);
288
+ let turnExtensions = 0;
261
289
  const maxContinuations = options.maxContinuations ?? 5;
262
290
  let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
263
291
  const totalUsage = { inputTokens: 0, outputTokens: 0 };
@@ -305,12 +333,12 @@ async function* agentLoop(messages, options) {
305
333
  const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
306
334
  const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
307
335
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
308
- const MAX_TOOLCALL_DELTA_EVENTS = 2e4;
336
+ const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
309
337
  let logicalTurnStartedAt = 0;
310
338
  let firstProviderEventAt;
311
339
  let providerDurationMs = 0;
312
340
  try {
313
- while (turn < maxTurns) {
341
+ while (turn < effectiveMaxTurns) {
314
342
  options.signal?.throwIfAborted();
315
343
  turn++;
316
344
  if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
@@ -381,6 +409,7 @@ async function* agentLoop(messages, options) {
381
409
  let lastEventType = "";
382
410
  let toolcallDeltaChars = 0;
383
411
  let toolcallDeltaCount = 0;
412
+ let toolcallNoProgressCount = 0;
384
413
  let runawayDetected = null;
385
414
  let attemptText = "";
386
415
  let lastYieldEndTime = Date.now();
@@ -421,6 +450,21 @@ async function* agentLoop(messages, options) {
421
450
  diag("stream_call", { nonStreaming: useNonStreamingFallback });
422
451
  streamCallStart = Date.now();
423
452
  providerAttemptStartedAt = streamCallStart;
453
+ let liveApiKey = options.apiKey;
454
+ let liveAccountId = options.accountId;
455
+ let liveProjectId = options.projectId;
456
+ if (options.resolveCredentials) {
457
+ try {
458
+ const fresh = await options.resolveCredentials();
459
+ liveApiKey = fresh.apiKey;
460
+ if (fresh.accountId !== void 0) liveAccountId = fresh.accountId;
461
+ if (fresh.projectId !== void 0) liveProjectId = fresh.projectId;
462
+ } catch (credErr) {
463
+ diag("credential_refresh_failed", {
464
+ error: (credErr instanceof Error ? credErr.message : String(credErr)).slice(0, 200)
465
+ });
466
+ }
467
+ }
424
468
  const result = stream({
425
469
  provider: options.provider,
426
470
  model: options.model,
@@ -432,12 +476,12 @@ async function* agentLoop(messages, options) {
432
476
  maxTokens: options.maxTokens,
433
477
  temperature: options.temperature,
434
478
  thinking: options.thinking,
435
- apiKey: options.apiKey,
479
+ apiKey: liveApiKey,
436
480
  baseUrl: options.baseUrl,
437
481
  signal: streamController.signal,
438
- accountId: options.accountId,
482
+ accountId: liveAccountId,
439
483
  transportSessionId: options.transportSessionId,
440
- projectId: options.projectId,
484
+ projectId: liveProjectId,
441
485
  cacheRetention: options.cacheRetention,
442
486
  promptCacheKey: options.promptCacheKey,
443
487
  serviceTier: options.serviceTier,
@@ -535,11 +579,13 @@ async function* agentLoop(messages, options) {
535
579
  const chunkChars = event.argsJson?.length ?? 0;
536
580
  toolcallDeltaChars += chunkChars;
537
581
  toolcallDeltaCount++;
538
- if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallDeltaCount > MAX_TOOLCALL_DELTA_EVENTS)) {
582
+ toolcallNoProgressCount = chunkChars > 0 ? 0 : toolcallNoProgressCount + 1;
583
+ if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallNoProgressCount > MAX_TOOLCALL_NO_PROGRESS_EVENTS)) {
539
584
  runawayDetected = {
540
585
  kind: toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS ? "chars" : "events",
541
586
  chars: toolcallDeltaChars,
542
- events: toolcallDeltaCount
587
+ events: toolcallDeltaCount,
588
+ noProgressEvents: toolcallNoProgressCount
543
589
  };
544
590
  diag("runaway_toolcall_detected", {
545
591
  ...runawayDetected,
@@ -721,11 +767,15 @@ async function* agentLoop(messages, options) {
721
767
  turn--;
722
768
  continue;
723
769
  }
724
- const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
770
+ const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.noProgressEvents} consecutive tool-call delta events without argument progress`;
725
771
  yield {
726
772
  type: "error",
727
- error: new Error(
728
- `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
773
+ error: new EZCoderAIError(
774
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Your conversation is preserved.`,
775
+ {
776
+ source: "provider",
777
+ hint: "Retry once. If it keeps happening, switch models and continue the preserved conversation."
778
+ }
729
779
  )
730
780
  };
731
781
  break;
@@ -752,7 +802,15 @@ async function* agentLoop(messages, options) {
752
802
  role: "assistant",
753
803
  content: [{ type: "text", text: attemptText }]
754
804
  });
755
- messages.push({ role: "user", content: PARTIAL_CONTINUATION_PROMPT });
805
+ messages.push({
806
+ role: "user",
807
+ content: PARTIAL_CONTINUATION_PROMPT,
808
+ provenance: {
809
+ source: "runtime",
810
+ kind: "continuation",
811
+ visibility: "hidden"
812
+ }
813
+ });
756
814
  preservedChars = attemptText.length;
757
815
  }
758
816
  diag("retry", {
@@ -871,6 +929,9 @@ async function* agentLoop(messages, options) {
871
929
  useNonStreamingFallback = false;
872
930
  totalUsage.inputTokens += response.usage.inputTokens;
873
931
  totalUsage.outputTokens += response.usage.outputTokens;
932
+ if (response.usage.reasoningTokens) {
933
+ totalUsage.reasoningTokens = (totalUsage.reasoningTokens ?? 0) + response.usage.reasoningTokens;
934
+ }
874
935
  if (response.usage.cacheRead) {
875
936
  totalUsage.cacheRead = (totalUsage.cacheRead ?? 0) + response.usage.cacheRead;
876
937
  }
@@ -922,7 +983,15 @@ async function* agentLoop(messages, options) {
922
983
  model: options.model
923
984
  });
924
985
  yield { type: "truncated", reason: "max_tokens", continued: true };
925
- messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
986
+ messages.push({
987
+ role: "user",
988
+ content: MAX_TOKENS_CONTINUATION_PROMPT,
989
+ provenance: {
990
+ source: "runtime",
991
+ kind: "continuation",
992
+ visibility: "hidden"
993
+ }
994
+ });
926
995
  continue;
927
996
  }
928
997
  yield { type: "truncated", reason: "max_tokens", continued: false };
@@ -998,6 +1067,7 @@ async function* agentLoop(messages, options) {
998
1067
  );
999
1068
  const executionResult = hasSequentialToolCall ? yield* executeToolCallsMixed(toolCalls, toolResults, executionOptions) : yield* executeToolCallsParallel(toolCalls, toolResults, executionOptions);
1000
1069
  messages.push({ role: "tool", content: executionResult.toolResults });
1070
+ yield { type: "checkpoint", turn };
1001
1071
  const toolsAborted = executionResult.aborted;
1002
1072
  if (fatalToolArgumentError) {
1003
1073
  if (fatalToolArgumentRecoverable && !toolArgumentAutoContinueUsed) {
@@ -1029,8 +1099,51 @@ async function* agentLoop(messages, options) {
1029
1099
  }
1030
1100
  }
1031
1101
  }
1032
- if (turn >= maxTurns) {
1033
- hitMaxTurns = true;
1102
+ if (turn >= effectiveMaxTurns) {
1103
+ let extended = false;
1104
+ if (options.onTurnBudgetExhausted && turnExtensions < maxTurnExtensions) {
1105
+ const extension = turnExtensions + 1;
1106
+ let granted = false;
1107
+ try {
1108
+ granted = await options.onTurnBudgetExhausted({
1109
+ turn,
1110
+ maxTurns: effectiveMaxTurns,
1111
+ extension
1112
+ });
1113
+ } catch {
1114
+ granted = false;
1115
+ }
1116
+ if (granted) {
1117
+ turnExtensions = extension;
1118
+ effectiveMaxTurns += maxTurns;
1119
+ extended = true;
1120
+ diag("turn_budget_extended", {
1121
+ turn,
1122
+ grantedTurns: effectiveMaxTurns,
1123
+ extension,
1124
+ provider: options.provider,
1125
+ model: options.model
1126
+ });
1127
+ yield {
1128
+ type: "turn_budget_extended",
1129
+ turn,
1130
+ grantedTurns: effectiveMaxTurns,
1131
+ extension
1132
+ };
1133
+ messages.push({
1134
+ role: "user",
1135
+ content: turnBudgetContinuationPrompt(),
1136
+ provenance: {
1137
+ source: "runtime",
1138
+ kind: "continuation",
1139
+ visibility: "hidden"
1140
+ }
1141
+ });
1142
+ }
1143
+ }
1144
+ if (!extended) {
1145
+ hitMaxTurns = true;
1146
+ }
1034
1147
  }
1035
1148
  }
1036
1149
  } finally {
@@ -1046,14 +1159,15 @@ async function* agentLoop(messages, options) {
1046
1159
  if (hitMaxTurns) {
1047
1160
  diag("max_turns_reached", {
1048
1161
  turn,
1049
- maxTurns,
1162
+ maxTurns: effectiveMaxTurns,
1163
+ extensions: turnExtensions,
1050
1164
  provider: options.provider,
1051
1165
  model: options.model
1052
1166
  });
1053
1167
  yield {
1054
1168
  type: "max_turns",
1055
1169
  totalTurns: turn,
1056
- maxTurns
1170
+ maxTurns: effectiveMaxTurns
1057
1171
  };
1058
1172
  }
1059
1173
  yield {
@@ -1089,7 +1203,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1089
1203
  try {
1090
1204
  const parsed = tool.parameters.parse(toolCall.args);
1091
1205
  const callerSignal = options.signal;
1092
- const toolTimeout = AbortSignal.timeout(3e5);
1206
+ const toolTimeout = AbortSignal.timeout(tool.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS);
1093
1207
  const ctx = {
1094
1208
  signal: callerSignal ? AbortSignal.any([callerSignal, toolTimeout]) : toolTimeout,
1095
1209
  toolCallId: toolCall.id,
@@ -1302,9 +1416,9 @@ function capToolResults(toolResults, maxToolResultChars) {
1302
1416
  const originalChars = toolResult.content.length;
1303
1417
  const headChars = Math.floor(max * 0.7);
1304
1418
  const tailChars = max - headChars;
1305
- const head = toolResult.content.slice(0, headChars);
1306
- const tail = toolResult.content.slice(-tailChars);
1307
- const omitted = originalChars - headChars - tailChars;
1419
+ const head = sliceHead(toolResult.content, headChars);
1420
+ const tail = sliceTail(toolResult.content, tailChars);
1421
+ const omitted = originalChars - head.length - tail.length;
1308
1422
  toolResult.content = head + `
1309
1423
 
1310
1424
  [... ${omitted} characters omitted ...]
@@ -1339,11 +1453,11 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1339
1453
  const headChars = Math.floor(fairShare * 0.7);
1340
1454
  const tailChars = fairShare - headChars;
1341
1455
  const omitted = originalChars - fairShare;
1342
- toolResult.content = toolResult.content.slice(0, headChars) + `
1456
+ toolResult.content = sliceHead(toolResult.content, headChars) + `
1343
1457
 
1344
1458
  [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1345
1459
 
1346
- ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1460
+ ` + sliceTail(toolResult.content, tailChars);
1347
1461
  toolResult.capped = {
1348
1462
  originalChars: toolResult.capped?.originalChars ?? originalChars,
1349
1463
  keptChars: toolResult.content.length,
@@ -1362,12 +1476,14 @@ function truncateToolResultText(text, maxChars) {
1362
1476
  if (text.length <= maxChars) return text;
1363
1477
  const tailChars = Math.min(Math.floor(maxChars * 0.3), 2e4);
1364
1478
  const headChars = Math.max(maxChars - tailChars, 0);
1365
- const omitted = text.length - headChars - tailChars;
1366
- return `${text.slice(0, headChars)}
1479
+ const head = sliceHead(text, headChars);
1480
+ const tail = sliceTail(text, tailChars);
1481
+ const omitted = text.length - head.length - tail.length;
1482
+ return `${head}
1367
1483
 
1368
1484
  [... ${omitted} characters omitted after context overflow ...]
1369
1485
 
1370
- ${text.slice(-tailChars)}`;
1486
+ ${tail}`;
1371
1487
  }
1372
1488
  function truncateOversizedToolResults(messages, maxChars) {
1373
1489
  if (maxChars <= 0) return false;