@prestyj/agent 5.9.0 → 5.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
20
20
  * batch runs in source order so stateful mutations cannot race each other.
21
21
  */
22
22
  executionMode?: ToolExecutionMode;
23
+ /**
24
+ * Overrides the loop's default per-tool timeout. A tool that owns a longer
25
+ * internal budget than the default must declare it here, or the loop cancels
26
+ * it first and the tool's own timeout — with its specific, actionable error
27
+ * message — becomes unreachable.
28
+ */
29
+ timeoutMs?: number;
23
30
  execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
24
31
  }
25
32
  interface AgentTextDeltaEvent {
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
70
77
  usage: Usage;
71
78
  timing: AgentTurnTiming;
72
79
  }
80
+ /**
81
+ * A safe point between steps: the assistant message and every tool result for
82
+ * this turn are now in the message array, and no provider call is in flight.
83
+ *
84
+ * Hosts that persist a transcript flush here. Without it a crash mid-run loses
85
+ * the WHOLE turn — including tool results whose side effects already landed on
86
+ * disk — because the only flush happens after the loop returns.
87
+ *
88
+ * Yielded immediately after tool results are appended, so it pairs with
89
+ * `turn_end` (which covers the assistant half) to cover every message.
90
+ */
91
+ interface AgentCheckpointEvent {
92
+ type: "checkpoint";
93
+ turn: number;
94
+ }
73
95
  interface AgentDoneEvent {
74
96
  type: "agent_done";
75
97
  totalTurns: number;
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
87
109
  totalTurns: number;
88
110
  maxTurns: number;
89
111
  }
112
+ /**
113
+ * Emitted when the loop was about to stop on an exhausted turn budget but the
114
+ * host granted an extension instead. The effective budget is raised and the
115
+ * loop continues with a continuation prompt, so this is NOT terminal — unlike
116
+ * `max_turns`, which still fires if the extended budget is also spent.
117
+ */
118
+ interface AgentTurnBudgetExtendedEvent {
119
+ type: "turn_budget_extended";
120
+ /** Turn number at which the budget was exhausted. */
121
+ turn: number;
122
+ /** New effective `maxTurns` after the extension. */
123
+ grantedTurns: number;
124
+ /** 1-based extension count for this run. */
125
+ extension: number;
126
+ }
90
127
  /**
91
128
  * Warning signal emitted when a turn ended on a non-clean stop reason —
92
129
  * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
148
185
  type: "follow_up_message";
149
186
  content: Message["content"];
150
187
  }
151
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
188
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
152
189
  interface TransformContextOptions {
153
190
  /** Force a transform after the provider reports context overflow. */
154
191
  force?: boolean;
@@ -168,6 +205,12 @@ interface AgentOptions {
168
205
  /** Control whether tools may/must be called, or select a named tool when supported. */
169
206
  toolChoice?: StreamOptions["toolChoice"];
170
207
  maxTurns?: number;
208
+ /**
209
+ * How many times `onTurnBudgetExhausted` may grant extra turns in one run.
210
+ * Each grant raises the effective budget by the original `maxTurns`.
211
+ * Default: 2. Set 0 to disable extensions entirely.
212
+ */
213
+ maxTurnExtensions?: number;
171
214
  maxTokens?: number;
172
215
  temperature?: number;
173
216
  thinking?: StreamOptions["thinking"];
@@ -234,6 +277,18 @@ interface AgentOptions {
234
277
  * on read.
235
278
  */
236
279
  getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
280
+ /**
281
+ * Consulted when a tool-running turn exhausts the turn budget mid-task,
282
+ * before the loop emits the terminal `max_turns` event. Return true to grant
283
+ * another `maxTurns` worth of turns; false (the default when unset) keeps
284
+ * today's hard cut-off. Hosts should only grant on evidence of progress —
285
+ * extending a spinning agent just buys it more tokens to spin with.
286
+ */
287
+ onTurnBudgetExhausted?: (ctx: {
288
+ turn: number;
289
+ maxTurns: number;
290
+ extension: number;
291
+ }) => Promise<boolean> | boolean;
237
292
  }
238
293
  interface AgentResult {
239
294
  message: AssistantMessage;
@@ -324,4 +379,13 @@ declare function isBillingError(err: unknown): boolean;
324
379
  declare function isUsageLimitError(err: unknown): boolean;
325
380
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
326
381
 
327
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
382
+ /**
383
+ * A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
384
+ * minutes prefilling a large prompt before it emits its first token. The
385
+ * first-event watchdog that protects hosted streams turns that into an abort
386
+ * → retry → cold-prefill loop that never converges, so it is disabled for
387
+ * loopback backends.
388
+ */
389
+ declare function isLocalBackendUrl(baseUrl?: string): boolean;
390
+
391
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
package/dist/index.d.ts CHANGED
@@ -20,6 +20,13 @@ interface AgentTool<T extends z.ZodType = z.ZodType> extends Tool {
20
20
  * batch runs in source order so stateful mutations cannot race each other.
21
21
  */
22
22
  executionMode?: ToolExecutionMode;
23
+ /**
24
+ * Overrides the loop's default per-tool timeout. A tool that owns a longer
25
+ * internal budget than the default must declare it here, or the loop cancels
26
+ * it first and the tool's own timeout — with its specific, actionable error
27
+ * message — becomes unreachable.
28
+ */
29
+ timeoutMs?: number;
23
30
  execute: (args: z.infer<T>, context: ToolContext) => ToolExecuteResult | Promise<ToolExecuteResult>;
24
31
  }
25
32
  interface AgentTextDeltaEvent {
@@ -70,6 +77,21 @@ interface AgentTurnEndEvent {
70
77
  usage: Usage;
71
78
  timing: AgentTurnTiming;
72
79
  }
80
+ /**
81
+ * A safe point between steps: the assistant message and every tool result for
82
+ * this turn are now in the message array, and no provider call is in flight.
83
+ *
84
+ * Hosts that persist a transcript flush here. Without it a crash mid-run loses
85
+ * the WHOLE turn — including tool results whose side effects already landed on
86
+ * disk — because the only flush happens after the loop returns.
87
+ *
88
+ * Yielded immediately after tool results are appended, so it pairs with
89
+ * `turn_end` (which covers the assistant half) to cover every message.
90
+ */
91
+ interface AgentCheckpointEvent {
92
+ type: "checkpoint";
93
+ turn: number;
94
+ }
73
95
  interface AgentDoneEvent {
74
96
  type: "agent_done";
75
97
  totalTurns: number;
@@ -87,6 +109,21 @@ interface AgentMaxTurnsEvent {
87
109
  totalTurns: number;
88
110
  maxTurns: number;
89
111
  }
112
+ /**
113
+ * Emitted when the loop was about to stop on an exhausted turn budget but the
114
+ * host granted an extension instead. The effective budget is raised and the
115
+ * loop continues with a continuation prompt, so this is NOT terminal — unlike
116
+ * `max_turns`, which still fires if the extended budget is also spent.
117
+ */
118
+ interface AgentTurnBudgetExtendedEvent {
119
+ type: "turn_budget_extended";
120
+ /** Turn number at which the budget was exhausted. */
121
+ turn: number;
122
+ /** New effective `maxTurns` after the extension. */
123
+ grantedTurns: number;
124
+ /** 1-based extension count for this run. */
125
+ extension: number;
126
+ }
90
127
  /**
91
128
  * Warning signal emitted when a turn ended on a non-clean stop reason —
92
129
  * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
@@ -148,7 +185,7 @@ interface AgentFollowUpMessageEvent {
148
185
  type: "follow_up_message";
149
186
  content: Message["content"];
150
187
  }
151
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
188
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentCheckpointEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTurnBudgetExtendedEvent | AgentTruncatedEvent | AgentErrorEvent;
152
189
  interface TransformContextOptions {
153
190
  /** Force a transform after the provider reports context overflow. */
154
191
  force?: boolean;
@@ -168,6 +205,12 @@ interface AgentOptions {
168
205
  /** Control whether tools may/must be called, or select a named tool when supported. */
169
206
  toolChoice?: StreamOptions["toolChoice"];
170
207
  maxTurns?: number;
208
+ /**
209
+ * How many times `onTurnBudgetExhausted` may grant extra turns in one run.
210
+ * Each grant raises the effective budget by the original `maxTurns`.
211
+ * Default: 2. Set 0 to disable extensions entirely.
212
+ */
213
+ maxTurnExtensions?: number;
171
214
  maxTokens?: number;
172
215
  temperature?: number;
173
216
  thinking?: StreamOptions["thinking"];
@@ -234,6 +277,18 @@ interface AgentOptions {
234
277
  * on read.
235
278
  */
236
279
  getFollowUpMessages?: () => Promise<Message[] | null> | Message[] | null;
280
+ /**
281
+ * Consulted when a tool-running turn exhausts the turn budget mid-task,
282
+ * before the loop emits the terminal `max_turns` event. Return true to grant
283
+ * another `maxTurns` worth of turns; false (the default when unset) keeps
284
+ * today's hard cut-off. Hosts should only grant on evidence of progress —
285
+ * extending a spinning agent just buys it more tokens to spin with.
286
+ */
287
+ onTurnBudgetExhausted?: (ctx: {
288
+ turn: number;
289
+ maxTurns: number;
290
+ extension: number;
291
+ }) => Promise<boolean> | boolean;
237
292
  }
238
293
  interface AgentResult {
239
294
  message: AssistantMessage;
@@ -324,4 +379,13 @@ declare function isBillingError(err: unknown): boolean;
324
379
  declare function isUsageLimitError(err: unknown): boolean;
325
380
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
326
381
 
327
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
382
+ /**
383
+ * A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
384
+ * minutes prefilling a large prompt before it emits its first token. The
385
+ * first-event watchdog that protects hosted streams turns that into an abort
386
+ * → retry → cold-prefill loop that never converges, so it is disabled for
387
+ * loopback backends.
388
+ */
389
+ declare function isLocalBackendUrl(baseUrl?: string): boolean;
390
+
391
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
package/dist/index.js CHANGED
@@ -8,9 +8,36 @@ import {
8
8
  EventStream,
9
9
  EZCoderAIError,
10
10
  isHardBillingMessage,
11
- redactValue
11
+ redactValue,
12
+ sliceHead,
13
+ sliceTail
12
14
  } from "@prestyj/ai";
15
+
16
+ // src/local-backend.ts
17
+ function isLocalBackendUrl(baseUrl) {
18
+ if (!baseUrl) return false;
19
+ let host;
20
+ try {
21
+ host = new URL(baseUrl).hostname.toLowerCase();
22
+ } catch {
23
+ return false;
24
+ }
25
+ if (host === "[::1]" || host === "::1") return true;
26
+ if (host === "localhost" || host.endsWith(".localhost")) return true;
27
+ if (host === "0.0.0.0") return true;
28
+ if (host.endsWith(".local")) return true;
29
+ const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
30
+ if (ipv4) {
31
+ const octets = ipv4.slice(1).map(Number);
32
+ if (octets.some((n) => n > 255)) return false;
33
+ return octets[0] === 127;
34
+ }
35
+ return false;
36
+ }
37
+
38
+ // src/agent-loop.ts
13
39
  var DEFAULT_MAX_TURNS = 300;
40
+ var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
14
41
  var _diagFn = null;
15
42
  function setStreamDiagnostic(fn) {
16
43
  _diagFn = fn;
@@ -213,6 +240,9 @@ function abortablePromise(promise, signal) {
213
240
  promise.then(resolveOnce, rejectOnce);
214
241
  });
215
242
  }
243
+ function turnBudgetContinuationPrompt() {
244
+ return "[You reached the turn limit for this segment but the work is not finished, so you have been granted more turns. Before continuing, state in one or two sentences what is already done and what remains, then keep going from there \u2014 do not restart work that is already complete.]";
245
+ }
216
246
  function abortableSleep(ms, signal) {
217
247
  if (signal?.aborted) return Promise.reject(createAbortError());
218
248
  return new Promise((resolve, reject) => {
@@ -234,6 +264,9 @@ function closeIterator(iterator) {
234
264
  }
235
265
  async function* agentLoop(messages, options) {
236
266
  const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
267
+ let effectiveMaxTurns = maxTurns;
268
+ const maxTurnExtensions = Math.max(0, options.maxTurnExtensions ?? 2);
269
+ let turnExtensions = 0;
237
270
  const maxContinuations = options.maxContinuations ?? 5;
238
271
  let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
239
272
  const totalUsage = { inputTokens: 0, outputTokens: 0 };
@@ -276,16 +309,17 @@ async function* agentLoop(messages, options) {
276
309
  const STREAM_THINKING_IDLE_TIMEOUT_MS = 3e5;
277
310
  const STREAM_THINKING_HARD_TIMEOUT_MS = 6e5;
278
311
  const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
279
- const isSakana = options.provider === "sakana";
280
- const firstEventTimeoutMs = isSakana ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
281
- const initialHardTimeoutMs = isSakana ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
312
+ const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
313
+ const localBackend = isLocalBackendUrl(options.baseUrl);
314
+ const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
315
+ const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
282
316
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
283
- const MAX_TOOLCALL_DELTA_EVENTS = 2e4;
317
+ const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
284
318
  let logicalTurnStartedAt = 0;
285
319
  let firstProviderEventAt;
286
320
  let providerDurationMs = 0;
287
321
  try {
288
- while (turn < maxTurns) {
322
+ while (turn < effectiveMaxTurns) {
289
323
  options.signal?.throwIfAborted();
290
324
  turn++;
291
325
  if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
@@ -306,7 +340,11 @@ async function* agentLoop(messages, options) {
306
340
  messages: messages.length,
307
341
  chars: msgChars,
308
342
  provider: options.provider,
309
- model: options.model
343
+ model: options.model,
344
+ thinking: options.thinking ?? "off",
345
+ firstEventTimeoutMs,
346
+ initialHardTimeoutMs,
347
+ localBackend
310
348
  });
311
349
  }
312
350
  if (firstTurn && options.getSteeringMessages) {
@@ -352,6 +390,7 @@ async function* agentLoop(messages, options) {
352
390
  let lastEventType = "";
353
391
  let toolcallDeltaChars = 0;
354
392
  let toolcallDeltaCount = 0;
393
+ let toolcallNoProgressCount = 0;
355
394
  let runawayDetected = null;
356
395
  let attemptText = "";
357
396
  let lastYieldEndTime = Date.now();
@@ -364,6 +403,7 @@ async function* agentLoop(messages, options) {
364
403
  if (useNonStreamingFallback) return;
365
404
  if (idleTimer) clearTimeout(idleTimer);
366
405
  const timeoutMs = hasReceivedEvent ? STREAM_IDLE_TIMEOUT_MS : hasReceivedThinking ? STREAM_THINKING_IDLE_TIMEOUT_MS : firstEventTimeoutMs;
406
+ if (!Number.isFinite(timeoutMs)) return;
367
407
  idleTimer = setTimeout(() => {
368
408
  diag("idle_timeout_fired", {
369
409
  events: streamEventCount,
@@ -505,11 +545,13 @@ async function* agentLoop(messages, options) {
505
545
  const chunkChars = event.argsJson?.length ?? 0;
506
546
  toolcallDeltaChars += chunkChars;
507
547
  toolcallDeltaCount++;
508
- if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallDeltaCount > MAX_TOOLCALL_DELTA_EVENTS)) {
548
+ toolcallNoProgressCount = chunkChars > 0 ? 0 : toolcallNoProgressCount + 1;
549
+ if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallNoProgressCount > MAX_TOOLCALL_NO_PROGRESS_EVENTS)) {
509
550
  runawayDetected = {
510
551
  kind: toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS ? "chars" : "events",
511
552
  chars: toolcallDeltaChars,
512
- events: toolcallDeltaCount
553
+ events: toolcallDeltaCount,
554
+ noProgressEvents: toolcallNoProgressCount
513
555
  };
514
556
  diag("runaway_toolcall_detected", {
515
557
  ...runawayDetected,
@@ -691,11 +733,15 @@ async function* agentLoop(messages, options) {
691
733
  turn--;
692
734
  continue;
693
735
  }
694
- const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
736
+ const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.noProgressEvents} consecutive tool-call delta events without argument progress`;
695
737
  yield {
696
738
  type: "error",
697
- error: new Error(
698
- `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
739
+ error: new EZCoderAIError(
740
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Your conversation is preserved.`,
741
+ {
742
+ source: "provider",
743
+ hint: "Retry once. If it keeps happening, switch models and continue the preserved conversation."
744
+ }
699
745
  )
700
746
  };
701
747
  break;
@@ -722,7 +768,15 @@ async function* agentLoop(messages, options) {
722
768
  role: "assistant",
723
769
  content: [{ type: "text", text: attemptText }]
724
770
  });
725
- messages.push({ role: "user", content: PARTIAL_CONTINUATION_PROMPT });
771
+ messages.push({
772
+ role: "user",
773
+ content: PARTIAL_CONTINUATION_PROMPT,
774
+ provenance: {
775
+ source: "runtime",
776
+ kind: "continuation",
777
+ visibility: "hidden"
778
+ }
779
+ });
726
780
  preservedChars = attemptText.length;
727
781
  }
728
782
  diag("retry", {
@@ -748,15 +802,29 @@ async function* agentLoop(messages, options) {
748
802
  continue;
749
803
  }
750
804
  if (transportFailure) {
805
+ const cause = malformed ? "malformed_stream" : socketDrop ? "socket_drop" : "stream_stall";
751
806
  diag("stall_exhausted", {
752
807
  stallRetries: MAX_STALL_RETRIES,
753
808
  provider: options.provider,
754
- model: options.model
809
+ model: options.model,
810
+ cause,
811
+ nonStreaming: useNonStreamingFallback,
812
+ events: streamEventCount,
813
+ eventTypes: eventTypeCounts,
814
+ lastEventType,
815
+ sinceLastEventMs: Date.now() - lastEventTime,
816
+ attemptDurationMs: Date.now() - streamCallStart,
817
+ maxConsumerLagMs
755
818
  });
756
819
  yield {
757
820
  type: "error",
758
- error: new Error(
759
- `The API provider's stream stalled ${MAX_STALL_RETRIES} times \u2014 the provider may be experiencing capacity issues. Your conversation is preserved. Send another message to retry.`
821
+ error: new EZCoderAIError(
822
+ `The connection to the API provider stopped responding after ${MAX_STALL_RETRIES} automatic retries. Your conversation is preserved.`,
823
+ {
824
+ source: "network",
825
+ hint: "Retry once. If it keeps happening on this device, disable any VPN or proxy and allow EZ Coder through firewall or antivirus web protection.",
826
+ cause: err
827
+ }
760
828
  )
761
829
  };
762
830
  break;
@@ -827,6 +895,9 @@ async function* agentLoop(messages, options) {
827
895
  useNonStreamingFallback = false;
828
896
  totalUsage.inputTokens += response.usage.inputTokens;
829
897
  totalUsage.outputTokens += response.usage.outputTokens;
898
+ if (response.usage.reasoningTokens) {
899
+ totalUsage.reasoningTokens = (totalUsage.reasoningTokens ?? 0) + response.usage.reasoningTokens;
900
+ }
830
901
  if (response.usage.cacheRead) {
831
902
  totalUsage.cacheRead = (totalUsage.cacheRead ?? 0) + response.usage.cacheRead;
832
903
  }
@@ -878,7 +949,15 @@ async function* agentLoop(messages, options) {
878
949
  model: options.model
879
950
  });
880
951
  yield { type: "truncated", reason: "max_tokens", continued: true };
881
- messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
952
+ messages.push({
953
+ role: "user",
954
+ content: MAX_TOKENS_CONTINUATION_PROMPT,
955
+ provenance: {
956
+ source: "runtime",
957
+ kind: "continuation",
958
+ visibility: "hidden"
959
+ }
960
+ });
882
961
  continue;
883
962
  }
884
963
  yield { type: "truncated", reason: "max_tokens", continued: false };
@@ -954,6 +1033,7 @@ async function* agentLoop(messages, options) {
954
1033
  );
955
1034
  const executionResult = hasSequentialToolCall ? yield* executeToolCallsMixed(toolCalls, toolResults, executionOptions) : yield* executeToolCallsParallel(toolCalls, toolResults, executionOptions);
956
1035
  messages.push({ role: "tool", content: executionResult.toolResults });
1036
+ yield { type: "checkpoint", turn };
957
1037
  const toolsAborted = executionResult.aborted;
958
1038
  if (fatalToolArgumentError) {
959
1039
  if (fatalToolArgumentRecoverable && !toolArgumentAutoContinueUsed) {
@@ -985,8 +1065,51 @@ async function* agentLoop(messages, options) {
985
1065
  }
986
1066
  }
987
1067
  }
988
- if (turn >= maxTurns) {
989
- hitMaxTurns = true;
1068
+ if (turn >= effectiveMaxTurns) {
1069
+ let extended = false;
1070
+ if (options.onTurnBudgetExhausted && turnExtensions < maxTurnExtensions) {
1071
+ const extension = turnExtensions + 1;
1072
+ let granted = false;
1073
+ try {
1074
+ granted = await options.onTurnBudgetExhausted({
1075
+ turn,
1076
+ maxTurns: effectiveMaxTurns,
1077
+ extension
1078
+ });
1079
+ } catch {
1080
+ granted = false;
1081
+ }
1082
+ if (granted) {
1083
+ turnExtensions = extension;
1084
+ effectiveMaxTurns += maxTurns;
1085
+ extended = true;
1086
+ diag("turn_budget_extended", {
1087
+ turn,
1088
+ grantedTurns: effectiveMaxTurns,
1089
+ extension,
1090
+ provider: options.provider,
1091
+ model: options.model
1092
+ });
1093
+ yield {
1094
+ type: "turn_budget_extended",
1095
+ turn,
1096
+ grantedTurns: effectiveMaxTurns,
1097
+ extension
1098
+ };
1099
+ messages.push({
1100
+ role: "user",
1101
+ content: turnBudgetContinuationPrompt(),
1102
+ provenance: {
1103
+ source: "runtime",
1104
+ kind: "continuation",
1105
+ visibility: "hidden"
1106
+ }
1107
+ });
1108
+ }
1109
+ }
1110
+ if (!extended) {
1111
+ hitMaxTurns = true;
1112
+ }
990
1113
  }
991
1114
  }
992
1115
  } finally {
@@ -1002,14 +1125,15 @@ async function* agentLoop(messages, options) {
1002
1125
  if (hitMaxTurns) {
1003
1126
  diag("max_turns_reached", {
1004
1127
  turn,
1005
- maxTurns,
1128
+ maxTurns: effectiveMaxTurns,
1129
+ extensions: turnExtensions,
1006
1130
  provider: options.provider,
1007
1131
  model: options.model
1008
1132
  });
1009
1133
  yield {
1010
1134
  type: "max_turns",
1011
1135
  totalTurns: turn,
1012
- maxTurns
1136
+ maxTurns: effectiveMaxTurns
1013
1137
  };
1014
1138
  }
1015
1139
  yield {
@@ -1045,7 +1169,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1045
1169
  try {
1046
1170
  const parsed = tool.parameters.parse(toolCall.args);
1047
1171
  const callerSignal = options.signal;
1048
- const toolTimeout = AbortSignal.timeout(3e5);
1172
+ const toolTimeout = AbortSignal.timeout(tool.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS);
1049
1173
  const ctx = {
1050
1174
  signal: callerSignal ? AbortSignal.any([callerSignal, toolTimeout]) : toolTimeout,
1051
1175
  toolCallId: toolCall.id,
@@ -1258,9 +1382,9 @@ function capToolResults(toolResults, maxToolResultChars) {
1258
1382
  const originalChars = toolResult.content.length;
1259
1383
  const headChars = Math.floor(max * 0.7);
1260
1384
  const tailChars = max - headChars;
1261
- const head = toolResult.content.slice(0, headChars);
1262
- const tail = toolResult.content.slice(-tailChars);
1263
- const omitted = originalChars - headChars - tailChars;
1385
+ const head = sliceHead(toolResult.content, headChars);
1386
+ const tail = sliceTail(toolResult.content, tailChars);
1387
+ const omitted = originalChars - head.length - tail.length;
1264
1388
  toolResult.content = head + `
1265
1389
 
1266
1390
  [... ${omitted} characters omitted ...]
@@ -1295,11 +1419,11 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1295
1419
  const headChars = Math.floor(fairShare * 0.7);
1296
1420
  const tailChars = fairShare - headChars;
1297
1421
  const omitted = originalChars - fairShare;
1298
- toolResult.content = toolResult.content.slice(0, headChars) + `
1422
+ toolResult.content = sliceHead(toolResult.content, headChars) + `
1299
1423
 
1300
1424
  [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1301
1425
 
1302
- ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1426
+ ` + sliceTail(toolResult.content, tailChars);
1303
1427
  toolResult.capped = {
1304
1428
  originalChars: toolResult.capped?.originalChars ?? originalChars,
1305
1429
  keptChars: toolResult.content.length,
@@ -1318,12 +1442,14 @@ function truncateToolResultText(text, maxChars) {
1318
1442
  if (text.length <= maxChars) return text;
1319
1443
  const tailChars = Math.min(Math.floor(maxChars * 0.3), 2e4);
1320
1444
  const headChars = Math.max(maxChars - tailChars, 0);
1321
- const omitted = text.length - headChars - tailChars;
1322
- return `${text.slice(0, headChars)}
1445
+ const head = sliceHead(text, headChars);
1446
+ const tail = sliceTail(text, tailChars);
1447
+ const omitted = text.length - head.length - tail.length;
1448
+ return `${head}
1323
1449
 
1324
1450
  [... ${omitted} characters omitted after context overflow ...]
1325
1451
 
1326
- ${text.slice(-tailChars)}`;
1452
+ ${tail}`;
1327
1453
  }
1328
1454
  function truncateOversizedToolResults(messages, maxChars) {
1329
1455
  if (maxChars <= 0) return false;
@@ -1575,6 +1701,7 @@ export {
1575
1701
  isAbortError,
1576
1702
  isBillingError,
1577
1703
  isContextOverflow,
1704
+ isLocalBackendUrl,
1578
1705
  isUsageLimitError,
1579
1706
  setStreamDiagnostic
1580
1707
  };