@prestyj/agent 5.8.0 → 5.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -26,6 +26,7 @@ __export(index_exports, {
26
26
  isAbortError: () => isAbortError,
27
27
  isBillingError: () => isBillingError,
28
28
  isContextOverflow: () => isContextOverflow,
29
+ isLocalBackendUrl: () => isLocalBackendUrl,
29
30
  isUsageLimitError: () => isUsageLimitError,
30
31
  setStreamDiagnostic: () => setStreamDiagnostic
31
32
  });
@@ -37,6 +38,30 @@ var import_ai2 = require("@prestyj/ai");
37
38
  // src/agent-loop.ts
38
39
  var import_zod = require("zod");
39
40
  var import_ai = require("@prestyj/ai");
41
+
42
+ // src/local-backend.ts
43
+ function isLocalBackendUrl(baseUrl) {
44
+ if (!baseUrl) return false;
45
+ let host;
46
+ try {
47
+ host = new URL(baseUrl).hostname.toLowerCase();
48
+ } catch {
49
+ return false;
50
+ }
51
+ if (host === "[::1]" || host === "::1") return true;
52
+ if (host === "localhost" || host.endsWith(".localhost")) return true;
53
+ if (host === "0.0.0.0") return true;
54
+ if (host.endsWith(".local")) return true;
55
+ const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
56
+ if (ipv4) {
57
+ const octets = ipv4.slice(1).map(Number);
58
+ if (octets.some((n) => n > 255)) return false;
59
+ return octets[0] === 127;
60
+ }
61
+ return false;
62
+ }
63
+
64
+ // src/agent-loop.ts
40
65
  var DEFAULT_MAX_TURNS = 300;
41
66
  var _diagFn = null;
42
67
  function setStreamDiagnostic(fn) {
@@ -215,7 +240,11 @@ function createAbortError() {
215
240
  }
216
241
  function abortablePromise(promise, signal) {
217
242
  if (!signal) return promise;
218
- if (signal.aborted) return Promise.reject(createAbortError());
243
+ if (signal.aborted) {
244
+ promise.catch(() => {
245
+ });
246
+ return Promise.reject(createAbortError());
247
+ }
219
248
  return new Promise((resolve, reject) => {
220
249
  let settled = false;
221
250
  const cleanup = () => signal.removeEventListener("abort", onAbort);
@@ -271,6 +300,7 @@ async function* agentLoop(messages, options) {
271
300
  let overloadRetries = 0;
272
301
  let emptyResponseRetries = 0;
273
302
  let stallRetries = 0;
303
+ let runawayToolcallRetries = 0;
274
304
  let overflowCompactionAttempts = 0;
275
305
  let toolResultTruncationAttempted = false;
276
306
  const invalidToolArgumentCounts = /* @__PURE__ */ new Map();
@@ -279,11 +309,16 @@ async function* agentLoop(messages, options) {
279
309
  const MAX_OVERLOAD_RETRIES = 10;
280
310
  const MAX_EMPTY_RESPONSE_RETRIES = 2;
281
311
  const MAX_STALL_RETRIES = 10;
312
+ const MAX_RUNAWAY_TOOLCALL_RETRIES = 2;
313
+ const RUNAWAY_TOOLCALL_RETRY_DELAY_MS = 1e3;
282
314
  const MAX_OVERFLOW_COMPACTIONS = 2;
283
315
  const STALL_RETRIES_BEFORE_NON_STREAMING = 2;
284
316
  const STALL_DELAY_MS = 1e3;
285
317
  const MIN_PARTIAL_PRESERVE_CHARS = 200;
286
318
  const PARTIAL_CONTINUATION_PROMPT = "[Your previous response was cut off by a connection failure. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
319
+ const MAX_OUTPUT_CONTINUATIONS = 2;
320
+ const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
321
+ let maxTokensContinuations = 0;
287
322
  const OVERLOAD_BASE_DELAY_MS = 2e3;
288
323
  const OVERLOAD_MAX_DELAY_MS = 3e4;
289
324
  const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
@@ -293,9 +328,10 @@ async function* agentLoop(messages, options) {
293
328
  const STREAM_THINKING_IDLE_TIMEOUT_MS = 3e5;
294
329
  const STREAM_THINKING_HARD_TIMEOUT_MS = 6e5;
295
330
  const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
296
- const isSakana = options.provider === "sakana";
297
- const firstEventTimeoutMs = isSakana ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
298
- const initialHardTimeoutMs = isSakana ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
331
+ const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
332
+ const localBackend = isLocalBackendUrl(options.baseUrl);
333
+ const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
334
+ const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
299
335
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
300
336
  const MAX_TOOLCALL_DELTA_EVENTS = 2e4;
301
337
  let logicalTurnStartedAt = 0;
@@ -323,7 +359,11 @@ async function* agentLoop(messages, options) {
323
359
  messages: messages.length,
324
360
  chars: msgChars,
325
361
  provider: options.provider,
326
- model: options.model
362
+ model: options.model,
363
+ thinking: options.thinking ?? "off",
364
+ firstEventTimeoutMs,
365
+ initialHardTimeoutMs,
366
+ localBackend
327
367
  });
328
368
  }
329
369
  if (firstTurn && options.getSteeringMessages) {
@@ -381,6 +421,7 @@ async function* agentLoop(messages, options) {
381
421
  if (useNonStreamingFallback) return;
382
422
  if (idleTimer) clearTimeout(idleTimer);
383
423
  const timeoutMs = hasReceivedEvent ? STREAM_IDLE_TIMEOUT_MS : hasReceivedThinking ? STREAM_THINKING_IDLE_TIMEOUT_MS : firstEventTimeoutMs;
424
+ if (!Number.isFinite(timeoutMs)) return;
384
425
  idleTimer = setTimeout(() => {
385
426
  diag("idle_timeout_fired", {
386
427
  events: streamEventCount,
@@ -686,11 +727,33 @@ async function* agentLoop(messages, options) {
686
727
  provider: options.provider,
687
728
  model: options.model
688
729
  });
730
+ if (runawayToolcallRetries < MAX_RUNAWAY_TOOLCALL_RETRIES) {
731
+ runawayToolcallRetries++;
732
+ const delayMs = RUNAWAY_TOOLCALL_RETRY_DELAY_MS * runawayToolcallRetries;
733
+ diag("retry", {
734
+ reason: "runaway_toolcall",
735
+ attempt: runawayToolcallRetries,
736
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
737
+ delayMs,
738
+ ...runawayDetected
739
+ });
740
+ yield {
741
+ type: "retry",
742
+ reason: "runaway_toolcall",
743
+ attempt: runawayToolcallRetries,
744
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
745
+ delayMs,
746
+ silent: true
747
+ };
748
+ await abortableSleep(delayMs, options.signal);
749
+ turn--;
750
+ continue;
751
+ }
689
752
  const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
690
753
  yield {
691
754
  type: "error",
692
755
  error: new Error(
693
- `The model glitched mid-tool-call and produced ${detail} without closing the call. This is usually an upstream model bug \u2014 try the same request again or switch models. Your conversation is preserved.`
756
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
694
757
  )
695
758
  };
696
759
  break;
@@ -743,15 +806,29 @@ async function* agentLoop(messages, options) {
743
806
  continue;
744
807
  }
745
808
  if (transportFailure) {
809
+ const cause = malformed ? "malformed_stream" : socketDrop ? "socket_drop" : "stream_stall";
746
810
  diag("stall_exhausted", {
747
811
  stallRetries: MAX_STALL_RETRIES,
748
812
  provider: options.provider,
749
- model: options.model
813
+ model: options.model,
814
+ cause,
815
+ nonStreaming: useNonStreamingFallback,
816
+ events: streamEventCount,
817
+ eventTypes: eventTypeCounts,
818
+ lastEventType,
819
+ sinceLastEventMs: Date.now() - lastEventTime,
820
+ attemptDurationMs: Date.now() - streamCallStart,
821
+ maxConsumerLagMs
750
822
  });
751
823
  yield {
752
824
  type: "error",
753
- error: new Error(
754
- `The API provider's stream stalled ${MAX_STALL_RETRIES} times \u2014 the provider may be experiencing capacity issues. Your conversation is preserved. Send another message to retry.`
825
+ error: new import_ai.EZCoderAIError(
826
+ `The connection to the API provider stopped responding after ${MAX_STALL_RETRIES} automatic retries. Your conversation is preserved.`,
827
+ {
828
+ source: "network",
829
+ hint: "Retry once. If it keeps happening on this device, disable any VPN or proxy and allow EZ Coder through firewall or antivirus web protection.",
830
+ cause: err
831
+ }
755
832
  )
756
833
  };
757
834
  break;
@@ -791,6 +868,7 @@ async function* agentLoop(messages, options) {
791
868
  }
792
869
  overloadRetries = 0;
793
870
  stallRetries = 0;
871
+ runawayToolcallRetries = 0;
794
872
  const contentArr = Array.isArray(response.message.content) ? response.message.content : null;
795
873
  const hasActionableContent = response.message.content !== "" && contentArr !== null && contentArr.some(
796
874
  (p) => p.type === "text" || p.type === "tool_call" || p.type === "server_tool_call"
@@ -862,6 +940,27 @@ async function* agentLoop(messages, options) {
862
940
  consecutivePauses = 0;
863
941
  const allToolCalls = extractToolCalls(response.message.content);
864
942
  if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
943
+ if (response.stopReason === "max_tokens") {
944
+ if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
945
+ maxTokensContinuations++;
946
+ diag("max_tokens_continuation", {
947
+ attempt: maxTokensContinuations,
948
+ maxAttempts: MAX_OUTPUT_CONTINUATIONS,
949
+ provider: options.provider,
950
+ model: options.model
951
+ });
952
+ yield { type: "truncated", reason: "max_tokens", continued: true };
953
+ messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
954
+ continue;
955
+ }
956
+ yield { type: "truncated", reason: "max_tokens", continued: false };
957
+ } else if (response.stopReason === "refusal" || response.stopReason === "error") {
958
+ yield {
959
+ type: "truncated",
960
+ reason: response.stopReason === "refusal" ? "refusal" : "provider_error",
961
+ continued: false
962
+ };
963
+ }
865
964
  if (options.getSteeringMessages) {
866
965
  const steering = await options.getSteeringMessages();
867
966
  if (steering && steering.length > 0) {
@@ -1228,16 +1327,22 @@ function capToolResults(toolResults, maxToolResultChars) {
1228
1327
  const max = Math.min(maxToolResultChars, hardMax);
1229
1328
  for (const toolResult of toolResults) {
1230
1329
  if (typeof toolResult.content !== "string" || toolResult.content.length <= max) continue;
1330
+ const originalChars = toolResult.content.length;
1231
1331
  const headChars = Math.floor(max * 0.7);
1232
1332
  const tailChars = max - headChars;
1233
1333
  const head = toolResult.content.slice(0, headChars);
1234
1334
  const tail = toolResult.content.slice(-tailChars);
1235
- const omitted = toolResult.content.length - headChars - tailChars;
1335
+ const omitted = originalChars - headChars - tailChars;
1236
1336
  toolResult.content = head + `
1237
1337
 
1238
1338
  [... ${omitted} characters omitted ...]
1239
1339
 
1240
1340
  ` + tail;
1341
+ toolResult.capped = {
1342
+ originalChars,
1343
+ keptChars: toolResult.content.length,
1344
+ scope: "per-result"
1345
+ };
1241
1346
  }
1242
1347
  }
1243
1348
  function capTurnToolResults(toolResults, maxTurnToolResultChars) {
@@ -1258,14 +1363,20 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1258
1363
  continue;
1259
1364
  }
1260
1365
  remaining -= fairShare;
1366
+ const originalChars = toolResult.content.length;
1261
1367
  const headChars = Math.floor(fairShare * 0.7);
1262
1368
  const tailChars = fairShare - headChars;
1263
- const omitted = toolResult.content.length - fairShare;
1369
+ const omitted = originalChars - fairShare;
1264
1370
  toolResult.content = toolResult.content.slice(0, headChars) + `
1265
1371
 
1266
1372
  [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1267
1373
 
1268
1374
  ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1375
+ toolResult.capped = {
1376
+ originalChars: toolResult.capped?.originalChars ?? originalChars,
1377
+ keptChars: toolResult.content.length,
1378
+ scope: "per-turn"
1379
+ };
1269
1380
  }
1270
1381
  }
1271
1382
  function normalizeToolResult(raw) {
@@ -1537,6 +1648,7 @@ var Agent = class {
1537
1648
  isAbortError,
1538
1649
  isBillingError,
1539
1650
  isContextOverflow,
1651
+ isLocalBackendUrl,
1540
1652
  isUsageLimitError,
1541
1653
  setStreamDiagnostic
1542
1654
  });