@prestyj/agent 5.10.0 → 5.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -63,6 +63,7 @@ function isLocalBackendUrl(baseUrl) {
63
63
 
64
64
  // src/agent-loop.ts
65
65
  var DEFAULT_MAX_TURNS = 300;
66
+ var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
66
67
  var _diagFn = null;
67
68
  function setStreamDiagnostic(fn) {
68
69
  _diagFn = fn;
@@ -82,7 +83,13 @@ function isContextOverflow(err) {
82
83
  if (overflowStatus === 402) return false;
83
84
  if (isBillingError(err)) return false;
84
85
  const msg = err.message.toLowerCase();
85
- return msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten") || msg.includes("token") && msg.includes("exceed");
86
+ if (msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten")) {
87
+ return true;
88
+ }
89
+ const rateLimited = overflowStatus === 429 || msg.includes("rate limit") || msg.includes("rate_limit") || msg.includes("too many requests");
90
+ const perUnitTime = msg.includes("per min") || msg.includes("/min") || msg.includes("per minute") || msg.includes("per hour") || msg.includes("per day") || msg.includes("tpm") || msg.includes("rpm");
91
+ if (rateLimited && perUnitTime) return false;
92
+ return msg.includes("token") && msg.includes("exceed");
86
93
  }
87
94
  function parseOverflowNumber(value) {
88
95
  return Number(value.replace(/[,_\s]/g, ""));
@@ -195,6 +202,17 @@ function isMalformedStream(err) {
195
202
  const msg = err.message;
196
203
  return /\bin JSON at position \d+/i.test(msg);
197
204
  }
205
+ var TIMEOUT_NAMES = /* @__PURE__ */ new Set(["TimeoutError", "ConnectTimeoutError", "HeadersTimeoutError"]);
206
+ var TIMEOUT_MESSAGES = [
207
+ /^request timed out\.?$/i,
208
+ /\brequest to [\w .-]+ timed out\b/i,
209
+ /\b(?:connection|socket|headers|stream) timed out\b/i
210
+ ];
211
+ function isBareTimeout(e) {
212
+ if (typeof e.name === "string" && TIMEOUT_NAMES.has(e.name)) return true;
213
+ if (typeof e.message !== "string") return false;
214
+ return TIMEOUT_MESSAGES.some((re) => re.test(e.message));
215
+ }
198
216
  function isTransportFailure(err) {
199
217
  const codes = /* @__PURE__ */ new Set([
200
218
  "ECONNRESET",
@@ -231,6 +249,8 @@ function isTransportFailure(err) {
231
249
  if (typeof e.message === "string") {
232
250
  for (const re of messages) if (re.test(e.message)) return true;
233
251
  }
252
+ const clientError = typeof e.status === "number" && e.status >= 400 && e.status < 500;
253
+ if (!clientError && isBareTimeout(e)) return true;
234
254
  cur = e.cause;
235
255
  }
236
256
  return false;
@@ -265,6 +285,9 @@ function abortablePromise(promise, signal) {
265
285
  promise.then(resolveOnce, rejectOnce);
266
286
  });
267
287
  }
288
+ function turnBudgetContinuationPrompt() {
289
+ return "[You reached the turn limit for this segment but the work is not finished, so you have been granted more turns. Before continuing, state in one or two sentences what is already done and what remains, then keep going from there \u2014 do not restart work that is already complete.]";
290
+ }
268
291
  function abortableSleep(ms, signal) {
269
292
  if (signal?.aborted) return Promise.reject(createAbortError());
270
293
  return new Promise((resolve, reject) => {
@@ -286,6 +309,9 @@ function closeIterator(iterator) {
286
309
  }
287
310
  async function* agentLoop(messages, options) {
288
311
  const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
312
+ let effectiveMaxTurns = maxTurns;
313
+ const maxTurnExtensions = Math.max(0, options.maxTurnExtensions ?? 2);
314
+ let turnExtensions = 0;
289
315
  const maxContinuations = options.maxContinuations ?? 5;
290
316
  let toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
291
317
  const totalUsage = { inputTokens: 0, outputTokens: 0 };
@@ -333,12 +359,12 @@ async function* agentLoop(messages, options) {
333
359
  const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
334
360
  const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
335
361
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
336
- const MAX_TOOLCALL_DELTA_EVENTS = 2e4;
362
+ const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
337
363
  let logicalTurnStartedAt = 0;
338
364
  let firstProviderEventAt;
339
365
  let providerDurationMs = 0;
340
366
  try {
341
- while (turn < maxTurns) {
367
+ while (turn < effectiveMaxTurns) {
342
368
  options.signal?.throwIfAborted();
343
369
  turn++;
344
370
  if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
@@ -409,6 +435,7 @@ async function* agentLoop(messages, options) {
409
435
  let lastEventType = "";
410
436
  let toolcallDeltaChars = 0;
411
437
  let toolcallDeltaCount = 0;
438
+ let toolcallNoProgressCount = 0;
412
439
  let runawayDetected = null;
413
440
  let attemptText = "";
414
441
  let lastYieldEndTime = Date.now();
@@ -449,6 +476,21 @@ async function* agentLoop(messages, options) {
449
476
  diag("stream_call", { nonStreaming: useNonStreamingFallback });
450
477
  streamCallStart = Date.now();
451
478
  providerAttemptStartedAt = streamCallStart;
479
+ let liveApiKey = options.apiKey;
480
+ let liveAccountId = options.accountId;
481
+ let liveProjectId = options.projectId;
482
+ if (options.resolveCredentials) {
483
+ try {
484
+ const fresh = await options.resolveCredentials();
485
+ liveApiKey = fresh.apiKey;
486
+ if (fresh.accountId !== void 0) liveAccountId = fresh.accountId;
487
+ if (fresh.projectId !== void 0) liveProjectId = fresh.projectId;
488
+ } catch (credErr) {
489
+ diag("credential_refresh_failed", {
490
+ error: (credErr instanceof Error ? credErr.message : String(credErr)).slice(0, 200)
491
+ });
492
+ }
493
+ }
452
494
  const result = (0, import_ai.stream)({
453
495
  provider: options.provider,
454
496
  model: options.model,
@@ -460,12 +502,12 @@ async function* agentLoop(messages, options) {
460
502
  maxTokens: options.maxTokens,
461
503
  temperature: options.temperature,
462
504
  thinking: options.thinking,
463
- apiKey: options.apiKey,
505
+ apiKey: liveApiKey,
464
506
  baseUrl: options.baseUrl,
465
507
  signal: streamController.signal,
466
- accountId: options.accountId,
508
+ accountId: liveAccountId,
467
509
  transportSessionId: options.transportSessionId,
468
- projectId: options.projectId,
510
+ projectId: liveProjectId,
469
511
  cacheRetention: options.cacheRetention,
470
512
  promptCacheKey: options.promptCacheKey,
471
513
  serviceTier: options.serviceTier,
@@ -563,11 +605,13 @@ async function* agentLoop(messages, options) {
563
605
  const chunkChars = event.argsJson?.length ?? 0;
564
606
  toolcallDeltaChars += chunkChars;
565
607
  toolcallDeltaCount++;
566
- if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallDeltaCount > MAX_TOOLCALL_DELTA_EVENTS)) {
608
+ toolcallNoProgressCount = chunkChars > 0 ? 0 : toolcallNoProgressCount + 1;
609
+ if (!runawayDetected && (toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS || toolcallNoProgressCount > MAX_TOOLCALL_NO_PROGRESS_EVENTS)) {
567
610
  runawayDetected = {
568
611
  kind: toolcallDeltaChars > MAX_TOOLCALL_DELTA_CHARS ? "chars" : "events",
569
612
  chars: toolcallDeltaChars,
570
- events: toolcallDeltaCount
613
+ events: toolcallDeltaCount,
614
+ noProgressEvents: toolcallNoProgressCount
571
615
  };
572
616
  diag("runaway_toolcall_detected", {
573
617
  ...runawayDetected,
@@ -749,11 +793,15 @@ async function* agentLoop(messages, options) {
749
793
  turn--;
750
794
  continue;
751
795
  }
752
- const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
796
+ const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.noProgressEvents} consecutive tool-call delta events without argument progress`;
753
797
  yield {
754
798
  type: "error",
755
- error: new Error(
756
- `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
799
+ error: new import_ai.EZCoderAIError(
800
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Your conversation is preserved.`,
801
+ {
802
+ source: "provider",
803
+ hint: "Retry once. If it keeps happening, switch models and continue the preserved conversation."
804
+ }
757
805
  )
758
806
  };
759
807
  break;
@@ -780,7 +828,15 @@ async function* agentLoop(messages, options) {
780
828
  role: "assistant",
781
829
  content: [{ type: "text", text: attemptText }]
782
830
  });
783
- messages.push({ role: "user", content: PARTIAL_CONTINUATION_PROMPT });
831
+ messages.push({
832
+ role: "user",
833
+ content: PARTIAL_CONTINUATION_PROMPT,
834
+ provenance: {
835
+ source: "runtime",
836
+ kind: "continuation",
837
+ visibility: "hidden"
838
+ }
839
+ });
784
840
  preservedChars = attemptText.length;
785
841
  }
786
842
  diag("retry", {
@@ -899,6 +955,9 @@ async function* agentLoop(messages, options) {
899
955
  useNonStreamingFallback = false;
900
956
  totalUsage.inputTokens += response.usage.inputTokens;
901
957
  totalUsage.outputTokens += response.usage.outputTokens;
958
+ if (response.usage.reasoningTokens) {
959
+ totalUsage.reasoningTokens = (totalUsage.reasoningTokens ?? 0) + response.usage.reasoningTokens;
960
+ }
902
961
  if (response.usage.cacheRead) {
903
962
  totalUsage.cacheRead = (totalUsage.cacheRead ?? 0) + response.usage.cacheRead;
904
963
  }
@@ -950,7 +1009,15 @@ async function* agentLoop(messages, options) {
950
1009
  model: options.model
951
1010
  });
952
1011
  yield { type: "truncated", reason: "max_tokens", continued: true };
953
- messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
1012
+ messages.push({
1013
+ role: "user",
1014
+ content: MAX_TOKENS_CONTINUATION_PROMPT,
1015
+ provenance: {
1016
+ source: "runtime",
1017
+ kind: "continuation",
1018
+ visibility: "hidden"
1019
+ }
1020
+ });
954
1021
  continue;
955
1022
  }
956
1023
  yield { type: "truncated", reason: "max_tokens", continued: false };
@@ -1026,6 +1093,7 @@ async function* agentLoop(messages, options) {
1026
1093
  );
1027
1094
  const executionResult = hasSequentialToolCall ? yield* executeToolCallsMixed(toolCalls, toolResults, executionOptions) : yield* executeToolCallsParallel(toolCalls, toolResults, executionOptions);
1028
1095
  messages.push({ role: "tool", content: executionResult.toolResults });
1096
+ yield { type: "checkpoint", turn };
1029
1097
  const toolsAborted = executionResult.aborted;
1030
1098
  if (fatalToolArgumentError) {
1031
1099
  if (fatalToolArgumentRecoverable && !toolArgumentAutoContinueUsed) {
@@ -1057,8 +1125,51 @@ async function* agentLoop(messages, options) {
1057
1125
  }
1058
1126
  }
1059
1127
  }
1060
- if (turn >= maxTurns) {
1061
- hitMaxTurns = true;
1128
+ if (turn >= effectiveMaxTurns) {
1129
+ let extended = false;
1130
+ if (options.onTurnBudgetExhausted && turnExtensions < maxTurnExtensions) {
1131
+ const extension = turnExtensions + 1;
1132
+ let granted = false;
1133
+ try {
1134
+ granted = await options.onTurnBudgetExhausted({
1135
+ turn,
1136
+ maxTurns: effectiveMaxTurns,
1137
+ extension
1138
+ });
1139
+ } catch {
1140
+ granted = false;
1141
+ }
1142
+ if (granted) {
1143
+ turnExtensions = extension;
1144
+ effectiveMaxTurns += maxTurns;
1145
+ extended = true;
1146
+ diag("turn_budget_extended", {
1147
+ turn,
1148
+ grantedTurns: effectiveMaxTurns,
1149
+ extension,
1150
+ provider: options.provider,
1151
+ model: options.model
1152
+ });
1153
+ yield {
1154
+ type: "turn_budget_extended",
1155
+ turn,
1156
+ grantedTurns: effectiveMaxTurns,
1157
+ extension
1158
+ };
1159
+ messages.push({
1160
+ role: "user",
1161
+ content: turnBudgetContinuationPrompt(),
1162
+ provenance: {
1163
+ source: "runtime",
1164
+ kind: "continuation",
1165
+ visibility: "hidden"
1166
+ }
1167
+ });
1168
+ }
1169
+ }
1170
+ if (!extended) {
1171
+ hitMaxTurns = true;
1172
+ }
1062
1173
  }
1063
1174
  }
1064
1175
  } finally {
@@ -1074,14 +1185,15 @@ async function* agentLoop(messages, options) {
1074
1185
  if (hitMaxTurns) {
1075
1186
  diag("max_turns_reached", {
1076
1187
  turn,
1077
- maxTurns,
1188
+ maxTurns: effectiveMaxTurns,
1189
+ extensions: turnExtensions,
1078
1190
  provider: options.provider,
1079
1191
  model: options.model
1080
1192
  });
1081
1193
  yield {
1082
1194
  type: "max_turns",
1083
1195
  totalTurns: turn,
1084
- maxTurns
1196
+ maxTurns: effectiveMaxTurns
1085
1197
  };
1086
1198
  }
1087
1199
  yield {
@@ -1117,7 +1229,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1117
1229
  try {
1118
1230
  const parsed = tool.parameters.parse(toolCall.args);
1119
1231
  const callerSignal = options.signal;
1120
- const toolTimeout = AbortSignal.timeout(3e5);
1232
+ const toolTimeout = AbortSignal.timeout(tool.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS);
1121
1233
  const ctx = {
1122
1234
  signal: callerSignal ? AbortSignal.any([callerSignal, toolTimeout]) : toolTimeout,
1123
1235
  toolCallId: toolCall.id,
@@ -1330,9 +1442,9 @@ function capToolResults(toolResults, maxToolResultChars) {
1330
1442
  const originalChars = toolResult.content.length;
1331
1443
  const headChars = Math.floor(max * 0.7);
1332
1444
  const tailChars = max - headChars;
1333
- const head = toolResult.content.slice(0, headChars);
1334
- const tail = toolResult.content.slice(-tailChars);
1335
- const omitted = originalChars - headChars - tailChars;
1445
+ const head = (0, import_ai.sliceHead)(toolResult.content, headChars);
1446
+ const tail = (0, import_ai.sliceTail)(toolResult.content, tailChars);
1447
+ const omitted = originalChars - head.length - tail.length;
1336
1448
  toolResult.content = head + `
1337
1449
 
1338
1450
  [... ${omitted} characters omitted ...]
@@ -1367,11 +1479,11 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1367
1479
  const headChars = Math.floor(fairShare * 0.7);
1368
1480
  const tailChars = fairShare - headChars;
1369
1481
  const omitted = originalChars - fairShare;
1370
- toolResult.content = toolResult.content.slice(0, headChars) + `
1482
+ toolResult.content = (0, import_ai.sliceHead)(toolResult.content, headChars) + `
1371
1483
 
1372
1484
  [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1373
1485
 
1374
- ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1486
+ ` + (0, import_ai.sliceTail)(toolResult.content, tailChars);
1375
1487
  toolResult.capped = {
1376
1488
  originalChars: toolResult.capped?.originalChars ?? originalChars,
1377
1489
  keptChars: toolResult.content.length,
@@ -1390,12 +1502,14 @@ function truncateToolResultText(text, maxChars) {
1390
1502
  if (text.length <= maxChars) return text;
1391
1503
  const tailChars = Math.min(Math.floor(maxChars * 0.3), 2e4);
1392
1504
  const headChars = Math.max(maxChars - tailChars, 0);
1393
- const omitted = text.length - headChars - tailChars;
1394
- return `${text.slice(0, headChars)}
1505
+ const head = (0, import_ai.sliceHead)(text, headChars);
1506
+ const tail = (0, import_ai.sliceTail)(text, tailChars);
1507
+ const omitted = text.length - head.length - tail.length;
1508
+ return `${head}
1395
1509
 
1396
1510
  [... ${omitted} characters omitted after context overflow ...]
1397
1511
 
1398
- ${text.slice(-tailChars)}`;
1512
+ ${tail}`;
1399
1513
  }
1400
1514
  function truncateOversizedToolResults(messages, maxChars) {
1401
1515
  if (maxChars <= 0) return false;