@prestyj/agent 5.8.0 → 5.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
87
87
  totalTurns: number;
88
88
  maxTurns: number;
89
89
  }
90
+ /**
91
+ * Warning signal emitted when a turn ended on a non-clean stop reason —
92
+ * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
93
+ * or a provider-reported `error` stop. Distinguishes a truncated/degraded
94
+ * completion from a clean one so hosts can warn the user instead of silently
95
+ * presenting incomplete output as done.
96
+ */
97
+ interface AgentTruncatedEvent {
98
+ type: "truncated";
99
+ reason: "max_tokens" | "refusal" | "provider_error";
100
+ /** True when the loop injected a continuation and will keep going. */
101
+ continued: boolean;
102
+ }
90
103
  interface AgentRetryEvent {
91
104
  type: "retry";
92
- reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
105
+ reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
93
106
  attempt: number;
94
107
  maxAttempts: number;
95
108
  delayMs: number;
@@ -135,7 +148,7 @@ interface AgentFollowUpMessageEvent {
135
148
  type: "follow_up_message";
136
149
  content: Message["content"];
137
150
  }
138
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
151
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
139
152
  interface TransformContextOptions {
140
153
  /** Force a transform after the provider reports context overflow. */
141
154
  force?: boolean;
package/dist/index.d.ts CHANGED
@@ -87,9 +87,22 @@ interface AgentMaxTurnsEvent {
87
87
  totalTurns: number;
88
88
  maxTurns: number;
89
89
  }
90
+ /**
91
+ * Warning signal emitted when a turn ended on a non-clean stop reason —
92
+ * `max_tokens` (output clipped at the model's output-token limit), `refusal`,
93
+ * or a provider-reported `error` stop. Distinguishes a truncated/degraded
94
+ * completion from a clean one so hosts can warn the user instead of silently
95
+ * presenting incomplete output as done.
96
+ */
97
+ interface AgentTruncatedEvent {
98
+ type: "truncated";
99
+ reason: "max_tokens" | "refusal" | "provider_error";
100
+ /** True when the loop injected a continuation and will keep going. */
101
+ continued: boolean;
102
+ }
90
103
  interface AgentRetryEvent {
91
104
  type: "retry";
92
- reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch";
105
+ reason: "overloaded" | "rate_limit" | "provider_error" | "empty_response" | "stream_stall" | "overflow_compact" | "tool_argument_glitch" | "runaway_toolcall";
93
106
  attempt: number;
94
107
  maxAttempts: number;
95
108
  delayMs: number;
@@ -135,7 +148,7 @@ interface AgentFollowUpMessageEvent {
135
148
  type: "follow_up_message";
136
149
  content: Message["content"];
137
150
  }
138
- type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentErrorEvent;
151
+ type AgentEvent = AgentTextDeltaEvent | AgentThinkingDeltaEvent | AgentToolCallStartEvent | AgentToolCallUpdateEvent | AgentToolCallEndEvent | AgentToolCallDeltaEvent | AgentServerToolCallEvent | AgentServerToolResultEvent | AgentSteeringMessageEvent | AgentFollowUpMessageEvent | AgentRetryEvent | AgentTurnEndEvent | AgentDoneEvent | AgentMaxTurnsEvent | AgentTruncatedEvent | AgentErrorEvent;
139
152
  interface TransformContextOptions {
140
153
  /** Force a transform after the provider reports context overflow. */
141
154
  force?: boolean;
package/dist/index.js CHANGED
@@ -188,7 +188,11 @@ function createAbortError() {
188
188
  }
189
189
  function abortablePromise(promise, signal) {
190
190
  if (!signal) return promise;
191
- if (signal.aborted) return Promise.reject(createAbortError());
191
+ if (signal.aborted) {
192
+ promise.catch(() => {
193
+ });
194
+ return Promise.reject(createAbortError());
195
+ }
192
196
  return new Promise((resolve, reject) => {
193
197
  let settled = false;
194
198
  const cleanup = () => signal.removeEventListener("abort", onAbort);
@@ -244,6 +248,7 @@ async function* agentLoop(messages, options) {
244
248
  let overloadRetries = 0;
245
249
  let emptyResponseRetries = 0;
246
250
  let stallRetries = 0;
251
+ let runawayToolcallRetries = 0;
247
252
  let overflowCompactionAttempts = 0;
248
253
  let toolResultTruncationAttempted = false;
249
254
  const invalidToolArgumentCounts = /* @__PURE__ */ new Map();
@@ -252,11 +257,16 @@ async function* agentLoop(messages, options) {
252
257
  const MAX_OVERLOAD_RETRIES = 10;
253
258
  const MAX_EMPTY_RESPONSE_RETRIES = 2;
254
259
  const MAX_STALL_RETRIES = 10;
260
+ const MAX_RUNAWAY_TOOLCALL_RETRIES = 2;
261
+ const RUNAWAY_TOOLCALL_RETRY_DELAY_MS = 1e3;
255
262
  const MAX_OVERFLOW_COMPACTIONS = 2;
256
263
  const STALL_RETRIES_BEFORE_NON_STREAMING = 2;
257
264
  const STALL_DELAY_MS = 1e3;
258
265
  const MIN_PARTIAL_PRESERVE_CHARS = 200;
259
266
  const PARTIAL_CONTINUATION_PROMPT = "[Your previous response was cut off by a connection failure. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
267
+ const MAX_OUTPUT_CONTINUATIONS = 2;
268
+ const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
269
+ let maxTokensContinuations = 0;
260
270
  const OVERLOAD_BASE_DELAY_MS = 2e3;
261
271
  const OVERLOAD_MAX_DELAY_MS = 3e4;
262
272
  const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
@@ -659,11 +669,33 @@ async function* agentLoop(messages, options) {
659
669
  provider: options.provider,
660
670
  model: options.model
661
671
  });
672
+ if (runawayToolcallRetries < MAX_RUNAWAY_TOOLCALL_RETRIES) {
673
+ runawayToolcallRetries++;
674
+ const delayMs = RUNAWAY_TOOLCALL_RETRY_DELAY_MS * runawayToolcallRetries;
675
+ diag("retry", {
676
+ reason: "runaway_toolcall",
677
+ attempt: runawayToolcallRetries,
678
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
679
+ delayMs,
680
+ ...runawayDetected
681
+ });
682
+ yield {
683
+ type: "retry",
684
+ reason: "runaway_toolcall",
685
+ attempt: runawayToolcallRetries,
686
+ maxAttempts: MAX_RUNAWAY_TOOLCALL_RETRIES,
687
+ delayMs,
688
+ silent: true
689
+ };
690
+ await abortableSleep(delayMs, options.signal);
691
+ turn--;
692
+ continue;
693
+ }
662
694
  const detail = runawayDetected.kind === "chars" ? `${(runawayDetected.chars / 1024).toFixed(0)} KB of tool-call arguments` : `${runawayDetected.events} tool-call delta events`;
663
695
  yield {
664
696
  type: "error",
665
697
  error: new Error(
666
- `The model glitched mid-tool-call and produced ${detail} without closing the call. This is usually an upstream model bug \u2014 try the same request again or switch models. Your conversation is preserved.`
698
+ `The model repeatedly failed to close a tool call after ${MAX_RUNAWAY_TOOLCALL_RETRIES} automatic retries (${detail}). Switch models and retry; your conversation is preserved.`
667
699
  )
668
700
  };
669
701
  break;
@@ -764,6 +796,7 @@ async function* agentLoop(messages, options) {
764
796
  }
765
797
  overloadRetries = 0;
766
798
  stallRetries = 0;
799
+ runawayToolcallRetries = 0;
767
800
  const contentArr = Array.isArray(response.message.content) ? response.message.content : null;
768
801
  const hasActionableContent = response.message.content !== "" && contentArr !== null && contentArr.some(
769
802
  (p) => p.type === "text" || p.type === "tool_call" || p.type === "server_tool_call"
@@ -835,6 +868,27 @@ async function* agentLoop(messages, options) {
835
868
  consecutivePauses = 0;
836
869
  const allToolCalls = extractToolCalls(response.message.content);
837
870
  if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
871
+ if (response.stopReason === "max_tokens") {
872
+ if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
873
+ maxTokensContinuations++;
874
+ diag("max_tokens_continuation", {
875
+ attempt: maxTokensContinuations,
876
+ maxAttempts: MAX_OUTPUT_CONTINUATIONS,
877
+ provider: options.provider,
878
+ model: options.model
879
+ });
880
+ yield { type: "truncated", reason: "max_tokens", continued: true };
881
+ messages.push({ role: "user", content: MAX_TOKENS_CONTINUATION_PROMPT });
882
+ continue;
883
+ }
884
+ yield { type: "truncated", reason: "max_tokens", continued: false };
885
+ } else if (response.stopReason === "refusal" || response.stopReason === "error") {
886
+ yield {
887
+ type: "truncated",
888
+ reason: response.stopReason === "refusal" ? "refusal" : "provider_error",
889
+ continued: false
890
+ };
891
+ }
838
892
  if (options.getSteeringMessages) {
839
893
  const steering = await options.getSteeringMessages();
840
894
  if (steering && steering.length > 0) {
@@ -1201,16 +1255,22 @@ function capToolResults(toolResults, maxToolResultChars) {
1201
1255
  const max = Math.min(maxToolResultChars, hardMax);
1202
1256
  for (const toolResult of toolResults) {
1203
1257
  if (typeof toolResult.content !== "string" || toolResult.content.length <= max) continue;
1258
+ const originalChars = toolResult.content.length;
1204
1259
  const headChars = Math.floor(max * 0.7);
1205
1260
  const tailChars = max - headChars;
1206
1261
  const head = toolResult.content.slice(0, headChars);
1207
1262
  const tail = toolResult.content.slice(-tailChars);
1208
- const omitted = toolResult.content.length - headChars - tailChars;
1263
+ const omitted = originalChars - headChars - tailChars;
1209
1264
  toolResult.content = head + `
1210
1265
 
1211
1266
  [... ${omitted} characters omitted ...]
1212
1267
 
1213
1268
  ` + tail;
1269
+ toolResult.capped = {
1270
+ originalChars,
1271
+ keptChars: toolResult.content.length,
1272
+ scope: "per-result"
1273
+ };
1214
1274
  }
1215
1275
  }
1216
1276
  function capTurnToolResults(toolResults, maxTurnToolResultChars) {
@@ -1231,14 +1291,20 @@ function capTurnToolResults(toolResults, maxTurnToolResultChars) {
1231
1291
  continue;
1232
1292
  }
1233
1293
  remaining -= fairShare;
1294
+ const originalChars = toolResult.content.length;
1234
1295
  const headChars = Math.floor(fairShare * 0.7);
1235
1296
  const tailChars = fairShare - headChars;
1236
- const omitted = toolResult.content.length - fairShare;
1297
+ const omitted = originalChars - fairShare;
1237
1298
  toolResult.content = toolResult.content.slice(0, headChars) + `
1238
1299
 
1239
1300
  [... ${omitted} characters trimmed: this turn's combined tool results exceeded the per-turn budget. Re-run this call alone with narrower filters or offset/limit if you need the omitted content ...]
1240
1301
 
1241
1302
  ` + (tailChars > 0 ? toolResult.content.slice(-tailChars) : "");
1303
+ toolResult.capped = {
1304
+ originalChars: toolResult.capped?.originalChars ?? originalChars,
1305
+ keptChars: toolResult.content.length,
1306
+ scope: "per-turn"
1307
+ };
1242
1308
  }
1243
1309
  }
1244
1310
  function normalizeToolResult(raw) {