@prestyj/agent 5.11.0 → 5.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -55,6 +55,13 @@ interface AgentToolCallEndEvent {
55
55
  details?: unknown;
56
56
  isError: boolean;
57
57
  durationMs: number;
58
+ /**
59
+ * Set only when the call failed schema validation: how many consecutive
60
+ * times this tool produced this same validation error. 1 means the model
61
+ * still has room to self-correct; 3 is the threshold that ends the turn.
62
+ * Logged so a retry loop shows up as a count instead of identical lines.
63
+ */
64
+ invalidArgAttempt?: number;
58
65
  }
59
66
  interface AgentTurnTiming {
60
67
  /** Logical turn start, before context transforms or provider retries. Unix epoch milliseconds. */
@@ -133,7 +140,7 @@ interface AgentTurnBudgetExtendedEvent {
133
140
  */
134
141
  interface AgentTruncatedEvent {
135
142
  type: "truncated";
136
- reason: "max_tokens" | "refusal" | "provider_error";
143
+ reason: "max_tokens" | "refusal" | "provider_error" | "empty_response";
137
144
  /** True when the loop injected a continuation and will keep going. */
138
145
  continued: boolean;
139
146
  }
@@ -215,6 +222,22 @@ interface AgentOptions {
215
222
  temperature?: number;
216
223
  thinking?: StreamOptions["thinking"];
217
224
  apiKey?: string;
225
+ /**
226
+ * Re-resolve the credential at the start of every turn. A run can span many
227
+ * minutes, and an OAuth grant refreshed by any process (another app window, a
228
+ * CLI session, the usage poller) invalidates the access token captured when
229
+ * the run began — so a pinned `apiKey` goes dead mid-run and every remaining
230
+ * turn fails with an authentication error. Returning the current credential
231
+ * here keeps a long run alive across rotations.
232
+ *
233
+ * Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
234
+ * resolver throws (the provider call then surfaces the real auth error).
235
+ */
236
+ resolveCredentials?: () => Promise<{
237
+ apiKey: string;
238
+ accountId?: string;
239
+ projectId?: string;
240
+ }>;
218
241
  baseUrl?: string;
219
242
  signal?: AbortSignal;
220
243
  accountId?: string;
@@ -374,7 +397,7 @@ declare function isBillingError(err: unknown): boolean;
374
397
  * plan running out of usage). Unlike a transient per-minute 429, this does NOT
375
398
  * clear with a quick retry — the user must wait for the window to reset — so the
376
399
  * loop surfaces it immediately instead of retrying for minutes. Matches the
377
- * canonical message gg-ai stamps onto the provider error.
400
+ * canonical message @prestyj/ai stamps onto the provider error.
378
401
  */
379
402
  declare function isUsageLimitError(err: unknown): boolean;
380
403
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
package/dist/index.d.ts CHANGED
@@ -55,6 +55,13 @@ interface AgentToolCallEndEvent {
55
55
  details?: unknown;
56
56
  isError: boolean;
57
57
  durationMs: number;
58
+ /**
59
+ * Set only when the call failed schema validation: how many consecutive
60
+ * times this tool produced this same validation error. 1 means the model
61
+ * still has room to self-correct; 3 is the threshold that ends the turn.
62
+ * Logged so a retry loop shows up as a count instead of identical lines.
63
+ */
64
+ invalidArgAttempt?: number;
58
65
  }
59
66
  interface AgentTurnTiming {
60
67
  /** Logical turn start, before context transforms or provider retries. Unix epoch milliseconds. */
@@ -133,7 +140,7 @@ interface AgentTurnBudgetExtendedEvent {
133
140
  */
134
141
  interface AgentTruncatedEvent {
135
142
  type: "truncated";
136
- reason: "max_tokens" | "refusal" | "provider_error";
143
+ reason: "max_tokens" | "refusal" | "provider_error" | "empty_response";
137
144
  /** True when the loop injected a continuation and will keep going. */
138
145
  continued: boolean;
139
146
  }
@@ -215,6 +222,22 @@ interface AgentOptions {
215
222
  temperature?: number;
216
223
  thinking?: StreamOptions["thinking"];
217
224
  apiKey?: string;
225
+ /**
226
+ * Re-resolve the credential at the start of every turn. A run can span many
227
+ * minutes, and an OAuth grant refreshed by any process (another app window, a
228
+ * CLI session, the usage poller) invalidates the access token captured when
229
+ * the run began — so a pinned `apiKey` goes dead mid-run and every remaining
230
+ * turn fails with an authentication error. Returning the current credential
231
+ * here keeps a long run alive across rotations.
232
+ *
233
+ * Falls back to `apiKey`/`accountId`/`projectId` when omitted or when the
234
+ * resolver throws (the provider call then surfaces the real auth error).
235
+ */
236
+ resolveCredentials?: () => Promise<{
237
+ apiKey: string;
238
+ accountId?: string;
239
+ projectId?: string;
240
+ }>;
218
241
  baseUrl?: string;
219
242
  signal?: AbortSignal;
220
243
  accountId?: string;
@@ -374,7 +397,7 @@ declare function isBillingError(err: unknown): boolean;
374
397
  * plan running out of usage). Unlike a transient per-minute 429, this does NOT
375
398
  * clear with a quick retry — the user must wait for the window to reset — so the
376
399
  * loop surfaces it immediately instead of retrying for minutes. Matches the
377
- * canonical message gg-ai stamps onto the provider error.
400
+ * canonical message @prestyj/ai stamps onto the provider error.
378
401
  */
379
402
  declare function isUsageLimitError(err: unknown): boolean;
380
403
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
package/dist/index.js CHANGED
@@ -57,7 +57,13 @@ function isContextOverflow(err) {
57
57
  if (overflowStatus === 402) return false;
58
58
  if (isBillingError(err)) return false;
59
59
  const msg = err.message.toLowerCase();
60
- return msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten") || msg.includes("token") && msg.includes("exceed");
60
+ if (msg.includes("prompt is too long") || msg.includes("prompt too long") || msg.includes("input is too long") || msg.includes("context_length_exceeded") || msg.includes("context_window_exceeded") || msg.includes("maximum context length") || msg.includes("exceeds model context window") || msg.includes("exceeds the context window") || msg.includes("content_too_large") || msg.includes("request_too_large") || msg.includes("reduce the length") || msg.includes("please shorten")) {
61
+ return true;
62
+ }
63
+ const rateLimited = overflowStatus === 429 || msg.includes("rate limit") || msg.includes("rate_limit") || msg.includes("too many requests");
64
+ const perUnitTime = msg.includes("per min") || msg.includes("/min") || msg.includes("per minute") || msg.includes("per hour") || msg.includes("per day") || msg.includes("tpm") || msg.includes("rpm");
65
+ if (rateLimited && perUnitTime) return false;
66
+ return msg.includes("token") && msg.includes("exceed");
61
67
  }
62
68
  function parseOverflowNumber(value) {
63
69
  return Number(value.replace(/[,_\s]/g, ""));
@@ -170,6 +176,17 @@ function isMalformedStream(err) {
170
176
  const msg = err.message;
171
177
  return /\bin JSON at position \d+/i.test(msg);
172
178
  }
179
+ var TIMEOUT_NAMES = /* @__PURE__ */ new Set(["TimeoutError", "ConnectTimeoutError", "HeadersTimeoutError"]);
180
+ var TIMEOUT_MESSAGES = [
181
+ /^request timed out\.?$/i,
182
+ /\brequest to [\w .-]+ timed out\b/i,
183
+ /\b(?:connection|socket|headers|stream) timed out\b/i
184
+ ];
185
+ function isBareTimeout(e) {
186
+ if (typeof e.name === "string" && TIMEOUT_NAMES.has(e.name)) return true;
187
+ if (typeof e.message !== "string") return false;
188
+ return TIMEOUT_MESSAGES.some((re) => re.test(e.message));
189
+ }
173
190
  function isTransportFailure(err) {
174
191
  const codes = /* @__PURE__ */ new Set([
175
192
  "ECONNRESET",
@@ -206,6 +223,8 @@ function isTransportFailure(err) {
206
223
  if (typeof e.message === "string") {
207
224
  for (const re of messages) if (re.test(e.message)) return true;
208
225
  }
226
+ const clientError = typeof e.status === "number" && e.status >= 400 && e.status < 500;
227
+ if (!clientError && isBareTimeout(e)) return true;
209
228
  cur = e.cause;
210
229
  }
211
230
  return false;
@@ -431,6 +450,21 @@ async function* agentLoop(messages, options) {
431
450
  diag("stream_call", { nonStreaming: useNonStreamingFallback });
432
451
  streamCallStart = Date.now();
433
452
  providerAttemptStartedAt = streamCallStart;
453
+ let liveApiKey = options.apiKey;
454
+ let liveAccountId = options.accountId;
455
+ let liveProjectId = options.projectId;
456
+ if (options.resolveCredentials) {
457
+ try {
458
+ const fresh = await options.resolveCredentials();
459
+ liveApiKey = fresh.apiKey;
460
+ if (fresh.accountId !== void 0) liveAccountId = fresh.accountId;
461
+ if (fresh.projectId !== void 0) liveProjectId = fresh.projectId;
462
+ } catch (credErr) {
463
+ diag("credential_refresh_failed", {
464
+ error: (credErr instanceof Error ? credErr.message : String(credErr)).slice(0, 200)
465
+ });
466
+ }
467
+ }
434
468
  const result = stream({
435
469
  provider: options.provider,
436
470
  model: options.model,
@@ -442,12 +476,12 @@ async function* agentLoop(messages, options) {
442
476
  maxTokens: options.maxTokens,
443
477
  temperature: options.temperature,
444
478
  thinking: options.thinking,
445
- apiKey: options.apiKey,
479
+ apiKey: liveApiKey,
446
480
  baseUrl: options.baseUrl,
447
481
  signal: streamController.signal,
448
- accountId: options.accountId,
482
+ accountId: liveAccountId,
449
483
  transportSessionId: options.transportSessionId,
450
- projectId: options.projectId,
484
+ projectId: liveProjectId,
451
485
  cacheRetention: options.cacheRetention,
452
486
  promptCacheKey: options.promptCacheKey,
453
487
  serviceTier: options.serviceTier,
@@ -891,6 +925,7 @@ async function* agentLoop(messages, options) {
891
925
  continue;
892
926
  }
893
927
  }
928
+ const emptyExhausted = !hasActionableContent;
894
929
  emptyResponseRetries = 0;
895
930
  useNonStreamingFallback = false;
896
931
  totalUsage.inputTokens += response.usage.inputTokens;
@@ -904,9 +939,11 @@ async function* agentLoop(messages, options) {
904
939
  if (response.usage.cacheWrite) {
905
940
  totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
906
941
  }
907
- messages.push(response.message);
908
- latestProviderUsage = response.usage;
909
- usageAnchorIndex = messages.length - 1;
942
+ if (!emptyExhausted) {
943
+ messages.push(response.message);
944
+ latestProviderUsage = response.usage;
945
+ usageAnchorIndex = messages.length - 1;
946
+ }
910
947
  const completedAt = Date.now();
911
948
  const outputTokensPerSecond = providerDurationMs > 0 && response.usage.outputTokens > 0 ? response.usage.outputTokens / (providerDurationMs / 1e3) : void 0;
912
949
  const timing = {
@@ -938,8 +975,15 @@ async function* agentLoop(messages, options) {
938
975
  }
939
976
  consecutivePauses = 0;
940
977
  const allToolCalls = extractToolCalls(response.message.content);
941
- if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
942
- if (response.stopReason === "max_tokens") {
978
+ if (emptyExhausted || response.stopReason !== "tool_use" && allToolCalls.length === 0) {
979
+ if (emptyExhausted) {
980
+ diag("empty_response_exhausted", {
981
+ provider: options.provider,
982
+ model: options.model,
983
+ stopReason: response.stopReason
984
+ });
985
+ yield { type: "truncated", reason: "empty_response", continued: false };
986
+ } else if (response.stopReason === "max_tokens") {
943
987
  if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
944
988
  maxTokensContinuations++;
945
989
  diag("max_tokens_continuation", {
@@ -1161,6 +1205,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1161
1205
  let resultContent;
1162
1206
  let details;
1163
1207
  let isError = false;
1208
+ let invalidArgAttempt;
1164
1209
  const tool = options.toolMap.get(toolCall.name);
1165
1210
  if (!tool) {
1166
1211
  resultContent = `Unknown tool: ${toolCall.name}`;
@@ -1198,6 +1243,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1198
1243
  const failureKey = `${toolCall.name}:${prettyError}`;
1199
1244
  const failureCount = (options.invalidToolArgumentCounts.get(failureKey) ?? 0) + 1;
1200
1245
  options.invalidToolArgumentCounts.set(failureKey, failureCount);
1246
+ invalidArgAttempt = failureCount;
1201
1247
  resultContent = `Invalid arguments for tool \`${toolCall.name}\`:
1202
1248
  ` + prettyError + "\nRe-issue the call with each field as the correct type.";
1203
1249
  if (failureCount >= 3) {
@@ -1228,7 +1274,8 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1228
1274
  result: toolResultPreview(resultContent),
1229
1275
  details,
1230
1276
  isError,
1231
- durationMs
1277
+ durationMs,
1278
+ ...invalidArgAttempt === void 0 ? {} : { invalidArgAttempt }
1232
1279
  });
1233
1280
  return { toolCallId: toolCall.id, content: resultContent, isError };
1234
1281
  }