@prestyj/agent 5.12.0 → 5.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -23,6 +23,8 @@ __export(index_exports, {
23
23
  Agent: () => Agent,
24
24
  AgentStream: () => AgentStream,
25
25
  agentLoop: () => agentLoop,
26
+ cancelledBeforeStartText: () => cancelledBeforeStartText,
27
+ indeterminateOutcomeText: () => indeterminateOutcomeText,
26
28
  isAbortError: () => isAbortError,
27
29
  isBillingError: () => isBillingError,
28
30
  isContextOverflow: () => isContextOverflow,
@@ -61,6 +63,60 @@ function isLocalBackendUrl(baseUrl) {
61
63
  return false;
62
64
  }
63
65
 
66
+ // src/output-ceiling.ts
67
+ var TTL_MS = 24 * 60 * 60 * 1e3;
68
+ var MIN_PLAUSIBLE_CEILING = 256;
69
+ var MAX_PLAUSIBLE_CEILING = 1e7;
70
+ var ceilings = /* @__PURE__ */ new Map();
71
+ function outputRouteKey(route) {
72
+ return `${route.provider}\0${route.baseUrl ?? "default"}\0${route.model}`;
73
+ }
74
+ function parseOutputTokenCeiling(err) {
75
+ if (!(err instanceof Error)) return null;
76
+ const msg = err.message;
77
+ if (!/max_tokens|max output tokens|output tokens|completion tokens/i.test(msg)) return null;
78
+ if (err.statusCode === 402) return null;
79
+ if (/credit|billing|payment|insufficient|balance|quota/i.test(msg)) return null;
80
+ const patterns = [
81
+ // Anthropic: "max_tokens: 100000 > 64000, which is the maximum allowed…"
82
+ /max_tokens:\s*\d+\s*>\s*(\d+)/i,
83
+ // OpenAI: "…this model supports at most 16384 completion tokens"
84
+ /(?:at most|maximum of|limit of)\s+([\d,_]+)\s*(?:output|completion)?\s*tokens/i,
85
+ // Generic: "max output tokens is 8192" / "max_tokens must be <= 4096"
86
+ /(?:max_tokens|max output tokens)\b[^\d]{0,30}?([\d,_]+)/i
87
+ ];
88
+ for (const pattern of patterns) {
89
+ const raw = msg.match(pattern)?.[1];
90
+ if (raw === void 0) continue;
91
+ const limit = Number(raw.replace(/[,_]/g, ""));
92
+ if (Number.isFinite(limit) && limit >= MIN_PLAUSIBLE_CEILING && limit <= MAX_PLAUSIBLE_CEILING) {
93
+ return limit;
94
+ }
95
+ }
96
+ return null;
97
+ }
98
+ function rememberOutputCeiling(key, limit) {
99
+ const known = outputTokenCeiling(key);
100
+ ceilings.set(key, {
101
+ limit: known === void 0 ? limit : Math.min(known, limit),
102
+ expiresAt: Date.now() + TTL_MS
103
+ });
104
+ }
105
+ function outputTokenCeiling(key) {
106
+ const entry = ceilings.get(key);
107
+ if (!entry) return void 0;
108
+ if (entry.expiresAt <= Date.now()) {
109
+ ceilings.delete(key);
110
+ return void 0;
111
+ }
112
+ return entry.limit;
113
+ }
114
+ function clampOutputTokens(key, requested) {
115
+ const ceiling = outputTokenCeiling(key);
116
+ if (ceiling === void 0) return requested;
117
+ return requested === void 0 ? ceiling : Math.min(requested, ceiling);
118
+ }
119
+
64
120
  // src/agent-loop.ts
65
121
  var DEFAULT_MAX_TURNS = 300;
66
122
  var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
@@ -345,6 +401,16 @@ async function* agentLoop(messages, options) {
345
401
  const MAX_OUTPUT_CONTINUATIONS = 2;
346
402
  const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
347
403
  let maxTokensContinuations = 0;
404
+ let providerCalls = 0;
405
+ let nonStreamingCalls = 0;
406
+ let warnedNonStreaming = false;
407
+ const MAX_OUTPUT_CEILING_RETRIES = 1;
408
+ let outputCeilingRetries = 0;
409
+ const ceilingKey = outputRouteKey({
410
+ provider: options.provider,
411
+ model: options.model,
412
+ baseUrl: options.baseUrl
413
+ });
348
414
  const OVERLOAD_BASE_DELAY_MS = 2e3;
349
415
  const OVERLOAD_MAX_DELAY_MS = 3e4;
350
416
  const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
@@ -474,6 +540,18 @@ async function* agentLoop(messages, options) {
474
540
  let streamIterator = null;
475
541
  try {
476
542
  diag("stream_call", { nonStreaming: useNonStreamingFallback });
543
+ providerCalls++;
544
+ if (useNonStreamingFallback) nonStreamingCalls++;
545
+ if (!warnedNonStreaming && nonStreamingCalls >= 3) {
546
+ warnedNonStreaming = true;
547
+ diag("non_streaming_session", {
548
+ nonStreamingCalls,
549
+ providerCalls,
550
+ provider: options.provider,
551
+ model: options.model,
552
+ impact: "streaming is disabled for this session after repeated stalls: failed turns are re-billed in full instead of resuming from partial output, and replies appear only when complete"
553
+ });
554
+ }
477
555
  streamCallStart = Date.now();
478
556
  providerAttemptStartedAt = streamCallStart;
479
557
  let liveApiKey = options.apiKey;
@@ -499,7 +577,9 @@ async function* agentLoop(messages, options) {
499
577
  serverTools: options.serverTools,
500
578
  toolChoice: options.toolChoice,
501
579
  webSearch: options.webSearch,
502
- maxTokens: options.maxTokens,
580
+ // Clamped to whatever ceiling this route has already rejected us for
581
+ // (identity when nothing has been learned).
582
+ maxTokens: clampOutputTokens(ceilingKey, options.maxTokens),
503
583
  temperature: options.temperature,
504
584
  thinking: options.thinking,
505
585
  apiKey: liveApiKey,
@@ -656,6 +736,29 @@ async function* agentLoop(messages, options) {
656
736
  });
657
737
  throw err;
658
738
  }
739
+ const statedCeiling = parseOutputTokenCeiling(err);
740
+ if (statedCeiling !== null) {
741
+ rememberOutputCeiling(ceilingKey, statedCeiling);
742
+ diag("output_ceiling_learned", {
743
+ ceiling: statedCeiling,
744
+ requested: options.maxTokens,
745
+ provider: options.provider,
746
+ model: options.model
747
+ });
748
+ if (outputCeilingRetries < MAX_OUTPUT_CEILING_RETRIES) {
749
+ outputCeilingRetries++;
750
+ yield {
751
+ type: "retry",
752
+ reason: "provider_error",
753
+ attempt: outputCeilingRetries,
754
+ maxAttempts: MAX_OUTPUT_CEILING_RETRIES,
755
+ delayMs: 0,
756
+ silent: true
757
+ };
758
+ turn--;
759
+ continue;
760
+ }
761
+ }
659
762
  if (isContextOverflow(err)) {
660
763
  const overflowDetails = extractContextOverflowDetails(err);
661
764
  diag("context_overflow_detected", {
@@ -951,6 +1054,7 @@ async function* agentLoop(messages, options) {
951
1054
  continue;
952
1055
  }
953
1056
  }
1057
+ const emptyExhausted = !hasActionableContent;
954
1058
  emptyResponseRetries = 0;
955
1059
  useNonStreamingFallback = false;
956
1060
  totalUsage.inputTokens += response.usage.inputTokens;
@@ -964,9 +1068,11 @@ async function* agentLoop(messages, options) {
964
1068
  if (response.usage.cacheWrite) {
965
1069
  totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
966
1070
  }
967
- messages.push(response.message);
968
- latestProviderUsage = response.usage;
969
- usageAnchorIndex = messages.length - 1;
1071
+ if (!emptyExhausted) {
1072
+ messages.push(response.message);
1073
+ latestProviderUsage = response.usage;
1074
+ usageAnchorIndex = messages.length - 1;
1075
+ }
970
1076
  const completedAt = Date.now();
971
1077
  const outputTokensPerSecond = providerDurationMs > 0 && response.usage.outputTokens > 0 ? response.usage.outputTokens / (providerDurationMs / 1e3) : void 0;
972
1078
  const timing = {
@@ -998,8 +1104,15 @@ async function* agentLoop(messages, options) {
998
1104
  }
999
1105
  consecutivePauses = 0;
1000
1106
  const allToolCalls = extractToolCalls(response.message.content);
1001
- if (response.stopReason !== "tool_use" && allToolCalls.length === 0) {
1002
- if (response.stopReason === "max_tokens") {
1107
+ if (emptyExhausted || response.stopReason !== "tool_use" && allToolCalls.length === 0) {
1108
+ if (emptyExhausted) {
1109
+ diag("empty_response_exhausted", {
1110
+ provider: options.provider,
1111
+ model: options.model,
1112
+ stopReason: response.stopReason
1113
+ });
1114
+ yield { type: "truncated", reason: "empty_response", continued: false };
1115
+ } else if (response.stopReason === "max_tokens") {
1003
1116
  if (maxTokensContinuations < MAX_OUTPUT_CONTINUATIONS) {
1004
1117
  maxTokensContinuations++;
1005
1118
  diag("max_tokens_continuation", {
@@ -1221,6 +1334,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1221
1334
  let resultContent;
1222
1335
  let details;
1223
1336
  let isError = false;
1337
+ let invalidArgAttempt;
1224
1338
  const tool = options.toolMap.get(toolCall.name);
1225
1339
  if (!tool) {
1226
1340
  resultContent = `Unknown tool: ${toolCall.name}`;
@@ -1258,6 +1372,7 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1258
1372
  const failureKey = `${toolCall.name}:${prettyError}`;
1259
1373
  const failureCount = (options.invalidToolArgumentCounts.get(failureKey) ?? 0) + 1;
1260
1374
  options.invalidToolArgumentCounts.set(failureKey, failureCount);
1375
+ invalidArgAttempt = failureCount;
1261
1376
  resultContent = `Invalid arguments for tool \`${toolCall.name}\`:
1262
1377
  ` + prettyError + "\nRe-issue the call with each field as the correct type.";
1263
1378
  if (failureCount >= 3) {
@@ -1274,6 +1389,8 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1274
1389
  toolCall.name
1275
1390
  );
1276
1391
  }
1392
+ } else if (options.signal?.aborted && isAbortError(err)) {
1393
+ resultContent = indeterminateOutcomeText(toolCall.name);
1277
1394
  } else {
1278
1395
  resultContent = (0, import_ai.redactValue)(err instanceof Error ? err.message : String(err));
1279
1396
  }
@@ -1288,7 +1405,8 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1288
1405
  result: toolResultPreview(resultContent),
1289
1406
  details,
1290
1407
  isError,
1291
- durationMs
1408
+ durationMs,
1409
+ ...invalidArgAttempt === void 0 ? {} : { invalidArgAttempt }
1292
1410
  });
1293
1411
  return { toolCallId: toolCall.id, content: resultContent, isError };
1294
1412
  }
@@ -1296,6 +1414,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1296
1414
  const eventStream = new import_ai.EventStream();
1297
1415
  const state = { finalized: false };
1298
1416
  const resultsById = /* @__PURE__ */ new Map();
1417
+ const dispatchedIds = /* @__PURE__ */ new Set();
1299
1418
  const abortHandler = () => eventStream.abort(new Error("aborted"));
1300
1419
  options.signal?.addEventListener("abort", abortHandler, { once: true });
1301
1420
  const phases = [];
@@ -1320,6 +1439,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1320
1439
  for (const phase of phases) {
1321
1440
  if (options.signal?.aborted) break;
1322
1441
  if (phase.sequential) {
1442
+ dispatchedIds.add(phase.sequential.id);
1323
1443
  const record = await executeSingleToolCall(
1324
1444
  phase.sequential,
1325
1445
  options,
@@ -1327,6 +1447,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1327
1447
  );
1328
1448
  resultsById.set(record.toolCallId, record);
1329
1449
  } else if (phase.parallel.length === 1) {
1450
+ dispatchedIds.add(phase.parallel[0].id);
1330
1451
  const record = await executeSingleToolCall(
1331
1452
  phase.parallel[0],
1332
1453
  options,
@@ -1336,6 +1457,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1336
1457
  } else {
1337
1458
  await Promise.all(
1338
1459
  phase.parallel.map(async (toolCall) => {
1460
+ dispatchedIds.add(toolCall.id);
1339
1461
  const record = await executeSingleToolCall(
1340
1462
  toolCall,
1341
1463
  options,
@@ -1366,7 +1488,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1366
1488
  options.signal?.removeEventListener("abort", abortHandler);
1367
1489
  state.finalized = true;
1368
1490
  }
1369
- const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1491
+ const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds);
1370
1492
  capToolResults(toolResults, options.maxToolResultChars);
1371
1493
  capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1372
1494
  return { toolResults, aborted };
@@ -1375,10 +1497,12 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
1375
1497
  const eventStream = new import_ai.EventStream();
1376
1498
  const state = { finalized: false };
1377
1499
  const resultsById = /* @__PURE__ */ new Map();
1500
+ const dispatchedIds = /* @__PURE__ */ new Set();
1378
1501
  const abortHandler = () => eventStream.abort(new Error("aborted"));
1379
1502
  options.signal?.addEventListener("abort", abortHandler, { once: true });
1380
1503
  Promise.all(
1381
1504
  toolCalls.map(async (toolCall) => {
1505
+ dispatchedIds.add(toolCall.id);
1382
1506
  const record = await executeSingleToolCall(
1383
1507
  toolCall,
1384
1508
  options,
@@ -1406,12 +1530,18 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
1406
1530
  options.signal?.removeEventListener("abort", abortHandler);
1407
1531
  state.finalized = true;
1408
1532
  }
1409
- const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1533
+ const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds);
1410
1534
  capToolResults(toolResults, options.maxToolResultChars);
1411
1535
  capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1412
1536
  return { toolResults, aborted };
1413
1537
  }
1414
- function buildToolResults(initialToolResults, toolCalls, resultsById) {
1538
+ function cancelledBeforeStartText(name) {
1539
+ return `\`${name}\` was cancelled before it started, so it had no effect. Safe to retry.`;
1540
+ }
1541
+ function indeterminateOutcomeText(name) {
1542
+ return `\`${name}\` started running and was cut off before it reported back, so its outcome is UNKNOWN \u2014 it may have completed. Check the real state (re-read the file, re-run a status command) before retrying it, and do not tell the user it did not happen.`;
1543
+ }
1544
+ function buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds) {
1415
1545
  const toolResults = [...initialToolResults];
1416
1546
  for (const toolCall of toolCalls) {
1417
1547
  const result = resultsById.get(toolCall.id);
@@ -1423,10 +1553,11 @@ function buildToolResults(initialToolResults, toolCalls, resultsById) {
1423
1553
  isError: result.isError || void 0
1424
1554
  });
1425
1555
  } else {
1556
+ const dispatched = dispatchedIds?.has(toolCall.id) ?? true;
1426
1557
  toolResults.push({
1427
1558
  type: "tool_result",
1428
1559
  toolCallId: toolCall.id,
1429
- content: "Tool execution was aborted.",
1560
+ content: dispatched ? indeterminateOutcomeText(toolCall.name) : cancelledBeforeStartText(toolCall.name),
1430
1561
  isError: true
1431
1562
  });
1432
1563
  }
@@ -1574,32 +1705,22 @@ function repairToolPairingAdjacent(messages) {
1574
1705
  const msg = messages[i];
1575
1706
  if (msg.role !== "assistant") continue;
1576
1707
  if (typeof msg.content === "string" || !Array.isArray(msg.content)) continue;
1577
- const toolCallIds = msg.content.filter((p) => p.type === "tool_call").map((p) => p.id);
1578
- if (toolCallIds.length === 0) continue;
1708
+ const orphanCalls = msg.content.filter((p) => p.type === "tool_call").map((p) => p);
1709
+ if (orphanCalls.length === 0) continue;
1710
+ const repaired = (call) => ({
1711
+ type: "tool_result",
1712
+ toolCallId: call.id,
1713
+ content: indeterminateOutcomeText(call.name),
1714
+ isError: true
1715
+ });
1579
1716
  const next = messages[i + 1];
1580
1717
  if (next?.role === "tool" && Array.isArray(next.content)) {
1581
1718
  const existingIds = new Set(next.content.map((r) => r.toolCallId));
1582
- const missing = toolCallIds.filter((id) => !existingIds.has(id));
1583
- if (missing.length > 0) {
1584
- for (const id of missing) {
1585
- next.content.push({
1586
- type: "tool_result",
1587
- toolCallId: id,
1588
- content: "Tool execution was interrupted.",
1589
- isError: true
1590
- });
1591
- }
1719
+ for (const call of orphanCalls) {
1720
+ if (!existingIds.has(call.id)) next.content.push(repaired(call));
1592
1721
  }
1593
1722
  } else {
1594
- messages.splice(i + 1, 0, {
1595
- role: "tool",
1596
- content: toolCallIds.map((id) => ({
1597
- type: "tool_result",
1598
- toolCallId: id,
1599
- content: "Tool execution was interrupted.",
1600
- isError: true
1601
- }))
1602
- });
1723
+ messages.splice(i + 1, 0, { role: "tool", content: orphanCalls.map(repaired) });
1603
1724
  }
1604
1725
  }
1605
1726
  const toolCallIdSet = /* @__PURE__ */ new Set();
@@ -1759,6 +1880,8 @@ var Agent = class {
1759
1880
  Agent,
1760
1881
  AgentStream,
1761
1882
  agentLoop,
1883
+ cancelledBeforeStartText,
1884
+ indeterminateOutcomeText,
1762
1885
  isAbortError,
1763
1886
  isBillingError,
1764
1887
  isContextOverflow,