@prestyj/agent 5.13.0 → 5.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -23,6 +23,8 @@ __export(index_exports, {
23
23
  Agent: () => Agent,
24
24
  AgentStream: () => AgentStream,
25
25
  agentLoop: () => agentLoop,
26
+ cancelledBeforeStartText: () => cancelledBeforeStartText,
27
+ indeterminateOutcomeText: () => indeterminateOutcomeText,
26
28
  isAbortError: () => isAbortError,
27
29
  isBillingError: () => isBillingError,
28
30
  isContextOverflow: () => isContextOverflow,
@@ -61,6 +63,60 @@ function isLocalBackendUrl(baseUrl) {
61
63
  return false;
62
64
  }
63
65
 
66
+ // src/output-ceiling.ts
67
+ var TTL_MS = 24 * 60 * 60 * 1e3;
68
+ var MIN_PLAUSIBLE_CEILING = 256;
69
+ var MAX_PLAUSIBLE_CEILING = 1e7;
70
+ var ceilings = /* @__PURE__ */ new Map();
71
+ function outputRouteKey(route) {
72
+ return `${route.provider}\0${route.baseUrl ?? "default"}\0${route.model}`;
73
+ }
74
+ function parseOutputTokenCeiling(err) {
75
+ if (!(err instanceof Error)) return null;
76
+ const msg = err.message;
77
+ if (!/max_tokens|max output tokens|output tokens|completion tokens/i.test(msg)) return null;
78
+ if (err.statusCode === 402) return null;
79
+ if (/credit|billing|payment|insufficient|balance|quota/i.test(msg)) return null;
80
+ const patterns = [
81
+ // Anthropic: "max_tokens: 100000 > 64000, which is the maximum allowed…"
82
+ /max_tokens:\s*\d+\s*>\s*(\d+)/i,
83
+ // OpenAI: "…this model supports at most 16384 completion tokens"
84
+ /(?:at most|maximum of|limit of)\s+([\d,_]+)\s*(?:output|completion)?\s*tokens/i,
85
+ // Generic: "max output tokens is 8192" / "max_tokens must be <= 4096"
86
+ /(?:max_tokens|max output tokens)\b[^\d]{0,30}?([\d,_]+)/i
87
+ ];
88
+ for (const pattern of patterns) {
89
+ const raw = msg.match(pattern)?.[1];
90
+ if (raw === void 0) continue;
91
+ const limit = Number(raw.replace(/[,_]/g, ""));
92
+ if (Number.isFinite(limit) && limit >= MIN_PLAUSIBLE_CEILING && limit <= MAX_PLAUSIBLE_CEILING) {
93
+ return limit;
94
+ }
95
+ }
96
+ return null;
97
+ }
98
+ function rememberOutputCeiling(key, limit) {
99
+ const known = outputTokenCeiling(key);
100
+ ceilings.set(key, {
101
+ limit: known === void 0 ? limit : Math.min(known, limit),
102
+ expiresAt: Date.now() + TTL_MS
103
+ });
104
+ }
105
+ function outputTokenCeiling(key) {
106
+ const entry = ceilings.get(key);
107
+ if (!entry) return void 0;
108
+ if (entry.expiresAt <= Date.now()) {
109
+ ceilings.delete(key);
110
+ return void 0;
111
+ }
112
+ return entry.limit;
113
+ }
114
+ function clampOutputTokens(key, requested) {
115
+ const ceiling = outputTokenCeiling(key);
116
+ if (ceiling === void 0) return requested;
117
+ return requested === void 0 ? ceiling : Math.min(requested, ceiling);
118
+ }
119
+
64
120
  // src/agent-loop.ts
65
121
  var DEFAULT_MAX_TURNS = 300;
66
122
  var DEFAULT_TOOL_TIMEOUT_MS = 3e5;
@@ -345,6 +401,16 @@ async function* agentLoop(messages, options) {
345
401
  const MAX_OUTPUT_CONTINUATIONS = 2;
346
402
  const MAX_TOKENS_CONTINUATION_PROMPT = "[Your previous response hit the output-token limit and was cut off. The text above is what was already delivered to the user. Continue exactly from where it stopped \u2014 do not repeat or restart it.]";
347
403
  let maxTokensContinuations = 0;
404
+ let providerCalls = 0;
405
+ let nonStreamingCalls = 0;
406
+ let warnedNonStreaming = false;
407
+ const MAX_OUTPUT_CEILING_RETRIES = 1;
408
+ let outputCeilingRetries = 0;
409
+ const ceilingKey = outputRouteKey({
410
+ provider: options.provider,
411
+ model: options.model,
412
+ baseUrl: options.baseUrl
413
+ });
348
414
  const OVERLOAD_BASE_DELAY_MS = 2e3;
349
415
  const OVERLOAD_MAX_DELAY_MS = 3e4;
350
416
  const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
@@ -474,6 +540,18 @@ async function* agentLoop(messages, options) {
474
540
  let streamIterator = null;
475
541
  try {
476
542
  diag("stream_call", { nonStreaming: useNonStreamingFallback });
543
+ providerCalls++;
544
+ if (useNonStreamingFallback) nonStreamingCalls++;
545
+ if (!warnedNonStreaming && nonStreamingCalls >= 3) {
546
+ warnedNonStreaming = true;
547
+ diag("non_streaming_session", {
548
+ nonStreamingCalls,
549
+ providerCalls,
550
+ provider: options.provider,
551
+ model: options.model,
552
+ impact: "streaming is disabled for this session after repeated stalls: failed turns are re-billed in full instead of resuming from partial output, and replies appear only when complete"
553
+ });
554
+ }
477
555
  streamCallStart = Date.now();
478
556
  providerAttemptStartedAt = streamCallStart;
479
557
  let liveApiKey = options.apiKey;
@@ -499,7 +577,9 @@ async function* agentLoop(messages, options) {
499
577
  serverTools: options.serverTools,
500
578
  toolChoice: options.toolChoice,
501
579
  webSearch: options.webSearch,
502
- maxTokens: options.maxTokens,
580
+ // Clamped to whatever ceiling this route has already rejected us for
581
+ // (identity when nothing has been learned).
582
+ maxTokens: clampOutputTokens(ceilingKey, options.maxTokens),
503
583
  temperature: options.temperature,
504
584
  thinking: options.thinking,
505
585
  apiKey: liveApiKey,
@@ -656,6 +736,29 @@ async function* agentLoop(messages, options) {
656
736
  });
657
737
  throw err;
658
738
  }
739
+ const statedCeiling = parseOutputTokenCeiling(err);
740
+ if (statedCeiling !== null) {
741
+ rememberOutputCeiling(ceilingKey, statedCeiling);
742
+ diag("output_ceiling_learned", {
743
+ ceiling: statedCeiling,
744
+ requested: options.maxTokens,
745
+ provider: options.provider,
746
+ model: options.model
747
+ });
748
+ if (outputCeilingRetries < MAX_OUTPUT_CEILING_RETRIES) {
749
+ outputCeilingRetries++;
750
+ yield {
751
+ type: "retry",
752
+ reason: "provider_error",
753
+ attempt: outputCeilingRetries,
754
+ maxAttempts: MAX_OUTPUT_CEILING_RETRIES,
755
+ delayMs: 0,
756
+ silent: true
757
+ };
758
+ turn--;
759
+ continue;
760
+ }
761
+ }
659
762
  if (isContextOverflow(err)) {
660
763
  const overflowDetails = extractContextOverflowDetails(err);
661
764
  diag("context_overflow_detected", {
@@ -1286,6 +1389,8 @@ async function executeSingleToolCall(toolCall, options, pushEvent) {
1286
1389
  toolCall.name
1287
1390
  );
1288
1391
  }
1392
+ } else if (options.signal?.aborted && isAbortError(err)) {
1393
+ resultContent = indeterminateOutcomeText(toolCall.name);
1289
1394
  } else {
1290
1395
  resultContent = (0, import_ai.redactValue)(err instanceof Error ? err.message : String(err));
1291
1396
  }
@@ -1309,6 +1414,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1309
1414
  const eventStream = new import_ai.EventStream();
1310
1415
  const state = { finalized: false };
1311
1416
  const resultsById = /* @__PURE__ */ new Map();
1417
+ const dispatchedIds = /* @__PURE__ */ new Set();
1312
1418
  const abortHandler = () => eventStream.abort(new Error("aborted"));
1313
1419
  options.signal?.addEventListener("abort", abortHandler, { once: true });
1314
1420
  const phases = [];
@@ -1333,6 +1439,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1333
1439
  for (const phase of phases) {
1334
1440
  if (options.signal?.aborted) break;
1335
1441
  if (phase.sequential) {
1442
+ dispatchedIds.add(phase.sequential.id);
1336
1443
  const record = await executeSingleToolCall(
1337
1444
  phase.sequential,
1338
1445
  options,
@@ -1340,6 +1447,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1340
1447
  );
1341
1448
  resultsById.set(record.toolCallId, record);
1342
1449
  } else if (phase.parallel.length === 1) {
1450
+ dispatchedIds.add(phase.parallel[0].id);
1343
1451
  const record = await executeSingleToolCall(
1344
1452
  phase.parallel[0],
1345
1453
  options,
@@ -1349,6 +1457,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1349
1457
  } else {
1350
1458
  await Promise.all(
1351
1459
  phase.parallel.map(async (toolCall) => {
1460
+ dispatchedIds.add(toolCall.id);
1352
1461
  const record = await executeSingleToolCall(
1353
1462
  toolCall,
1354
1463
  options,
@@ -1379,7 +1488,7 @@ async function* executeToolCallsMixed(toolCalls, initialToolResults, options) {
1379
1488
  options.signal?.removeEventListener("abort", abortHandler);
1380
1489
  state.finalized = true;
1381
1490
  }
1382
- const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1491
+ const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds);
1383
1492
  capToolResults(toolResults, options.maxToolResultChars);
1384
1493
  capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1385
1494
  return { toolResults, aborted };
@@ -1388,10 +1497,12 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
1388
1497
  const eventStream = new import_ai.EventStream();
1389
1498
  const state = { finalized: false };
1390
1499
  const resultsById = /* @__PURE__ */ new Map();
1500
+ const dispatchedIds = /* @__PURE__ */ new Set();
1391
1501
  const abortHandler = () => eventStream.abort(new Error("aborted"));
1392
1502
  options.signal?.addEventListener("abort", abortHandler, { once: true });
1393
1503
  Promise.all(
1394
1504
  toolCalls.map(async (toolCall) => {
1505
+ dispatchedIds.add(toolCall.id);
1395
1506
  const record = await executeSingleToolCall(
1396
1507
  toolCall,
1397
1508
  options,
@@ -1419,12 +1530,18 @@ async function* executeToolCallsParallel(toolCalls, initialToolResults, options)
1419
1530
  options.signal?.removeEventListener("abort", abortHandler);
1420
1531
  state.finalized = true;
1421
1532
  }
1422
- const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById);
1533
+ const toolResults = buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds);
1423
1534
  capToolResults(toolResults, options.maxToolResultChars);
1424
1535
  capTurnToolResults(toolResults, options.maxTurnToolResultChars);
1425
1536
  return { toolResults, aborted };
1426
1537
  }
1427
- function buildToolResults(initialToolResults, toolCalls, resultsById) {
1538
+ function cancelledBeforeStartText(name) {
1539
+ return `\`${name}\` was cancelled before it started, so it had no effect. Safe to retry.`;
1540
+ }
1541
+ function indeterminateOutcomeText(name) {
1542
+ return `\`${name}\` started running and was cut off before it reported back, so its outcome is UNKNOWN \u2014 it may have completed. Check the real state (re-read the file, re-run a status command) before retrying it, and do not tell the user it did not happen.`;
1543
+ }
1544
+ function buildToolResults(initialToolResults, toolCalls, resultsById, dispatchedIds) {
1428
1545
  const toolResults = [...initialToolResults];
1429
1546
  for (const toolCall of toolCalls) {
1430
1547
  const result = resultsById.get(toolCall.id);
@@ -1436,10 +1553,11 @@ function buildToolResults(initialToolResults, toolCalls, resultsById) {
1436
1553
  isError: result.isError || void 0
1437
1554
  });
1438
1555
  } else {
1556
+ const dispatched = dispatchedIds?.has(toolCall.id) ?? true;
1439
1557
  toolResults.push({
1440
1558
  type: "tool_result",
1441
1559
  toolCallId: toolCall.id,
1442
- content: "Tool execution was aborted.",
1560
+ content: dispatched ? indeterminateOutcomeText(toolCall.name) : cancelledBeforeStartText(toolCall.name),
1443
1561
  isError: true
1444
1562
  });
1445
1563
  }
@@ -1587,32 +1705,22 @@ function repairToolPairingAdjacent(messages) {
1587
1705
  const msg = messages[i];
1588
1706
  if (msg.role !== "assistant") continue;
1589
1707
  if (typeof msg.content === "string" || !Array.isArray(msg.content)) continue;
1590
- const toolCallIds = msg.content.filter((p) => p.type === "tool_call").map((p) => p.id);
1591
- if (toolCallIds.length === 0) continue;
1708
+ const orphanCalls = msg.content.filter((p) => p.type === "tool_call").map((p) => p);
1709
+ if (orphanCalls.length === 0) continue;
1710
+ const repaired = (call) => ({
1711
+ type: "tool_result",
1712
+ toolCallId: call.id,
1713
+ content: indeterminateOutcomeText(call.name),
1714
+ isError: true
1715
+ });
1592
1716
  const next = messages[i + 1];
1593
1717
  if (next?.role === "tool" && Array.isArray(next.content)) {
1594
1718
  const existingIds = new Set(next.content.map((r) => r.toolCallId));
1595
- const missing = toolCallIds.filter((id) => !existingIds.has(id));
1596
- if (missing.length > 0) {
1597
- for (const id of missing) {
1598
- next.content.push({
1599
- type: "tool_result",
1600
- toolCallId: id,
1601
- content: "Tool execution was interrupted.",
1602
- isError: true
1603
- });
1604
- }
1719
+ for (const call of orphanCalls) {
1720
+ if (!existingIds.has(call.id)) next.content.push(repaired(call));
1605
1721
  }
1606
1722
  } else {
1607
- messages.splice(i + 1, 0, {
1608
- role: "tool",
1609
- content: toolCallIds.map((id) => ({
1610
- type: "tool_result",
1611
- toolCallId: id,
1612
- content: "Tool execution was interrupted.",
1613
- isError: true
1614
- }))
1615
- });
1723
+ messages.splice(i + 1, 0, { role: "tool", content: orphanCalls.map(repaired) });
1616
1724
  }
1617
1725
  }
1618
1726
  const toolCallIdSet = /* @__PURE__ */ new Set();
@@ -1772,6 +1880,8 @@ var Agent = class {
1772
1880
  Agent,
1773
1881
  AgentStream,
1774
1882
  agentLoop,
1883
+ cancelledBeforeStartText,
1884
+ indeterminateOutcomeText,
1775
1885
  isAbortError,
1776
1886
  isBillingError,
1777
1887
  isContextOverflow,