@prestyj/agent 5.24.0 → 5.25.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -374,6 +374,35 @@ function closeIterator(iterator) {
374
374
  Promise.resolve(iterator.return()).catch(() => {
375
375
  });
376
376
  }
377
+ var PREFILL_TIMEOUT_MS_PER_1K_TOKENS = 640;
378
+ var STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
379
+ var STREAM_FIRST_EVENT_TIMEOUT_MAX_MS = 18e4;
380
+ var STREAM_FIRST_EVENT_TIMEOUT_SCALE_MIN_TOKENS = 2e4;
381
+ function scaledFirstEventTimeoutMs(promptTokens) {
382
+ if (promptTokens < STREAM_FIRST_EVENT_TIMEOUT_SCALE_MIN_TOKENS) return null;
383
+ return Math.min(
384
+ STREAM_FIRST_EVENT_TIMEOUT_MAX_MS,
385
+ STREAM_FIRST_EVENT_TIMEOUT_MS + promptTokens / 1e3 * PREFILL_TIMEOUT_MS_PER_1K_TOKENS
386
+ );
387
+ }
388
+ var CACHE_HEALTH_MIN_PROMPT_TOKENS = 4e4;
389
+ var CACHE_HEALTH_LOW_RATIO = 0.5;
390
+ function assessCacheHealth(usage) {
391
+ const input = usage.inputTokens ?? 0;
392
+ const cacheRead = usage.cacheRead ?? 0;
393
+ const cacheWrite = usage.cacheWrite ?? 0;
394
+ const promptTokens = input + cacheRead + cacheWrite;
395
+ if (promptTokens < CACHE_HEALTH_MIN_PROMPT_TOKENS) {
396
+ return { ratio: null, promptTokens, cacheRead, low: false };
397
+ }
398
+ const ratio = cacheRead / promptTokens;
399
+ return {
400
+ ratio,
401
+ promptTokens,
402
+ cacheRead,
403
+ low: ratio < CACHE_HEALTH_LOW_RATIO
404
+ };
405
+ }
377
406
  async function* agentLoop(messages, options) {
378
407
  const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
379
408
  let effectiveMaxTurns = maxTurns;
@@ -415,6 +444,7 @@ async function* agentLoop(messages, options) {
415
444
  let providerCalls = 0;
416
445
  let nonStreamingCalls = 0;
417
446
  let warnedNonStreaming = false;
447
+ let warnedPromptCacheMiss = false;
418
448
  const MAX_OUTPUT_CEILING_RETRIES = 1;
419
449
  let outputCeilingRetries = 0;
420
450
  const ceilingKey = outputRouteKey({
@@ -424,7 +454,6 @@ async function* agentLoop(messages, options) {
424
454
  });
425
455
  const OVERLOAD_BASE_DELAY_MS = 2e3;
426
456
  const OVERLOAD_MAX_DELAY_MS = 3e4;
427
- const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
428
457
  const STREAM_IDLE_TIMEOUT_MS = 9e4;
429
458
  const STREAM_HARD_TIMEOUT_MS = 9e4;
430
459
  const STREAM_OUTPUT_HARD_TIMEOUT_MS = 3e5;
@@ -433,8 +462,8 @@ async function* agentLoop(messages, options) {
433
462
  const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
434
463
  const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
435
464
  const localBackend = isLocalBackendUrl(options.baseUrl);
436
- const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
437
- const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
465
+ const baseFirstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
466
+ const baseHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
438
467
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
439
468
  const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
440
469
  let logicalTurnStartedAt = 0;
@@ -446,17 +475,28 @@ async function* agentLoop(messages, options) {
446
475
  turn++;
447
476
  if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
448
477
  toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
449
- if (_diagFn) {
450
- let msgChars = 0;
451
- for (const m of messages) {
452
- if (typeof m.content === "string") msgChars += m.content.length;
453
- else if (Array.isArray(m.content)) {
454
- for (const p of m.content) {
455
- if ("text" in p && typeof p.text === "string") msgChars += p.text.length;
456
- if ("content" in p && typeof p.content === "string") msgChars += p.content.length;
457
- }
478
+ let msgChars = 0;
479
+ for (const m of messages) {
480
+ if (typeof m.content === "string") msgChars += m.content.length;
481
+ else if (Array.isArray(m.content)) {
482
+ for (const p of m.content) {
483
+ if ("text" in p && typeof p.text === "string") msgChars += p.text.length;
484
+ if ("content" in p && typeof p.content === "string") msgChars += p.content.length;
458
485
  }
459
486
  }
487
+ }
488
+ let firstEventTimeoutMs;
489
+ let initialHardTimeoutMs;
490
+ if (baseFirstEventTimeoutMs === STREAM_FIRST_EVENT_TIMEOUT_MS) {
491
+ const promptTokens = Math.ceil(msgChars / 3);
492
+ const scaled = scaledFirstEventTimeoutMs(promptTokens);
493
+ firstEventTimeoutMs = scaled ?? baseFirstEventTimeoutMs;
494
+ initialHardTimeoutMs = Math.max(baseHardTimeoutMs, firstEventTimeoutMs + 3e4);
495
+ } else {
496
+ firstEventTimeoutMs = baseFirstEventTimeoutMs;
497
+ initialHardTimeoutMs = baseHardTimeoutMs;
498
+ }
499
+ if (_diagFn) {
460
500
  diag("turn_start", {
461
501
  turn,
462
502
  messages: messages.length,
@@ -1079,6 +1119,27 @@ async function* agentLoop(messages, options) {
1079
1119
  if (response.usage.cacheWrite) {
1080
1120
  totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
1081
1121
  }
1122
+ const cacheHealth = assessCacheHealth(response.usage);
1123
+ if (cacheHealth.ratio !== null) {
1124
+ diag("cache_health", {
1125
+ promptTokens: cacheHealth.promptTokens,
1126
+ cacheRead: cacheHealth.cacheRead,
1127
+ ratio: Math.round(cacheHealth.ratio * 100) / 100,
1128
+ provider: options.provider,
1129
+ model: options.model
1130
+ });
1131
+ if (cacheHealth.low && !warnedPromptCacheMiss) {
1132
+ warnedPromptCacheMiss = true;
1133
+ diag("prompt_cache_miss", {
1134
+ promptTokens: cacheHealth.promptTokens,
1135
+ cacheRead: cacheHealth.cacheRead,
1136
+ ratio: Math.round(cacheHealth.ratio * 100) / 100,
1137
+ provider: options.provider,
1138
+ model: options.model,
1139
+ impact: "large prompts are consistently served mostly uncached \u2014 every turn re-prefills the whole context; if this persists, lower the provider's compaction latency cap (resolveCompactionPolicy)"
1140
+ });
1141
+ }
1142
+ }
1082
1143
  if (!emptyExhausted) {
1083
1144
  messages.push(response.message);
1084
1145
  latestProviderUsage = response.usage;