@prestyj/agent 5.24.0 → 5.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -346,6 +346,35 @@ function closeIterator(iterator) {
346
346
  Promise.resolve(iterator.return()).catch(() => {
347
347
  });
348
348
  }
349
+ var PREFILL_TIMEOUT_MS_PER_1K_TOKENS = 640;
350
+ var STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
351
+ var STREAM_FIRST_EVENT_TIMEOUT_MAX_MS = 18e4;
352
+ var STREAM_FIRST_EVENT_TIMEOUT_SCALE_MIN_TOKENS = 2e4;
353
+ function scaledFirstEventTimeoutMs(promptTokens) {
354
+ if (promptTokens < STREAM_FIRST_EVENT_TIMEOUT_SCALE_MIN_TOKENS) return null;
355
+ return Math.min(
356
+ STREAM_FIRST_EVENT_TIMEOUT_MAX_MS,
357
+ STREAM_FIRST_EVENT_TIMEOUT_MS + promptTokens / 1e3 * PREFILL_TIMEOUT_MS_PER_1K_TOKENS
358
+ );
359
+ }
360
+ var CACHE_HEALTH_MIN_PROMPT_TOKENS = 4e4;
361
+ var CACHE_HEALTH_LOW_RATIO = 0.5;
362
+ function assessCacheHealth(usage) {
363
+ const input = usage.inputTokens ?? 0;
364
+ const cacheRead = usage.cacheRead ?? 0;
365
+ const cacheWrite = usage.cacheWrite ?? 0;
366
+ const promptTokens = input + cacheRead + cacheWrite;
367
+ if (promptTokens < CACHE_HEALTH_MIN_PROMPT_TOKENS) {
368
+ return { ratio: null, promptTokens, cacheRead, low: false };
369
+ }
370
+ const ratio = cacheRead / promptTokens;
371
+ return {
372
+ ratio,
373
+ promptTokens,
374
+ cacheRead,
375
+ low: ratio < CACHE_HEALTH_LOW_RATIO
376
+ };
377
+ }
349
378
  async function* agentLoop(messages, options) {
350
379
  const maxTurns = options.maxTurns ?? DEFAULT_MAX_TURNS;
351
380
  let effectiveMaxTurns = maxTurns;
@@ -387,6 +416,7 @@ async function* agentLoop(messages, options) {
387
416
  let providerCalls = 0;
388
417
  let nonStreamingCalls = 0;
389
418
  let warnedNonStreaming = false;
419
+ let warnedPromptCacheMiss = false;
390
420
  const MAX_OUTPUT_CEILING_RETRIES = 1;
391
421
  let outputCeilingRetries = 0;
392
422
  const ceilingKey = outputRouteKey({
@@ -396,7 +426,6 @@ async function* agentLoop(messages, options) {
396
426
  });
397
427
  const OVERLOAD_BASE_DELAY_MS = 2e3;
398
428
  const OVERLOAD_MAX_DELAY_MS = 3e4;
399
- const STREAM_FIRST_EVENT_TIMEOUT_MS = 45e3;
400
429
  const STREAM_IDLE_TIMEOUT_MS = 9e4;
401
430
  const STREAM_HARD_TIMEOUT_MS = 9e4;
402
431
  const STREAM_OUTPUT_HARD_TIMEOUT_MS = 3e5;
@@ -405,8 +434,8 @@ async function* agentLoop(messages, options) {
405
434
  const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
406
435
  const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
407
436
  const localBackend = isLocalBackendUrl(options.baseUrl);
408
- const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
409
- const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
437
+ const baseFirstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
438
+ const baseHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
410
439
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
411
440
  const MAX_TOOLCALL_NO_PROGRESS_EVENTS = 2e4;
412
441
  let logicalTurnStartedAt = 0;
@@ -418,17 +447,28 @@ async function* agentLoop(messages, options) {
418
447
  turn++;
419
448
  if (logicalTurnStartedAt === 0) logicalTurnStartedAt = Date.now();
420
449
  toolMap = new Map((options.tools ?? []).map((t) => [t.name, t]));
421
- if (_diagFn) {
422
- let msgChars = 0;
423
- for (const m of messages) {
424
- if (typeof m.content === "string") msgChars += m.content.length;
425
- else if (Array.isArray(m.content)) {
426
- for (const p of m.content) {
427
- if ("text" in p && typeof p.text === "string") msgChars += p.text.length;
428
- if ("content" in p && typeof p.content === "string") msgChars += p.content.length;
429
- }
450
+ let msgChars = 0;
451
+ for (const m of messages) {
452
+ if (typeof m.content === "string") msgChars += m.content.length;
453
+ else if (Array.isArray(m.content)) {
454
+ for (const p of m.content) {
455
+ if ("text" in p && typeof p.text === "string") msgChars += p.text.length;
456
+ if ("content" in p && typeof p.content === "string") msgChars += p.content.length;
430
457
  }
431
458
  }
459
+ }
460
+ let firstEventTimeoutMs;
461
+ let initialHardTimeoutMs;
462
+ if (baseFirstEventTimeoutMs === STREAM_FIRST_EVENT_TIMEOUT_MS) {
463
+ const promptTokens = Math.ceil(msgChars / 3);
464
+ const scaled = scaledFirstEventTimeoutMs(promptTokens);
465
+ firstEventTimeoutMs = scaled ?? baseFirstEventTimeoutMs;
466
+ initialHardTimeoutMs = Math.max(baseHardTimeoutMs, firstEventTimeoutMs + 3e4);
467
+ } else {
468
+ firstEventTimeoutMs = baseFirstEventTimeoutMs;
469
+ initialHardTimeoutMs = baseHardTimeoutMs;
470
+ }
471
+ if (_diagFn) {
432
472
  diag("turn_start", {
433
473
  turn,
434
474
  messages: messages.length,
@@ -1051,6 +1091,27 @@ async function* agentLoop(messages, options) {
1051
1091
  if (response.usage.cacheWrite) {
1052
1092
  totalUsage.cacheWrite = (totalUsage.cacheWrite ?? 0) + response.usage.cacheWrite;
1053
1093
  }
1094
+ const cacheHealth = assessCacheHealth(response.usage);
1095
+ if (cacheHealth.ratio !== null) {
1096
+ diag("cache_health", {
1097
+ promptTokens: cacheHealth.promptTokens,
1098
+ cacheRead: cacheHealth.cacheRead,
1099
+ ratio: Math.round(cacheHealth.ratio * 100) / 100,
1100
+ provider: options.provider,
1101
+ model: options.model
1102
+ });
1103
+ if (cacheHealth.low && !warnedPromptCacheMiss) {
1104
+ warnedPromptCacheMiss = true;
1105
+ diag("prompt_cache_miss", {
1106
+ promptTokens: cacheHealth.promptTokens,
1107
+ cacheRead: cacheHealth.cacheRead,
1108
+ ratio: Math.round(cacheHealth.ratio * 100) / 100,
1109
+ provider: options.provider,
1110
+ model: options.model,
1111
+ impact: "large prompts are consistently served mostly uncached \u2014 every turn re-prefills the whole context; if this persists, lower the provider's compaction latency cap (resolveCompactionPolicy)"
1112
+ });
1113
+ }
1114
+ }
1054
1115
  if (!emptyExhausted) {
1055
1116
  messages.push(response.message);
1056
1117
  latestProviderUsage = response.usage;