@prestyj/agent 5.9.0 → 5.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -324,4 +324,13 @@ declare function isBillingError(err: unknown): boolean;
324
324
  declare function isUsageLimitError(err: unknown): boolean;
325
325
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
326
326
 
327
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
327
+ /**
328
+ * A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
329
+ * minutes prefilling a large prompt before it emits its first token. The
330
+ * first-event watchdog that protects hosted streams turns that into an abort
331
+ * → retry → cold-prefill loop that never converges, so it is disabled for
332
+ * loopback backends.
333
+ */
334
+ declare function isLocalBackendUrl(baseUrl?: string): boolean;
335
+
336
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
package/dist/index.d.ts CHANGED
@@ -324,4 +324,13 @@ declare function isBillingError(err: unknown): boolean;
324
324
  declare function isUsageLimitError(err: unknown): boolean;
325
325
  declare function agentLoop(messages: Message[], options: AgentOptions): AsyncGenerator<AgentEvent, AgentResult>;
326
326
 
327
- export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isUsageLimitError, setStreamDiagnostic };
327
+ /**
328
+ * A local inference server (llama.cpp, vLLM, Ollama, LM Studio) can spend
329
+ * minutes prefilling a large prompt before it emits its first token. The
330
+ * first-event watchdog that protects hosted streams turns that into an abort
331
+ * → retry → cold-prefill loop that never converges, so it is disabled for
332
+ * loopback backends.
333
+ */
334
+ declare function isLocalBackendUrl(baseUrl?: string): boolean;
335
+
336
+ export { Agent, type AgentDoneEvent, type AgentErrorEvent, type AgentEvent, type AgentFollowUpMessageEvent, type AgentOptions, type AgentResult, type AgentRetryEvent, type AgentServerToolCallEvent, type AgentServerToolResultEvent, type AgentSteeringMessageEvent, AgentStream, type AgentTextDeltaEvent, type AgentThinkingDeltaEvent, type AgentTool, type AgentToolCallDeltaEvent, type AgentToolCallEndEvent, type AgentToolCallStartEvent, type AgentToolCallUpdateEvent, type AgentTurnEndEvent, type AgentTurnTiming, type StreamDiagnosticFn, type StructuredToolResult, type ToolContext, type ToolExecuteResult, type ToolExecutionMode, type TransformContextOptions, agentLoop, isAbortError, isBillingError, isContextOverflow, isLocalBackendUrl, isUsageLimitError, setStreamDiagnostic };
package/dist/index.js CHANGED
@@ -10,6 +10,30 @@ import {
10
10
  isHardBillingMessage,
11
11
  redactValue
12
12
  } from "@prestyj/ai";
13
+
14
+ // src/local-backend.ts
15
+ function isLocalBackendUrl(baseUrl) {
16
+ if (!baseUrl) return false;
17
+ let host;
18
+ try {
19
+ host = new URL(baseUrl).hostname.toLowerCase();
20
+ } catch {
21
+ return false;
22
+ }
23
+ if (host === "[::1]" || host === "::1") return true;
24
+ if (host === "localhost" || host.endsWith(".localhost")) return true;
25
+ if (host === "0.0.0.0") return true;
26
+ if (host.endsWith(".local")) return true;
27
+ const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
28
+ if (ipv4) {
29
+ const octets = ipv4.slice(1).map(Number);
30
+ if (octets.some((n) => n > 255)) return false;
31
+ return octets[0] === 127;
32
+ }
33
+ return false;
34
+ }
35
+
36
+ // src/agent-loop.ts
13
37
  var DEFAULT_MAX_TURNS = 300;
14
38
  var _diagFn = null;
15
39
  function setStreamDiagnostic(fn) {
@@ -276,9 +300,10 @@ async function* agentLoop(messages, options) {
276
300
  const STREAM_THINKING_IDLE_TIMEOUT_MS = 3e5;
277
301
  const STREAM_THINKING_HARD_TIMEOUT_MS = 6e5;
278
302
  const NON_STREAMING_HARD_TIMEOUT_MS = 3e5;
279
- const isSakana = options.provider === "sakana";
280
- const firstEventTimeoutMs = isSakana ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
281
- const initialHardTimeoutMs = isSakana ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
303
+ const usesSilentReasoningBudget = options.provider === "sakana" || options.provider === "openai" && options.thinking != null;
304
+ const localBackend = isLocalBackendUrl(options.baseUrl);
305
+ const firstEventTimeoutMs = localBackend ? Number.POSITIVE_INFINITY : usesSilentReasoningBudget ? STREAM_THINKING_IDLE_TIMEOUT_MS : STREAM_FIRST_EVENT_TIMEOUT_MS;
306
+ const initialHardTimeoutMs = localBackend || usesSilentReasoningBudget ? STREAM_THINKING_HARD_TIMEOUT_MS : STREAM_HARD_TIMEOUT_MS;
282
307
  const MAX_TOOLCALL_DELTA_CHARS = 1e6;
283
308
  const MAX_TOOLCALL_DELTA_EVENTS = 2e4;
284
309
  let logicalTurnStartedAt = 0;
@@ -306,7 +331,11 @@ async function* agentLoop(messages, options) {
306
331
  messages: messages.length,
307
332
  chars: msgChars,
308
333
  provider: options.provider,
309
- model: options.model
334
+ model: options.model,
335
+ thinking: options.thinking ?? "off",
336
+ firstEventTimeoutMs,
337
+ initialHardTimeoutMs,
338
+ localBackend
310
339
  });
311
340
  }
312
341
  if (firstTurn && options.getSteeringMessages) {
@@ -364,6 +393,7 @@ async function* agentLoop(messages, options) {
364
393
  if (useNonStreamingFallback) return;
365
394
  if (idleTimer) clearTimeout(idleTimer);
366
395
  const timeoutMs = hasReceivedEvent ? STREAM_IDLE_TIMEOUT_MS : hasReceivedThinking ? STREAM_THINKING_IDLE_TIMEOUT_MS : firstEventTimeoutMs;
396
+ if (!Number.isFinite(timeoutMs)) return;
367
397
  idleTimer = setTimeout(() => {
368
398
  diag("idle_timeout_fired", {
369
399
  events: streamEventCount,
@@ -748,15 +778,29 @@ async function* agentLoop(messages, options) {
748
778
  continue;
749
779
  }
750
780
  if (transportFailure) {
781
+ const cause = malformed ? "malformed_stream" : socketDrop ? "socket_drop" : "stream_stall";
751
782
  diag("stall_exhausted", {
752
783
  stallRetries: MAX_STALL_RETRIES,
753
784
  provider: options.provider,
754
- model: options.model
785
+ model: options.model,
786
+ cause,
787
+ nonStreaming: useNonStreamingFallback,
788
+ events: streamEventCount,
789
+ eventTypes: eventTypeCounts,
790
+ lastEventType,
791
+ sinceLastEventMs: Date.now() - lastEventTime,
792
+ attemptDurationMs: Date.now() - streamCallStart,
793
+ maxConsumerLagMs
755
794
  });
756
795
  yield {
757
796
  type: "error",
758
- error: new Error(
759
- `The API provider's stream stalled ${MAX_STALL_RETRIES} times \u2014 the provider may be experiencing capacity issues. Your conversation is preserved. Send another message to retry.`
797
+ error: new EZCoderAIError(
798
+ `The connection to the API provider stopped responding after ${MAX_STALL_RETRIES} automatic retries. Your conversation is preserved.`,
799
+ {
800
+ source: "network",
801
+ hint: "Retry once. If it keeps happening on this device, disable any VPN or proxy and allow EZ Coder through firewall or antivirus web protection.",
802
+ cause: err
803
+ }
760
804
  )
761
805
  };
762
806
  break;
@@ -1575,6 +1619,7 @@ export {
1575
1619
  isAbortError,
1576
1620
  isBillingError,
1577
1621
  isContextOverflow,
1622
+ isLocalBackendUrl,
1578
1623
  isUsageLimitError,
1579
1624
  setStreamDiagnostic
1580
1625
  };