@dudousxd/nestjs-agent-core 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,6 +1,74 @@
1
1
  import { StandardSchemaV1 } from '@standard-schema/spec';
2
2
  import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
3
3
 
4
+ /**
5
+ * Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
6
+ * classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
7
+ * the server rolled the tool's work back — retrying THAT class is safe, unlike a tool's general
8
+ * business failure, which stays a one-shot outcome (no durable step retries: a tool may not be
9
+ * idempotent). See `runAgentLoop`'s `tool:<call.id>` step body and `AgentRunSteps.tool` — both wrap
10
+ * `registry.invoke(...)` with {@link invokeWithTransientRetry} so a retry never becomes a new
11
+ * checkpoint; history still shows exactly one step per tool call.
12
+ */
13
+ /**
14
+ * Default transient-tool-error classifier: true for a recognized MySQL/Postgres/SQLite
15
+ * lock-contention shape (by driver `code`/`errno`/`sqlState`, or a matching message), checked on
16
+ * the error itself and one level of `cause` (drivers commonly wrap the original error). A plain
17
+ * `Error` with none of these markers — any other business failure — is `false`.
18
+ */
19
+ declare function isTransientToolError(error: unknown): boolean;
20
+ /** Total attempts (initial try + retries) when `toolTransientRetry` doesn't set `attempts`. */
21
+ declare const DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
22
+ /** Backoff base in ms — the wait between attempt N and N+1 is `backoffMs * N`. */
23
+ declare const DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
24
+ /** The host-configurable half of the policy — everything except the (non-wire-safe) `classify` fn. */
25
+ interface ToolTransientRetryOptions {
26
+ /** Total attempts (initial try + retries). Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS}. */
27
+ attempts?: number;
28
+ /** Backoff base in ms. Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS}. */
29
+ backoffMs?: number;
30
+ /** Overrides the default classifier — widen or narrow which errors are treated as transient. */
31
+ classify?: (error: unknown) => boolean;
32
+ }
33
+ /** `false` disables transient retry entirely — a tool's own thrown error surfaces immediately. */
34
+ type ToolTransientRetrySetting = ToolTransientRetryOptions | false;
35
+ /** Just the wire-safe (numeric) half of a resolved policy — what a dispatched envelope carries. */
36
+ interface ToolTransientRetryNumbers {
37
+ attempts: number;
38
+ backoffMs: number;
39
+ }
40
+ /**
41
+ * Resolves the numeric half of `toolTransientRetry` for the dispatched wire envelope: `false` when
42
+ * explicitly disabled, else concrete `{ attempts, backoffMs }` (defaults filled in) — never
43
+ * `undefined`, so the dispatched handler always gets a definite answer instead of re-deriving its
44
+ * own default. The `classify` function never rides this — it isn't wire-safe; the dispatched
45
+ * handler resolves its own `classify` from its local module options (see `AgentRunSteps.tool`).
46
+ */
47
+ declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
48
+ interface InvokeWithTransientRetryOptions {
49
+ /**
50
+ * Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so a retry
51
+ * never swallows one — same rule the loop's tool catch already applies. Undefined for a call site
52
+ * with no such notion (e.g. the dispatched step handler, which has no workflow ctx of its own).
53
+ */
54
+ isControlFlowError?: (error: unknown) => boolean;
55
+ /**
56
+ * Called before each wait-and-retry, with the 1-based ordinal of the attempt that just failed and
57
+ * the error it threw. The call site uses this to emit the `tool.retry` diagnostics point event —
58
+ * `invokeWithTransientRetry` itself carries no tool identity (name/callId), only the thunk.
59
+ */
60
+ onRetry?: (attempt: number, error: unknown) => void;
61
+ }
62
+ /**
63
+ * Retries `fn` in place — never a new durable step/checkpoint, just repeated attempts inside
64
+ * whichever step body already wraps this call. `setting: false` runs `fn` once, unwrapped (no
65
+ * classify/backoff bookkeeping at all). Otherwise: try; on a thrown error, rethrow immediately if
66
+ * it's a recognized control-flow signal, else if the (possibly custom) classifier calls it
67
+ * transient AND attempts remain, wait `backoffMs * attemptNumber` and retry; otherwise rethrow the
68
+ * error as-is.
69
+ */
70
+ declare function invokeWithTransientRetry<T>(fn: () => Promise<T>, setting: ToolTransientRetrySetting, options?: InvokeWithTransientRetryOptions): Promise<T>;
71
+
4
72
  /** Who is driving the turn. Roles + tenant come from the host app (nestjs-context/authz). */
5
73
  interface Actor {
6
74
  id: string;
@@ -312,6 +380,15 @@ interface ToolStepEnvelope {
312
380
  ctx: ToolStepCtx;
313
381
  /** Applied INSIDE the handler (`withToolTimeout`) — never as a durable step `timeoutMs`. */
314
382
  timeoutMs?: number;
383
+ /**
384
+ * The numeric half of `toolTransientRetry` (resolved by the loop from `AgentLoopDeps`, always a
385
+ * definite value — `false` when disabled, else concrete `{ attempts, backoffMs }` with defaults
386
+ * already filled in) — never `undefined`, so the dispatched handler gets the SAME policy the
387
+ * loop would have used locally. The `classify` function is deliberately absent: it isn't
388
+ * wire-safe, so the handler resolves its own from its local module options (see
389
+ * `AgentRunSteps.tool`) instead of trying to serialize a function.
390
+ */
391
+ transientRetry: ToolTransientRetryNumbers | false;
315
392
  }
316
393
 
317
394
  /**
@@ -1266,6 +1343,16 @@ interface AgentLoopDeps {
1266
1343
  * Undefined → no timeout.
1267
1344
  */
1268
1345
  toolTimeoutMs?: number;
1346
+ /**
1347
+ * Retries a tool's own invocation, in place, when it throws a classified-transient error (a DB
1348
+ * deadlock, a lock-wait timeout, a serialization failure — see `isTransientToolError`) — never a
1349
+ * new durable step/checkpoint, just repeated attempts inside the same `tool:<call.id>` step body.
1350
+ * Default ON (`{ attempts: 2, backoffMs: 150 }` with the default classifier) when undefined; set
1351
+ * `{ classify }` to widen/narrow which errors count as transient, or `false` to disable entirely.
1352
+ * A tool's other (non-transient) failures are unaffected — they remain a one-shot business
1353
+ * outcome, exactly as before.
1354
+ */
1355
+ toolTransientRetry?: ToolTransientRetrySetting;
1269
1356
  /**
1270
1357
  * When set, after the final turn the loop makes one extra model call to propose up to this many
1271
1358
  * short follow-up questions, stored on the assistant message's `followUps`. Costs an extra call
@@ -1406,6 +1493,18 @@ interface AgentRetrieved {
1406
1493
  /** How many passages the retriever returned. */
1407
1494
  count: number;
1408
1495
  }
1496
+ /**
1497
+ * A transient-classified tool error being retried in place (no new checkpoint) — see
1498
+ * `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
1499
+ */
1500
+ interface AgentToolRetry {
1501
+ toolName: string;
1502
+ toolCallId: string;
1503
+ /** 1-based ordinal of the attempt that just failed and is about to be retried. */
1504
+ attempt: number;
1505
+ /** The failed attempt's error message. */
1506
+ message: string;
1507
+ }
1409
1508
  /** START payload of an `aviary:agent:llm.turn:*` span — one model call within a run. */
1410
1509
  interface AgentLlmTurnSpan {
1411
1510
  runId: string;
@@ -1446,6 +1545,7 @@ declare module '@dudousxd/nestjs-diagnostics' {
1446
1545
  'run.failed': AgentRunFailed;
1447
1546
  delegated: AgentDelegated;
1448
1547
  retrieved: AgentRetrieved;
1548
+ 'tool.retry': AgentToolRetry;
1449
1549
  'llm.turn': AgentLlmTurnSpan;
1450
1550
  'tool.execution': AgentToolExecutionSpan;
1451
1551
  retrieval: AgentRetrievalSpan;
@@ -1461,6 +1561,7 @@ declare function publishAgentRunFinished(payload: AgentRunFinished): void;
1461
1561
  declare function publishAgentRunFailed(payload: AgentRunFailed): void;
1462
1562
  declare function publishAgentDelegated(payload: AgentDelegated): void;
1463
1563
  declare function publishAgentRetrieved(payload: AgentRetrieved): void;
1564
+ declare function publishAgentToolRetry(payload: AgentToolRetry): void;
1464
1565
  /**
1465
1566
  * Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
1466
1567
  * `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
@@ -1474,7 +1575,7 @@ declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
1474
1575
  /** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
1475
1576
  type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
1476
1577
  /**
1477
- * All 8 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
1578
+ * All 9 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
1478
1579
  * subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
1479
1580
  * {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
1480
1581
  * error in both directions: an extra/misspelled entry fails this array's own
@@ -1496,4 +1597,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
1496
1597
  */
1497
1598
  declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
1498
1599
 
1499
- export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
1600
+ export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, invokeWithTransientRetry, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
package/dist/index.d.ts CHANGED
@@ -1,6 +1,74 @@
1
1
  import { StandardSchemaV1 } from '@standard-schema/spec';
2
2
  import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
3
3
 
4
+ /**
5
+ * Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
6
+ * classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
7
+ * the server rolled the tool's work back — retrying THAT class is safe, unlike a tool's general
8
+ * business failure, which stays a one-shot outcome (no durable step retries: a tool may not be
9
+ * idempotent). See `runAgentLoop`'s `tool:<call.id>` step body and `AgentRunSteps.tool` — both wrap
10
+ * `registry.invoke(...)` with {@link invokeWithTransientRetry} so a retry never becomes a new
11
+ * checkpoint; history still shows exactly one step per tool call.
12
+ */
13
+ /**
14
+ * Default transient-tool-error classifier: true for a recognized MySQL/Postgres/SQLite
15
+ * lock-contention shape (by driver `code`/`errno`/`sqlState`, or a matching message), checked on
16
+ * the error itself and one level of `cause` (drivers commonly wrap the original error). A plain
17
+ * `Error` with none of these markers — any other business failure — is `false`.
18
+ */
19
+ declare function isTransientToolError(error: unknown): boolean;
20
+ /** Total attempts (initial try + retries) when `toolTransientRetry` doesn't set `attempts`. */
21
+ declare const DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
22
+ /** Backoff base in ms — the wait between attempt N and N+1 is `backoffMs * N`. */
23
+ declare const DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
24
+ /** The host-configurable half of the policy — everything except the (non-wire-safe) `classify` fn. */
25
+ interface ToolTransientRetryOptions {
26
+ /** Total attempts (initial try + retries). Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS}. */
27
+ attempts?: number;
28
+ /** Backoff base in ms. Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS}. */
29
+ backoffMs?: number;
30
+ /** Overrides the default classifier — widen or narrow which errors are treated as transient. */
31
+ classify?: (error: unknown) => boolean;
32
+ }
33
+ /** `false` disables transient retry entirely — a tool's own thrown error surfaces immediately. */
34
+ type ToolTransientRetrySetting = ToolTransientRetryOptions | false;
35
+ /** Just the wire-safe (numeric) half of a resolved policy — what a dispatched envelope carries. */
36
+ interface ToolTransientRetryNumbers {
37
+ attempts: number;
38
+ backoffMs: number;
39
+ }
40
+ /**
41
+ * Resolves the numeric half of `toolTransientRetry` for the dispatched wire envelope: `false` when
42
+ * explicitly disabled, else concrete `{ attempts, backoffMs }` (defaults filled in) — never
43
+ * `undefined`, so the dispatched handler always gets a definite answer instead of re-deriving its
44
+ * own default. The `classify` function never rides this — it isn't wire-safe; the dispatched
45
+ * handler resolves its own `classify` from its local module options (see `AgentRunSteps.tool`).
46
+ */
47
+ declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
48
+ interface InvokeWithTransientRetryOptions {
49
+ /**
50
+ * Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so a retry
51
+ * never swallows one — same rule the loop's tool catch already applies. Undefined for a call site
52
+ * with no such notion (e.g. the dispatched step handler, which has no workflow ctx of its own).
53
+ */
54
+ isControlFlowError?: (error: unknown) => boolean;
55
+ /**
56
+ * Called before each wait-and-retry, with the 1-based ordinal of the attempt that just failed and
57
+ * the error it threw. The call site uses this to emit the `tool.retry` diagnostics point event —
58
+ * `invokeWithTransientRetry` itself carries no tool identity (name/callId), only the thunk.
59
+ */
60
+ onRetry?: (attempt: number, error: unknown) => void;
61
+ }
62
+ /**
63
+ * Retries `fn` in place — never a new durable step/checkpoint, just repeated attempts inside
64
+ * whichever step body already wraps this call. `setting: false` runs `fn` once, unwrapped (no
65
+ * classify/backoff bookkeeping at all). Otherwise: try; on a thrown error, rethrow immediately if
66
+ * it's a recognized control-flow signal, else if the (possibly custom) classifier calls it
67
+ * transient AND attempts remain, wait `backoffMs * attemptNumber` and retry; otherwise rethrow the
68
+ * error as-is.
69
+ */
70
+ declare function invokeWithTransientRetry<T>(fn: () => Promise<T>, setting: ToolTransientRetrySetting, options?: InvokeWithTransientRetryOptions): Promise<T>;
71
+
4
72
  /** Who is driving the turn. Roles + tenant come from the host app (nestjs-context/authz). */
5
73
  interface Actor {
6
74
  id: string;
@@ -312,6 +380,15 @@ interface ToolStepEnvelope {
312
380
  ctx: ToolStepCtx;
313
381
  /** Applied INSIDE the handler (`withToolTimeout`) — never as a durable step `timeoutMs`. */
314
382
  timeoutMs?: number;
383
+ /**
384
+ * The numeric half of `toolTransientRetry` (resolved by the loop from `AgentLoopDeps`, always a
385
+ * definite value — `false` when disabled, else concrete `{ attempts, backoffMs }` with defaults
386
+ * already filled in) — never `undefined`, so the dispatched handler gets the SAME policy the
387
+ * loop would have used locally. The `classify` function is deliberately absent: it isn't
388
+ * wire-safe, so the handler resolves its own from its local module options (see
389
+ * `AgentRunSteps.tool`) instead of trying to serialize a function.
390
+ */
391
+ transientRetry: ToolTransientRetryNumbers | false;
315
392
  }
316
393
 
317
394
  /**
@@ -1266,6 +1343,16 @@ interface AgentLoopDeps {
1266
1343
  * Undefined → no timeout.
1267
1344
  */
1268
1345
  toolTimeoutMs?: number;
1346
+ /**
1347
+ * Retries a tool's own invocation, in place, when it throws a classified-transient error (a DB
1348
+ * deadlock, a lock-wait timeout, a serialization failure — see `isTransientToolError`) — never a
1349
+ * new durable step/checkpoint, just repeated attempts inside the same `tool:<call.id>` step body.
1350
+ * Default ON (`{ attempts: 2, backoffMs: 150 }` with the default classifier) when undefined; set
1351
+ * `{ classify }` to widen/narrow which errors count as transient, or `false` to disable entirely.
1352
+ * A tool's other (non-transient) failures are unaffected — they remain a one-shot business
1353
+ * outcome, exactly as before.
1354
+ */
1355
+ toolTransientRetry?: ToolTransientRetrySetting;
1269
1356
  /**
1270
1357
  * When set, after the final turn the loop makes one extra model call to propose up to this many
1271
1358
  * short follow-up questions, stored on the assistant message's `followUps`. Costs an extra call
@@ -1406,6 +1493,18 @@ interface AgentRetrieved {
1406
1493
  /** How many passages the retriever returned. */
1407
1494
  count: number;
1408
1495
  }
1496
+ /**
1497
+ * A transient-classified tool error being retried in place (no new checkpoint) — see
1498
+ * `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
1499
+ */
1500
+ interface AgentToolRetry {
1501
+ toolName: string;
1502
+ toolCallId: string;
1503
+ /** 1-based ordinal of the attempt that just failed and is about to be retried. */
1504
+ attempt: number;
1505
+ /** The failed attempt's error message. */
1506
+ message: string;
1507
+ }
1409
1508
  /** START payload of an `aviary:agent:llm.turn:*` span — one model call within a run. */
1410
1509
  interface AgentLlmTurnSpan {
1411
1510
  runId: string;
@@ -1446,6 +1545,7 @@ declare module '@dudousxd/nestjs-diagnostics' {
1446
1545
  'run.failed': AgentRunFailed;
1447
1546
  delegated: AgentDelegated;
1448
1547
  retrieved: AgentRetrieved;
1548
+ 'tool.retry': AgentToolRetry;
1449
1549
  'llm.turn': AgentLlmTurnSpan;
1450
1550
  'tool.execution': AgentToolExecutionSpan;
1451
1551
  retrieval: AgentRetrievalSpan;
@@ -1461,6 +1561,7 @@ declare function publishAgentRunFinished(payload: AgentRunFinished): void;
1461
1561
  declare function publishAgentRunFailed(payload: AgentRunFailed): void;
1462
1562
  declare function publishAgentDelegated(payload: AgentDelegated): void;
1463
1563
  declare function publishAgentRetrieved(payload: AgentRetrieved): void;
1564
+ declare function publishAgentToolRetry(payload: AgentToolRetry): void;
1464
1565
  /**
1465
1566
  * Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
1466
1567
  * `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
@@ -1474,7 +1575,7 @@ declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
1474
1575
  /** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
1475
1576
  type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
1476
1577
  /**
1477
- * All 8 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
1578
+ * All 9 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
1478
1579
  * subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
1479
1580
  * {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
1480
1581
  * error in both directions: an extra/misspelled entry fails this array's own
@@ -1496,4 +1597,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
1496
1597
  */
1497
1598
  declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
1498
1599
 
1499
- export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
1600
+ export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, invokeWithTransientRetry, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
package/dist/index.js CHANGED
@@ -363,6 +363,10 @@ function publishAgentRetrieved(payload) {
363
363
  emit("agent", "retrieved", payload);
364
364
  }
365
365
  __name(publishAgentRetrieved, "publishAgentRetrieved");
366
+ function publishAgentToolRetry(payload) {
367
+ emit("agent", "tool.retry", payload);
368
+ }
369
+ __name(publishAgentToolRetry, "publishAgentToolRetry");
366
370
  var AGENT_SPAN_EVENTS = [
367
371
  "llm.turn",
368
372
  "tool.execution",
@@ -377,13 +381,96 @@ var AGENT_DIAGNOSTIC_EVENTS = [
377
381
  "run.finished",
378
382
  "run.failed",
379
383
  "delegated",
380
- "retrieved"
384
+ "retrieved",
385
+ "tool.retry"
381
386
  ];
382
387
  function agentDiagnosticKey(event) {
383
388
  return `agent:${event}`;
384
389
  }
385
390
  __name(agentDiagnosticKey, "agentDiagnosticKey");
386
391
 
392
+ // src/tool-retry.ts
393
+ function hasTransientShape(error) {
394
+ if (typeof error !== "object" || error === null) {
395
+ return false;
396
+ }
397
+ const code = "code" in error ? error.code : void 0;
398
+ const errno = "errno" in error ? error.errno : void 0;
399
+ const sqlState = "sqlState" in error ? error.sqlState : void 0;
400
+ if (code === 1213 || code === 1205 || errno === 1213 || errno === 1205) {
401
+ return true;
402
+ }
403
+ if (code === "ER_LOCK_DEADLOCK" || code === "ER_LOCK_WAIT_TIMEOUT") {
404
+ return true;
405
+ }
406
+ if (code === "40001" || code === "40P01" || sqlState === "40001" || sqlState === "40P01") {
407
+ return true;
408
+ }
409
+ if (code === "SQLITE_BUSY") {
410
+ return true;
411
+ }
412
+ const message = "message" in error ? error.message : void 0;
413
+ return typeof message === "string" && /deadlock|lock wait timeout|serialization failure/i.test(message);
414
+ }
415
+ __name(hasTransientShape, "hasTransientShape");
416
+ function isTransientToolError(error) {
417
+ if (hasTransientShape(error)) {
418
+ return true;
419
+ }
420
+ if (typeof error === "object" && error !== null && "cause" in error) {
421
+ const cause = error.cause;
422
+ if (cause !== void 0 && cause !== error && hasTransientShape(cause)) {
423
+ return true;
424
+ }
425
+ }
426
+ return false;
427
+ }
428
+ __name(isTransientToolError, "isTransientToolError");
429
+ var DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
430
+ var DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
431
+ function resolveToolTransientRetryNumbers(setting) {
432
+ if (setting === false) {
433
+ return false;
434
+ }
435
+ return {
436
+ attempts: setting?.attempts ?? DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS,
437
+ backoffMs: setting?.backoffMs ?? DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS
438
+ };
439
+ }
440
+ __name(resolveToolTransientRetryNumbers, "resolveToolTransientRetryNumbers");
441
+ function delay(ms) {
442
+ return new Promise((resolve) => {
443
+ setTimeout(resolve, ms);
444
+ });
445
+ }
446
+ __name(delay, "delay");
447
+ async function invokeWithTransientRetry(fn, setting, options) {
448
+ if (setting === false) {
449
+ return fn();
450
+ }
451
+ const attempts = setting.attempts ?? DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS;
452
+ const backoffMs = setting.backoffMs ?? DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS;
453
+ const classify = setting.classify ?? isTransientToolError;
454
+ let attempt = 1;
455
+ for (; ; ) {
456
+ try {
457
+ return await fn();
458
+ } catch (error) {
459
+ if (options?.isControlFlowError?.(error) === true) {
460
+ throw error;
461
+ }
462
+ const attemptsRemain = attempt < attempts;
463
+ if (!attemptsRemain || !classify(error)) {
464
+ throw error;
465
+ }
466
+ options?.onRetry?.(attempt, error);
467
+ await delay(backoffMs * attempt);
468
+ attempt += 1;
469
+ }
470
+ }
471
+ }
472
+ __name(invokeWithTransientRetry, "invokeWithTransientRetry");
473
+
387
474
  // src/agent-loop.ts
388
475
  function resolveCostUsd(usage, reportedCostUsd, price) {
389
476
  if (reportedCostUsd !== void 0) {
@@ -942,7 +1029,10 @@ ${buildContextBlock(passages)}`;
942
1029
  ctx: stepCtx,
943
1030
  ...deps.toolTimeoutMs !== void 0 ? {
944
1031
  timeoutMs: deps.toolTimeoutMs
945
- } : {}
1032
+ } : {},
1033
+ // Numeric-only: the handler applies withToolTimeout AND its own local `classify` — see
1034
+ // ToolStepEnvelope.transientRetry.
1035
+ transientRetry: resolveToolTransientRetryNumbers(deps.toolTransientRetry)
946
1036
  };
947
1037
  output = await hooks.dispatchTool(call, envelope);
948
1038
  } else {
@@ -950,7 +1040,19 @@ ${buildContextBlock(passages)}`;
950
1040
  toolCallId: call.id,
951
1041
  toolName: call.name,
952
1042
  toolType
953
- }, () => deps.registry.invoke(call.name, call.input, ctx, deps.rolesPolicy)));
1043
+ }, () => invokeWithTransientRetry(() => deps.registry.invoke(call.name, call.input, ctx, deps.rolesPolicy), deps.toolTransientRetry ?? {}, {
1044
+ ...hooks.isControlFlowError !== void 0 ? {
1045
+ isControlFlowError: hooks.isControlFlowError
1046
+ } : {},
1047
+ onRetry: /* @__PURE__ */ __name((attempt, retryError) => {
1048
+ publishAgentToolRetry({
1049
+ toolName: call.name,
1050
+ toolCallId: call.id,
1051
+ attempt,
1052
+ message: retryError instanceof Error ? retryError.message : String(retryError)
1053
+ });
1054
+ }, "onRetry")
1055
+ })));
954
1056
  output = deps.toolTimeoutMs !== void 0 ? await withToolTimeout(invocation, deps.toolTimeoutMs, call.name) : await invocation;
955
1057
  }
956
1058
  const executionMs = Date.now() - startedAt2;
@@ -1072,6 +1174,8 @@ export {
1072
1174
  AGENT_TOOL_REGISTRY,
1073
1175
  AgentRegistry,
1074
1176
  AgentStreamError,
1177
+ DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS,
1178
+ DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS,
1075
1179
  DefaultRolesPolicy,
1076
1180
  QuotaExceededError,
1077
1181
  ToolForbiddenError,
@@ -1088,6 +1192,8 @@ export {
1088
1192
  estimateCost,
1089
1193
  filterToolsByAllowList,
1090
1194
  filterToolsByRole,
1195
+ invokeWithTransientRetry,
1196
+ isTransientToolError,
1091
1197
  publishAgentDelegated,
1092
1198
  publishAgentMessage,
1093
1199
  publishAgentQuotaExceeded,
@@ -1096,6 +1202,8 @@ export {
1096
1202
  publishAgentRunFinished,
1097
1203
  publishAgentRunStarted,
1098
1204
  publishAgentToolCall,
1205
+ publishAgentToolRetry,
1206
+ resolveToolTransientRetryNumbers,
1099
1207
  runAgentLoop,
1100
1208
  seedModelPrices,
1101
1209
  traceLlmTurn,