@dudousxd/nestjs-agent-core 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +125 -7
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +114 -2
- package/dist/index.d.ts +114 -2
- package/dist/index.js +119 -7
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.cts
CHANGED
|
@@ -1,6 +1,74 @@
|
|
|
1
1
|
import { StandardSchemaV1 } from '@standard-schema/spec';
|
|
2
2
|
import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
|
|
3
3
|
|
|
4
|
+
/**
|
|
5
|
+
* Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
|
|
6
|
+
* classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
|
|
7
|
+
* the server rolled the tool's work back — retrying THAT class is safe, unlike a tool's general
|
|
8
|
+
* business failure, which stays a one-shot outcome (no durable step retries: a tool may not be
|
|
9
|
+
* idempotent). See `runAgentLoop`'s `tool:<call.id>` step body and `AgentRunSteps.tool` — both wrap
|
|
10
|
+
* `registry.invoke(...)` with {@link invokeWithTransientRetry} so a retry never becomes a new
|
|
11
|
+
* checkpoint; history still shows exactly one step per tool call.
|
|
12
|
+
*/
|
|
13
|
+
/**
|
|
14
|
+
* Default transient-tool-error classifier: true for a recognized MySQL/Postgres/SQLite
|
|
15
|
+
* lock-contention shape (by driver `code`/`errno`/`sqlState`, or a matching message), checked on
|
|
16
|
+
* the error itself and one level of `cause` (drivers commonly wrap the original error). A plain
|
|
17
|
+
* `Error` with none of these markers — any other business failure — is `false`.
|
|
18
|
+
*/
|
|
19
|
+
declare function isTransientToolError(error: unknown): boolean;
|
|
20
|
+
/** Total attempts (initial try + retries) when `toolTransientRetry` doesn't set `attempts`. */
|
|
21
|
+
declare const DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
|
|
22
|
+
/** Backoff base in ms — the wait between attempt N and N+1 is `backoffMs * N`. */
|
|
23
|
+
declare const DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
|
|
24
|
+
/** The host-configurable half of the policy — everything except the (non-wire-safe) `classify` fn. */
|
|
25
|
+
interface ToolTransientRetryOptions {
|
|
26
|
+
/** Total attempts (initial try + retries). Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS}. */
|
|
27
|
+
attempts?: number;
|
|
28
|
+
/** Backoff base in ms. Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS}. */
|
|
29
|
+
backoffMs?: number;
|
|
30
|
+
/** Overrides the default classifier — widen or narrow which errors are treated as transient. */
|
|
31
|
+
classify?: (error: unknown) => boolean;
|
|
32
|
+
}
|
|
33
|
+
/** `false` disables transient retry entirely — a tool's own thrown error surfaces immediately. */
|
|
34
|
+
type ToolTransientRetrySetting = ToolTransientRetryOptions | false;
|
|
35
|
+
/** Just the wire-safe (numeric) half of a resolved policy — what a dispatched envelope carries. */
|
|
36
|
+
interface ToolTransientRetryNumbers {
|
|
37
|
+
attempts: number;
|
|
38
|
+
backoffMs: number;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Resolves the numeric half of `toolTransientRetry` for the dispatched wire envelope: `false` when
|
|
42
|
+
* explicitly disabled, else concrete `{ attempts, backoffMs }` (defaults filled in) — never
|
|
43
|
+
* `undefined`, so the dispatched handler always gets a definite answer instead of re-deriving its
|
|
44
|
+
* own default. The `classify` function never rides this — it isn't wire-safe; the dispatched
|
|
45
|
+
* handler resolves its own `classify` from its local module options (see `AgentRunSteps.tool`).
|
|
46
|
+
*/
|
|
47
|
+
declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
|
|
48
|
+
interface InvokeWithTransientRetryOptions {
|
|
49
|
+
/**
|
|
50
|
+
* Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so a retry
|
|
51
|
+
* never swallows one — same rule the loop's tool catch already applies. Undefined for a call site
|
|
52
|
+
* with no such notion (e.g. the dispatched step handler, which has no workflow ctx of its own).
|
|
53
|
+
*/
|
|
54
|
+
isControlFlowError?: (error: unknown) => boolean;
|
|
55
|
+
/**
|
|
56
|
+
* Called before each wait-and-retry, with the 1-based ordinal of the attempt that just failed and
|
|
57
|
+
* the error it threw. The call site uses this to emit the `tool.retry` diagnostics point event —
|
|
58
|
+
* `invokeWithTransientRetry` itself carries no tool identity (name/callId), only the thunk.
|
|
59
|
+
*/
|
|
60
|
+
onRetry?: (attempt: number, error: unknown) => void;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Retries `fn` in place — never a new durable step/checkpoint, just repeated attempts inside
|
|
64
|
+
* whichever step body already wraps this call. `setting: false` runs `fn` once, unwrapped (no
|
|
65
|
+
* classify/backoff bookkeeping at all). Otherwise: try; on a thrown error, rethrow immediately if
|
|
66
|
+
* it's a recognized control-flow signal, else if the (possibly custom) classifier calls it
|
|
67
|
+
* transient AND attempts remain, wait `backoffMs * attemptNumber` and retry; otherwise rethrow the
|
|
68
|
+
* error as-is.
|
|
69
|
+
*/
|
|
70
|
+
declare function invokeWithTransientRetry<T>(fn: () => Promise<T>, setting: ToolTransientRetrySetting, options?: InvokeWithTransientRetryOptions): Promise<T>;
|
|
71
|
+
|
|
4
72
|
/** Who is driving the turn. Roles + tenant come from the host app (nestjs-context/authz). */
|
|
5
73
|
interface Actor {
|
|
6
74
|
id: string;
|
|
@@ -312,6 +380,15 @@ interface ToolStepEnvelope {
|
|
|
312
380
|
ctx: ToolStepCtx;
|
|
313
381
|
/** Applied INSIDE the handler (`withToolTimeout`) — never as a durable step `timeoutMs`. */
|
|
314
382
|
timeoutMs?: number;
|
|
383
|
+
/**
|
|
384
|
+
* The numeric half of `toolTransientRetry` (resolved by the loop from `AgentLoopDeps`, always a
|
|
385
|
+
* definite value — `false` when disabled, else concrete `{ attempts, backoffMs }` with defaults
|
|
386
|
+
* already filled in) — never `undefined`, so the dispatched handler gets the SAME policy the
|
|
387
|
+
* loop would have used locally. The `classify` function is deliberately absent: it isn't
|
|
388
|
+
* wire-safe, so the handler resolves its own from its local module options (see
|
|
389
|
+
* `AgentRunSteps.tool`) instead of trying to serialize a function.
|
|
390
|
+
*/
|
|
391
|
+
transientRetry: ToolTransientRetryNumbers | false;
|
|
315
392
|
}
|
|
316
393
|
|
|
317
394
|
/**
|
|
@@ -559,6 +636,12 @@ interface RecordToolCallInput {
|
|
|
559
636
|
toolType: 'read' | 'action';
|
|
560
637
|
input: unknown;
|
|
561
638
|
status: ToolCallStatus;
|
|
639
|
+
/**
|
|
640
|
+
* The run (turn) this tool call belongs to — enables a governance surface to deep-link a tool
|
|
641
|
+
* call out to its trace waterfall. Optional so a caller predating this (or a store's own
|
|
642
|
+
* synthetic tool calls) can omit it; the store persists it as `null` when absent.
|
|
643
|
+
*/
|
|
644
|
+
runId?: string;
|
|
562
645
|
}
|
|
563
646
|
interface UpdateToolCallInput {
|
|
564
647
|
toolCallId: string;
|
|
@@ -879,6 +962,8 @@ interface ToolCallActivityRow {
|
|
|
879
962
|
status: string;
|
|
880
963
|
threadId: string;
|
|
881
964
|
createdAt: string;
|
|
965
|
+
/** The run this call belongs to, for a trace deep-link; `null` for a call recorded before this shipped. */
|
|
966
|
+
runId: string | null;
|
|
882
967
|
}
|
|
883
968
|
/** A recent thread with rolled-up activity. */
|
|
884
969
|
interface ThreadActivityRow {
|
|
@@ -949,6 +1034,8 @@ interface PendingApprovalRow {
|
|
|
949
1034
|
agentName: string | null;
|
|
950
1035
|
/** ISO timestamp. */
|
|
951
1036
|
requestedAt: string;
|
|
1037
|
+
/** The run this call belongs to, for a trace deep-link; `null` for a call recorded before this shipped. */
|
|
1038
|
+
runId: string | null;
|
|
952
1039
|
}
|
|
953
1040
|
/** Governance rollup for one tool over a range. */
|
|
954
1041
|
interface ToolStatRow {
|
|
@@ -1004,6 +1091,7 @@ interface RunWhere {
|
|
|
1004
1091
|
agentName?: string;
|
|
1005
1092
|
status?: string;
|
|
1006
1093
|
errorCode?: string;
|
|
1094
|
+
threadId?: string;
|
|
1007
1095
|
fromDay?: string;
|
|
1008
1096
|
toDay?: string;
|
|
1009
1097
|
}
|
|
@@ -1255,6 +1343,16 @@ interface AgentLoopDeps {
|
|
|
1255
1343
|
* Undefined → no timeout.
|
|
1256
1344
|
*/
|
|
1257
1345
|
toolTimeoutMs?: number;
|
|
1346
|
+
/**
|
|
1347
|
+
* Retries a tool's own invocation, in place, when it throws a classified-transient error (a DB
|
|
1348
|
+
* deadlock, a lock-wait timeout, a serialization failure — see `isTransientToolError`) — never a
|
|
1349
|
+
* new durable step/checkpoint, just repeated attempts inside the same `tool:<call.id>` step body.
|
|
1350
|
+
* Default ON (`{ attempts: 2, backoffMs: 150 }` with the default classifier) when undefined; set
|
|
1351
|
+
* `{ classify }` to widen/narrow which errors count as transient, or `false` to disable entirely.
|
|
1352
|
+
* A tool's other (non-transient) failures are unaffected — they remain a one-shot business
|
|
1353
|
+
* outcome, exactly as before.
|
|
1354
|
+
*/
|
|
1355
|
+
toolTransientRetry?: ToolTransientRetrySetting;
|
|
1258
1356
|
/**
|
|
1259
1357
|
* When set, after the final turn the loop makes one extra model call to propose up to this many
|
|
1260
1358
|
* short follow-up questions, stored on the assistant message's `followUps`. Costs an extra call
|
|
@@ -1395,6 +1493,18 @@ interface AgentRetrieved {
|
|
|
1395
1493
|
/** How many passages the retriever returned. */
|
|
1396
1494
|
count: number;
|
|
1397
1495
|
}
|
|
1496
|
+
/**
|
|
1497
|
+
* A transient-classified tool error being retried in place (no new checkpoint) — see
|
|
1498
|
+
* `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
|
|
1499
|
+
*/
|
|
1500
|
+
interface AgentToolRetry {
|
|
1501
|
+
toolName: string;
|
|
1502
|
+
toolCallId: string;
|
|
1503
|
+
/** 1-based ordinal of the attempt that just failed and is about to be retried. */
|
|
1504
|
+
attempt: number;
|
|
1505
|
+
/** The failed attempt's error message. */
|
|
1506
|
+
message: string;
|
|
1507
|
+
}
|
|
1398
1508
|
/** START payload of an `aviary:agent:llm.turn:*` span — one model call within a run. */
|
|
1399
1509
|
interface AgentLlmTurnSpan {
|
|
1400
1510
|
runId: string;
|
|
@@ -1435,6 +1545,7 @@ declare module '@dudousxd/nestjs-diagnostics' {
|
|
|
1435
1545
|
'run.failed': AgentRunFailed;
|
|
1436
1546
|
delegated: AgentDelegated;
|
|
1437
1547
|
retrieved: AgentRetrieved;
|
|
1548
|
+
'tool.retry': AgentToolRetry;
|
|
1438
1549
|
'llm.turn': AgentLlmTurnSpan;
|
|
1439
1550
|
'tool.execution': AgentToolExecutionSpan;
|
|
1440
1551
|
retrieval: AgentRetrievalSpan;
|
|
@@ -1450,6 +1561,7 @@ declare function publishAgentRunFinished(payload: AgentRunFinished): void;
|
|
|
1450
1561
|
declare function publishAgentRunFailed(payload: AgentRunFailed): void;
|
|
1451
1562
|
declare function publishAgentDelegated(payload: AgentDelegated): void;
|
|
1452
1563
|
declare function publishAgentRetrieved(payload: AgentRetrieved): void;
|
|
1564
|
+
declare function publishAgentToolRetry(payload: AgentToolRetry): void;
|
|
1453
1565
|
/**
|
|
1454
1566
|
* Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
|
|
1455
1567
|
* `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
|
|
@@ -1463,7 +1575,7 @@ declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
|
|
|
1463
1575
|
/** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
|
|
1464
1576
|
type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
|
|
1465
1577
|
/**
|
|
1466
|
-
* All
|
|
1578
|
+
* All 9 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
|
|
1467
1579
|
* subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
|
|
1468
1580
|
* {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
|
|
1469
1581
|
* error in both directions: an extra/misspelled entry fails this array's own
|
|
@@ -1485,4 +1597,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
|
|
|
1485
1597
|
*/
|
|
1486
1598
|
declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
|
|
1487
1599
|
|
|
1488
|
-
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
|
|
1600
|
+
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, invokeWithTransientRetry, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,6 +1,74 @@
|
|
|
1
1
|
import { StandardSchemaV1 } from '@standard-schema/spec';
|
|
2
2
|
import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
|
|
3
3
|
|
|
4
|
+
/**
|
|
5
|
+
* Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
|
|
6
|
+
* classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
|
|
7
|
+
* the server rolled the tool's work back — retrying THAT class is safe, unlike a tool's general
|
|
8
|
+
* business failure, which stays a one-shot outcome (no durable step retries: a tool may not be
|
|
9
|
+
* idempotent). See `runAgentLoop`'s `tool:<call.id>` step body and `AgentRunSteps.tool` — both wrap
|
|
10
|
+
* `registry.invoke(...)` with {@link invokeWithTransientRetry} so a retry never becomes a new
|
|
11
|
+
* checkpoint; history still shows exactly one step per tool call.
|
|
12
|
+
*/
|
|
13
|
+
/**
|
|
14
|
+
* Default transient-tool-error classifier: true for a recognized MySQL/Postgres/SQLite
|
|
15
|
+
* lock-contention shape (by driver `code`/`errno`/`sqlState`, or a matching message), checked on
|
|
16
|
+
* the error itself and one level of `cause` (drivers commonly wrap the original error). A plain
|
|
17
|
+
* `Error` with none of these markers — any other business failure — is `false`.
|
|
18
|
+
*/
|
|
19
|
+
declare function isTransientToolError(error: unknown): boolean;
|
|
20
|
+
/** Total attempts (initial try + retries) when `toolTransientRetry` doesn't set `attempts`. */
|
|
21
|
+
declare const DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
|
|
22
|
+
/** Backoff base in ms — the wait between attempt N and N+1 is `backoffMs * N`. */
|
|
23
|
+
declare const DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
|
|
24
|
+
/** The host-configurable half of the policy — everything except the (non-wire-safe) `classify` fn. */
|
|
25
|
+
interface ToolTransientRetryOptions {
|
|
26
|
+
/** Total attempts (initial try + retries). Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS}. */
|
|
27
|
+
attempts?: number;
|
|
28
|
+
/** Backoff base in ms. Defaults to {@link DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS}. */
|
|
29
|
+
backoffMs?: number;
|
|
30
|
+
/** Overrides the default classifier — widen or narrow which errors are treated as transient. */
|
|
31
|
+
classify?: (error: unknown) => boolean;
|
|
32
|
+
}
|
|
33
|
+
/** `false` disables transient retry entirely — a tool's own thrown error surfaces immediately. */
|
|
34
|
+
type ToolTransientRetrySetting = ToolTransientRetryOptions | false;
|
|
35
|
+
/** Just the wire-safe (numeric) half of a resolved policy — what a dispatched envelope carries. */
|
|
36
|
+
interface ToolTransientRetryNumbers {
|
|
37
|
+
attempts: number;
|
|
38
|
+
backoffMs: number;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Resolves the numeric half of `toolTransientRetry` for the dispatched wire envelope: `false` when
|
|
42
|
+
* explicitly disabled, else concrete `{ attempts, backoffMs }` (defaults filled in) — never
|
|
43
|
+
* `undefined`, so the dispatched handler always gets a definite answer instead of re-deriving its
|
|
44
|
+
* own default. The `classify` function never rides this — it isn't wire-safe; the dispatched
|
|
45
|
+
* handler resolves its own `classify` from its local module options (see `AgentRunSteps.tool`).
|
|
46
|
+
*/
|
|
47
|
+
declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
|
|
48
|
+
interface InvokeWithTransientRetryOptions {
|
|
49
|
+
/**
|
|
50
|
+
* Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so a retry
|
|
51
|
+
* never swallows one — same rule the loop's tool catch already applies. Undefined for a call site
|
|
52
|
+
* with no such notion (e.g. the dispatched step handler, which has no workflow ctx of its own).
|
|
53
|
+
*/
|
|
54
|
+
isControlFlowError?: (error: unknown) => boolean;
|
|
55
|
+
/**
|
|
56
|
+
* Called before each wait-and-retry, with the 1-based ordinal of the attempt that just failed and
|
|
57
|
+
* the error it threw. The call site uses this to emit the `tool.retry` diagnostics point event —
|
|
58
|
+
* `invokeWithTransientRetry` itself carries no tool identity (name/callId), only the thunk.
|
|
59
|
+
*/
|
|
60
|
+
onRetry?: (attempt: number, error: unknown) => void;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Retries `fn` in place — never a new durable step/checkpoint, just repeated attempts inside
|
|
64
|
+
* whichever step body already wraps this call. `setting: false` runs `fn` once, unwrapped (no
|
|
65
|
+
* classify/backoff bookkeeping at all). Otherwise: try; on a thrown error, rethrow immediately if
|
|
66
|
+
* it's a recognized control-flow signal, else if the (possibly custom) classifier calls it
|
|
67
|
+
* transient AND attempts remain, wait `backoffMs * attemptNumber` and retry; otherwise rethrow the
|
|
68
|
+
* error as-is.
|
|
69
|
+
*/
|
|
70
|
+
declare function invokeWithTransientRetry<T>(fn: () => Promise<T>, setting: ToolTransientRetrySetting, options?: InvokeWithTransientRetryOptions): Promise<T>;
|
|
71
|
+
|
|
4
72
|
/** Who is driving the turn. Roles + tenant come from the host app (nestjs-context/authz). */
|
|
5
73
|
interface Actor {
|
|
6
74
|
id: string;
|
|
@@ -312,6 +380,15 @@ interface ToolStepEnvelope {
|
|
|
312
380
|
ctx: ToolStepCtx;
|
|
313
381
|
/** Applied INSIDE the handler (`withToolTimeout`) — never as a durable step `timeoutMs`. */
|
|
314
382
|
timeoutMs?: number;
|
|
383
|
+
/**
|
|
384
|
+
* The numeric half of `toolTransientRetry` (resolved by the loop from `AgentLoopDeps`, always a
|
|
385
|
+
* definite value — `false` when disabled, else concrete `{ attempts, backoffMs }` with defaults
|
|
386
|
+
* already filled in) — never `undefined`, so the dispatched handler gets the SAME policy the
|
|
387
|
+
* loop would have used locally. The `classify` function is deliberately absent: it isn't
|
|
388
|
+
* wire-safe, so the handler resolves its own from its local module options (see
|
|
389
|
+
* `AgentRunSteps.tool`) instead of trying to serialize a function.
|
|
390
|
+
*/
|
|
391
|
+
transientRetry: ToolTransientRetryNumbers | false;
|
|
315
392
|
}
|
|
316
393
|
|
|
317
394
|
/**
|
|
@@ -559,6 +636,12 @@ interface RecordToolCallInput {
|
|
|
559
636
|
toolType: 'read' | 'action';
|
|
560
637
|
input: unknown;
|
|
561
638
|
status: ToolCallStatus;
|
|
639
|
+
/**
|
|
640
|
+
* The run (turn) this tool call belongs to — enables a governance surface to deep-link a tool
|
|
641
|
+
* call out to its trace waterfall. Optional so a caller predating this (or a store's own
|
|
642
|
+
* synthetic tool calls) can omit it; the store persists it as `null` when absent.
|
|
643
|
+
*/
|
|
644
|
+
runId?: string;
|
|
562
645
|
}
|
|
563
646
|
interface UpdateToolCallInput {
|
|
564
647
|
toolCallId: string;
|
|
@@ -879,6 +962,8 @@ interface ToolCallActivityRow {
|
|
|
879
962
|
status: string;
|
|
880
963
|
threadId: string;
|
|
881
964
|
createdAt: string;
|
|
965
|
+
/** The run this call belongs to, for a trace deep-link; `null` for a call recorded before this shipped. */
|
|
966
|
+
runId: string | null;
|
|
882
967
|
}
|
|
883
968
|
/** A recent thread with rolled-up activity. */
|
|
884
969
|
interface ThreadActivityRow {
|
|
@@ -949,6 +1034,8 @@ interface PendingApprovalRow {
|
|
|
949
1034
|
agentName: string | null;
|
|
950
1035
|
/** ISO timestamp. */
|
|
951
1036
|
requestedAt: string;
|
|
1037
|
+
/** The run this call belongs to, for a trace deep-link; `null` for a call recorded before this shipped. */
|
|
1038
|
+
runId: string | null;
|
|
952
1039
|
}
|
|
953
1040
|
/** Governance rollup for one tool over a range. */
|
|
954
1041
|
interface ToolStatRow {
|
|
@@ -1004,6 +1091,7 @@ interface RunWhere {
|
|
|
1004
1091
|
agentName?: string;
|
|
1005
1092
|
status?: string;
|
|
1006
1093
|
errorCode?: string;
|
|
1094
|
+
threadId?: string;
|
|
1007
1095
|
fromDay?: string;
|
|
1008
1096
|
toDay?: string;
|
|
1009
1097
|
}
|
|
@@ -1255,6 +1343,16 @@ interface AgentLoopDeps {
|
|
|
1255
1343
|
* Undefined → no timeout.
|
|
1256
1344
|
*/
|
|
1257
1345
|
toolTimeoutMs?: number;
|
|
1346
|
+
/**
|
|
1347
|
+
* Retries a tool's own invocation, in place, when it throws a classified-transient error (a DB
|
|
1348
|
+
* deadlock, a lock-wait timeout, a serialization failure — see `isTransientToolError`) — never a
|
|
1349
|
+
* new durable step/checkpoint, just repeated attempts inside the same `tool:<call.id>` step body.
|
|
1350
|
+
* Default ON (`{ attempts: 2, backoffMs: 150 }` with the default classifier) when undefined; set
|
|
1351
|
+
* `{ classify }` to widen/narrow which errors count as transient, or `false` to disable entirely.
|
|
1352
|
+
* A tool's other (non-transient) failures are unaffected — they remain a one-shot business
|
|
1353
|
+
* outcome, exactly as before.
|
|
1354
|
+
*/
|
|
1355
|
+
toolTransientRetry?: ToolTransientRetrySetting;
|
|
1258
1356
|
/**
|
|
1259
1357
|
* When set, after the final turn the loop makes one extra model call to propose up to this many
|
|
1260
1358
|
* short follow-up questions, stored on the assistant message's `followUps`. Costs an extra call
|
|
@@ -1395,6 +1493,18 @@ interface AgentRetrieved {
|
|
|
1395
1493
|
/** How many passages the retriever returned. */
|
|
1396
1494
|
count: number;
|
|
1397
1495
|
}
|
|
1496
|
+
/**
|
|
1497
|
+
* A transient-classified tool error being retried in place (no new checkpoint) — see
|
|
1498
|
+
* `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
|
|
1499
|
+
*/
|
|
1500
|
+
interface AgentToolRetry {
|
|
1501
|
+
toolName: string;
|
|
1502
|
+
toolCallId: string;
|
|
1503
|
+
/** 1-based ordinal of the attempt that just failed and is about to be retried. */
|
|
1504
|
+
attempt: number;
|
|
1505
|
+
/** The failed attempt's error message. */
|
|
1506
|
+
message: string;
|
|
1507
|
+
}
|
|
1398
1508
|
/** START payload of an `aviary:agent:llm.turn:*` span — one model call within a run. */
|
|
1399
1509
|
interface AgentLlmTurnSpan {
|
|
1400
1510
|
runId: string;
|
|
@@ -1435,6 +1545,7 @@ declare module '@dudousxd/nestjs-diagnostics' {
|
|
|
1435
1545
|
'run.failed': AgentRunFailed;
|
|
1436
1546
|
delegated: AgentDelegated;
|
|
1437
1547
|
retrieved: AgentRetrieved;
|
|
1548
|
+
'tool.retry': AgentToolRetry;
|
|
1438
1549
|
'llm.turn': AgentLlmTurnSpan;
|
|
1439
1550
|
'tool.execution': AgentToolExecutionSpan;
|
|
1440
1551
|
retrieval: AgentRetrievalSpan;
|
|
@@ -1450,6 +1561,7 @@ declare function publishAgentRunFinished(payload: AgentRunFinished): void;
|
|
|
1450
1561
|
declare function publishAgentRunFailed(payload: AgentRunFailed): void;
|
|
1451
1562
|
declare function publishAgentDelegated(payload: AgentDelegated): void;
|
|
1452
1563
|
declare function publishAgentRetrieved(payload: AgentRetrieved): void;
|
|
1564
|
+
declare function publishAgentToolRetry(payload: AgentToolRetry): void;
|
|
1453
1565
|
/**
|
|
1454
1566
|
* Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
|
|
1455
1567
|
* `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
|
|
@@ -1463,7 +1575,7 @@ declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
|
|
|
1463
1575
|
/** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
|
|
1464
1576
|
type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
|
|
1465
1577
|
/**
|
|
1466
|
-
* All
|
|
1578
|
+
* All 9 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
|
|
1467
1579
|
* subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
|
|
1468
1580
|
* {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
|
|
1469
1581
|
* error in both directions: an extra/misspelled entry fails this array's own
|
|
@@ -1485,4 +1597,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
|
|
|
1485
1597
|
*/
|
|
1486
1598
|
declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
|
|
1487
1599
|
|
|
1488
|
-
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
|
|
1600
|
+
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, type ThreadActivityRow, type ThreadDetail, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByRole, invokeWithTransientRetry, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, withToolTimeout };
|
package/dist/index.js
CHANGED
|
@@ -363,6 +363,10 @@ function publishAgentRetrieved(payload) {
|
|
|
363
363
|
emit("agent", "retrieved", payload);
|
|
364
364
|
}
|
|
365
365
|
__name(publishAgentRetrieved, "publishAgentRetrieved");
|
|
366
|
+
function publishAgentToolRetry(payload) {
|
|
367
|
+
emit("agent", "tool.retry", payload);
|
|
368
|
+
}
|
|
369
|
+
__name(publishAgentToolRetry, "publishAgentToolRetry");
|
|
366
370
|
var AGENT_SPAN_EVENTS = [
|
|
367
371
|
"llm.turn",
|
|
368
372
|
"tool.execution",
|
|
@@ -377,13 +381,96 @@ var AGENT_DIAGNOSTIC_EVENTS = [
|
|
|
377
381
|
"run.finished",
|
|
378
382
|
"run.failed",
|
|
379
383
|
"delegated",
|
|
380
|
-
"retrieved"
|
|
384
|
+
"retrieved",
|
|
385
|
+
"tool.retry"
|
|
381
386
|
];
|
|
382
387
|
function agentDiagnosticKey(event) {
|
|
383
388
|
return `agent:${event}`;
|
|
384
389
|
}
|
|
385
390
|
__name(agentDiagnosticKey, "agentDiagnosticKey");
|
|
386
391
|
|
|
392
|
+
// src/tool-retry.ts
|
|
393
|
+
function hasTransientShape(error) {
|
|
394
|
+
if (typeof error !== "object" || error === null) {
|
|
395
|
+
return false;
|
|
396
|
+
}
|
|
397
|
+
const code = "code" in error ? error.code : void 0;
|
|
398
|
+
const errno = "errno" in error ? error.errno : void 0;
|
|
399
|
+
const sqlState = "sqlState" in error ? error.sqlState : void 0;
|
|
400
|
+
if (code === 1213 || code === 1205 || errno === 1213 || errno === 1205) {
|
|
401
|
+
return true;
|
|
402
|
+
}
|
|
403
|
+
if (code === "ER_LOCK_DEADLOCK" || code === "ER_LOCK_WAIT_TIMEOUT") {
|
|
404
|
+
return true;
|
|
405
|
+
}
|
|
406
|
+
if (code === "40001" || code === "40P01" || sqlState === "40001" || sqlState === "40P01") {
|
|
407
|
+
return true;
|
|
408
|
+
}
|
|
409
|
+
if (code === "SQLITE_BUSY") {
|
|
410
|
+
return true;
|
|
411
|
+
}
|
|
412
|
+
const message = "message" in error ? error.message : void 0;
|
|
413
|
+
return typeof message === "string" && /deadlock|lock wait timeout|serialization failure/i.test(message);
|
|
414
|
+
}
|
|
415
|
+
__name(hasTransientShape, "hasTransientShape");
|
|
416
|
+
function isTransientToolError(error) {
|
|
417
|
+
if (hasTransientShape(error)) {
|
|
418
|
+
return true;
|
|
419
|
+
}
|
|
420
|
+
if (typeof error === "object" && error !== null && "cause" in error) {
|
|
421
|
+
const cause = error.cause;
|
|
422
|
+
if (cause !== void 0 && cause !== error && hasTransientShape(cause)) {
|
|
423
|
+
return true;
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
return false;
|
|
427
|
+
}
|
|
428
|
+
__name(isTransientToolError, "isTransientToolError");
|
|
429
|
+
var DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS = 2;
|
|
430
|
+
var DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS = 150;
|
|
431
|
+
function resolveToolTransientRetryNumbers(setting) {
|
|
432
|
+
if (setting === false) {
|
|
433
|
+
return false;
|
|
434
|
+
}
|
|
435
|
+
return {
|
|
436
|
+
attempts: setting?.attempts ?? DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS,
|
|
437
|
+
backoffMs: setting?.backoffMs ?? DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS
|
|
438
|
+
};
|
|
439
|
+
}
|
|
440
|
+
__name(resolveToolTransientRetryNumbers, "resolveToolTransientRetryNumbers");
|
|
441
|
+
function delay(ms) {
|
|
442
|
+
return new Promise((resolve) => {
|
|
443
|
+
setTimeout(resolve, ms);
|
|
444
|
+
});
|
|
445
|
+
}
|
|
446
|
+
__name(delay, "delay");
|
|
447
|
+
async function invokeWithTransientRetry(fn, setting, options) {
|
|
448
|
+
if (setting === false) {
|
|
449
|
+
return fn();
|
|
450
|
+
}
|
|
451
|
+
const attempts = setting.attempts ?? DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS;
|
|
452
|
+
const backoffMs = setting.backoffMs ?? DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS;
|
|
453
|
+
const classify = setting.classify ?? isTransientToolError;
|
|
454
|
+
let attempt = 1;
|
|
455
|
+
for (; ; ) {
|
|
456
|
+
try {
|
|
457
|
+
return await fn();
|
|
458
|
+
} catch (error) {
|
|
459
|
+
if (options?.isControlFlowError?.(error) === true) {
|
|
460
|
+
throw error;
|
|
461
|
+
}
|
|
462
|
+
const attemptsRemain = attempt < attempts;
|
|
463
|
+
if (!attemptsRemain || !classify(error)) {
|
|
464
|
+
throw error;
|
|
465
|
+
}
|
|
466
|
+
options?.onRetry?.(attempt, error);
|
|
467
|
+
await delay(backoffMs * attempt);
|
|
468
|
+
attempt += 1;
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
__name(invokeWithTransientRetry, "invokeWithTransientRetry");
|
|
473
|
+
|
|
387
474
|
// src/agent-loop.ts
|
|
388
475
|
function resolveCostUsd(usage, reportedCostUsd, price) {
|
|
389
476
|
if (reportedCostUsd !== void 0) {
|
|
@@ -785,7 +872,8 @@ ${buildContextBlock(passages)}`;
|
|
|
785
872
|
input: {
|
|
786
873
|
query: input.userText
|
|
787
874
|
},
|
|
788
|
-
status: "auto_executed"
|
|
875
|
+
status: "auto_executed",
|
|
876
|
+
runId: hooks.runId
|
|
789
877
|
});
|
|
790
878
|
await deps.store.updateToolCall({
|
|
791
879
|
toolCallId,
|
|
@@ -834,7 +922,8 @@ ${buildContextBlock(passages)}`;
|
|
|
834
922
|
toolName: call.name,
|
|
835
923
|
toolType: "read",
|
|
836
924
|
input: call.input,
|
|
837
|
-
status: "auto_executed"
|
|
925
|
+
status: "auto_executed",
|
|
926
|
+
runId: hooks.runId
|
|
838
927
|
}));
|
|
839
928
|
const overDepth = (input.delegationDepth ?? 0) >= MAX_DELEGATION_DEPTH;
|
|
840
929
|
publishAgentDelegated({
|
|
@@ -876,7 +965,8 @@ ${buildContextBlock(passages)}`;
|
|
|
876
965
|
toolName: call.name,
|
|
877
966
|
toolType: "action",
|
|
878
967
|
input: call.input,
|
|
879
|
-
status: "pending_approval"
|
|
968
|
+
status: "pending_approval",
|
|
969
|
+
runId: hooks.runId
|
|
880
970
|
}));
|
|
881
971
|
const decision = await hooks.awaitApproval(call, ctx);
|
|
882
972
|
deciderRef = decision.executedByRef ?? input.actor.id;
|
|
@@ -913,7 +1003,8 @@ ${buildContextBlock(passages)}`;
|
|
|
913
1003
|
toolName: call.name,
|
|
914
1004
|
toolType: "read",
|
|
915
1005
|
input: call.input,
|
|
916
|
-
status: "auto_executed"
|
|
1006
|
+
status: "auto_executed",
|
|
1007
|
+
runId: hooks.runId
|
|
917
1008
|
}));
|
|
918
1009
|
}
|
|
919
1010
|
const startedAt2 = Date.now();
|
|
@@ -938,7 +1029,10 @@ ${buildContextBlock(passages)}`;
|
|
|
938
1029
|
ctx: stepCtx,
|
|
939
1030
|
...deps.toolTimeoutMs !== void 0 ? {
|
|
940
1031
|
timeoutMs: deps.toolTimeoutMs
|
|
941
|
-
} : {}
|
|
1032
|
+
} : {},
|
|
1033
|
+
// Numeric-only: the handler applies withToolTimeout AND its own local `classify` — see
|
|
1034
|
+
// ToolStepEnvelope.transientRetry.
|
|
1035
|
+
transientRetry: resolveToolTransientRetryNumbers(deps.toolTransientRetry)
|
|
942
1036
|
};
|
|
943
1037
|
output = await hooks.dispatchTool(call, envelope);
|
|
944
1038
|
} else {
|
|
@@ -946,7 +1040,19 @@ ${buildContextBlock(passages)}`;
|
|
|
946
1040
|
toolCallId: call.id,
|
|
947
1041
|
toolName: call.name,
|
|
948
1042
|
toolType
|
|
949
|
-
}, () => deps.registry.invoke(call.name, call.input, ctx, deps.rolesPolicy)
|
|
1043
|
+
}, () => invokeWithTransientRetry(() => deps.registry.invoke(call.name, call.input, ctx, deps.rolesPolicy), deps.toolTransientRetry ?? {}, {
|
|
1044
|
+
...hooks.isControlFlowError !== void 0 ? {
|
|
1045
|
+
isControlFlowError: hooks.isControlFlowError
|
|
1046
|
+
} : {},
|
|
1047
|
+
onRetry: /* @__PURE__ */ __name((attempt, retryError) => {
|
|
1048
|
+
publishAgentToolRetry({
|
|
1049
|
+
toolName: call.name,
|
|
1050
|
+
toolCallId: call.id,
|
|
1051
|
+
attempt,
|
|
1052
|
+
message: retryError instanceof Error ? retryError.message : String(retryError)
|
|
1053
|
+
});
|
|
1054
|
+
}, "onRetry")
|
|
1055
|
+
})));
|
|
950
1056
|
output = deps.toolTimeoutMs !== void 0 ? await withToolTimeout(invocation, deps.toolTimeoutMs, call.name) : await invocation;
|
|
951
1057
|
}
|
|
952
1058
|
const executionMs = Date.now() - startedAt2;
|
|
@@ -1068,6 +1174,8 @@ export {
|
|
|
1068
1174
|
AGENT_TOOL_REGISTRY,
|
|
1069
1175
|
AgentRegistry,
|
|
1070
1176
|
AgentStreamError,
|
|
1177
|
+
DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS,
|
|
1178
|
+
DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS,
|
|
1071
1179
|
DefaultRolesPolicy,
|
|
1072
1180
|
QuotaExceededError,
|
|
1073
1181
|
ToolForbiddenError,
|
|
@@ -1084,6 +1192,8 @@ export {
|
|
|
1084
1192
|
estimateCost,
|
|
1085
1193
|
filterToolsByAllowList,
|
|
1086
1194
|
filterToolsByRole,
|
|
1195
|
+
invokeWithTransientRetry,
|
|
1196
|
+
isTransientToolError,
|
|
1087
1197
|
publishAgentDelegated,
|
|
1088
1198
|
publishAgentMessage,
|
|
1089
1199
|
publishAgentQuotaExceeded,
|
|
@@ -1092,6 +1202,8 @@ export {
|
|
|
1092
1202
|
publishAgentRunFinished,
|
|
1093
1203
|
publishAgentRunStarted,
|
|
1094
1204
|
publishAgentToolCall,
|
|
1205
|
+
publishAgentToolRetry,
|
|
1206
|
+
resolveToolTransientRetryNumbers,
|
|
1095
1207
|
runAgentLoop,
|
|
1096
1208
|
seedModelPrices,
|
|
1097
1209
|
traceLlmTurn,
|