@dudousxd/nestjs-agent-core 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1,6 +1,272 @@
1
1
  import { StandardSchemaV1 } from '@standard-schema/spec';
2
2
  import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
3
3
 
4
+ /**
5
+ * Asking the USER a structured question, and waiting for the answer.
6
+ *
7
+ * `awaitApproval` collects a yes/no about work already proposed; this collects the scope BEFORE the
8
+ * work. Two surfaces produce it — a configured intake (`AgentLoopDeps.intake`) and the model-callable
9
+ * `ask` tool (`AgentLoopDeps.ask`) — and they deliberately produce the SAME {@link
10
+ * ElicitationRequest}, persist through the same tool-call row, and resume through the same
11
+ * `tool:<runId>:<callId>` signal. A consumer cannot tell which one asked, and should not have to.
12
+ */
13
+
14
+ /** One choice a question offers. */
15
+ interface ElicitationOption {
16
+ /** Stable identifier submitted back. Never shown to the user. */
17
+ value: string;
18
+ /** What the user reads. */
19
+ label: string;
20
+ /**
21
+ * A single character a UI may bind as a keyboard shortcut for this option. Advisory — nothing in
22
+ * the library reads it, and a client is free to render its own.
23
+ */
24
+ hotkey?: string;
25
+ }
26
+ /** One question in a set. */
27
+ interface ElicitationQuestion {
28
+ /** Unique within its request; the key answers come back under. */
29
+ id: string;
30
+ prompt: string;
31
+ options: ElicitationOption[];
32
+ /** More than one option may be chosen. Omit → single choice. */
33
+ multiple?: boolean;
34
+ /**
35
+ * The options already picked for the user. The claim this whole surface makes is that confirming
36
+ * is enough, so a question with no defaults is a question the user must stop and think about —
37
+ * which is the case the design is trying to avoid. Empty/omitted is allowed and means exactly
38
+ * that: submitting without answering leaves this question unanswered.
39
+ */
40
+ defaults?: string[];
41
+ /** Accept values that are not among `options` (a typed-in answer). Omit → options only. */
42
+ allowFreeText?: boolean;
43
+ }
44
+ /**
45
+ * A question set awaiting a human. Identical in shape whether an `@Agent`'s configured intake or
46
+ * the model's `ask` tool authored it — `source` records which, for audit, not for control flow.
47
+ *
48
+ * `questions.length` is known when the request is written, which is what lets a client render
49
+ * "Question 1 of 3" without guessing whether a fourth is coming.
50
+ */
51
+ interface ElicitationRequest {
52
+ /** The tool-call id this request is persisted under, and the signal it is answered through. */
53
+ id: string;
54
+ source: 'intake' | 'ask';
55
+ /** What the assistant says above the form. */
56
+ preamble?: string;
57
+ questions: ElicitationQuestion[];
58
+ }
59
+ /** What a human sent back for an {@link ElicitationRequest}. */
60
+ interface ElicitationReply {
61
+ /**
62
+ * questionId → chosen values. A question whose id is ABSENT takes the request's own `defaults` —
63
+ * that is what makes "just submit" mean "yes, your pre-picked answers". A present-but-empty array
64
+ * is an explicit "none of these" and does NOT fall back.
65
+ */
66
+ answers: Record<string, string[]>;
67
+ /**
68
+ * The user declined to answer and told the agent to proceed on its own assumptions. Distinct from
69
+ * confirming the defaults even though the resulting values are the same: one is a decision the
70
+ * user made, the other is one they refused to make, and only the first is evidence of intent.
71
+ */
72
+ skipped?: boolean;
73
+ /** Opaque ref of WHO answered, when it wasn't the run's own actor. */
74
+ answeredByRef?: string;
75
+ }
76
+ /** A settled elicitation: what the agent proceeds on, and how it got there. */
77
+ interface ElicitationOutcome {
78
+ /** One entry per question, in request order — always present, so a caller never re-applies defaults. */
79
+ answers: Record<string, string[]>;
80
+ skipped: boolean;
81
+ /** Question ids filled from the request's `defaults` rather than by the human. */
82
+ defaulted: string[];
83
+ }
84
+ /**
85
+ * Read whatever the human channel delivered as an {@link ElicitationReply}.
86
+ *
87
+ * A question set is persisted as an `action` tool call in `pending_approval` — that is what puts it
88
+ * in the approvals inbox a deployment already has, instead of needing one of its own. The cost of
89
+ * that choice is that the thing which comes back may be a {@link Decision} someone pressed
90
+ * Approve/Reject on rather than a set of answers, and a `Decision` carries no `answers` at all.
91
+ *
92
+ * Approve means every question keeps its own pre-picked `defaults`, which is exactly what "just
93
+ * submit" already means on this surface; Reject is the same declining-to-answer a skip is. Neither
94
+ * reading is a guess — a yes/no channel cannot say more than that, and saying it here is what lets
95
+ * one inbox settle both kinds of pending work.
96
+ *
97
+ * Returns the reply UNCHANGED when it already carries answers, so the common path allocates nothing
98
+ * and a caller can identity-compare.
99
+ */
100
+ declare function normalizeElicitationReply(reply: ElicitationReply | Decision): ElicitationReply;
101
+ /**
102
+ * Settle a reply against the request it answers: fill every unanswered question from its own
103
+ * `defaults`, drop submitted values that aren't on offer, and collapse a single-choice question to
104
+ * one value.
105
+ *
106
+ * PURE, and deliberately so. Both of its inputs are already journaled by the time the loop calls it
107
+ * — the request came from module config or from an `llm:<i>` checkpoint, the reply from the signal
108
+ * checkpoint — so every process replaying the turn reaches the same values without a checkpoint of
109
+ * its own. Resolving defaults in the HTTP layer instead would put them behind a store read that a
110
+ * replay would have to repeat.
111
+ */
112
+ declare function resolveElicitation(request: ElicitationRequest, raw: ElicitationReply | Decision): ElicitationOutcome;
113
+ /**
114
+ * What a settled elicitation looks like to everyone downstream: the model reading it back as a tool
115
+ * result, the thread reader rendering it, the auditor asking what the agent was told to do. One
116
+ * shape for both surfaces — nothing here records which of them asked.
117
+ */
118
+ interface ElicitationResult extends ElicitationOutcome {
119
+ /** The questions against the chosen LABELS, so a reader (and a model) can act on it. */
120
+ summary: string;
121
+ }
122
+ /** {@link resolveElicitation} plus its human-readable rendering. Pure, for the same reason. */
123
+ declare function settleElicitation(request: ElicitationRequest, reply: ElicitationReply | Decision): ElicitationResult;
124
+ /**
125
+ * The answers as the model reads them: the question's own prompt against the chosen options' LABELS,
126
+ * not their opaque `value`s — a model shown `{"scope":["b"]}` has been told nothing.
127
+ */
128
+ declare function renderElicitationAnswers(request: ElicitationRequest, outcome: ElicitationOutcome): string;
129
+ /** The reserved tool name the model calls to ask the user something. */
130
+ declare const ASK_TOOL_NAME = "ask";
131
+ /** What the model must supply when it calls `ask`. */
132
+ interface AskToolInput {
133
+ preamble?: string;
134
+ questions: ElicitationQuestion[];
135
+ }
136
+ /** How many questions one `ask` may carry. A form the user has to scroll is a form they skip. */
137
+ declare const MAX_ASK_QUESTIONS = 5;
138
+ /**
139
+ * The `ask` tool's input schema, hand-written rather than borrowed from a validation library: core
140
+ * depends on no validator, and the schema has to carry a JSON Schema a provider can constrain
141
+ * generation against. It publishes one through the Standard JSON Schema extension
142
+ * (`~standard.jsonSchema.input`), which is the path the AI SDK adapter already recognises for
143
+ * Valibot / ArkType / Zod 4.
144
+ */
145
+ declare const askInputSchema: StandardSchemaV1<unknown, AskToolInput>;
146
+ /**
147
+ * What the model is told the `ask` tool is for. Written to discourage the two failure modes that
148
+ * make a clarifying question worse than a guess: asking about something the conversation already
149
+ * settled, and asking without saying what you would have done.
150
+ */
151
+ declare const ASK_TOOL_DESCRIPTION = "Ask the user to settle the scope of the work before you do it. Use it when a reasonable person would produce a materially different result depending on the answer \u2014 not to confirm something the conversation already says. Every question must pre-pick the answer you would choose, so the user can confirm instead of deciding. The user may decline, in which case you proceed on those pre-picked answers.";
152
+ /**
153
+ * The `ask` tool as the model sees it. NOT a `ToolSpec` and never registered: `ask` has no handler,
154
+ * because the loop settles it against a human instead of invoking anything. Keeping it out of the
155
+ * `ToolRegistry` is also what keeps the kind decision off a process-local lookup — see
156
+ * `claimToolCall`.
157
+ */
158
+ declare function askToolDefinition(): ToolDefinition;
159
+ /** A question set an `@Agent` asks before it starts working. See `AgentLoopDeps.intake`. */
160
+ interface AgentIntake {
161
+ questions: ElicitationQuestion[];
162
+ /** What the assistant says above the form. Omit → {@link DEFAULT_INTAKE_PREAMBLE}. */
163
+ preamble?: string;
164
+ /**
165
+ * `'thread-start'` (default) asks once, on the first turn of a thread; `'every-turn'` asks before
166
+ * every turn. Both are decided from what `load:thread` recorded about the thread when the turn
167
+ * began, never from anything this process happens to know — by the time a replay reaches the
168
+ * question, the thread already holds the assistant message the first attempt wrote.
169
+ */
170
+ when?: 'thread-start' | 'every-turn';
171
+ }
172
+ declare const DEFAULT_INTAKE_PREAMBLE = "A few questions before I start. I have pre-picked what I would choose, so confirming is enough.";
173
+
174
+ /**
175
+ * The ceiling on how much of a thread rides into a turn. Without one, `runAgentLoop` maps EVERY
176
+ * message the store returns into the model's messages, so a long-lived thread grows until the
177
+ * provider rejects the request — and every turn before that one pays for the whole transcript.
178
+ * `@dudousxd/nestjs-agent-core` ships `windowHistory` as the built-in; anything satisfying this SPI
179
+ * works. Wire it as `AgentLoopDeps.historyPolicy`, or via `AgentModule.forRoot({ history })` /
180
+ * `@Agent({ history })`.
181
+ */
182
+
183
+ /** Whose history this is — enough for a policy to size the window per agent or per actor. */
184
+ interface HistoryPolicyContext {
185
+ threadId: string;
186
+ actor: Actor;
187
+ /** The agent running this turn. Undefined → the default agent. */
188
+ agentName?: string;
189
+ }
190
+ /** How a policy split the thread: what rides into the turn, and what the ceiling left out. */
191
+ interface HistorySelection {
192
+ /** Sent to the model, oldest-first. */
193
+ keep: ModelMessage[];
194
+ /** Left out, oldest-first. Folded into a leading summary when the policy implements `summarize`. */
195
+ drop: ModelMessage[];
196
+ }
197
+ /** A stand-in for the messages a window left out, plus what producing it cost. */
198
+ interface HistorySummary {
199
+ /** Prose the loop folds into the window as a leading `system` message. */
200
+ text: string;
201
+ /**
202
+ * What the summarizer spent, when it called a model. Recorded as a `history_summary` usage row, so
203
+ * a ceiling on context cost cannot itself become spend nothing accounts for. Omit for a summarizer
204
+ * that calls no model (a rollup of tool names, a digest the app already stored).
205
+ */
206
+ usage?: MessageUsage;
207
+ /** Accounting label for the model that produced it; falls back to `AgentLoopDeps.modelId`. */
208
+ modelId?: string;
209
+ }
210
+ interface HistoryPolicy {
211
+ /**
212
+ * The most messages {@link select} can ever keep, where the ceiling can be stated as a row count.
213
+ *
214
+ * A hint the loop hands to the store, which then reads that many of the thread's newest rows
215
+ * rather than its whole transcript (see `ThreadTurnReader`). Declaring it is a PROMISE about
216
+ * `select`: that it keeps at most this many messages, and that they are the NEWEST ones — so a
217
+ * window of this size is indistinguishable, to `select`, from the full transcript. A policy that
218
+ * can keep more than this, or can keep something older than the newest `maxMessages`, must omit it
219
+ * rather than select over rows the store was never asked for.
220
+ *
221
+ * Omit where the ceiling is not a row count at all — a token budget alone cannot name one, since
222
+ * one message can be four tokens or forty thousand. Omitting costs only the bound on the READ; the
223
+ * prompt is identical either way.
224
+ */
225
+ readonly maxMessages?: number;
226
+ /**
227
+ * Decide what the model sees.
228
+ *
229
+ * MUST be a pure function of `messages`. The loop calls it INSIDE the `load:thread` checkpoint and
230
+ * records its result there, so the ceiling bounds the journal as well as the prompt: what the
231
+ * checkpoint holds is the selection rather than the store's whole `ThreadDetail`, on a payload
232
+ * every replay re-reads. It still adds no position of its own — a policy cannot change the name or
233
+ * position of a single existing checkpoint.
234
+ *
235
+ * Read a clock, a feature flag or a database here and the resumed run windows differently from the
236
+ * one that suspended: the model gets a different prompt, and any step whose existence depends on
237
+ * the split lands at a position the history has no room for. Anything non-deterministic belongs in
238
+ * {@link summarize}, which has a checkpoint of its own.
239
+ *
240
+ * The newest message must always be in `keep` — dropping it leaves the turn with nothing to answer.
241
+ */
242
+ select(messages: ModelMessage[], ctx: HistoryPolicyContext): HistorySelection;
243
+ /**
244
+ * Fold the dropped messages into prose the model reads in their place, prepended to the window as
245
+ * a `system` message. Optional — without it, dropped messages are simply gone, and `load:thread`
246
+ * does not record them either: a summarizer is the only thing that ever reads them back.
247
+ *
248
+ * Runs inside the loop's `history:summarize` checkpoint, so it may call a model or hit the
249
+ * network: the first attempt's result is journaled and every replay reads it back instead of
250
+ * re-summarizing. It runs once per RUN (not per model step), and only when `select` actually
251
+ * dropped something.
252
+ */
253
+ summarize?(dropped: ModelMessage[], ctx: HistoryPolicyContext): Promise<HistorySummary>;
254
+ }
255
+ /**
256
+ * The window an agent asks for declaratively — `AgentModule.forRoot({ history })` and
257
+ * `@Agent({ history })`. Plain data, so it can live in a decorator's metadata; the NestJS layer
258
+ * turns it into a `windowHistory` policy. A consumer needing anything the window can't express
259
+ * supplies a {@link HistoryPolicy} instead.
260
+ */
261
+ interface AgentHistoryWindow {
262
+ /** Keep at most this many of the newest messages. */
263
+ maxMessages?: number;
264
+ /** Keep the newest messages whose estimated tokens fit this budget. */
265
+ maxTokens?: number;
266
+ /** Fold what the window left out into a leading summary — one extra model call per run. */
267
+ summarize?: boolean;
268
+ }
269
+
4
270
  /**
5
271
  * Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
6
272
  * classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
@@ -47,9 +313,10 @@ interface ToolTransientRetryNumbers {
47
313
  declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
48
314
  interface InvokeWithTransientRetryOptions {
49
315
  /**
50
- * Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so a retry
51
- * never swallows one — same rule the loop's tool catch already applies. Undefined for a call site
52
- * with no such notion (e.g. the dispatched step handler, which has no workflow ctx of its own).
316
+ * Widens what counts as a control-flow signal (durable suspend / continue-as-new) so a retry never
317
+ * swallows one — same rule the loop's tool catch applies. An ADDITION to the built-in marker check
318
+ * ({@link isControlFlowSignal}), which already covers every signal the durable runtimes raise;
319
+ * supply this only for a runner whose signals carry no marker.
53
320
  */
54
321
  isControlFlowError?: (error: unknown) => boolean;
55
322
  /**
@@ -76,13 +343,44 @@ interface Actor {
76
343
  roles?: string[];
77
344
  tenantRef?: string;
78
345
  }
79
- type ToolKind = 'read' | 'action' | 'agent';
346
+ type ToolKind = 'read' | 'action' | 'agent' | 'ask' | 'skill' | 'memory';
347
+ /**
348
+ * One agent->agent edge on {@link AgentDefinition.delegatesTo}. A bare name is the awaited
349
+ * delegation that has always existed; the object form is how an author says this one runs in the
350
+ * background.
351
+ */
352
+ type AgentDelegation = string | {
353
+ agent: string;
354
+ detached?: boolean;
355
+ };
356
+ /**
357
+ * Where a detached sub-agent run posts its answer: the thread that delegated it, and the
358
+ * `agent`-kind tool call that started it. Carried on the child run's own {@link AgentRunInput},
359
+ * because by the time the child finishes the parent turn is over and nothing is holding the
360
+ * address.
361
+ */
362
+ interface DetachedDelivery {
363
+ threadId: string;
364
+ toolCallId: string;
365
+ }
80
366
  /**
81
367
  * Declared shape of a tool.
82
368
  * - `read` auto-executes.
83
369
  * - `action` never auto-executes — requires HITL approval.
84
370
  * - `agent` delegates to another named agent (durable: a child workflow; inline: a nested loop),
85
371
  * handled at the loop level — NOT via a handler. Carries `targetAgent`.
372
+ * - `ask` puts a question set to the user and waits for the answers (see `elicitation.ts`).
373
+ * Handled at the loop level and never registered, so no `ToolSpec` carries this kind:
374
+ * the only tool that has it is the built-in `ask`, whose definition the loop supplies.
375
+ * - `skill` reads one of the procedures offered in the turn's `<skills>` catalog (see `skills.ts`).
376
+ * Auto-executes like a read and performs nothing, but is served by the LOOP from the
377
+ * catalog the journal holds rather than by a registered handler — so, like `ask`, no
378
+ * `ToolSpec` carries this kind.
379
+ * - `memory` records one fact about the actor for later turns (see `memory.ts`). Served by the LOOP
380
+ * and never registered, like `skill`. It WRITES, but auto-executes rather than asking
381
+ * for approval: an agent may only ever write the scope of the actor it is running for,
382
+ * so the blast radius of a bad one is the prompt of the person who was talking, and the
383
+ * remedy is the read-back that lets them delete it.
86
384
  */
87
385
  interface ToolSpec {
88
386
  name: string;
@@ -96,6 +394,20 @@ interface ToolSpec {
96
394
  inputSchema: StandardSchemaV1;
97
395
  /** For `kind: 'agent'` — the name of the agent to delegate to. */
98
396
  targetAgent?: string;
397
+ /**
398
+ * For `kind: 'agent'` — start the delegation and let the calling turn END, instead of holding it
399
+ * open until the delegate answers. The call's result is a {@link DetachedDelegationReceipt}, and
400
+ * the answer arrives later as its own message in the same thread (see
401
+ * {@link AgentRunInput.deliverTo}).
402
+ *
403
+ * Authored per EDGE, never chosen by the model: a model that can decide to detach can decide to
404
+ * detach the one thing the user is sitting there waiting for, and it has no way to know which that
405
+ * is. The person wiring `A -> B` does.
406
+ *
407
+ * Settled into the call's `persist:toolcall` checkpoint alongside `targetAgent`, so every replay
408
+ * reads the branch back rather than re-deciding it against a registry that may have changed.
409
+ */
410
+ detached?: boolean;
99
411
  /** Roles allowed to invoke. Undefined → defaults applied by RolesPolicy (e.g. ADMIN-only). */
100
412
  roles?: string[];
101
413
  /**
@@ -176,7 +488,14 @@ interface MessageUsage {
176
488
  */
177
489
  costUsd?: number | null;
178
490
  }
179
- type UsagePurpose = 'chat' | 'follow_ups';
491
+ /**
492
+ * What a usage row was spent ON. `chat` is a model step of the turn itself; `follow_ups` is the
493
+ * extra call that proposes follow-up questions; `history_summary` is the extra call a
494
+ * {@link import('./spi/history-policy.js').HistoryPolicy} makes to fold windowed-out messages into a
495
+ * summary — so bounding context cost never becomes spend nothing accounts for; `structured_output`
496
+ * is the formatting pass that restates a finished answer as `AgentLoopDeps.outputSchema` requires.
497
+ */
498
+ type UsagePurpose = 'chat' | 'follow_ups' | 'history_summary' | 'structured_output';
180
499
  interface QuotaState {
181
500
  usedTokens: number;
182
501
  limitTokens: number;
@@ -194,6 +513,12 @@ interface QuotaView {
194
513
  withinLimit: boolean;
195
514
  costUsd: number;
196
515
  }
516
+ /**
517
+ * Anything a human sends back into a parked run: a {@link Decision} on an action tool, or an
518
+ * `ElicitationReply` answering a question set. Both travel the same `tool:<runId>:<toolCallId>`
519
+ * signal, so the runner that delivers them does not need to know which it is carrying.
520
+ */
521
+ type HumanReply = Decision | ElicitationReply;
197
522
  /** A human decision on a pending action tool call. */
198
523
  interface Decision {
199
524
  approved: boolean;
@@ -273,9 +598,19 @@ interface AgentRunInput {
273
598
  agentName?: string;
274
599
  /**
275
600
  * How many agent→agent delegations deep this run already is (0 for a top-level turn). The runner
276
- * increments it for each child run; the loop refuses to delegate past {@link MAX_DELEGATION_DEPTH}.
601
+ * increments it for each child run; the loop refuses to delegate past its depth ceiling.
277
602
  */
278
603
  delegationDepth?: number;
604
+ /**
605
+ * The named agents already on this delegation chain, root first — what {@link delegationDepth}
606
+ * counts, spelled out. The runner appends its own agent's name for each child it starts.
607
+ *
608
+ * A count can only say a chain is LONG. This says whether it is going in circles, and how often:
609
+ * an agent that appears here is one the chain has already passed through, so a delegation back to
610
+ * it is a cycle by inspection rather than by proxy. A run whose runner does not supply it falls
611
+ * back to the depth ceiling alone.
612
+ */
613
+ delegationPath?: readonly string[];
279
614
  /**
280
615
  * When set, this run streams into ANOTHER run's sink instead of its own. A sub-agent run carries
281
616
  * its top-level ancestor's runId here so its tokens (and its pending action-tool frames) land in
@@ -283,6 +618,18 @@ interface AgentRunInput {
283
618
  * approve, a sub-agent's HITL action. Propagated unchanged down the delegation chain.
284
619
  */
285
620
  sinkRunId?: string;
621
+ /**
622
+ * Set on a DETACHED sub-agent run: the thread and tool call this run answers into when it
623
+ * finishes. Its presence is also what makes a run detached from the inside — it has no ancestor
624
+ * sink to stream into, so nothing else distinguishes it from a top-level turn.
625
+ */
626
+ deliverTo?: DetachedDelivery;
627
+ /**
628
+ * The run that started this one (a delegation's parent). Recorded with the run so a governance
629
+ * surface can roll a delegation's cost up to the turn that asked for it; a detached child is
630
+ * otherwise a row with nothing pointing at it.
631
+ */
632
+ parentRunId?: string;
286
633
  /**
287
634
  * Re-run the last exchange instead of adding a new message: the loop truncates everything after
288
635
  * the thread's last user message and re-answers it (no `userText` is appended). Used by a
@@ -305,10 +652,51 @@ interface AgentDefinition {
305
652
  systemPrompt?: string | PromptBuilder;
306
653
  /** Allow-list of tool names this agent may use (subset of all registered tools). */
307
654
  tools?: string[];
308
- /** Names of other agents this agent may hand off to (auto-registered as `agent`-kind tools). */
309
- delegatesTo?: string[];
655
+ /**
656
+ * Other agents this agent may hand off to (auto-registered as `agent`-kind tools). A bare name
657
+ * is the awaited form; `{ agent, detached: true }` starts the delegate and lets this agent's turn
658
+ * finish without its answer — see {@link ToolSpec.detached}.
659
+ */
660
+ delegatesTo?: AgentDelegation[];
310
661
  modelId?: string;
311
662
  maxSteps?: number;
663
+ /**
664
+ * How deep delegation may nest below this agent. Undefined → {@link MAX_DELEGATION_DEPTH}.
665
+ *
666
+ * Bounds the CHAIN, not the fan-out: how many agents a turn delegates to is the model's.
667
+ */
668
+ maxDelegationDepth?: number;
669
+ /**
670
+ * How many times one agent may appear on a single delegation chain.
671
+ * Undefined → {@link DEFAULT_MAX_AGENT_APPEARANCES}.
672
+ */
673
+ maxAgentAppearances?: number;
674
+ /**
675
+ * This agent's own ceiling on how much of a thread rides into its turn, overriding the
676
+ * module-wide one. A persona that reasons over a long back-and-forth and one that answers a single
677
+ * question from a page context want very different windows.
678
+ */
679
+ history?: AgentHistoryWindow;
680
+ /**
681
+ * Constrain this agent's final answer to a schema. A live schema INSTANCE, so it is resolved from
682
+ * DI on whichever process runs the turn and never travels on `AgentRunInput` — a Standard Schema
683
+ * cannot survive the JSON hop into a durable workflow, which is why there is no per-request
684
+ * override on the HTTP surface.
685
+ */
686
+ outputSchema?: StandardSchemaV1;
687
+ /**
688
+ * Extra model calls allowed to fix an answer that failed {@link outputSchema}. Undefined → 1.
689
+ */
690
+ outputRepairAttempts?: number;
691
+ /**
692
+ * Questions this agent puts to the user BEFORE it starts working. Authored, so the turn pays no
693
+ * model call to produce them and a client knows the total up front. Undefined → no intake.
694
+ */
695
+ intake?: AgentIntake;
696
+ /**
697
+ * Whether this agent is offered the built-in `ask` tool. Undefined → the module-wide setting.
698
+ */
699
+ ask?: boolean;
312
700
  }
313
701
  /**
314
702
  * The read-model the `GET agents` endpoint returns to a client — the safe public subset of an
@@ -352,6 +740,8 @@ interface StoredMessage {
352
740
  attachments?: MessageAttachment[];
353
741
  followUps?: string[];
354
742
  usage?: MessageUsage;
743
+ /** The run (turn) that produced this message; absent on a row written before this was recorded. */
744
+ runId?: string;
355
745
  createdAt: string;
356
746
  }
357
747
  interface ThreadDetail extends ThreadSummary {
@@ -370,6 +760,13 @@ interface LlmStepEnvelope {
370
760
  messages: ModelMessage[];
371
761
  /** The turn's actor — the handler re-derives tool definitions from it (definitionsFor). */
372
762
  actor: Actor;
763
+ /**
764
+ * Hold this call's stream frames rather than writing them to the run's sink, and return them on
765
+ * the result. Set by the loop when an output processor has to see the whole answer before the
766
+ * subscriber does — the dispatched handler streams to a worker-side sink the loop cannot
767
+ * interpose on, so the instruction has to ride the envelope. Absent → stream live, as before.
768
+ */
769
+ bufferOutput?: boolean;
373
770
  }
374
771
  /**
375
772
  * The serializable subset of `AiToolCtx` — everything except `host` (re-attached handler-side
@@ -455,6 +852,23 @@ declare const AGENT_ATTACHMENT_STAGING: unique symbol;
455
852
  * {@link import('./spi/approval-port.js').AgentApprovalPort}.
456
853
  */
457
854
  declare const AGENT_APPROVAL_PORT: unique symbol;
855
+ /**
856
+ * The resolved `SkillsConfig` (provider + scope resolver + ceiling), or `undefined` where the host
857
+ * configured no skills. Bound once and read by BOTH the loop's deps and the listing endpoint, so
858
+ * what a user can invoke and what the model can reach are the same resolution.
859
+ */
860
+ declare const AGENT_SKILLS: unique symbol;
861
+ /**
862
+ * The resolved `MemoryConfig` (provider + scope resolver + ceilings), or `undefined` where the host
863
+ * configured no memory. Bound once and read by BOTH the loop's deps and the read-back endpoint, so
864
+ * what a person is shown and what the model was shown are the same resolution.
865
+ */
866
+ declare const AGENT_MEMORY: unique symbol;
867
+ /**
868
+ * A shared, mutable list `SkillDiscoveryService` fills with `@Skill`-decorated providers. Read
869
+ * lazily by the bound {@link AGENT_SKILLS} provider — discovery runs after DI has built it.
870
+ */
871
+ declare const AGENT_SKILL_SOURCES: unique symbol;
458
872
 
459
873
  /**
460
874
  * Per-invocation context handed to a tool handler. Host-supplied bits are optional. Identity lives
@@ -555,6 +969,14 @@ interface ModelTurnArgs {
555
969
  /** The model writes streamed text deltas here as it generates them. */
556
970
  sink: SinkWriter;
557
971
  abortSignal?: AbortSignal;
972
+ /**
973
+ * Constrain this turn's reply to a schema (a provider's JSON/response-format mode). Set only on
974
+ * the loop's structured-output formatting pass, which always sends `tools: []` — most providers
975
+ * refuse a response format and a tool set in the same request, and the ones that accept both stop
976
+ * calling tools. A provider that cannot constrain generation may ignore this: the loop validates
977
+ * the reply against the same schema either way, so ignoring it costs reliability, not safety.
978
+ */
979
+ outputSchema?: StandardSchemaV1;
558
980
  }
559
981
  /** The outcome of ONE assistant turn. The loop — not the model — drives tool execution. */
560
982
  interface ModelTurnResult {
@@ -574,6 +996,41 @@ interface ModelTurnResult {
574
996
  * governance read-model uses it verbatim; otherwise it estimates from tokens × the pricing table.
575
997
  */
576
998
  costUsd?: number;
999
+ /**
1000
+ * The reply already parsed, for a provider that constrained generation against
1001
+ * {@link ModelTurnArgs.outputSchema} and therefore has the value in hand. A fast path only — the
1002
+ * loop validates it against the schema regardless, so omitting it (and leaving the loop to read
1003
+ * the JSON out of `text`) is always correct.
1004
+ */
1005
+ object?: unknown;
1006
+ }
1007
+ /**
1008
+ * A model turn whose live frames were HELD instead of streamed, because an output processor has to
1009
+ * see the whole answer before anything downstream does. Produced by whatever ran the model call —
1010
+ * the loop itself, or the dispatched-step handler when the envelope asked for it — and released to
1011
+ * the real sink by the loop once the gate has passed.
1012
+ */
1013
+ interface BufferedModelTurnResult extends ModelTurnResult {
1014
+ /** The NDJSON stream lines the call produced, in write order. */
1015
+ bufferedFrames?: string[];
1016
+ /**
1017
+ * The transformed PREFIX an incremental gate already wrote to the run's sink while the call was
1018
+ * streaming. Its presence — not the chain configured on the process that reads it back — is what
1019
+ * tells the gate step which release this turn still owes, so a run that resumes under a
1020
+ * re-declared chain cannot flush the same text twice. Empty string when an incremental gate ran
1021
+ * and released nothing; absent when none ran.
1022
+ */
1023
+ releasedText?: string;
1024
+ /**
1025
+ * The refusal an incremental gate reached from a prefix, carried on the RESULT rather than thrown
1026
+ * from the call. The loop raises it only after the turn's `persist:usage` and `quota:bump`
1027
+ * checkpoints — those tokens were genuinely spent, and a gate that hid its own cost would let a
1028
+ * mis-tuned chain burn a budget invisibly.
1029
+ */
1030
+ gateRejection?: {
1031
+ processor: string;
1032
+ reason: string;
1033
+ };
577
1034
  }
578
1035
  /**
579
1036
  * Thin wrapper over the actual LLM. The concrete impl (e.g. Vercel AI SDK `streamText`
@@ -644,9 +1101,41 @@ type AgentStreamEvent = {
644
1101
  kind: 'tool-output-error';
645
1102
  id: string;
646
1103
  error: string;
1104
+ }
1105
+ /**
1106
+ * The run has put a question set to the user and is parked until someone answers it (or skips).
1107
+ * Written by the LOOP for both elicitation surfaces — the configured intake and the model's `ask`
1108
+ * tool — so a client renders one form either way rather than learning to recognise a tool name.
1109
+ * The matching `tool-output` frame, under the same `id`, carries the settled answers.
1110
+ */
1111
+ | {
1112
+ kind: 'elicitation';
1113
+ id: string;
1114
+ request: ElicitationRequest;
1115
+ }
1116
+ /**
1117
+ * Someone stopped this run. The stream's LAST frame, written by the runner that settled the
1118
+ * cancel, immediately before a normal `end()` — never a `fail()`, because a cancel is not an
1119
+ * error and a client that retries on a failed stream must not retry this.
1120
+ *
1121
+ * A run that simply ends wrote everything it had; one that ends after this frame did not, and the
1122
+ * difference is the whole point: without it a reader cannot tell a truncated answer from a
1123
+ * complete one. Consumers that predate the frame ignore it and see the `end()` they always saw.
1124
+ */
1125
+ | {
1126
+ kind: 'cancelled';
647
1127
  };
648
1128
  /** Encode one event as an NDJSON line (`{...}\n`) for {@link SinkWriter.write}. */
649
1129
  declare function encodeStreamEvent(event: AgentStreamEvent): Uint8Array;
1130
+ /**
1131
+ * Read one NDJSON line back, or `null` when the line is not a stream event at all.
1132
+ *
1133
+ * `null` covers a genuinely opaque chunk, not just malformed JSON: the sink is a byte channel, so a
1134
+ * model provider is free to write anything into it and some write bare text. A caller that has to
1135
+ * CLASSIFY a chunk — the output gate, which may only forward what it can prove is not the answer —
1136
+ * treats an unreadable frame as unclassifiable rather than guessing.
1137
+ */
1138
+ declare function decodeStreamEvent(line: string): AgentStreamEvent | null;
650
1139
 
651
1140
  interface CreateThreadInput {
652
1141
  actor: Actor;
@@ -665,6 +1154,14 @@ interface AppendMessageInput {
665
1154
  attachments?: MessageAttachment[];
666
1155
  followUps?: string[];
667
1156
  usage?: MessageUsage;
1157
+ /**
1158
+ * The run (turn) that produced this message. Without it a consumer can only guess which turn a
1159
+ * message belongs to by comparing timestamps against the run's `startedAt`, and that guess breaks
1160
+ * the moment a turn is regenerated — the replaced answer is truncated away, so the times no longer
1161
+ * line up 1:1. Optional so a caller predating this (and a host that appends messages outside a
1162
+ * run) can omit it; the store persists it as `null` when absent.
1163
+ */
1164
+ runId?: string;
668
1165
  }
669
1166
  interface RecordToolCallInput {
670
1167
  toolCallId: string;
@@ -709,16 +1206,67 @@ interface RecordRunStartInput {
709
1206
  threadId: string;
710
1207
  actorRef: string;
711
1208
  agentName?: string;
1209
+ /**
1210
+ * The run that started this one, for a delegation's child run. The parent->child edge exists in
1211
+ * the durable runtime's own journal, but only there: a governance surface reading run rows alone
1212
+ * cannot roll a delegation's cost up to the turn that asked for it, and a DETACHED child outlives
1213
+ * its parent's turn entirely, so nothing in the transcript pairs them either.
1214
+ *
1215
+ * Optional, and a store that persists nothing for it still works — it loses the tree, not the run.
1216
+ */
1217
+ parentRunId?: string;
712
1218
  /** sha256 hex of the run's resolved (pre-RAG) system prompt — identifies the prompt VERSION. */
713
1219
  promptHash?: string;
714
1220
  }
715
1221
  interface RecordRunEndInput {
716
1222
  runId: string;
717
- status: 'completed' | 'failed';
1223
+ /**
1224
+ * `cancelled` is a THIRD terminal, not a flavour of `failed`: someone asked the run to stop and it
1225
+ * did, which is the control working. A consumer computing a failure rate over these rows has to be
1226
+ * able to leave it out — counting a user pressing Stop as an error pages whoever is on call for
1227
+ * model failures. It carries no `errorCode`/`errorMessage`, since there is nothing to diagnose.
1228
+ */
1229
+ status: 'completed' | 'failed' | 'cancelled';
718
1230
  durationMs?: number;
719
1231
  errorCode?: string;
720
1232
  errorMessage?: string;
721
1233
  }
1234
+ /** Which thread, and how many of its newest messages, {@link ThreadTurnReader.loadThreadForTurn} reads. */
1235
+ interface ThreadTurnQuery {
1236
+ threadId: string;
1237
+ /** Omitted reads every message; `0` reads none. */
1238
+ messageLimit?: number;
1239
+ }
1240
+ /** What a turn reads off a thread — a bounded window, not the transcript. */
1241
+ interface ThreadTurnPage {
1242
+ title: string;
1243
+ defaultAgent: string | null;
1244
+ /** Whether the THREAD has ever been answered, not whether {@link messages} holds an answer. */
1245
+ hasAssistantMessage: boolean;
1246
+ /** Oldest first, carrying only the fields a model turn reads. */
1247
+ messages: StoredMessage[];
1248
+ }
1249
+ /**
1250
+ * A store that can hand a turn the WINDOW it is about to send, instead of the thread's transcript.
1251
+ *
1252
+ * {@link AgentStore.getThread} materializes every message row, every attachment and every tool
1253
+ * output a thread ever recorded, and the run then journals what it loaded — so a long thread pays
1254
+ * for its whole history on every turn and again on every replay, to send a prompt bounded to its
1255
+ * last few messages. This read is bounded by the database (`order by created_at desc limit ?`),
1256
+ * projected to the columns a model turn actually reads.
1257
+ *
1258
+ * `hasAssistantMessage` is answered over the WHOLE thread, never the page: it answers "has this
1259
+ * conversation been answered before?" — what a `thread-start` intake asks — and a thread whose window
1260
+ * happens to hold only the user's last questions has still been answered. `null` for a thread that is
1261
+ * unknown or soft-deleted, matching `getThread`.
1262
+ *
1263
+ * Probed STRUCTURALLY rather than declared on {@link AgentStore}, the same way `defaultAgentForThread`
1264
+ * is: it is an optimization a store either offers or does not, and one that predates it still answers
1265
+ * correctly through the full read.
1266
+ */
1267
+ interface ThreadTurnReader {
1268
+ loadThreadForTurn(query: ThreadTurnQuery): Promise<ThreadTurnPage | null>;
1269
+ }
722
1270
  /** ORM-agnostic persistence. Refs are string ids; adapters may add real relations. */
723
1271
  interface AgentStore {
724
1272
  createThread(input: CreateThreadInput): Promise<ThreadSummary>;
@@ -769,11 +1317,15 @@ interface AgentStore {
769
1317
  */
770
1318
  ownerOfToolCall(toolCallId: string): Promise<string | null>;
771
1319
  /**
772
- * The runId currently streaming the thread a tool call belongs to (its thread's `activeStreamId`),
773
- * or `null` if the call or its active run is unknown. HITL approve / reject route the decision to
774
- * THIS run, derived server-side from the tool call alone — so a decision reaches the exact run
1320
+ * The run awaiting a decision on `toolCallId`: the call's OWN `runId` when the row carries one,
1321
+ * else the thread's `activeStreamId`. Both HITL approve/reject and an elicitation answer route
1322
+ * through this, derived server-side from the tool call alone — so a decision reaches the exact run
775
1323
  * awaiting it, including a sub-agent's own child run, which the client never sees and could not
776
1324
  * name. No client-supplied runId is trusted (or needed).
1325
+ *
1326
+ * The row's own runId comes FIRST because `activeStreamId` names whichever run is streaming the
1327
+ * thread right now, and that is only the same run while a thread holds exactly one. The fallback
1328
+ * is for rows written before tool calls recorded a runId, which have nothing else to answer with.
777
1329
  */
778
1330
  runForToolCall(toolCallId: string): Promise<string | null>;
779
1331
  /**
@@ -783,9 +1335,43 @@ interface AgentStore {
783
1335
  */
784
1336
  ownerOfActiveStream(runId: string): Promise<string | null>;
785
1337
  appendMessage(input: AppendMessageInput): Promise<StoredMessage>;
1338
+ /**
1339
+ * Attach a turn's settled tool RESULTS to a message that was already appended, replacing whatever
1340
+ * it held. A message's tool calls are known when it is written and their outputs are not, but a
1341
+ * thread reader pairs the two off THAT MESSAGE — so an output that only ever reaches the tool-call
1342
+ * table leaves every call on a reopened thread looking like a tool still running.
1343
+ *
1344
+ * Required rather than optional: a store that silently declines this renders a finished turn as a
1345
+ * permanently in-flight one, with nothing logged and nothing to notice. A missing method should
1346
+ * fail to compile instead.
1347
+ */
1348
+ setMessageToolResults(messageId: string, results: ToolResult[]): Promise<void>;
786
1349
  truncateFrom(threadId: string, messageId: string): Promise<void>;
787
1350
  recordToolCall(input: RecordToolCallInput): Promise<void>;
788
1351
  updateToolCall(input: UpdateToolCallInput): Promise<void>;
1352
+ /**
1353
+ * OPTIONAL: of `mediaIds`, the ones a message that still exists — in a thread owned by
1354
+ * `actorRef` — still carries as an attachment. The inverse of
1355
+ * {@link import('./attachment-staging.js').AttachmentStagingStore.list}: the host can enumerate
1356
+ * the media it staged but cannot see a transcript, and this side sees every transcript but never
1357
+ * holds the bytes, so neither can decide alone what is safe to delete.
1358
+ *
1359
+ * DERIVED, not tracked. A reference is not permanent: `truncateFrom` deletes messages — which is
1360
+ * exactly what regenerating a turn does — so media that was referenced becomes unreferenced
1361
+ * again. A flag set when a message is sent would never be unset by that delete, and the bytes
1362
+ * would be pinned for ever with nothing pointing at them. Answering from the surviving message
1363
+ * rows on every call is the only form of this that stays true after a truncation.
1364
+ *
1365
+ * Scoped to one actor, like every other read on this surface: media referenced only by ANOTHER
1366
+ * actor's thread is reported unreferenced here, so this can never be turned into a probe for what
1367
+ * exists in someone else's conversation. A host pairs it with its own per-actor inventory, so the
1368
+ * candidate ids are already the caller's own.
1369
+ *
1370
+ * Returns each id at most once, in the order asked. Absent on a store that predates this — a
1371
+ * caller must treat the absence as "cannot answer" and collect NOTHING, never as "nothing is
1372
+ * referenced", which would delete every attachment the actor ever sent.
1373
+ */
1374
+ referencedMediaIds?(actorRef: string, mediaIds: readonly string[]): Promise<string[]>;
789
1375
  recordUsage(input: RecordUsageInput): Promise<void>;
790
1376
  /**
791
1377
  * The actor's spend for `day` (UTC): total tokens plus the summed provider-reported USD cost.
@@ -879,6 +1465,163 @@ interface Retriever {
879
1465
  retrieve(query: string, options?: RetrieveOptions): Promise<Passage[]>;
880
1466
  }
881
1467
 
1468
+ /**
1469
+ * The two seams that see a turn's traffic to and from the model: an {@link InputProcessor} rewrites
1470
+ * the prompt on its way out, an {@link OutputProcessor} inspects the answer on its way back and may
1471
+ * redact, replace or refuse it. Wire them as `AgentLoopDeps.inputProcessors` /
1472
+ * `outputProcessors`, or via `AgentModule.forRoot({ inputProcessors, outputProcessors })`.
1473
+ *
1474
+ * WHAT THESE ARE NOT: a second way to decide what enters the context. `HistoryPolicy` owns
1475
+ * SELECTION — which of the thread's messages ride into the turn, and what stands in for the rest.
1476
+ * Processors run on whatever selection produced and own TRANSFORMATION — what those messages say.
1477
+ * The distinction is load-bearing rather than stylistic: selection is contractually pure and is what
1478
+ * the `load:thread` checkpoint records, so it bounds the journal as well as the prompt, while a
1479
+ * processor is allowed to call a model, runs in a checkpoint of its own per step, and rewrites a
1480
+ * DERIVED prompt that never becomes the thread's own memory of what was said. Splitting the same
1481
+ * decision across both means neither can be reasoned about alone, and the cheap one stops being the
1482
+ * whole answer to "why did this turn cost that much".
1483
+ */
1484
+
1485
+ /** Which turn, and which model step of it, a processor is looking at. */
1486
+ interface ProcessorContext {
1487
+ threadId: string;
1488
+ actor: Actor;
1489
+ /** The agent running this turn. Undefined → the default agent. */
1490
+ agentName?: string;
1491
+ /** 0-based model step within the run — the same index the `llm:<step>` checkpoint carries. */
1492
+ step: number;
1493
+ }
1494
+ /** Everything the model is about to be sent, as the previous processor in the chain left it. */
1495
+ interface ProcessedPrompt {
1496
+ /** The composed system prompt (agent base + contributors + any injected retrieval block). */
1497
+ system: string;
1498
+ /** The turn's messages, oldest-first, already through the history ceiling. */
1499
+ messages: ModelMessage[];
1500
+ }
1501
+ /**
1502
+ * Rewrites the prompt before each model call of a turn — masking identifiers, stamping a policy
1503
+ * preamble, collapsing an oversized tool result. Runs on EVERY step, not once per run, because the
1504
+ * transcript grows between steps: a redactor that only saw the opening prompt would wave through
1505
+ * whatever a tool result carried back.
1506
+ *
1507
+ * Runs inside the loop's `process:input:<step>` checkpoint and its result is journaled, so a
1508
+ * processor may call a model or hit the network — a resumed run reads back the prompt the suspended
1509
+ * attempt built rather than composing a different one.
1510
+ */
1511
+ interface InputProcessor {
1512
+ /** Identifies this processor in a failure. Keep it stable — it is user-visible on an error. */
1513
+ readonly name: string;
1514
+ process(prompt: ProcessedPrompt, ctx: ProcessorContext): ProcessedPrompt | Promise<ProcessedPrompt>;
1515
+ }
1516
+ /** One model step's answer, as the previous processor in the chain left it. */
1517
+ interface ModelAnswer {
1518
+ /** The assembled assistant text for this step. */
1519
+ text: string;
1520
+ /** The tool calls the same step asked for. Read-only context — a processor cannot change them. */
1521
+ toolCalls: readonly ToolCallRequest[];
1522
+ }
1523
+ /**
1524
+ * What an {@link OutputProcessor} decided. `pass` hands the text to the next processor unchanged;
1525
+ * `replace` hands it on rewritten (a redaction is a replacement); `reject` ends the chain AND the
1526
+ * run — the text is never streamed, never persisted, and the caller gets an
1527
+ * {@link OutputRejectedError} rather than an answer.
1528
+ */
1529
+ type OutputVerdict = {
1530
+ action: 'pass';
1531
+ } | {
1532
+ action: 'replace';
1533
+ text: string;
1534
+ } | {
1535
+ action: 'reject';
1536
+ reason: string;
1537
+ };
1538
+ /**
1539
+ * Characters an incremental gate keeps holding at the end of the transformed answer, when a
1540
+ * processor declares {@link IncrementalGating} without naming its own window. Wide enough for the
1541
+ * patterns a redactor is usually written against — an SSN, an email address, a card number — and
1542
+ * deliberately not wider: the window IS the answer's minimum latency tail, since those characters
1543
+ * are only released once the whole-answer pass runs.
1544
+ */
1545
+ declare const DEFAULT_INCREMENTAL_LOOKBACK_CHARS = 64;
1546
+ /**
1547
+ * A processor's statement that its verdict on a PREFIX of an answer is worth acting on, which is
1548
+ * what lets the loop release that prefix to the reader instead of holding the whole answer.
1549
+ *
1550
+ * Declaring it is a promise about every prefix `P` of the final answer, and the loop takes it at
1551
+ * face value on the streaming path:
1552
+ *
1553
+ * 1. REJECTION IS PREFIX-DECIDABLE. The refusal this processor would return for the whole answer is
1554
+ * already returned for the first prefix that contains the reason. A refusal that only emerges
1555
+ * from the complete answer still fails the run, but by then the reader has seen a prefix — there
1556
+ * is no un-sending bytes, and that is the cost of opting in.
1557
+ * 2. REPLACEMENT IS PREFIX-STABLE UP TO THE WINDOW. Once a character of `process(P)`'s output is
1558
+ * more than {@link lookbackChars} from the end of that output, it never changes as `P` grows.
1559
+ *
1560
+ * A broken second promise is caught, not tolerated: the whole-answer pass stays authoritative, and
1561
+ * the loop raises a `ProcessorFailedError` when its result is not an extension of what the chain
1562
+ * already released. So a window too short for a pattern fails loudly rather than streaming the text
1563
+ * it was supposed to redact.
1564
+ */
1565
+ interface IncrementalGating {
1566
+ /**
1567
+ * How many characters of this processor's own output stay held back. Undefined →
1568
+ * {@link DEFAULT_INCREMENTAL_LOOKBACK_CHARS}. Set it to the length of the longest pattern the
1569
+ * processor can act on — anything shorter is a run that fails on the pattern it was written for.
1570
+ */
1571
+ readonly lookbackChars?: number;
1572
+ }
1573
+ /**
1574
+ * Inspects each model step's answer before anything downstream sees it — before it reaches the live
1575
+ * stream, before it is persisted, before it becomes the next step's context.
1576
+ *
1577
+ * Gating an answer and streaming it as it is generated are mutually exclusive, so registering ANY
1578
+ * output processor switches the turn's model call off the run's sink: nothing reaches the
1579
+ * subscriber until this chain has ruled on it — see `AgentLoopDeps.outputProcessors`. How much of
1580
+ * that costs the reader is what {@link incremental} decides.
1581
+ *
1582
+ * Runs inside the loop's `process:output:<step>` checkpoint (which also performs the release), so a
1583
+ * processor may call a model — a moderation pass is the motivating case — and a replay reads the
1584
+ * verdict back instead of re-deciding it.
1585
+ */
1586
+ interface OutputProcessor {
1587
+ /** Identifies this processor in a rejection or a failure. User-visible; keep it stable. */
1588
+ readonly name: string;
1589
+ /**
1590
+ * Opt this processor into gating a PREFIX, so the turn keeps streaming — see
1591
+ * {@link IncrementalGating} for what declaring it promises. Undefined means the whole answer,
1592
+ * which is what a processor written against the complete text needs and therefore the only safe
1593
+ * default; a chain is incremental only when EVERY member of it declares this, so a neighbour can
1594
+ * never downgrade what another author was given.
1595
+ */
1596
+ readonly incremental?: IncrementalGating;
1597
+ process(answer: ModelAnswer, ctx: ProcessorContext): OutputVerdict | Promise<OutputVerdict>;
1598
+ }
1599
+ /**
1600
+ * The run ended because an output processor refused the answer — NOT because the model failed. The
1601
+ * two must stay distinguishable: a model failure is worth retrying and worth paging someone about,
1602
+ * a refusal is the control doing its job. Both runners map this to the `output_rejected` stream
1603
+ * error code.
1604
+ */
1605
+ declare class OutputRejectedError extends Error {
1606
+ /** {@link OutputProcessor.name} of the processor that refused. */
1607
+ readonly processor: string;
1608
+ /** The reason it gave, verbatim. */
1609
+ readonly reason: string;
1610
+ constructor(processor: string, reason: string);
1611
+ }
1612
+ /**
1613
+ * A processor threw. Wrapped so it can never be mistaken for the model call failing — the loop's
1614
+ * only other source of failure at that point in the turn — and so the failure names the processor
1615
+ * that produced it instead of surfacing a bare `TypeError` from someone else's code.
1616
+ */
1617
+ declare class ProcessorFailedError extends Error {
1618
+ /** Which seam it was on, so a reader knows whether the prompt or the answer was in flight. */
1619
+ readonly phase: 'input' | 'output';
1620
+ /** The processor's `name`. */
1621
+ readonly processor: string;
1622
+ constructor(phase: 'input' | 'output', processor: string, cause: unknown);
1623
+ }
1624
+
882
1625
  /**
883
1626
  * Turns text into embedding vectors — the sibling of {@link import('./model-provider.js').ModelProvider}
884
1627
  * for the retrieval side. Batched (`texts` → one vector each, same order) so ingestion can embed many
@@ -919,8 +1662,12 @@ interface AgentRunner {
919
1662
  start(input: AgentRunInput): Promise<{
920
1663
  runId: string;
921
1664
  }>;
922
- /** Deliver a HITL decision for a pending action tool call. */
923
- signal(runId: string, toolCallId: string, decision: Decision): Promise<void>;
1665
+ /**
1666
+ * Deliver a human's reply to a parked tool call — a {@link import('../types.js').Decision} on an
1667
+ * action tool, or an `ElicitationReply` answering a question set. One channel for both, because
1668
+ * both park the run the same way and a runner has no reason to tell them apart.
1669
+ */
1670
+ signal(runId: string, toolCallId: string, reply: HumanReply): Promise<void>;
924
1671
  cancel(runId: string): Promise<void>;
925
1672
  }
926
1673
 
@@ -1048,7 +1795,11 @@ interface RecentRunRow {
1048
1795
  threadId: string;
1049
1796
  actorRef: string;
1050
1797
  agentName: string | null;
1051
- /** 'running' | 'completed' | 'failed'. */
1798
+ /**
1799
+ * `'running'` | `'completed'` | `'failed'` | `'cancelled'`. `cancelled` is a terminal of its own —
1800
+ * someone asked the run to stop and it did — so a consumer computing a failure rate over these
1801
+ * rows has to be able to leave it out rather than fold it into `failed`.
1802
+ */
1052
1803
  status: string;
1053
1804
  durationMs: number | null;
1054
1805
  errorCode: string | null;
@@ -1058,6 +1809,14 @@ interface RecentRunRow {
1058
1809
  startedAt: string;
1059
1810
  /** sha256 hex of the run's resolved (pre-RAG) system prompt; `null` for a run recorded before this shipped. */
1060
1811
  promptHash: string | null;
1812
+ /**
1813
+ * The run that delegated this one; `null` for a turn nobody delegated, and for any run recorded
1814
+ * before the column existed. Pairing a child with its parent is what lets a console draw the
1815
+ * delegation tree and roll a child's cost up to the turn that asked for it — and for a DETACHED
1816
+ * child it is the only pairing there is, since it outlives its parent's turn and the transcript
1817
+ * holds no other link.
1818
+ */
1819
+ parentRunId: string | null;
1061
1820
  }
1062
1821
  /** One tool call awaiting a HITL decision, for the cross-thread approvals inbox. */
1063
1822
  interface PendingApprovalRow {
@@ -1335,13 +2094,57 @@ interface StageAttachmentInput {
1335
2094
  sizeBytes: number;
1336
2095
  actor: Actor;
1337
2096
  }
2097
+ /**
2098
+ * All a chat turn may say about a file it attaches: the id of something already staged. Everything
2099
+ * else about the attachment (url, content type, name) is read back from the staging store, never
2100
+ * from the request — see {@link AttachmentStagingStore.resolve}.
2101
+ */
2102
+ interface AttachmentRef {
2103
+ mediaId: string;
2104
+ }
2105
+ /** Input to {@link AttachmentStagingStore.resolve} — the claimed id plus who is claiming it. */
2106
+ interface ResolveAttachmentInput {
2107
+ mediaId: string;
2108
+ actor: Actor;
2109
+ }
2110
+ /**
2111
+ * One entry in an actor's staged-media inventory: enough to show the file and to decide whether it
2112
+ * has aged out, and nothing more.
2113
+ *
2114
+ * Carries no url, unlike {@link MessageAttachment}. A url is minted per turn by
2115
+ * {@link AttachmentStagingStore.resolve} precisely so it can be short-lived; a listing that handed
2116
+ * one back would mint a fetchable url for every entry on the page, for a caller that asked to see
2117
+ * a file list. Whoever needs the bytes goes through `resolve` and is checked there.
2118
+ */
2119
+ interface StagedAttachment {
2120
+ mediaId: string;
2121
+ /** Original filename, as `stage` received it. */
2122
+ name: string;
2123
+ contentType: string;
2124
+ sizeBytes: number;
2125
+ /** ISO-8601 UTC instant the bytes were staged — what an age threshold is measured against. */
2126
+ createdAt: string;
2127
+ }
2128
+ /** Input to {@link AttachmentStagingStore.list} — whose inventory, and how much of it. */
2129
+ interface ListStagedAttachmentsInput {
2130
+ actor: Actor;
2131
+ /**
2132
+ * Only media staged strictly before this ISO-8601 UTC instant. This is how a caller expresses
2133
+ * "old enough to be worth looking at": freshly staged media belongs to an upload the user has
2134
+ * not sent yet, which is in flight rather than garbage. The instant comes from the caller
2135
+ * because how long a composer may sit open is the host's knowledge, not this library's.
2136
+ */
2137
+ stagedBefore?: string;
2138
+ /** Cap on entries returned, newest first. */
2139
+ limit?: number;
2140
+ }
1338
2141
  /**
1339
2142
  * Optional upload-side seam for message attachments (an image/PDF a user attaches to a chat
1340
2143
  * message before the model ever sees it). The lib never fetches bytes itself — {@link MessageAttachment.url}
1341
2144
  * must already be reachable by the model provider — so something has to turn an uploaded file into
1342
2145
  * that URL first. A store adapter (or a thin wrapper over the host's own media pipeline) implements
1343
2146
  * this; consumers inject via `AGENT_ATTACHMENT_STAGING`. Unbound, the optional `POST /agent/attachments`
1344
- * upload controller is never mounted.
2147
+ * upload controller is never mounted, and a chat turn cannot carry attachments at all.
1345
2148
  */
1346
2149
  interface AttachmentStagingStore {
1347
2150
  /**
@@ -1350,10 +2153,44 @@ interface AttachmentStagingStore {
1350
2153
  * returned url must be reachable by the model provider.
1351
2154
  */
1352
2155
  stage(input: StageAttachmentInput): Promise<MessageAttachment>;
1353
- }
1354
-
1355
- /**
1356
- * Console-side HITL decisions. Implemented by the agent runtime (the nestjs package binds it to
2156
+ /**
2157
+ * Turn a `mediaId` a chat turn claims back into the attachment to send with it, or `null` when
2158
+ * the id is unknown OR is not this actor's. The two cases are deliberately indistinguishable, so
2159
+ * the chat endpoint cannot be used to probe which media ids exist.
2160
+ *
2161
+ * SECURITY: this is the ONLY source of a url the model provider will be asked to fetch. The chat
2162
+ * endpoint discards every other field a client sends and rebuilds each attachment from here — an
2163
+ * implementation that echoes back a url taken from the request reopens the SSRF this seam exists
2164
+ * to close. Called once per turn, so a short-lived presigned url is minted fresh rather than
2165
+ * replayed stale.
2166
+ */
2167
+ resolve(input: ResolveAttachmentInput): Promise<MessageAttachment | null>;
2168
+ /**
2169
+ * OPTIONAL: the actor's staged media, newest first. The host staged these bytes, so only the host
2170
+ * can enumerate them — this library holds references, never an inventory.
2171
+ *
2172
+ * Pairs with {@link import('./agent-store.js').AgentStore.referencedMediaIds}, which answers the
2173
+ * half only the library can: of these ids, which a live message still carries. Together they are
2174
+ * a sweep — inventory minus references, restricted to entries old enough not to be an upload in
2175
+ * flight. Deleting whatever is left is the host's call and the host's alone; this library never
2176
+ * removes a host's bytes.
2177
+ *
2178
+ * Scoped to `input.actor` without exception. An implementation that ignores it turns a file list
2179
+ * into a way to read someone else's documents.
2180
+ *
2181
+ * `stagedBefore` and `limit` are the store's to apply — but a caller on a delete path must not
2182
+ * assume they were, since getting that wrong deletes files. `AgentService.collectableAttachments`
2183
+ * re-applies the age cut on the results for that reason.
2184
+ *
2185
+ * Absent on a store that predates this: there is no inventory to fall back on, so listing and
2186
+ * collection are simply unavailable (the read surface answers 501) rather than quietly empty —
2187
+ * an empty inventory and an unanswerable one look identical and mean opposite things.
2188
+ */
2189
+ list?(input: ListStagedAttachmentsInput): Promise<StagedAttachment[]>;
2190
+ }
2191
+
2192
+ /**
2193
+ * Console-side HITL decisions. Implemented by the agent runtime (the nestjs package binds it to
1357
2194
  * the same signal path chat approvals use); the dashboard injects it OPTIONALLY — absent = the
1358
2195
  * approvals inbox renders read-only.
1359
2196
  */
@@ -1450,7 +2287,7 @@ declare function dayBoundsUtc(range: GovernanceRange): {
1450
2287
  */
1451
2288
  declare function isToolEnabled(spec: ToolSpec, handler?: ToolHandler): Promise<boolean>;
1452
2289
  /**
1453
- * Zeroth filter layer: drop tools this deployment has turned off, before anyone asks who may call
2290
+ * Second filter layer: drop tools this deployment has turned off, before anyone asks who may call
1454
2291
  * them. A disabled tool is absent, not forbidden — the difference matters, because "forbidden"
1455
2292
  * tells the model (and the user reading a refusal) that the capability exists.
1456
2293
  */
@@ -1464,18 +2301,1011 @@ declare function filterToolsByEnabled<T extends {
1464
2301
  */
1465
2302
  declare function canActorUseTool(actor: Actor, handler?: ToolHandler): Promise<boolean>;
1466
2303
  /**
1467
- * Third filter layer: drop tools whose own `canUse` refuses this actor. Runs after the app-wide
2304
+ * Fourth filter layer: drop tools whose own `canUse` refuses this actor. Runs after the app-wide
1468
2305
  * `RolesPolicy`, and is additive to it — a tool can narrow who reaches it, never widen.
1469
2306
  */
1470
2307
  declare function filterToolsByCanUse<T extends {
1471
2308
  spec: ToolSpec;
1472
2309
  handler?: ToolHandler;
1473
2310
  }>(entries: T[], actor: Actor): Promise<T[]>;
1474
- /** First filter layer: drop tools the actor's role may not invoke. `can` may be async (authz). */
2311
+ /** Third filter layer: drop tools the actor's role may not invoke. `can` may be async (authz). */
1475
2312
  declare function filterToolsByRole(tools: ToolSpec[], actor: Actor, policy: RolesPolicy): Promise<ToolSpec[]>;
1476
- /** Second filter layer: if the agent pins an allow-list, keep only those tool names. */
2313
+ /**
2314
+ * First filter layer: if the agent pins an allow-list, keep only those tool names. Pure and
2315
+ * synchronous, which is why it runs ahead of the three gates that may do I/O — see
2316
+ * `ToolRegistry.definitionsFor`.
2317
+ */
1477
2318
  declare function filterToolsByAllowList(tools: ToolSpec[], allowedTools: string[] | undefined): ToolSpec[];
1478
2319
 
2320
+ /**
2321
+ * The built-in {@link HistoryPolicy}: keep the newest messages that fit a count and/or a token
2322
+ * budget, and (optionally) fold the rest into a summary. See `./spi/history-policy.ts` for the seam
2323
+ * itself and the determinism contract `select` has to hold to.
2324
+ */
2325
+
2326
+ /**
2327
+ * Rough token count for one message: ~4 characters per token over its content and its serialized
2328
+ * tool calls/results, plus a small per-message envelope allowance for the role and part framing.
2329
+ *
2330
+ * A heuristic on purpose. A real tokenizer is model-specific and would drag a provider dependency
2331
+ * into core, while a budget only has to be approximately right to keep a thread clear of the
2332
+ * provider's hard limit — and it must be a pure function, because it runs inside `select`. Pass
2333
+ * `estimate` to {@link windowHistory} to substitute a real one.
2334
+ */
2335
+ declare function estimateMessageTokens(message: ModelMessage): number;
2336
+ interface WindowHistoryOptions {
2337
+ /** Keep at most this many of the newest messages. Omit → no count limit. */
2338
+ maxMessages?: number;
2339
+ /** Keep the newest messages whose estimated tokens fit this budget. Omit → no token limit. */
2340
+ maxTokens?: number;
2341
+ /** Substitute for {@link estimateMessageTokens}. Must be pure — it runs inside `select`. */
2342
+ estimate?: (message: ModelMessage) => number;
2343
+ /** Folds the dropped messages into a leading summary. Omit → they are simply gone. */
2344
+ summarize?: HistoryPolicy['summarize'];
2345
+ }
2346
+ /**
2347
+ * Keep the newest messages that fit. Both limits apply when both are set — whichever cuts more wins.
2348
+ * Neither set is a policy that keeps everything, which is what an unconfigured loop already does.
2349
+ *
2350
+ * A naive slice is safe here because a tool exchange is ONE message: the loop persists a call's
2351
+ * results onto the same assistant message that made them (`toolCalls` + `toolResults`), and each
2352
+ * model adapter expands that into the assistant/tool pair the provider wants. So a cut can't orphan
2353
+ * a tool result from its call, the way it could against a provider-shaped transcript.
2354
+ */
2355
+ declare function windowHistory(options: WindowHistoryOptions): HistoryPolicy;
2356
+ /**
2357
+ * What a summarizer is told to produce when none is supplied. Written for a reader who will continue
2358
+ * the conversation without seeing the messages themselves, so it asks for the parts a later turn
2359
+ * still has to act on rather than a readable recap.
2360
+ */
2361
+ declare const DEFAULT_HISTORY_SUMMARY_INSTRUCTION = "Summarize this conversation for an assistant that will continue it without seeing these messages. Preserve decisions made, facts and constraints the user stated, identifiers and names mentioned, and anything left unresolved. Omit pleasantries. Reply with the summary only \u2014 no preamble, no headings.";
2362
+ /**
2363
+ * A {@link HistoryPolicy.summarize} backed by one extra, non-streamed model call — what
2364
+ * `AgentModule.forRoot({ history: { summarize: true } })` binds. Writes to a discarding sink so the
2365
+ * summary's tokens never reach the user's live stream, and reports its usage so the call lands in
2366
+ * the spend read-model like any other.
2367
+ */
2368
+ declare function summarizeWithModel(model: ModelProvider, instruction?: string): NonNullable<HistoryPolicy['summarize']>;
2369
+
2370
+ /**
2371
+ * Running the processor chains, and the stream buffering an output gate requires. See
2372
+ * `./spi/processors.ts` for the seams themselves and the boundary against `HistoryPolicy`.
2373
+ */
2374
+
2375
+ /**
2376
+ * Holds a model call's stream frames instead of letting them reach the subscriber. `end`/`fail` are
2377
+ * swallowed for the same reason `childSinkWriter` swallows them: the loop owns the run's stream
2378
+ * lifecycle across however many steps the turn takes, and the model call is one step of it.
2379
+ *
2380
+ * Frames are kept as the decoded NDJSON LINES rather than the raw `Uint8Array`s so the buffer can
2381
+ * ride a durable checkpoint — the gate has to survive a suspend between the model call and the
2382
+ * verdict, and bytes do not round-trip through JSON.
2383
+ */
2384
+ interface FrameBuffer {
2385
+ writer: SinkWriter;
2386
+ /** Everything written so far, one entry per NDJSON line, in write order. */
2387
+ frames(): string[];
2388
+ }
2389
+ declare function createFrameBuffer(): FrameBuffer;
2390
+ /**
2391
+ * The chunks a passed gate releases to the live stream: the held frames, with every text frame
2392
+ * collapsed into ONE carrying `text` — the answer as the chain left it, which is the only version
2393
+ * anything downstream is allowed to see.
2394
+ *
2395
+ * A frame that does not decode is dropped along with them. It cannot be forwarded, because a gate
2396
+ * that forwards bytes it cannot classify is not a gate: a provider writing bare text into the sink
2397
+ * (rather than the `AgentStreamEvent` vocabulary) would otherwise stream the ungated answer past
2398
+ * the processor that was supposed to hold it. That is the cost of an output gate for a provider
2399
+ * outside the encoded vocabulary, and it is stated in `AgentLoopDeps.outputProcessors`.
2400
+ */
2401
+ declare function releaseGatedFrames(frames: readonly string[], text: string): Uint8Array[];
2402
+ /**
2403
+ * Fold the prompt through each processor in order, every one seeing what the previous one produced.
2404
+ * A throw is wrapped so it cannot read as the model call failing — see {@link ProcessorFailedError}.
2405
+ */
2406
+ declare function runInputProcessors(processors: readonly InputProcessor[], prompt: ProcessedPrompt, ctx: ProcessorContext): Promise<ProcessedPrompt>;
2407
+ /** A settled output chain: the answer as the chain left it, plus who refused it if anyone did. */
2408
+ interface OutputGateResult {
2409
+ /** The text every downstream consumer sees — the stream, the persisted message, the next step. */
2410
+ text: string;
2411
+ /** Set only on a refusal; the run ends with an `OutputRejectedError` naming these. */
2412
+ rejection?: {
2413
+ processor: string;
2414
+ reason: string;
2415
+ };
2416
+ }
2417
+ /**
2418
+ * Fold the answer through each processor in order. The FIRST rejection ends the chain: a later
2419
+ * processor has nothing to add about text that is never going anywhere, and running it would bill
2420
+ * a moderation call for a turn already refused.
2421
+ */
2422
+ declare function runOutputProcessors(processors: readonly OutputProcessor[], answer: ModelAnswer, ctx: ProcessorContext): Promise<OutputGateResult>;
2423
+ /**
2424
+ * Put each suggestion through the output chain on its own, and keep what survives.
2425
+ *
2426
+ * A follow-up is model-generated text that gets persisted and rendered, so the gate has to see it.
2427
+ * What it does NOT get is the chain's usual answer to a refusal — ending the run. The suggestions
2428
+ * are produced AFTER the turn's answer has already cleared the same chain, and retracting a cleared
2429
+ * answer because a question nobody asked for was refused would make a run's outcome depend on a
2430
+ * by-product. Dropping the suggestion removes it just as completely, which is what the refusal was
2431
+ * for. A `replace` is honoured; a processor that empties one drops it too, since an empty suggestion
2432
+ * is nothing to render.
2433
+ *
2434
+ * One suggestion at a time rather than the joined list, so a refusal is confined to the one that
2435
+ * earned it and a `replace` cannot smear across a neighbour.
2436
+ */
2437
+ declare function gateFollowUps(processors: readonly OutputProcessor[], followUps: readonly string[], ctx: ProcessorContext): Promise<string[]>;
2438
+ /**
2439
+ * How much of a turn's stream an output chain costs the reader.
2440
+ *
2441
+ * `off` — nothing registered, the model writes straight to the run's sink.
2442
+ * `whole` — at least one processor needs the complete answer, so every frame is held until the
2443
+ * chain has passed and the answer arrives as one `text` frame.
2444
+ * `incremental` — every processor declared {@link IncrementalGating}, so a lookback-bounded prefix
2445
+ * is released while the call streams.
2446
+ */
2447
+ type OutputGateMode = 'off' | 'whole' | 'incremental';
2448
+ /**
2449
+ * ALL or nothing: one undeclared processor puts the whole chain on the whole-answer path. Its
2450
+ * author wrote `process` against the complete text, and a chain that ran it on a prefix because a
2451
+ * neighbour opted in would be handing it an input it never agreed to read.
2452
+ */
2453
+ declare function resolveOutputGateMode(processors: readonly OutputProcessor[]): OutputGateMode;
2454
+ /**
2455
+ * The chain's window is the WIDEST any member asked for: a release safe for the shortest-sighted
2456
+ * processor is not safe for the one that matches longer patterns, and the gate makes one release
2457
+ * decision for all of them.
2458
+ *
2459
+ * An undeclared processor contributes nothing rather than the default — it has no window, because a
2460
+ * chain containing one never reaches the incremental path at all.
2461
+ */
2462
+ declare function resolveGateLookback(processors: readonly OutputProcessor[]): number;
2463
+ /** A live gate over one model call: the sink the model writes to, plus what it let through. */
2464
+ interface IncrementalGate {
2465
+ /** Hand this to the model instead of the run's writer. */
2466
+ writer: SinkWriter;
2467
+ /** Resolves once every chunk handed to {@link writer} has been ruled on. */
2468
+ settled(): Promise<void>;
2469
+ /** The transformed prefix already written to the run's sink. */
2470
+ released(): string;
2471
+ /** Set once a prefix was refused; nothing further was released after that. */
2472
+ rejection(): OutputGateResult['rejection'];
2473
+ }
2474
+ /**
2475
+ * Releases a model call's answer to `writer` as it arrives, holding back the last `lookbackChars`
2476
+ * characters of what the chain produces so a processor can still change them.
2477
+ *
2478
+ * The chain runs over the whole PREFIX accumulated so far rather than over each new chunk, so a
2479
+ * processor always sees well-formed text and never has to reassemble a pattern split across
2480
+ * frames — which is what makes {@link IncrementalGating}'s promise a claim about prefixes. It runs
2481
+ * once per streamed `text` frame.
2482
+ *
2483
+ * Only an EXTENSION of what the reader already has is ever written: a chain whose output stops
2484
+ * agreeing with its earlier output has broken its promise, and the loop's whole-answer pass reports
2485
+ * that rather than this gate papering over it by re-sending a different answer.
2486
+ *
2487
+ * Frames that are not text are forwarded live, in arrival order. They are not the answer, so the
2488
+ * gate does not own them — but it does drop any frame it cannot decode, for the same reason
2489
+ * {@link releaseGatedFrames} does: a gate that forwards bytes it cannot classify is not a gate.
2490
+ */
2491
+ declare function createIncrementalGate(options: {
2492
+ processors: readonly OutputProcessor[];
2493
+ ctx: ProcessorContext;
2494
+ lookbackChars: number;
2495
+ writer: SinkWriter;
2496
+ }): IncrementalGate;
2497
+ /**
2498
+ * The tail an incremental gate still owes the reader once the whole-answer pass has settled: the
2499
+ * authoritative text minus the prefix already released.
2500
+ *
2501
+ * Throws when the settled answer is not an extension of that prefix. That check is the reason the
2502
+ * final pass stays authoritative for both the stream and the store rather than the stream being
2503
+ * stitched together from per-prefix results: agreement between what was streamed and what was
2504
+ * stored becomes structural instead of assumed.
2505
+ */
2506
+ declare function gateTail(processors: readonly OutputProcessor[], released: string, text: string): string;
2507
+
2508
+ /**
2509
+ * Constraining a turn's ANSWER to a schema. See `AgentLoopDeps.outputSchema` for where this sits in
2510
+ * the loop and why it is a separate model call rather than a constraint on the turn's own calls.
2511
+ */
2512
+
2513
+ /**
2514
+ * The turn produced an answer that does not satisfy `outputSchema`, and the bounded repair attempts
2515
+ * did not fix it. A DEFINED outcome of asking for structured output, not a crash: it carries the
2516
+ * validation issues and the text that failed them, so a caller can log what the model actually said
2517
+ * instead of guessing from a parse error. Both runners map it to the `structured_output_invalid`
2518
+ * stream error code.
2519
+ */
2520
+ declare class StructuredOutputError extends Error {
2521
+ /** Why it failed the schema. Empty only when the text was not JSON at all. */
2522
+ readonly issues: readonly StandardSchemaV1.Issue[];
2523
+ /** The last text the model produced, verbatim. */
2524
+ readonly text: string;
2525
+ /** How many model calls were spent trying (1 = the formatting pass, no repairs). */
2526
+ readonly attempts: number;
2527
+ constructor(issues: readonly StandardSchemaV1.Issue[], text: string, attempts: number);
2528
+ }
2529
+ /** A validated answer, or the issues that stopped it being one. */
2530
+ type StructuredOutcome<T> = {
2531
+ ok: true;
2532
+ value: T;
2533
+ } | {
2534
+ ok: false;
2535
+ issues: readonly StandardSchemaV1.Issue[];
2536
+ };
2537
+ /**
2538
+ * Pull a JSON value out of a model reply, tolerating the code fences and lead-in prose a provider
2539
+ * that cannot constrain its own generation still emits. `undefined` when there is no JSON in there
2540
+ * at all — distinct from a JSON `null`, which is a value the schema may well accept.
2541
+ */
2542
+ declare function extractJson(text: string): unknown;
2543
+ /**
2544
+ * Validate one attempt. The provider's own parsed `object` is preferred when it reported one (a
2545
+ * provider that constrained generation already did the parse), but it is validated all the same —
2546
+ * "the provider says it matched" is not the same claim as "it matches", and a provider that ignored
2547
+ * `outputSchema` entirely must fail here rather than downstream.
2548
+ */
2549
+ declare function validateStructured<T>(schema: StandardSchemaV1<unknown, T>, text: string, reported: unknown): Promise<StructuredOutcome<T>>;
2550
+ /**
2551
+ * What the formatting pass is told to do. Written for a model that is being shown a finished answer
2552
+ * and asked to restate it — never to extend or improve it, which would make the structured result
2553
+ * disagree with the prose the user already read.
2554
+ */
2555
+ declare const DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION = "Restate the assistant's final answer as a single JSON value matching the required schema. Use only information already present in the conversation \u2014 add nothing, and answer nothing that was not asked. Reply with ONLY the JSON: no prose, no explanation, no code fences.";
2556
+ /** Appends the previous attempt's validation issues, so a repair call knows what to fix. */
2557
+ declare function repairInstruction(instruction: string, issues: readonly StandardSchemaV1.Issue[]): string;
2558
+
2559
+ /**
2560
+ * Skills: an authored procedure the model pulls in when a task calls for it, instead of every
2561
+ * instruction living in the system prompt.
2562
+ *
2563
+ * WHAT A SKILL IS NOT: an agent. An `@Agent` is WHO is answering — its persona, its tools, its
2564
+ * history ceiling, its output schema. A skill is HOW one particular task is done, and any agent may
2565
+ * pull one in. That is why a skill carries no model, no tool list and no schema: the moment it did,
2566
+ * the two would be the same thing wearing different names, and a consumer would have to choose
2567
+ * between them for reasons nobody could state.
2568
+ *
2569
+ * WHAT IT COSTS THE PROMPT: one line per skill. The catalog block below carries names, scopes and
2570
+ * descriptions only; a BODY reaches the model as a tool result, on the transcript, where the
2571
+ * `HistoryPolicy` ceiling already governs it. So skills are not a fourth thing competing for the
2572
+ * system block with the agent's own prompt, its contributors and injected retrieval — see
2573
+ * `AgentLoopDeps.skills` for how the four compose.
2574
+ */
2575
+
2576
+ /** How a skill is identified in the catalog, to the model and to a `/`-autocomplete alike. */
2577
+ interface SkillSummary {
2578
+ /** Unique within a scope. The handle the model passes to the `skill` tool. */
2579
+ name: string;
2580
+ /** One line: what task this covers, and therefore when to load it. Read by the MODEL to choose. */
2581
+ description: string;
2582
+ /** The opaque scope token this skill is published at — see {@link ScopeResolver}. */
2583
+ scope: string;
2584
+ }
2585
+ /** A skill with the instructions themselves. Only ever materialized when something loads it. */
2586
+ interface Skill extends SkillSummary {
2587
+ body: string;
2588
+ }
2589
+ /** Whose turn is asking, and therefore which scopes apply. */
2590
+ interface ScopeContext {
2591
+ actor: Actor;
2592
+ threadId: string;
2593
+ /** The agent running this turn. Undefined → the default agent. */
2594
+ agentName?: string;
2595
+ pageContext?: PageContext;
2596
+ }
2597
+ /**
2598
+ * The skills-facing name for {@link ScopeContext}, so a `SkillProvider`'s signature reads in its own
2599
+ * vocabulary. Deliberately the SAME type rather than a parallel one: skills and memory resolve
2600
+ * scopes through one `ScopeResolver`, and a deployment that could answer "which scopes does this
2601
+ * actor have" twice would eventually answer it differently.
2602
+ */
2603
+ type SkillContext = ScopeContext;
2604
+ /** The scope token every deployment has: skills nobody narrowed. */
2605
+ declare const GLOBAL_SCOPE = "global";
2606
+ /** The token for one actor's own skills. */
2607
+ declare function actorScope(actor: Actor): string;
2608
+ /** The token for a tenant's skills — `Actor.tenantRef` is the only tenant key this library knows. */
2609
+ declare function tenantScope(tenantRef: string): string;
2610
+ /**
2611
+ * Which scopes a turn may draw skills from, MOST SPECIFIC FIRST. Precedence is the order: a skill
2612
+ * from an earlier token outranks a same-named one from a later token, and the model is shown which
2613
+ * of them won.
2614
+ *
2615
+ * A HOST-SUPPLIED FUNCTION rather than an enum this library owns, because the axes a deployment
2616
+ * scopes by are the deployment's own. `Actor` gives an id and a tenant; it does not give a sector, a
2617
+ * squadron, a base, a shift — and every one of those is a real axis in some consumer. An enum here
2618
+ * would make each of them a schema change in a library that has no business knowing they exist,
2619
+ * while a token is a string a host mints for itself. Return `['sector:logistics', 'tenant:base-7',
2620
+ * 'global']` and precedence follows, with nothing in this package edited.
2621
+ *
2622
+ * MUST be a pure function of its context. It runs inside the `skills:catalog` checkpoint and its
2623
+ * result is journaled, so a resumed run reads back the scopes the first attempt resolved rather than
2624
+ * asking a membership table that may have changed since — which would otherwise let a run's prompt
2625
+ * differ from the one its journal records. Read a database here at your peril; read it in the
2626
+ * provider, whose answer is journaled at the same position.
2627
+ */
2628
+ interface ScopeResolver {
2629
+ resolve(ctx: SkillContext): string[] | Promise<string[]>;
2630
+ }
2631
+ /**
2632
+ * The scopes derivable from an {@link Actor} alone — the common case, so a consumer wiring skills
2633
+ * for the first time supplies no resolver at all: the actor's own, their tenant's (when they have
2634
+ * one), and the deployment's. A host adding an axis of its own replaces this rather than extending
2635
+ * it, since the ORDER is the precedence and only the host knows where its axis belongs.
2636
+ */
2637
+ declare const defaultScopeResolver: ScopeResolver;
2638
+ /**
2639
+ * Where skills come from. Deliberately two calls rather than one: `list` is asked on EVERY turn and
2640
+ * must stay cheap, while a body is read only when the model decides it needs that procedure — the
2641
+ * whole point of the surface. A provider over a table selects name/description/scope for the first
2642
+ * and one row for the second.
2643
+ *
2644
+ * This library owns no skill table. A skill's real scoping axes, its authoring UI and its audit
2645
+ * trail are all the host's, and a host that related its own `Sector` entity into a table this
2646
+ * package created at boot would be writing migrations against a schema the boot-time heal also
2647
+ * edits. Tokens split it the other way round: the library owns the contract (what a scope means, how
2648
+ * precedence works, what is journaled), the host owns the rows.
2649
+ */
2650
+ /** Arguments to {@link SkillProvider.list}. */
2651
+ interface ListSkillsInput {
2652
+ scopes: readonly string[];
2653
+ ctx: SkillContext;
2654
+ }
2655
+ /**
2656
+ * Arguments to {@link SkillProvider.load}. An object rather than positional arguments because
2657
+ * `name` and `scope` are both strings: transposed, a positional call compiles clean, returns `null`,
2658
+ * and the skill silently fails to load.
2659
+ */
2660
+ interface LoadSkillInput {
2661
+ name: string;
2662
+ scope: string;
2663
+ ctx: SkillContext;
2664
+ }
2665
+ interface SkillProvider {
2666
+ /**
2667
+ * Every skill visible at `scopes`, in any order — this library sorts and resolves precedence. A
2668
+ * provider MAY return skills outside `scopes`; they are dropped rather than trusted, so a filter
2669
+ * bug in a host cannot widen what an actor is offered.
2670
+ */
2671
+ list(input: ListSkillsInput): SkillSummary[] | Promise<SkillSummary[]>;
2672
+ /**
2673
+ * The instructions for one skill, or `null` when it is gone. Called only from inside the loading
2674
+ * checkpoint, so the body it returns is journaled and every replay reads THAT text back — an
2675
+ * edited skill never rewrites the prompt of a run already in flight.
2676
+ */
2677
+ load(input: LoadSkillInput): string | null | Promise<string | null>;
2678
+ }
2679
+ /** A source of skills that is a fixed list — `@Skill()`-decorated classes, or plain config. */
2680
+ declare function staticSkillProvider(skills: readonly Skill[]): SkillProvider;
2681
+ /**
2682
+ * Read several sources as one. The order is the tie-break and nothing else: precedence between
2683
+ * skills is by SCOPE, so two sources only ever compete when they publish the same name at the same
2684
+ * scope, and then the earlier source wins. Wire the host's own provider first — a row it can edit is
2685
+ * a better answer than a body that needs a deploy to change.
2686
+ */
2687
+ declare function compositeSkillProvider(providers: readonly SkillProvider[]): SkillProvider;
2688
+ /**
2689
+ * One skill as it is offered — to the model in the catalog block, and to a client over
2690
+ * `GET /agent/skills`. Both read the same list, built by the same call, so what a user can type
2691
+ * after a `/` and what the model can reach cannot drift apart.
2692
+ */
2693
+ interface SkillCatalogEntry {
2694
+ name: string;
2695
+ description: string;
2696
+ /** The scope token it resolved from — the provenance a user is entitled to see. */
2697
+ scope: string;
2698
+ /**
2699
+ * Scope tokens of same-named skills this one outranks, widest last. Present ONLY when something
2700
+ * was shadowed, so a reader can tell "there is no org default" from "there is one and yours wins".
2701
+ * The model is shown it for the same reason: a silent override is indistinguishable from an
2702
+ * instruction nobody wrote, and the user can only be told "your setting differs from the org
2703
+ * default" by something that knows both existed.
2704
+ */
2705
+ shadows?: string[];
2706
+ }
2707
+ /** What a turn resolved — journaled whole, so the prompt is reconstructible from the journal alone. */
2708
+ interface SkillOffer {
2709
+ /** The scope tokens this turn drew from, most specific first. */
2710
+ scopes: string[];
2711
+ /** The skills offered, most specific first then by name. */
2712
+ entries: SkillCatalogEntry[];
2713
+ /** Applicable skills `maxSkills` left out. Non-zero means the catalog is not the whole truth. */
2714
+ omitted: number;
2715
+ }
2716
+ /**
2717
+ * How many skills a catalog offers before it starts leaving some out. A ceiling on the SYSTEM
2718
+ * PROMPT's share, not on how many a deployment may have: the block costs one line each, and a
2719
+ * hundred lines of "here is something you could load" crowds out the agent's own instructions while
2720
+ * making the choice harder rather than easier.
2721
+ */
2722
+ declare const DEFAULT_MAX_SKILLS = 20;
2723
+ /** How a turn reaches its skills. See `AgentLoopDeps.skills`. */
2724
+ interface SkillsConfig {
2725
+ provider: SkillProvider;
2726
+ /** Undefined → {@link defaultScopeResolver}: the actor's own, their tenant's, the deployment's. */
2727
+ scopes?: ScopeResolver;
2728
+ /** Undefined → {@link DEFAULT_MAX_SKILLS}. */
2729
+ maxSkills?: number;
2730
+ }
2731
+ /**
2732
+ * Resolve a provider's skills against an ordered scope list: most specific wins, and the loser's
2733
+ * scope is recorded rather than discarded.
2734
+ *
2735
+ * Pure, and separately exported, because two callers must reach the identical answer — the loop, so
2736
+ * the model is offered it, and the listing endpoint, so a user is. A second implementation of
2737
+ * "which skills apply" is a second answer, and the two diverge the first time either is edited.
2738
+ */
2739
+ declare function resolveSkillCatalog(summaries: readonly SkillSummary[], scopes: readonly string[], maxSkills?: number): Omit<SkillOffer, 'scopes'>;
2740
+ /** Resolve the scopes and the catalog for one turn. Called INSIDE the loop's `skills:catalog` step. */
2741
+ declare function offerSkills(config: SkillsConfig, ctx: SkillContext): Promise<SkillOffer>;
2742
+ /**
2743
+ * The catalog as the model reads it. One line per skill — name, scope, description, and what it
2744
+ * overrides — because the entire claim of progressive disclosure is that choosing what to read costs
2745
+ * far less than reading everything.
2746
+ */
2747
+ declare function buildSkillsBlock(entries: readonly SkillCatalogEntry[]): string;
2748
+ /** The reserved name of the built-in skill-loading tool. */
2749
+ declare const SKILL_TOOL_NAME = "skill";
2750
+ /** What the model passes to the `skill` tool. */
2751
+ interface SkillToolInput {
2752
+ name: string;
2753
+ }
2754
+ /**
2755
+ * The `skill` tool's input schema. Hand-written for the same reason `askInputSchema` is: core
2756
+ * depends on no validator and the schema has to publish a JSON Schema a provider can constrain
2757
+ * generation against.
2758
+ *
2759
+ * The available names are deliberately NOT an enum here. A per-turn enum would make the tool's
2760
+ * definition depend on the catalog — and the dispatched llm step re-derives the tool list on
2761
+ * whichever worker serves it, from ITS OWN provider, which is exactly the process-local lookup this
2762
+ * design keeps off the turn's decisions. The names live in the prompt, where the journal holds them;
2763
+ * a name that is not in the catalog comes back as an ordinary tool failure listing the ones that are.
2764
+ */
2765
+ declare const skillInputSchema: StandardSchemaV1<unknown, SkillToolInput>;
2766
+ declare const SKILL_TOOL_DESCRIPTION = "Read one of the procedures listed in <skills>. Call it before doing a task a listed skill covers, and follow what it returns. It performs nothing and changes nothing \u2014 it only gives you instructions you do not yet have.";
2767
+ /**
2768
+ * The `skill` tool as the model sees it. NOT a `ToolSpec` and never registered, exactly like `ask`:
2769
+ * it has no handler, because the loop serves it from the catalog the journal holds. Keeping it out
2770
+ * of the `ToolRegistry` is also what keeps its kind off a process-local lookup — see `claimToolCall`.
2771
+ */
2772
+ declare function skillToolDefinition(): ToolDefinition;
2773
+ /**
2774
+ * Append the built-in `skill` definition to a turn's tool list. Exported because the dispatched llm
2775
+ * step re-derives the tool list on a worker and has to reach the same list the loop would have.
2776
+ */
2777
+ declare function withSkillTool({ tools, enabled, }: {
2778
+ tools: ToolDefinition[];
2779
+ enabled: boolean;
2780
+ }): ToolDefinition[];
2781
+ /** What a `skill` call resolved to — the body, or why it did not. */
2782
+ type SkillLoadOutcome = {
2783
+ ok: true;
2784
+ skill: Skill;
2785
+ shadows?: string[];
2786
+ } | {
2787
+ ok: false;
2788
+ error: string;
2789
+ };
2790
+ /**
2791
+ * Serve one `skill` call against the catalog THIS TURN was offered.
2792
+ *
2793
+ * The catalog is the authorization boundary, not just a menu: a name the turn was never offered is
2794
+ * refused here, so a model that invents one (or repeats one it saw in another actor's thread) cannot
2795
+ * reach a body through a provider that would happily serve it. Because the catalog came out of the
2796
+ * `skills:catalog` checkpoint, that boundary is the one the journal records rather than the one this
2797
+ * process's provider currently believes in.
2798
+ */
2799
+ declare function loadSkill(config: SkillsConfig, offer: SkillOffer, name: string, ctx: SkillContext): Promise<SkillLoadOutcome>;
2800
+ /** Who is trying to write a skill. */
2801
+ interface SkillAuthor {
2802
+ /**
2803
+ * `'human'` is a person acting through a UI; `'agent'` is anything else — a tool, a turn, a
2804
+ * batch job. The distinction is the whole of the rule below, so it is not inferable and has to
2805
+ * be stated by the caller.
2806
+ */
2807
+ kind: 'human' | 'agent';
2808
+ actorRef?: string;
2809
+ }
2810
+ /** A request to author a skill at a scope. */
2811
+ interface SkillWriteRequest {
2812
+ /** The scope token the skill would be published at. */
2813
+ scope: string;
2814
+ actor: Actor;
2815
+ /** The actor's resolved scopes, most specific first — the same list a turn draws from. */
2816
+ scopes: readonly string[];
2817
+ author: SkillAuthor;
2818
+ /**
2819
+ * The host's own answer to "may this person administer that scope" — a sector lead editing their
2820
+ * sector's skills, an operator editing the deployment's. This library cannot know it: it has no
2821
+ * notion of who administers `sector:logistics`, and inventing one would be a second, weaker
2822
+ * authorization model next to the host's real one. Undefined/false → a wider scope is refused.
2823
+ */
2824
+ elevated?: boolean;
2825
+ }
2826
+ type SkillWriteVerdict = {
2827
+ allowed: true;
2828
+ } | {
2829
+ allowed: false;
2830
+ reason: string;
2831
+ };
2832
+ /**
2833
+ * May this author publish a skill at this scope?
2834
+ *
2835
+ * A skill's body is instructions the model follows, which makes writing one at a scope an edit to
2836
+ * everyone in that scope's system prompt. The rules follow from that, and only the third is
2837
+ * interesting:
2838
+ *
2839
+ * 1. You may only write into a scope you are yourself in. Writing into a scope you are not in is
2840
+ * authoring for people you have no relationship with.
2841
+ * 2. Your OWN scope is yours. A per-actor skill affects exactly one prompt — the author's.
2842
+ * 3. NOTHING BUT A HUMAN MAY WRITE ABOVE ITS OWN SCOPE, whatever `elevated` says. An agent that can
2843
+ * write a `tenant:` skill is an agent whose prompt anyone in the tenant can edit by talking to
2844
+ * it: the user asks for something, the turn writes the instruction, and every later turn for
2845
+ * every other user follows it. That is prompt injection with persistence, and no elevation a
2846
+ * host could grant makes it a different shape. An agent that has genuinely learned something
2847
+ * worth sharing proposes it; a person publishes it.
2848
+ * 4. A human writing above their own scope needs the host to say so (`elevated`).
2849
+ */
2850
+ declare function skillWriteVerdict(request: SkillWriteRequest): SkillWriteVerdict;
2851
+
2852
+ /**
2853
+ * Memory: what the assistant has concluded about a person or an organisation, carried across turns
2854
+ * and across threads.
2855
+ *
2856
+ * HOW A READER TELLS IT FROM RAG. Retrieval answers "what do the documents say"; memory answers
2857
+ * "what did I decide about you". A passage is content a person authored and can correct at its
2858
+ * source, and it is cited. A memory has no source to go and fix: it is the agent's own inference,
2859
+ * which is why every record carries an {@link MemoryOrigin} (who wrote it, in which conversation),
2860
+ * why the block tells the model to treat it as fallible, and why `forget` is a REQUIRED method on
2861
+ * the provider rather than an optional one. A wrong document is a content problem. A wrong memory is
2862
+ * the assistant being confidently wrong about someone with nobody aware it is there, so the whole
2863
+ * design is arranged around making it visible and removable.
2864
+ *
2865
+ * WHAT IT COSTS THE PROMPT. One line per memory, capped by `maxMemories`, and each line capped by
2866
+ * `maxFactChars` at WRITE time — so the block a turn can carry is the product of two numbers an
2867
+ * operator sets, rather than however much the model felt like writing down. Unlike a skill, a memory
2868
+ * has no body/catalog split: a fact that cannot be stated in a line is not a memory, it is a
2869
+ * document, and documents are retrieval's job.
2870
+ *
2871
+ * WHAT IT IS NOT: recall over a TRANSCRIPT. Searching what was said earlier is retrieval, and this
2872
+ * library already has `Retriever`/`Reranker` for that.
2873
+ *
2874
+ * RECALL OVER THE MEMORIES THEMSELVES IS A DIFFERENT QUESTION, and it is answered here. The prompt
2875
+ * budget must be bounded; the store has no reason to be. A person accumulates preferences over
2876
+ * months and an organisation publishes facts for everyone in it, so the applicable set outgrows
2877
+ * `maxMemories` quickly — and once it does, WHICH of them the block carries is a decision somebody
2878
+ * has to make. Making it by scope starves the widest scopes first, which is precisely backwards: the
2879
+ * facts that apply to the most people are the ones nobody sees. So a provider MAY supply
2880
+ * {@link MemoryProvider.search}, and where it does, the block is filled by relevance to the turn
2881
+ * (see {@link resolveMemoryDigest}) with scope still a hard filter and never a ranking signal.
2882
+ * `maxMemories` then means "how many matter right now" rather than "how many a person may have".
2883
+ */
2884
+
2885
+ /** Who wrote a memory, and out of what. Kept so a person reading it back can ask "says who?". */
2886
+ interface MemoryOrigin {
2887
+ /** `'agent'` — a turn concluded it. `'human'` — a person wrote it in the host's own console. */
2888
+ author: 'agent' | 'human';
2889
+ /**
2890
+ * The conversation the conclusion was drawn in. MAY name a thread whose messages the history
2891
+ * ceiling has since dropped: a memory deliberately outlives its source, so this is a pointer that
2892
+ * is allowed to dangle, and a read-back that cannot resolve it says so rather than hiding the row.
2893
+ */
2894
+ threadId?: string;
2895
+ runId?: string;
2896
+ actorRef?: string;
2897
+ }
2898
+ /** One thing the assistant believes, at one scope. */
2899
+ interface MemoryRecord {
2900
+ /** The host's row id — what `forget` takes and what a read-back offers a delete button for. */
2901
+ id: string;
2902
+ /**
2903
+ * What the fact is ABOUT. The handle that makes a conflict mechanically detectable: two memories
2904
+ * sharing a key at different scopes are one question answered twice, and the narrower wins. A
2905
+ * keyless store could only ever hope its facts did not contradict each other.
2906
+ */
2907
+ key: string;
2908
+ /** The fact itself, as the model reads it. */
2909
+ text: string;
2910
+ /** The opaque scope token it is held at — see `ScopeResolver` in `skills.ts`. */
2911
+ scope: string;
2912
+ origin: MemoryOrigin;
2913
+ /** ISO-8601, so the ceiling can keep the newest with a string comparison it can do purely. */
2914
+ updatedAt: string;
2915
+ /**
2916
+ * Always-on: this fact is in the block whether or not the turn is about it. The difference between
2917
+ * working memory and recall, drawn per record — "they report on the calendar year" must not depend
2918
+ * on the turn mentioning dates, while "they prefer the shorter runway" can wait until it comes up.
2919
+ *
2920
+ * A PROPERTY OF THE ROW, NOT OF A WRITE. This library reads it and never sets it, the same way it
2921
+ * never mints an `id`: an agent deciding its own conclusions are always-on is an agent deciding
2922
+ * how much of every future prompt it gets, so pinning is an operator's act in the host's console.
2923
+ * That is also why {@link StoreMemoryInput} carries no `pinned` — the `remember` tool has no
2924
+ * parameter to refuse. A host whose upsert drops the flag has silently unpinned the row; preserve
2925
+ * it across an upsert on (`scope`, `key`).
2926
+ */
2927
+ pinned?: boolean;
2928
+ }
2929
+ /** A same-key memory a narrower scope outranked, with the value it holds. */
2930
+ interface OverriddenMemory {
2931
+ scope: string;
2932
+ text: string;
2933
+ /**
2934
+ * Who asserted the beaten value. Travels because precedence is blind to it: an agent's own
2935
+ * inference at `actor:` outranks an administrator's published policy at `global`, and an agent
2936
+ * that knew only that a wider value existed could not tell the two apart — one is a stale guess,
2937
+ * the other is what the organisation decided.
2938
+ */
2939
+ author: MemoryOrigin['author'];
2940
+ }
2941
+ /** One memory as the model and a read-back both meet it. */
2942
+ interface MemoryDigestEntry extends MemoryRecord {
2943
+ /**
2944
+ * Same-key memories this one outranks, widest last. Present ONLY where something was outranked.
2945
+ *
2946
+ * Carries the beaten TEXT, which is where memory departs from a skill's `shadows`. A skill only
2947
+ * has to say which scope it beat — the agent follows one procedure either way. A memory is a
2948
+ * VALUE, and an agent that knows only that a wider value existed cannot tell the user what the
2949
+ * difference is; it can only choose, silently, which is the thing this is meant to prevent.
2950
+ */
2951
+ overrides?: OverriddenMemory[];
2952
+ }
2953
+ /** What a turn resolved — journaled whole, so its prompt is reconstructible from its journal alone. */
2954
+ interface MemoryDigest {
2955
+ /** The scope tokens this turn drew from, most specific first. */
2956
+ scopes: string[];
2957
+ /** The memories in the block, most specific first then by key. */
2958
+ entries: MemoryDigestEntry[];
2959
+ /**
2960
+ * Applicable memories `maxMemories` left out. Non-zero means the block is not the whole truth —
2961
+ * and under {@link recalled} it counts what the SEARCH offered and the ceiling dropped, not what
2962
+ * the store holds, because nothing asked the store for a total.
2963
+ */
2964
+ omitted: number;
2965
+ /**
2966
+ * Of {@link omitted}, how many were ALWAYS-ON. Reported separately because it means something else
2967
+ * entirely: ordinary omission is the budget doing its job, while an always-on memory the ceiling
2968
+ * dropped is a deployment's own standing policies having silently stopped reaching any prompt.
2969
+ * Non-zero is a misconfiguration — more was pinned than the block holds — and the fix is to unpin
2970
+ * something or raise `maxMemories`, not to wait for it to come back.
2971
+ */
2972
+ pinnedOmitted: number;
2973
+ /**
2974
+ * Whether the entries were selected by relevance to this turn rather than read whole. The model is
2975
+ * told (see {@link buildMemoryBlock}), because a partial set read as a whole one turns an absence
2976
+ * into evidence: "you never told me that" about a fact it simply was not shown.
2977
+ *
2978
+ * Optional so a digest journaled before recall existed reads back as the plain scoped read it was.
2979
+ */
2980
+ recalled?: boolean;
2981
+ }
2982
+ /** What a write asks the host to store. `id` and `updatedAt` are the host's to mint. */
2983
+ interface MemoryFact {
2984
+ key: string;
2985
+ text: string;
2986
+ scope: string;
2987
+ origin: MemoryOrigin;
2988
+ }
2989
+ /** Arguments to {@link MemoryProvider.list}. */
2990
+ interface ListMemoriesInput {
2991
+ /** The scope tokens that apply to this turn, most specific first. */
2992
+ scopes: readonly string[];
2993
+ ctx: ScopeContext;
2994
+ }
2995
+ /**
2996
+ * Arguments to {@link MemoryProvider.search}.
2997
+ *
2998
+ * `scopes` GATES, AND IT GATES FIRST. A search that ranks before it filters is a cross-tenant leak
2999
+ * wearing a relevance score: the nearest neighbour to "what is our rollback policy" is another
3000
+ * base's rollback policy. Filter in the query, not after it. Records returned outside `scopes` are
3001
+ * dropped rather than trusted, so a mistake here costs throughput rather than privacy — but the
3002
+ * drop is a backstop, not the boundary.
3003
+ */
3004
+ interface SearchMemoriesInput {
3005
+ /** The scope tokens that apply to this turn, most specific first. A HARD FILTER. */
3006
+ scopes: readonly string[];
3007
+ /** What this turn is about. Never blank — a blank query is served by `list` instead. */
3008
+ query: string;
3009
+ /**
3010
+ * At most this many distinct KEYS may be ranked in, which is the block's own unit: precedence
3011
+ * resolves a key to one line, so `limit` keys is `limit` lines. Pinned records are returned in
3012
+ * addition and do not count against it.
3013
+ */
3014
+ limit: number;
3015
+ ctx: ScopeContext;
3016
+ }
3017
+ /** Arguments to {@link MemoryProvider.forget}. */
3018
+ interface ForgetMemoryInput {
3019
+ id: string;
3020
+ ctx: ScopeContext;
3021
+ }
3022
+ /**
3023
+ * Arguments to {@link MemoryProvider.write}. An object rather than positional arguments because
3024
+ * `key`, `text` and `scope` are all strings: transposed, a positional call compiles clean and writes
3025
+ * a fact whose key is its value, at a scope nobody meant.
3026
+ */
3027
+ interface StoreMemoryInput extends MemoryFact {
3028
+ ctx: ScopeContext;
3029
+ }
3030
+ /**
3031
+ * Where memories live. The host owns the rows, for the same reason it owns skill rows: the axes a
3032
+ * deployment scopes by, the console that edits them and the audit trail are all the host's, and a
3033
+ * table this package created at boot would put a consumer's migrations in the path of the schema
3034
+ * heal that manages the `agent_*` tables.
3035
+ *
3036
+ * `forget` is REQUIRED, unlike `write`. A deployment may reasonably populate memory from its own
3037
+ * pipeline and offer the agent no write tool; no deployment may reasonably hold conclusions about a
3038
+ * person that the person cannot have deleted. Making it optional would make forgetting a wiring
3039
+ * choice, and it is not one.
3040
+ */
3041
+ interface MemoryProvider {
3042
+ /**
3043
+ * Every memory visible at `scopes`, in any order — this library resolves precedence and the
3044
+ * ceiling. Asked on EVERY turn, so keep it cheap. A provider MAY return records outside `scopes`;
3045
+ * they are dropped rather than trusted, so a filter bug in a host cannot widen what an actor sees.
3046
+ */
3047
+ list(input: ListMemoriesInput): MemoryRecord[] | Promise<MemoryRecord[]>;
3048
+ /** Delete one record by id. `false` where there was nothing by that id to delete. */
3049
+ forget(input: ForgetMemoryInput): boolean | Promise<boolean>;
3050
+ /**
3051
+ * Upsert on (`scope`, `key`) and return the stored record. Omit to serve memory read-only — the
3052
+ * `remember` tool is then never offered, and a turn's tool list is the same on every pod.
3053
+ */
3054
+ write?(input: StoreMemoryInput): MemoryRecord | Promise<MemoryRecord>;
3055
+ /**
3056
+ * The memories worth putting in front of a turn about `query`, MOST RELEVANT FIRST. Omit and every
3057
+ * turn is served by {@link list} — a deployment with twenty memories must not have to stand up an
3058
+ * index to keep working, and one with two thousand uses whatever it already runs (pgvector, a
3059
+ * full-text index, a hybrid service). The host owns the index for the same reason it owns the
3060
+ * rows.
3061
+ *
3062
+ * THREE CLAUSES, and the third is the one that is easy to miss:
3063
+ *
3064
+ * 1. Filter to `scopes` before ranking — see {@link SearchMemoriesInput}.
3065
+ * 2. Rank the KEYS visible at those scopes and take the best `limit` of them.
3066
+ * 3. Return EVERY record sharing a returned key, plus every `pinned` record at those scopes.
3067
+ *
3068
+ * Clause three is what keeps precedence intact. Precedence resolves a conflict between two records
3069
+ * at one key, narrower winning and the beaten value riding along; a search that returned the
3070
+ * `global` half of a conflict and not the `actor:` half would render the org default as the answer
3071
+ * — the exact failure memory exists to prevent, and one nothing downstream can detect. It is one
3072
+ * query either way:
3073
+ *
3074
+ * ```sql
3075
+ * SELECT * FROM agent_memory
3076
+ * WHERE scope IN (:scopes)
3077
+ * AND (pinned OR key IN (SELECT key FROM agent_memory
3078
+ * WHERE scope IN (:scopes)
3079
+ * ORDER BY embedding <=> :queryVector
3080
+ * LIMIT :limit))
3081
+ * ```
3082
+ */
3083
+ search?(input: SearchMemoriesInput): MemoryRecord[] | Promise<MemoryRecord[]>;
3084
+ }
3085
+ /**
3086
+ * How many memories the block carries. The same ceiling shape as `maxSkills`, and for the same
3087
+ * reason: a prompt that grows with how much the agent has written down is a prompt whose cost nobody
3088
+ * set. A budget on the PROMPT and never on the store — with a provider that can
3089
+ * {@link MemoryProvider.search}, it is how many matter right now rather than how many a person may
3090
+ * have.
3091
+ */
3092
+ declare const DEFAULT_MAX_MEMORIES = 20;
3093
+ /**
3094
+ * How long one memory may be, enforced when it is WRITTEN rather than when it is rendered. Enforcing
3095
+ * it at render time would make the prompt disagree with the store; enforcing it at write time makes
3096
+ * the block's whole ceiling a multiplication an operator can do — `maxMemories × maxFactChars` — and
3097
+ * pushes back on the model at the moment it is writing an essay instead of a fact.
3098
+ */
3099
+ declare const DEFAULT_MAX_FACT_CHARS = 240;
3100
+ /** How a turn reaches its memory. See `AgentLoopDeps.memory`. */
3101
+ interface MemoryConfig {
3102
+ provider: MemoryProvider;
3103
+ /** Undefined → `defaultScopeResolver`: the actor's own, their tenant's, the deployment's. */
3104
+ scopes?: ScopeResolver;
3105
+ /** Undefined → {@link DEFAULT_MAX_MEMORIES}. Always-on memories are taken from it first. */
3106
+ maxMemories?: number;
3107
+ /** Undefined → {@link DEFAULT_MAX_FACT_CHARS}. */
3108
+ maxFactChars?: number;
3109
+ }
3110
+ /**
3111
+ * Resolve a provider's records against an ordered scope list: most specific wins a key, and the
3112
+ * losers' values are carried rather than discarded.
3113
+ *
3114
+ * Pure, and separately exported, because the loop and the read-back endpoint must reach the same
3115
+ * answer — a person has to be shown what the model was shown, or "see what it believes about you"
3116
+ * means nothing.
3117
+ */
3118
+ /** Arguments to {@link resolveMemoryDigest}. */
3119
+ interface ResolveMemoryDigestInput {
3120
+ records: readonly MemoryRecord[];
3121
+ /** The scope tokens that apply, most specific first — the precedence order. */
3122
+ scopes: readonly string[];
3123
+ /** Undefined → {@link DEFAULT_MAX_MEMORIES}. */
3124
+ maxMemories?: number;
3125
+ /**
3126
+ * Whether `records` arrived most-relevant-first, from {@link MemoryProvider.search}. A key's rank
3127
+ * is its best-placed record's. Undefined/false → the ceiling selects by scope and recency, which
3128
+ * is the right order only while nothing is being starved.
3129
+ */
3130
+ ranked?: boolean;
3131
+ }
3132
+ declare function resolveMemoryDigest({ records, scopes, maxMemories, ranked, }: ResolveMemoryDigestInput): Omit<MemoryDigest, 'scopes' | 'recalled'>;
3133
+ /** Arguments to {@link offerMemories}. */
3134
+ interface OfferMemoriesInput {
3135
+ config: MemoryConfig;
3136
+ ctx: ScopeContext;
3137
+ /**
3138
+ * What this turn is about — the loop passes the user's own message, and nothing else. Given, and
3139
+ * with a provider that has an index, the block is filled by relevance to it. Absent, the scope is
3140
+ * read whole: that is what the read-back endpoint wants, since a person must be shown every belief
3141
+ * held about them rather than the slice one turn happened to need.
3142
+ */
3143
+ query?: string;
3144
+ }
3145
+ /**
3146
+ * Resolve the scopes and the digest for one turn. Called INSIDE the loop's `memory:digest` step, and
3147
+ * that placement is the whole of its determinism: a search is the most re-derivable thing in this
3148
+ * library — the index moves, a neighbour is written, embeddings are recomputed — so its result has
3149
+ * to be part of the checkpoint every replay reads back, never something a replaying process asks
3150
+ * again.
3151
+ */
3152
+ declare function offerMemories({ config, ctx, query, }: OfferMemoriesInput): Promise<MemoryDigest>;
3153
+ /**
3154
+ * The block as the model reads it.
3155
+ *
3156
+ * THREE THINGS IT HAS TO DO THAT A SKILLS CATALOG DOES NOT.
3157
+ *
3158
+ * It must frame each note by WHO ASSERTED IT, which is why there are two sections rather than one
3159
+ * sentence over the whole list. An unlabelled fact in a system prompt reads with the authority of an
3160
+ * instruction, and the hazard of an agent's own inference is exactly that authority — so those are
3161
+ * hedged. But a memory an administrator published for a whole organisation is not the agent's guess,
3162
+ * and telling the model to "prefer what the user says now" about it hands any user an override of
3163
+ * their organisation's policy by asserting the opposite. `MemoryOrigin.author` is the axis, and it
3164
+ * is the only one that matters here: scope says who a fact applies to, not who decided it.
3165
+ *
3166
+ * Where a narrower scope won, it must print the value it beat, so the model can tell the user their
3167
+ * setting differs from the org's instead of quietly applying one of them.
3168
+ *
3169
+ * And a block that is a SELECTION must say so. A model reading a partial set as a whole one turns an
3170
+ * absence into evidence — "you never told me that" about a fact it simply was not shown.
3171
+ */
3172
+ /** Arguments to {@link buildMemoryBlock}. */
3173
+ interface BuildMemoryBlockInput {
3174
+ entries: readonly MemoryDigestEntry[];
3175
+ /** Whether this deployment offers `remember` — the block names the tool only where it exists. */
3176
+ writable: boolean;
3177
+ /** Whether applicable memories are missing from `entries` — the ceiling bit, or recall selected. */
3178
+ partial: boolean;
3179
+ }
3180
+ declare function buildMemoryBlock({ entries, writable, partial }: BuildMemoryBlockInput): string;
3181
+ /** The reserved name of the built-in memory-writing tool. */
3182
+ declare const REMEMBER_TOOL_NAME = "remember";
3183
+ /** What the model passes to the `remember` tool. */
3184
+ interface RememberToolInput {
3185
+ key: string;
3186
+ fact: string;
3187
+ }
3188
+ /**
3189
+ * The `remember` tool's input schema. Hand-written for the same reason `askInputSchema` and
3190
+ * `skillInputSchema` are: core depends on no validator and has to publish a JSON Schema a provider
3191
+ * can constrain generation against.
3192
+ *
3193
+ * THERE IS NO SCOPE PARAMETER, and that is the enforcement rather than a simplification. An agent
3194
+ * may only ever write the actor it is running for (see {@link memoryWriteVerdict}), so a scope
3195
+ * argument could only ever be a request that gets refused — and a refusable request is one a model
3196
+ * will keep making, and one a future reader will be tempted to make grantable.
3197
+ */
3198
+ declare const rememberInputSchema: StandardSchemaV1<unknown, RememberToolInput>;
3199
+ declare const REMEMBER_TOOL_DESCRIPTION = "Record one durable fact about this user under a key, so later conversations start knowing it. For things that will still be true another day \u2014 how they work, what their organisation requires, a correction they made. Not for what this conversation is about, and not for anything you were not told or could not reasonably infer. The user can read and delete everything you record here.";
3200
+ /**
3201
+ * The `remember` tool as the model sees it. NOT a `ToolSpec` and never registered, exactly like
3202
+ * `ask` and `skill`: it has no handler, because the loop serves it against the digest the journal
3203
+ * holds. Keeping it out of the `ToolRegistry` is also what keeps its kind off a process-local
3204
+ * lookup — see `claimToolCall`.
3205
+ */
3206
+ declare function rememberToolDefinition(): ToolDefinition;
3207
+ /**
3208
+ * Append the built-in `remember` definition to a turn's tool list. Exported because the dispatched
3209
+ * llm step re-derives the tool list on a worker and has to reach the same list the loop would have.
3210
+ */
3211
+ /** Arguments to {@link withMemoryTool}. */
3212
+ interface WithMemoryToolInput {
3213
+ tools: ToolDefinition[];
3214
+ /** Whether this deployment's provider can write at all — module config, uniform across its pods. */
3215
+ enabled: boolean;
3216
+ }
3217
+ declare function withMemoryTool({ tools, enabled }: WithMemoryToolInput): ToolDefinition[];
3218
+ /** Who is trying to write a memory. */
3219
+ interface MemoryAuthor {
3220
+ /**
3221
+ * `'human'` is a person acting through a UI; `'agent'` is anything else — a turn, a tool, a batch
3222
+ * job. The distinction is the whole of rule three below, so it is not inferable and has to be
3223
+ * stated by the caller.
3224
+ */
3225
+ kind: 'human' | 'agent';
3226
+ actorRef?: string;
3227
+ }
3228
+ /** A request to write a memory at a scope. */
3229
+ interface MemoryWriteRequest {
3230
+ scope: string;
3231
+ actor: Actor;
3232
+ /** The actor's resolved scopes, most specific first — the same list the turn drew its digest from. */
3233
+ scopes: readonly string[];
3234
+ author: MemoryAuthor;
3235
+ /**
3236
+ * The host's own answer to "may this person administer that scope". This library cannot know it,
3237
+ * and inventing an answer would be a second, weaker authorization model next to the host's real
3238
+ * one. Undefined/false → a wider scope is refused.
3239
+ */
3240
+ elevated?: boolean;
3241
+ }
3242
+ type MemoryVerdict = {
3243
+ allowed: true;
3244
+ } | {
3245
+ allowed: false;
3246
+ reason: string;
3247
+ };
3248
+ /**
3249
+ * May this author write a memory at this scope?
3250
+ *
3251
+ * The same four rules as `skillWriteVerdict`, deliberately duplicated rather than shared: they are
3252
+ * the same rules about two different things, and folding them into one function would mean a future
3253
+ * change to how skills are authored silently changing who may edit what the assistant believes about
3254
+ * a person. Rule three is the one that matters, and it bites harder here than it does for skills:
3255
+ *
3256
+ * 1. You may only write into a scope you are yourself in.
3257
+ * 2. Your OWN scope is yours. A memory at `actor:<you>` affects exactly one prompt — your own.
3258
+ * 3. NOTHING BUT A HUMAN MAY WRITE ABOVE ITS OWN SCOPE, whatever `elevated` says. A skill an agent
3259
+ * could publish at `tenant:` is a procedure anyone in the tenant can edit by talking to the
3260
+ * assistant; a MEMORY it could publish there is a fact everyone in the tenant is then answered
3261
+ * from, with no document to inspect and nobody aware it was written. An agent that has genuinely
3262
+ * learned something the organisation should know proposes it; a person publishes it.
3263
+ * 4. A human writing above their own scope needs the host to say so (`elevated`).
3264
+ */
3265
+ declare function memoryWriteVerdict(request: MemoryWriteRequest): MemoryVerdict;
3266
+ /**
3267
+ * May this actor delete this memory?
3268
+ *
3269
+ * Narrower than the write rule on purpose, and it takes no `elevated` flag. Deleting what the
3270
+ * assistant believes about YOU needs no permission from anyone — that is the point of the read-back
3271
+ * — so the endpoint that serves it must not be able to ask a host a question it might answer no to.
3272
+ * Deleting what it believes about a tenant is an administrative act on shared state, which belongs
3273
+ * in the host's console with the same elevation a write there needs.
3274
+ */
3275
+ /** A request to delete one memory. */
3276
+ interface MemoryForgetRequest {
3277
+ record: Pick<MemoryRecord, 'scope'>;
3278
+ actor: Actor;
3279
+ }
3280
+ declare function memoryForgetVerdict({ record, actor }: MemoryForgetRequest): MemoryVerdict;
3281
+ /** What a `remember` call resolved to — the stored record, or why nothing was stored. */
3282
+ type MemoryWriteOutcome = {
3283
+ ok: true;
3284
+ record: MemoryRecord;
3285
+ } | {
3286
+ ok: false;
3287
+ error: string;
3288
+ };
3289
+ /**
3290
+ * Serve one `remember` call against the digest THIS TURN resolved.
3291
+ *
3292
+ * The digest's `scopes` are the authorization boundary, not the resolver: they came out of the
3293
+ * `memory:digest` checkpoint, so a write is checked against the scopes the run recorded rather than
3294
+ * the ones a replaying process's resolver would produce now. A membership table edited mid-run
3295
+ * cannot retroactively widen — or narrow — what a turn already in flight is allowed to write.
3296
+ */
3297
+ /** Arguments to {@link writeMemory}. */
3298
+ interface WriteMemoryInput {
3299
+ config: MemoryConfig;
3300
+ /** The digest THIS TURN journaled — the authorization boundary, never a fresh resolution. */
3301
+ digest: MemoryDigest;
3302
+ /** The `remember` call as the model made it. */
3303
+ call: RememberToolInput;
3304
+ ctx: ScopeContext;
3305
+ runId: string;
3306
+ }
3307
+ declare function writeMemory({ config, digest, call, ctx, runId, }: WriteMemoryInput): Promise<MemoryWriteOutcome>;
3308
+
1479
3309
  /** Holds the named agent definitions registered via `AgentModule.forFeature([...])`. */
1480
3310
  declare class AgentRegistry {
1481
3311
  private readonly definitions;
@@ -1485,6 +3315,121 @@ declare class AgentRegistry {
1485
3315
  list(): AgentDefinition[];
1486
3316
  }
1487
3317
 
3318
+ /**
3319
+ * Is this the durable runtime refusing a checkpoint position, rather than anything the agent did?
3320
+ *
3321
+ * Matched by NAME because this package cannot import either class: the in-process engine throws
3322
+ * `@dudousxd/nestjs-durable-core`'s `NonDeterminismError` and the thin BullMQ worker throws
3323
+ * `@dudousxd/nestjs-durable-worker`'s `NondeterminismError` — the same contract under two spellings
3324
+ * in two packages, neither of which core depends on. Same cross-runtime reasoning as the
3325
+ * `isControlFlowError` hook, which exists for exactly this reason on the suspend path.
3326
+ *
3327
+ * Suffix rather than equality, because a runtime is free to qualify the name: the sibling Adonis
3328
+ * stack raises a `WorkflowNondeterminismError` from its remote-replay path, which an equality check
3329
+ * would wave through as an ordinary tool failure. `replay-integrity.contract.spec.ts` pins this
3330
+ * against the real exported classes, so a rename upstream fails a test instead of quietly
3331
+ * disarming the guard.
3332
+ *
3333
+ * Callers must let these through untouched. Every `catch` in a workflow body reacts by writing more
3334
+ * checkpoints (a toolfail, a run-end, a deactivate), and on a journal that has already diverged each
3335
+ * of those asks for a position the history cannot supply — so the recovery attempt raises its own
3336
+ * refusal, and THAT is the error the operator reads: a message pointing at the wrong seq, naming
3337
+ * checkpoints from the recovery path rather than the two that actually disagreed.
3338
+ */
3339
+ declare function isReplayIntegrityError(error: unknown): boolean;
3340
+
3341
+ /**
3342
+ * Is this the runner unwinding the turn (a durable suspend / continue-as-new) rather than something
3343
+ * the agent did? Callers must rethrow it untouched: every `catch` in the loop reacts by writing more
3344
+ * checkpoints, and a suspend recorded as a tool failure leaves a journal the resumed replay cannot
3345
+ * line up with.
3346
+ *
3347
+ * Detected from the error itself so a host that never wired `AgentLoopHooks.isControlFlowError` still
3348
+ * gets its suspends through — that hook is an override for a runner whose signals carry no marker,
3349
+ * not the only line of defence. Checked as a stamped property, NOT `instanceof`: the classes differ
3350
+ * per runtime, which is why the marker exists.
3351
+ */
3352
+ declare function isControlFlowSignal(error: unknown): boolean;
3353
+
3354
+ /** An agent->agent edge with its awaited-or-not resolved, whichever form it was authored in. */
3355
+ interface ResolvedDelegation {
3356
+ agent: string;
3357
+ detached: boolean;
3358
+ }
3359
+ /** Read one {@link AgentDelegation} entry in either of its forms. */
3360
+ declare function normalizeDelegation(entry: AgentDelegation): ResolvedDelegation;
3361
+ /**
3362
+ * What a detached delegation hands the model in place of an answer. Its `status` is the whole point:
3363
+ * a model that is given an object shaped like a result will report one, and this run has none yet.
3364
+ */
3365
+ interface DetachedDelegationReceipt {
3366
+ detached: true;
3367
+ status: 'started';
3368
+ /** The agent now working on it. */
3369
+ agent: string;
3370
+ /** The run doing the work — what a client subscribes to, and what stamps the delivered message. */
3371
+ runId: string;
3372
+ /**
3373
+ * The same fact in the only vocabulary the model reliably acts on: prose in a tool result. The
3374
+ * structured fields above are for a client; this is for the turn that has to explain itself to a
3375
+ * user without claiming a result it does not hold.
3376
+ */
3377
+ note: string;
3378
+ }
3379
+ /** How a detached delegation ended, written back onto its tool-call row once the run settles. */
3380
+ interface DetachedDelegationOutcome {
3381
+ detached: true;
3382
+ status: 'delivered' | 'failed' | 'cancelled';
3383
+ agent: string;
3384
+ runId: string;
3385
+ /** The delivered answer. Present on `delivered` only. */
3386
+ text?: string;
3387
+ /** Why it did not deliver. Present on `failed` only. */
3388
+ error?: string;
3389
+ }
3390
+ /** The receipt an `agent`-kind call returns the moment its delegate is under way. */
3391
+ declare function detachedStarted(args: {
3392
+ agent: string;
3393
+ runId: string;
3394
+ }): DetachedDelegationReceipt;
3395
+ /** The outcome written onto the delegating tool call when a detached run finishes its answer. */
3396
+ declare function detachedDelivered(args: {
3397
+ agent: string;
3398
+ runId: string;
3399
+ text: string;
3400
+ }): DetachedDelegationOutcome;
3401
+ /**
3402
+ * The outcome written onto the delegating tool call when a detached run never produced an answer.
3403
+ * Written by the RUNNER, not the loop: a run that crashed or was stopped cannot record its own
3404
+ * ending, and a delegation card that stays "started" for ever is the one state a reader cannot act
3405
+ * on.
3406
+ */
3407
+ declare function detachedUnsettled(args: {
3408
+ agent: string;
3409
+ runId: string;
3410
+ status: 'failed' | 'cancelled';
3411
+ error?: string;
3412
+ }): DetachedDelegationOutcome;
3413
+ /**
3414
+ * Settle the delegation a detached run was started for when that run ends with no answer.
3415
+ *
3416
+ * Two writes, because a reader needs both: the tool-call row so a governance surface stops counting
3417
+ * it as in flight, and a MESSAGE in the delegating thread so the person who asked finds out. A
3418
+ * delivered answer already arrives as a message; without this, the failure is the one outcome that
3419
+ * silently never does, and the conversation shows "started" for ever.
3420
+ *
3421
+ * Called by the RUNNER, never the loop — a run that crashed or was stopped cannot record its own
3422
+ * ending. Skipped entirely when the thread is gone, for the same reason delivery is.
3423
+ */
3424
+ declare function settleUnsettledDelegation(args: {
3425
+ store: AgentStore;
3426
+ delivery: DetachedDelivery;
3427
+ agent: string;
3428
+ runId: string;
3429
+ status: 'failed' | 'cancelled';
3430
+ error?: string;
3431
+ }): Promise<void>;
3432
+
1488
3433
  /** Thrown when an actor invokes a tool their role is not allowed. */
1489
3434
  declare class ToolForbiddenError extends Error {
1490
3435
  readonly toolName: string;
@@ -1529,10 +3474,14 @@ declare class ToolRegistry {
1529
3474
  allSpecs(): ToolSpec[];
1530
3475
  /**
1531
3476
  * The tools to offer the model for this actor+agent, after the four filter layers: what this
1532
- * deployment has enabled, what this actor's role allows, what each tool's own `canUse` allows
1533
- * this actor, and finally what this agent pinned.
3477
+ * agent pinned, what this deployment has enabled, what this actor's role allows, and what each
3478
+ * tool's own `canUse` allows this actor.
1534
3479
  *
1535
3480
  * Every layer only ever removes tools, so no arrangement of them can widen what a turn reaches.
3481
+ * The agent's allow-list therefore goes FIRST, even though it is the narrowest statement: it is a
3482
+ * pure name-set match, while each of the three below it may be a round trip — an authz service, a
3483
+ * feature-flag store, an MCP server — and this runs once per model step. Asking those about a
3484
+ * tool the allow-list has already excluded is a call whose answer nothing reads.
1536
3485
  */
1537
3486
  definitionsFor(actor: Actor, policy: RolesPolicy, allowedTools?: string[]): Promise<ToolDefinition[]>;
1538
3487
  /**
@@ -1549,7 +3498,7 @@ declare class DefaultRolesPolicy implements RolesPolicy {
1549
3498
  can(actor: Actor, tool: ToolSpec): boolean;
1550
3499
  }
1551
3500
 
1552
- interface AgentLoopDeps {
3501
+ interface AgentLoopDeps<TOutput = unknown> {
1553
3502
  model: ModelProvider;
1554
3503
  store: AgentStore;
1555
3504
  registry: ToolRegistry;
@@ -1571,6 +3520,26 @@ interface AgentLoopDeps {
1571
3520
  */
1572
3521
  promptContributors?: PromptContributor[];
1573
3522
  maxSteps?: number;
3523
+ /**
3524
+ * How deep agent→agent delegation may nest before the loop refuses further hops.
3525
+ * Defaults to {@link MAX_DELEGATION_DEPTH}.
3526
+ *
3527
+ * It bounds NESTING, never fan-out: how many agents a turn delegates to is the model's, one tool
3528
+ * call each, and nothing here caps that. What it guards is a hop the model cannot see — a
3529
+ * `delegatesTo` cycle (A→B→A), where each agent is making one reasonable call and the recursion
3530
+ * is a property of the wiring rather than of any decision.
3531
+ */
3532
+ maxDelegationDepth?: number;
3533
+ /**
3534
+ * How many times one agent may appear on a single delegation chain.
3535
+ * Defaults to {@link DEFAULT_MAX_AGENT_APPEARANCES}.
3536
+ *
3537
+ * This is the guard the depth ceiling was a proxy for, made exact: the loop compares the target
3538
+ * against {@link AgentRunInput.delegationPath} and knows whether the chain has been here before,
3539
+ * and how often. A chain of eight DISTINCT agents is long, not looping, and no longer refused for
3540
+ * resembling one.
3541
+ */
3542
+ maxAgentAppearances?: number;
1574
3543
  /** Optional host handle threaded to tool ctx (e.g. an ORM EntityManager). */
1575
3544
  host?: unknown;
1576
3545
  /** Agent-level tool allow-list. Undefined → all tools (after role filtering). */
@@ -1611,13 +3580,174 @@ interface AgentLoopDeps {
1611
3580
  * message/step) and reused for every step's estimate. Undefined → `costUsd` is always `null`.
1612
3581
  */
1613
3582
  pricingStore?: AgentPricingStore;
3583
+ /**
3584
+ * Bounds how much of the thread rides into the turn. Undefined → the WHOLE thread, every message
3585
+ * the store holds, which is unbounded: a long-lived thread eventually exceeds the provider's
3586
+ * context limit, and pays for the full transcript on every turn up to that point. See
3587
+ * `windowHistory` for the built-in.
3588
+ */
3589
+ historyPolicy?: HistoryPolicy;
3590
+ /**
3591
+ * Rewrites the prompt before EACH model call of the turn, in order (see {@link InputProcessor}).
3592
+ * Transformation only — `historyPolicy` owns which messages are there in the first place. Adds
3593
+ * one `process:input:<step>` checkpoint per step; empty/undefined adds none.
3594
+ */
3595
+ inputProcessors?: InputProcessor[];
3596
+ /**
3597
+ * Inspects each model step's answer before the stream, the store or the next step sees it, and
3598
+ * may redact, replace or refuse it (see {@link OutputProcessor}).
3599
+ *
3600
+ * REGISTERING ONE TAKES THE MODEL CALL OFF THE RUN'S SINK for the turn — a gate cannot run after
3601
+ * the answer has already reached the reader. What the chain costs the subscriber depends on what
3602
+ * it declares:
3603
+ *
3604
+ * - Any processor WITHOUT `incremental` → the whole answer is buffered and released as one `text`
3605
+ * frame once the chain has passed. No token-by-token text, and (for a provider that writes bytes
3606
+ * outside the `AgentStreamEvent` vocabulary) no frame the gate cannot classify.
3607
+ * - EVERY processor with `incremental` → the chain runs over the growing prefix and releases it as
3608
+ * it arrives, holding back the widest `lookbackChars` any of them asked for. The whole-answer
3609
+ * pass still runs and is still authoritative for the stream and the store.
3610
+ *
3611
+ * Both add one `process:output:<step>` checkpoint per step, in the same position, so a chain can
3612
+ * change its declaration without moving a checkpoint. Empty/undefined adds none, and the turn
3613
+ * streams exactly as it always did.
3614
+ */
3615
+ outputProcessors?: OutputProcessor[];
3616
+ /**
3617
+ * Constrain the turn's answer to a schema, returned validated as `object` on the loop's result and
3618
+ * recorded on the assistant message as a synthetic `structured_output` tool call (the same shape
3619
+ * inject-mode retrieval uses, so no store gains a column for it).
3620
+ *
3621
+ * HOW IT COMPOSES WITH TOOL CALLING: as a separate formatting pass, ALWAYS. The turn runs its
3622
+ * model→tools iteration exactly as it would without a schema; once a step comes back with no tool
3623
+ * calls, one extra non-streamed call (`structured:<step>`, `tools: []`, `outputSchema` set)
3624
+ * restates that answer as the schema. Most providers cannot serve a response format and a tool set
3625
+ * in one request, and skipping the pass for an agent that happens to have no tools would make the
3626
+ * checkpoint sequence depend on a tool-registry lookup — the registry of whichever process is
3627
+ * replaying — which is exactly how a run ends up asking for a position its history has no room
3628
+ * for. So the pass is unconditional, and it costs one model call per turn (recorded as
3629
+ * `structured_output` usage).
3630
+ */
3631
+ outputSchema?: StandardSchemaV1<unknown, TOutput>;
3632
+ /** Overrides the formatting pass's system prompt. Undefined → `DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION`. */
3633
+ outputInstruction?: string;
3634
+ /**
3635
+ * How many extra model calls may try to fix an answer that failed `outputSchema`, each shown the
3636
+ * previous attempt's validation issues. Undefined → 1; `0` → fail on the first invalid reply.
3637
+ * Bounded because a model that cannot satisfy a schema usually cannot satisfy it on the fourth
3638
+ * try either, and every attempt is billed.
3639
+ */
3640
+ outputRepairAttempts?: number;
3641
+ /**
3642
+ * Show the formatting pass the turn's whole transcript instead of just the question and the
3643
+ * answer. For an agent whose answer cannot be restated from its own words — one that reports on
3644
+ * rows a tool returned and names only their total in the prose, say.
3645
+ *
3646
+ * OFF by default because the pass is a translation, and a translation needs the thing being
3647
+ * translated. The transcript it would otherwise carry is the turn's entire prompt a second time,
3648
+ * at no discount: the pass swaps the system block for the schema instruction, and the system block
3649
+ * is the prompt cache's prefix, so nothing of the first call's cache survives into it.
3650
+ */
3651
+ outputFromTranscript?: boolean;
3652
+ /**
3653
+ * A question set the agent puts to the user BEFORE it starts working — collecting the scope, as
3654
+ * against `awaitApproval`, which sanctions work already proposed. The questions are AUTHORED, so
3655
+ * the turn spends no model call producing them and a client knows the total ("Question 1 of 3")
3656
+ * the moment the form appears.
3657
+ *
3658
+ * The turn parks on the answers exactly as it parks on an approval, and the request persists as
3659
+ * the same tool-call row the model's `ask` writes — see {@link ask}. Undefined → no intake, and a
3660
+ * turn's checkpoint sequence is byte-identical to one that never had the option.
3661
+ */
3662
+ intake?: AgentIntake;
3663
+ /**
3664
+ * Offer the model the built-in `ask` tool, so it can put its own question set to the user when it
3665
+ * judges the scope is missing — the same surface as {@link intake}, minus the "known in advance".
3666
+ *
3667
+ * `ask` is NOT a registered tool: it has no handler (the loop settles it against a human), and
3668
+ * keeping it out of the `ToolRegistry` is what keeps its kind out of a process-local lookup. The
3669
+ * loop appends its definition to the turn's tool list from THIS flag, which is module config and
3670
+ * therefore uniform across a deployment. Undefined/false → the model never sees it.
3671
+ */
3672
+ ask?: boolean;
3673
+ /**
3674
+ * Authored procedures the model may pull in when a task calls for one, resolved per turn against
3675
+ * the actor's scopes — see `skills.ts`. Undefined → no catalog block, no `skill` tool, and a
3676
+ * turn's checkpoint sequence is byte-identical to one that never had the option.
3677
+ *
3678
+ * HOW IT COMPOSES WITH THE OTHER FOUR THINGS THAT WRITE THE PROMPT. The system block is assembled
3679
+ * in one fixed order — the agent's base prompt, then each `promptContributors` section, then
3680
+ * {@link memory}, then the injected retrieval block, then the skills CATALOG — and skills are the
3681
+ * cheapest of the five by construction: the catalog is one line per skill and nothing else. The
3682
+ * instructions themselves arrive as a tool RESULT, on the transcript, which means the budget they
3683
+ * draw on is the one `historyPolicy` already governs rather than a private allowance of their own.
3684
+ *
3685
+ * A BODY IS NOT LIKE A MEMORY, which does ride the system block. A body is read BECAUSE the model
3686
+ * went and asked for it, so the transcript — what this conversation happened to pull in — is where
3687
+ * it belongs. A memory is worthless unless it is in front of the model on the turn nobody thought
3688
+ * to look for it, and it is affordable there only because it has no body to carry.
3689
+ */
3690
+ skills?: SkillsConfig;
3691
+ /**
3692
+ * What the assistant has previously concluded about the actor and their organisation, resolved per
3693
+ * turn against the same scope tokens skills use — see `memory.ts`. Undefined → no memory block, no
3694
+ * `remember` tool, and a turn's checkpoint sequence is byte-identical to one that never had the
3695
+ * option.
3696
+ *
3697
+ * WHERE IT SITS IN THE PROMPT. The system block is assembled most-durable-first: the agent's base
3698
+ * prompt (the same for everyone, every turn), then `promptContributors`, then MEMORY (the same for
3699
+ * this person, every turn), then the injected retrieval block (this question only), then the
3700
+ * skills catalog (a menu rather than an instruction, so a reader meets instructions before
3701
+ * options).
3702
+ *
3703
+ * WHY IT IS IN THE SYSTEM BLOCK AT ALL, when a skill's body deliberately is not. A body is read
3704
+ * BECAUSE the model went and asked for it; a memory is worthless unless it is in front of the
3705
+ * model on the turn nobody thought to look for it — "they report in nautical miles" only works
3706
+ * unprompted. What makes that affordable is that a memory has no body: `maxMemories` lines, each
3707
+ * capped at `maxFactChars` when it is written, so the block's ceiling is a product of two numbers
3708
+ * an operator set rather than however much the model felt like writing down.
3709
+ */
3710
+ memory?: MemoryConfig;
1614
3711
  }
3712
+ /** One task's outcome under {@link AgentLoopHooks.parallel}, reported instead of thrown. */
3713
+ type SettledTask<T> = {
3714
+ ok: true;
3715
+ value: T;
3716
+ } | {
3717
+ ok: false;
3718
+ error: unknown;
3719
+ };
3720
+ /**
3721
+ * The {@link AgentLoopHooks.parallel} implementation for a runner whose checkpoint positions are
3722
+ * handed out on the CALL (both durable primitives are: `ctx.localStep` and `ctx.step` take their
3723
+ * position before their first `await`). Every task is invoked here, synchronously and in list
3724
+ * order, before any of them is awaited — which is what fixes the block of positions the tasks
3725
+ * occupy, whatever order they then settle in. Nothing rejects: the caller decides what an
3726
+ * individual failure means.
3727
+ */
3728
+ declare function settleAll<T>(tasks: readonly (() => Promise<T>)[]): Promise<SettledTask<T>[]>;
1615
3729
  interface AgentLoopHooks {
1616
3730
  runId: string;
1617
3731
  /** A writer for this run's live token stream (data plane). */
1618
3732
  openSink(): SinkWriter | Promise<SinkWriter>;
1619
3733
  /** HITL gate for an action tool. Inline resolves a pending promise; durable awaits a signal. */
1620
3734
  awaitApproval(call: ToolCallRequest, ctx: AiToolCtx): Promise<Decision>;
3735
+ /**
3736
+ * Park the run on a question set and resolve with what the human sent back. The SAME wait
3737
+ * `awaitApproval` is — the durable runner maps both to `ctx.waitForSignal` on
3738
+ * `tool:<runId>:<callId>`, so an answer and an approval reach a parked run through one path.
3739
+ *
3740
+ * Optional, and its absence changes no checkpoint: the loop falls back to `awaitApproval` and
3741
+ * reads the decision as "the user confirmed the pre-picked answers" (approved) or "the user
3742
+ * skipped" (rejected). That is the honest reduction of a yes/no channel, and it means a host that
3743
+ * only ever implemented approval still runs an elicitation to completion instead of hanging.
3744
+ *
3745
+ * Declared as `HumanReply` rather than `ElicitationReply` because that is what the channel really
3746
+ * carries: a question set is parked as a `pending_approval` action, so a `Decision` can arrive on
3747
+ * this wait from the approvals inbox even where the host implements it. The loop reduces one to
3748
+ * the other — see {@link normalizeElicitationReply} — so an implementer never has to.
3749
+ */
3750
+ awaitAnswers?(request: ElicitationRequest, ctx: AiToolCtx): Promise<HumanReply>;
1621
3751
  /**
1622
3752
  * Run another named agent and return its answer. Provided only when the host wired multi-agent
1623
3753
  * support (durable → child workflow, inline → nested loop). Exposed to tools as `ctx.runAgent`.
@@ -1625,6 +3755,27 @@ interface AgentLoopHooks {
1625
3755
  runAgent?(agentName: string, task: string): Promise<{
1626
3756
  text: string;
1627
3757
  }>;
3758
+ /**
3759
+ * Start another named agent and return its run id WITHOUT waiting for it, so the calling turn can
3760
+ * finish while the delegate is still working. The durable runner maps this to `ctx.startChild`
3761
+ * (checkpointed as `spawn:<id>`; no suspend, unlike the `ctx.child` behind {@link runAgent}); the
3762
+ * inline runner to a nested loop nobody awaits.
3763
+ *
3764
+ * `toolCallId` is the delegation's own call, which the started run carries as its delivery
3765
+ * address: it posts its answer back into the calling thread against that row.
3766
+ *
3767
+ * Absent -> a delegation the journal declares detached is AWAITED instead. The loop writes the
3768
+ * same checkpoint names either way (see {@link delegateToolCall}), so a runner that cannot detach
3769
+ * still answers the user; only the runner's own positions differ, and a given runner always makes
3770
+ * the same choice for the same call.
3771
+ */
3772
+ startAgent?(args: {
3773
+ agentName: string;
3774
+ task: string;
3775
+ toolCallId: string;
3776
+ }): Promise<{
3777
+ runId: string;
3778
+ }>;
1628
3779
  /**
1629
3780
  * Checkpoint wrapper. Inline = call fn directly; durable = ctx.localStep(name, fn) — the
1630
3781
  * in-process checkpointed primitive (`name` is a checkpoint identity, not a worker group).
@@ -1635,10 +3786,12 @@ interface AgentLoopHooks {
1635
3786
  /**
1636
3787
  * Dispatch the model turn as a routed remote step. When present the loop uses it INSTEAD of
1637
3788
  * running the model inline under hooks.step. Must resolve to exactly what deps.model.runTurn
1638
- * resolves. The durable runner enriches the envelope with sink routing (sinkRunId/childSink)
1639
- * on its side — core stays sink-topology-agnostic.
3789
+ * resolves, plus `bufferedFrames` when (and only when) the envelope set `bufferOutput` — the
3790
+ * handler streams to a worker-side sink the loop cannot wrap, so an output gate depends on it
3791
+ * honouring that flag. The durable runner enriches the envelope with sink routing
3792
+ * (sinkRunId/childSink) on its side — core stays sink-topology-agnostic.
1640
3793
  */
1641
- dispatchLlm?(index: number, envelope: LlmStepEnvelope): Promise<ModelTurnResult>;
3794
+ dispatchLlm?(index: number, envelope: LlmStepEnvelope): Promise<BufferedModelTurnResult>;
1642
3795
  /**
1643
3796
  * Dispatch one tool execution as a routed remote step. Replaces ONLY the registry.invoke call
1644
3797
  * (+ its timeout, applied handler-side); all persist steps around it stay local.
@@ -1646,14 +3799,83 @@ interface AgentLoopHooks {
1646
3799
  dispatchTool?(call: ToolCallRequest, envelope: ToolStepEnvelope): Promise<unknown>;
1647
3800
  /**
1648
3801
  * Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so the loop
1649
- * rethrows them untouched instead of recording a failure. Inline runners omit it — they have no
1650
- * control-flow exceptions.
3802
+ * rethrows them untouched instead of recording a failure. An OVERRIDE, not the gate: the loop
3803
+ * already recognizes any signal carrying the durable runtime's `Symbol.for` marker on its own (see
3804
+ * {@link isControlFlowSignal}), so a host that omits this — an inline runner with no control-flow
3805
+ * exceptions, or one that simply forgot — never turns a suspend into a persisted tool failure.
3806
+ * Supply it for a runner whose signals carry no marker.
1651
3807
  */
1652
3808
  isControlFlowError?(error: unknown): boolean;
3809
+ /**
3810
+ * Run `tasks` concurrently, resolving once EVERY one has settled — one outcome per task, in INPUT
3811
+ * order, never rejecting. Supplying it is a statement about the runner's checkpointing: each task
3812
+ * MUST be invoked synchronously, in list order, before any is awaited, so a runner that hands out
3813
+ * positions on the call assigns them in that order regardless of which task finishes first.
3814
+ * {@link settleAll} is exactly that, and is what both bundled runners pass.
3815
+ *
3816
+ * Waiting for ALL of them is the other half of the contract. A durable runner unwinds a turn by
3817
+ * THROWING (a dispatched step suspends), and a sibling abandoned part-way through its own dispatch
3818
+ * is a tool nobody ever runs.
3819
+ *
3820
+ * Absent → the loop runs a turn's tool calls one at a time. That is the honest answer for a runner
3821
+ * whose positions are assigned anywhere other than the call, and it is the only behaviour that
3822
+ * existed before, so nothing gains concurrency without saying so.
3823
+ */
3824
+ parallel?<T>(tasks: readonly (() => Promise<T>)[]): Promise<SettledTask<T>[]>;
3825
+ /**
3826
+ * Has someone asked this run to stop?
3827
+ *
3828
+ * Answered at the points where stopping is safe and cheap — between steps, before the next model
3829
+ * call, before the turn's tools are dispatched — and NEVER consulted anywhere else. Two properties
3830
+ * make it safe to ask a live question inside a replayed loop body:
3831
+ *
3832
+ * 1. THE ANSWER IS JOURNALED. The loop only ever calls this inside a checkpoint, so the first
3833
+ * process to reach a given position writes the answer there and every later replay reads it
3834
+ * back. A cancel that arrives between two replays is therefore seen at the first position the
3835
+ * history does NOT yet hold, and cannot change the branch a replayed position already took.
3836
+ * 2. THE POSITIONS ARE PATCHED IN. They exist only for a run that {@link patched} admits to the
3837
+ * `agent:cancellation` shape, so a run already in flight when a deployment gained this keeps
3838
+ * replaying against the sequence it recorded.
3839
+ *
3840
+ * Undefined → the run cannot be cancelled and takes not one extra checkpoint, which is the honest
3841
+ * answer for a host with nowhere to record the request.
3842
+ *
3843
+ * WHAT IT CANNOT REACH: a turn parked on a human (`awaitApproval`/`awaitAnswers`) is suspended
3844
+ * INSIDE a position the journal already holds, so no observation of any kind fires there. Stopping
3845
+ * a parked run is the runner's job, through whatever hard cancel its runtime has.
3846
+ */
3847
+ cancelled?(): Promise<boolean>;
3848
+ /**
3849
+ * Does this run take the loop shape guarded by `id`? A runner replaying against recorded
3850
+ * checkpoints answers `false` for a run that started before the shape changed, so that run keeps
3851
+ * replaying the shape its history holds (the durable runtime's `ctx.patched`, which consumes a
3852
+ * position for a new run and gives it back to an old one). Absent → `true`: a runner that records
3853
+ * no positions has no older shape to preserve.
3854
+ */
3855
+ patched?(id: string): Promise<boolean>;
1653
3856
  }
1654
3857
  declare class QuotaExceededError extends Error {
1655
3858
  constructor();
1656
3859
  }
3860
+ /**
3861
+ * Someone asked this run to stop, and it did. NOT a failure — it is the outcome a user pressing Stop
3862
+ * is entitled to, and a consumer whose reliability numbers count `failed` runs must be able to leave
3863
+ * it out. Thrown by the loop at the point it observed the cancel, and settled by the runner, which
3864
+ * records the run `cancelled` and ends the stream with a `cancelled` frame rather than failing it.
3865
+ *
3866
+ * Carries no message detail on purpose: there is nothing to diagnose, and a cancel reads the same
3867
+ * whether it came from a Stop button, an operator console, or a deployment draining.
3868
+ */
3869
+ declare class RunCancelledError extends Error {
3870
+ constructor();
3871
+ }
3872
+ /**
3873
+ * The stream error code a failed run surfaces to its subscriber. A refusal by an output processor
3874
+ * and an answer that never satisfied `outputSchema` are the CONTROLS working, not the model
3875
+ * breaking: a client that retries on `run_failed` must not retry either of them, and neither should
3876
+ * page whoever is on call for model failures.
3877
+ */
3878
+ declare function agentFailureCode(error: unknown): string;
1657
3879
  /** Reject if `work` doesn't settle within `ms`. The underlying work is left to finish on its own. */
1658
3880
  declare function withToolTimeout<T>(work: Promise<T>, ms: number, toolName: string): Promise<T>;
1659
3881
  /**
@@ -1672,14 +3894,26 @@ declare function traceToolExecution<T>(runId: string, call: {
1672
3894
  toolName: string;
1673
3895
  toolType: 'read' | 'action';
1674
3896
  }, run: () => Promise<T>): Promise<T>;
3897
+ /**
3898
+ * Append the built-in `ask` definition to a turn's tool list. Exported because the dispatched llm
3899
+ * step re-derives the tool list on a worker and has to reach the same list the loop would have.
3900
+ */
3901
+ declare function withAskTool({ tools, ask, }: {
3902
+ tools: ToolDefinition[];
3903
+ ask: boolean | undefined;
3904
+ }): ToolDefinition[];
3905
+ /** What a turn answered with: the assistant text, plus the validated `outputSchema` value if any. */
3906
+ interface AgentLoopResult<TOutput = unknown> {
3907
+ text: string;
3908
+ /** Present only when `AgentLoopDeps.outputSchema` was set — the validated structured answer. */
3909
+ object?: TOutput;
3910
+ }
1675
3911
  /**
1676
3912
  * The provider-agnostic agent turn, reused by both the inline and durable runners.
1677
3913
  * It drives the model→tools→model iteration; the runner supplies the `step`/`awaitApproval`
1678
3914
  * hooks that make the same loop body either in-process or a replay-safe durable workflow.
1679
3915
  */
1680
- declare function runAgentLoop(deps: AgentLoopDeps, input: AgentRunInput, hooks: AgentLoopHooks): Promise<{
1681
- text: string;
1682
- }>;
3916
+ declare function runAgentLoop<TOutput = unknown>(deps: AgentLoopDeps<TOutput>, input: AgentRunInput, hooks: AgentLoopHooks): Promise<AgentLoopResult<TOutput>>;
1683
3917
 
1684
3918
  /** Payloads carried on each `aviary:agent:*` channel. */
1685
3919
  interface AgentRunStarted {
@@ -1724,6 +3958,8 @@ interface AgentDelegated {
1724
3958
  runId: string;
1725
3959
  fromAgent?: string;
1726
3960
  toAgent: string;
3961
+ /** The delegate was STARTED, not awaited — this run's turn ended without its answer. */
3962
+ detached?: boolean;
1727
3963
  }
1728
3964
  interface AgentRetrieved {
1729
3965
  runId: string;
@@ -1731,6 +3967,64 @@ interface AgentRetrieved {
1731
3967
  /** How many passages the retriever returned. */
1732
3968
  count: number;
1733
3969
  }
3970
+ /**
3971
+ * The skills a turn was offered, published once per run at the `skills:catalog` checkpoint. Metadata
3972
+ * only — counts and the block's size, never a skill's name or its body.
3973
+ */
3974
+ interface AgentSkillsResolved {
3975
+ runId: string;
3976
+ /** How many scope tokens the resolver returned for this turn. */
3977
+ scopes: number;
3978
+ /** Skills in the catalog block the model was shown. */
3979
+ offered: number;
3980
+ /** Applicable skills the `maxSkills` ceiling left out — non-zero means the catalog is partial. */
3981
+ omitted: number;
3982
+ /**
3983
+ * Characters the catalog block added to the system prompt. The WHOLE of what skills cost it: a
3984
+ * skill's body never enters the system block, it arrives as a tool result on the transcript.
3985
+ */
3986
+ promptChars: number;
3987
+ }
3988
+ /**
3989
+ * The memories a turn was shown, published once per run at the `memory:digest` checkpoint. Metadata
3990
+ * only — counts and the block's size, never a memory's key or its text. A key is as revealing as a
3991
+ * fact (`diagnosis`, `clearance`), so it does not travel on a diagnostics channel either.
3992
+ */
3993
+ interface AgentMemoryResolved {
3994
+ runId: string;
3995
+ /** How many scope tokens the resolver returned for this turn. */
3996
+ scopes: number;
3997
+ /** Memories in the block the model was shown. */
3998
+ offered: number;
3999
+ /** Applicable memories the `maxMemories` ceiling left out — non-zero means the block is partial. */
4000
+ omitted: number;
4001
+ /**
4002
+ * Of `omitted`, how many were ALWAYS-ON. The one to alert on: ordinary omission is the budget
4003
+ * working, while a dropped always-on memory is a deployment's own standing policies having stopped
4004
+ * reaching any prompt — more was pinned than the block holds.
4005
+ */
4006
+ pinnedOmitted: number;
4007
+ /**
4008
+ * Whether the block was selected by relevance to this turn rather than read whole. Distinguishes
4009
+ * "this deployment holds few memories" from "this turn drew twenty out of two thousand", which
4010
+ * `offered` alone reports identically.
4011
+ */
4012
+ recalled: boolean;
4013
+ /** Characters the memory block added to the system prompt. The WHOLE of what memory cost it. */
4014
+ promptChars: number;
4015
+ }
4016
+ /**
4017
+ * A turn writing one memory — the event an operator watches to see the agent's own write volume,
4018
+ * which is the risk surface memory has and retrieval does not. The scope token travels (it is what
4019
+ * says whose prompt just changed, and `run.started` already carries an actor id); the key and the
4020
+ * text do not.
4021
+ */
4022
+ interface AgentMemoryWritten {
4023
+ runId: string;
4024
+ scope: string;
4025
+ /** Length of the stored fact in characters — never the fact itself. */
4026
+ chars: number;
4027
+ }
1734
4028
  /**
1735
4029
  * A transient-classified tool error being retried in place (no new checkpoint) — see
1736
4030
  * `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
@@ -1771,6 +4065,17 @@ interface AgentFollowUpsSpan {
1771
4065
  /** How many follow-up questions were requested. */
1772
4066
  count: number;
1773
4067
  }
4068
+ /**
4069
+ * START payload of an `aviary:agent:structured-output:*` span — the formatting pass that restates a
4070
+ * finished answer as `AgentLoopDeps.outputSchema`. `attempt` is 0 for the pass itself and counts up
4071
+ * for each bounded repair, so a run that needed three tries is visible as three spans.
4072
+ */
4073
+ interface AgentStructuredOutputSpan {
4074
+ runId: string;
4075
+ /** Zero-based model-call index of the final turn whose answer is being restated. */
4076
+ step: number;
4077
+ attempt: number;
4078
+ }
1774
4079
  /** Declaration-merge so `emit('agent', ...)`, `trace('agent', ...)` and telescope infer the agent payloads. */
1775
4080
  declare module '@dudousxd/nestjs-diagnostics' {
1776
4081
  interface ChannelRegistry {
@@ -1783,11 +4088,15 @@ declare module '@dudousxd/nestjs-diagnostics' {
1783
4088
  'run.failed': AgentRunFailed;
1784
4089
  delegated: AgentDelegated;
1785
4090
  retrieved: AgentRetrieved;
4091
+ 'skills.resolved': AgentSkillsResolved;
4092
+ 'memory.resolved': AgentMemoryResolved;
4093
+ 'memory.written': AgentMemoryWritten;
1786
4094
  'tool.retry': AgentToolRetry;
1787
4095
  'llm.turn': AgentLlmTurnSpan;
1788
4096
  'tool.execution': AgentToolExecutionSpan;
1789
4097
  retrieval: AgentRetrievalSpan;
1790
4098
  'follow-ups': AgentFollowUpsSpan;
4099
+ 'structured-output': AgentStructuredOutputSpan;
1791
4100
  };
1792
4101
  }
1793
4102
  }
@@ -1800,6 +4109,9 @@ declare function publishAgentRunFailed(payload: AgentRunFailed): void;
1800
4109
  declare function publishAgentDelegated(payload: AgentDelegated): void;
1801
4110
  declare function publishAgentRetrieved(payload: AgentRetrieved): void;
1802
4111
  declare function publishAgentToolRetry(payload: AgentToolRetry): void;
4112
+ declare function publishAgentSkillsResolved(payload: AgentSkillsResolved): void;
4113
+ declare function publishAgentMemoryResolved(payload: AgentMemoryResolved): void;
4114
+ declare function publishAgentMemoryWritten(payload: AgentMemoryWritten): void;
1803
4115
  /**
1804
4116
  * Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
1805
4117
  * `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
@@ -1807,13 +4119,13 @@ declare function publishAgentToolRetry(payload: AgentToolRetry): void;
1807
4119
  * nothing to subscribe to on their base channels, and claiming their keys would be meaningless
1808
4120
  * (the generic bridge only records point traffic).
1809
4121
  */
1810
- type AgentSpanEvent = 'llm.turn' | 'tool.execution' | 'retrieval' | 'follow-ups';
4122
+ type AgentSpanEvent = 'llm.turn' | 'tool.execution' | 'retrieval' | 'follow-ups' | 'structured-output';
1811
4123
  /** All span-only events, in a stable order — for a future span recorder to derive sub-channels from. */
1812
4124
  declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
1813
4125
  /** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
1814
4126
  type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
1815
4127
  /**
1816
- * All 9 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
4128
+ * All 12 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
1817
4129
  * subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
1818
4130
  * {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
1819
4131
  * error in both directions: an extra/misspelled entry fails this array's own
@@ -1835,4 +4147,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
1835
4147
  */
1836
4148
  declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
1837
4149
 
1838
- export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type ApprovalWhere, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type DetailThreadRef, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceRunDetail, type GovernanceThreadDetail, type GovernanceThreadDetailQuery, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunToolCallRow, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, THREAD_DETAIL_CONTENT_CHARS, type ThreadActivityRow, type ThreadDetail, type ThreadMessageRow, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadUsageRollup, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolDisabledError, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, canActorUseTool, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByCanUse, filterToolsByEnabled, filterToolsByRole, invokeWithTransientRetry, isToolEnabled, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, rollupThreadUsage, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, truncateDetailContent, withToolTimeout };
4150
+ export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MEMORY, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SKILLS, AGENT_SKILL_SOURCES, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, ASK_TOOL_DESCRIPTION, ASK_TOOL_NAME, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDelegation, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentHistoryWindow, type AgentIntake, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentLoopResult, type AgentMemoryResolved, type AgentMemoryWritten, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSkillsResolved, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentStructuredOutputSpan, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type ApprovalWhere, type AskToolInput, type AttachmentRef, type AttachmentStagingStore, type BufferedModelTurnResult, type BuildMemoryBlockInput, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_HISTORY_SUMMARY_INSTRUCTION, DEFAULT_INCREMENTAL_LOOKBACK_CHARS, DEFAULT_INTAKE_PREAMBLE, DEFAULT_MAX_FACT_CHARS, DEFAULT_MAX_MEMORIES, DEFAULT_MAX_SKILLS, DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type DetachedDelegationOutcome, type DetachedDelegationReceipt, type DetachedDelivery, type DetailThreadRef, type ElicitationOption, type ElicitationOutcome, type ElicitationQuestion, type ElicitationReply, type ElicitationRequest, type ElicitationResult, type EmbeddingProvider, type ForgetMemoryInput, type FrameBuffer, GLOBAL_SCOPE, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceRunDetail, type GovernanceThreadDetail, type GovernanceThreadDetailQuery, type GovernanceUsageInput, type HistoryPolicy, type HistoryPolicyContext, type HistorySelection, type HistorySummary, type HumanReply, type IncrementalGate, type IncrementalGating, type InputProcessor, type InvokeWithTransientRetryOptions, type ListMemoriesInput, type ListSkillsInput, type ListStagedAttachmentsInput, type LlmStepEnvelope, type LoadSkillInput, MAX_ASK_QUESTIONS, type MemoryAuthor, type MemoryConfig, type MemoryDigest, type MemoryDigestEntry, type MemoryFact, type MemoryForgetRequest, type MemoryOrigin, type MemoryProvider, type MemoryRecord, type MemoryVerdict, type MemoryWriteOutcome, type MemoryWriteRequest, type MessageAttachment, type MessageRole, type MessageUsage, type ModelAnswer, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type OfferMemoriesInput, type OutputGateMode, type OutputGateResult, type OutputProcessor, OutputRejectedError, type OutputVerdict, type OverriddenMemory, type PageContext, type Passage, type PendingApprovalRow, type ProcessedPrompt, type ProcessorContext, ProcessorFailedError, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, REMEMBER_TOOL_DESCRIPTION, REMEMBER_TOOL_NAME, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RememberToolInput, type RerankOptions, type Reranker, type ResolveAttachmentInput, type ResolveMemoryDigestInput, type ResolvedDelegation, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, RunCancelledError, type RunErrorBreakdownRow, type RunMetrics, type RunToolCallRow, type RunTrendPoint, type RunWhere, SKILL_TOOL_DESCRIPTION, SKILL_TOOL_NAME, type ScopeContext, type ScopeResolver, type SearchMemoriesInput, type SettledTask, type SinkWriter, type Skill, type SkillAuthor, type SkillCatalogEntry, type SkillContext, type SkillLoadOutcome, type SkillOffer, type SkillProvider, type SkillSummary, type SkillToolInput, type SkillWriteRequest, type SkillWriteVerdict, type SkillsConfig, type StageAttachmentInput, type StagedAttachment, type StoreMemoryInput, type StoredMessage, type StreamError, type StructuredOutcome, StructuredOutputError, THREAD_DETAIL_CONTENT_CHARS, type ThreadActivityRow, type ThreadDetail, type ThreadMessageRow, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadTurnPage, type ThreadTurnQuery, type ThreadTurnReader, type ThreadUsageRollup, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolDisabledError, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, type WindowHistoryOptions, type WithMemoryToolInput, type WriteMemoryInput, actorScope, agentDiagnosticKey, agentFailureCode, askInputSchema, askToolDefinition, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, buildMemoryBlock, buildSkillsBlock, canActorUseTool, compositeSkillProvider, createFrameBuffer, createIncrementalGate, dayBoundsUtc, decodeStreamEvent, defaultScopeResolver, detachedDelivered, detachedStarted, detachedUnsettled, encodeStreamEvent, estimateCost, estimateMessageTokens, extractJson, filterToolsByAllowList, filterToolsByCanUse, filterToolsByEnabled, filterToolsByRole, gateFollowUps, gateTail, invokeWithTransientRetry, isControlFlowSignal, isReplayIntegrityError, isToolEnabled, isTransientToolError, loadSkill, memoryForgetVerdict, memoryWriteVerdict, normalizeDelegation, normalizeElicitationReply, offerMemories, offerSkills, publishAgentDelegated, publishAgentMemoryResolved, publishAgentMemoryWritten, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentSkillsResolved, publishAgentToolCall, publishAgentToolRetry, releaseGatedFrames, rememberInputSchema, rememberToolDefinition, renderElicitationAnswers, repairInstruction, resolveElicitation, resolveGateLookback, resolveMemoryDigest, resolveOutputGateMode, resolveSkillCatalog, resolveToolTransientRetryNumbers, rollupThreadUsage, runAgentLoop, runInputProcessors, runOutputProcessors, seedModelPrices, settleAll, settleElicitation, settleUnsettledDelegation, skillInputSchema, skillToolDefinition, skillWriteVerdict, staticSkillProvider, summarizeWithModel, tenantScope, traceLlmTurn, traceToolExecution, truncateDetailContent, validateStructured, windowHistory, withAskTool, withMemoryTool, withSkillTool, withToolTimeout, writeMemory };