@dudousxd/nestjs-agent-core 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -2
- package/dist/index.cjs +3190 -637
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +2350 -38
- package/dist/index.d.ts +2350 -38
- package/dist/index.js +3105 -631
- package/dist/index.js.map +1 -1
- package/package.json +3 -2
package/dist/index.d.cts
CHANGED
|
@@ -1,6 +1,272 @@
|
|
|
1
1
|
import { StandardSchemaV1 } from '@standard-schema/spec';
|
|
2
2
|
import { ChannelRegistry } from '@dudousxd/nestjs-diagnostics';
|
|
3
3
|
|
|
4
|
+
/**
|
|
5
|
+
* Asking the USER a structured question, and waiting for the answer.
|
|
6
|
+
*
|
|
7
|
+
* `awaitApproval` collects a yes/no about work already proposed; this collects the scope BEFORE the
|
|
8
|
+
* work. Two surfaces produce it — a configured intake (`AgentLoopDeps.intake`) and the model-callable
|
|
9
|
+
* `ask` tool (`AgentLoopDeps.ask`) — and they deliberately produce the SAME {@link
|
|
10
|
+
* ElicitationRequest}, persist through the same tool-call row, and resume through the same
|
|
11
|
+
* `tool:<runId>:<callId>` signal. A consumer cannot tell which one asked, and should not have to.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** One choice a question offers. */
|
|
15
|
+
interface ElicitationOption {
|
|
16
|
+
/** Stable identifier submitted back. Never shown to the user. */
|
|
17
|
+
value: string;
|
|
18
|
+
/** What the user reads. */
|
|
19
|
+
label: string;
|
|
20
|
+
/**
|
|
21
|
+
* A single character a UI may bind as a keyboard shortcut for this option. Advisory — nothing in
|
|
22
|
+
* the library reads it, and a client is free to render its own.
|
|
23
|
+
*/
|
|
24
|
+
hotkey?: string;
|
|
25
|
+
}
|
|
26
|
+
/** One question in a set. */
|
|
27
|
+
interface ElicitationQuestion {
|
|
28
|
+
/** Unique within its request; the key answers come back under. */
|
|
29
|
+
id: string;
|
|
30
|
+
prompt: string;
|
|
31
|
+
options: ElicitationOption[];
|
|
32
|
+
/** More than one option may be chosen. Omit → single choice. */
|
|
33
|
+
multiple?: boolean;
|
|
34
|
+
/**
|
|
35
|
+
* The options already picked for the user. The claim this whole surface makes is that confirming
|
|
36
|
+
* is enough, so a question with no defaults is a question the user must stop and think about —
|
|
37
|
+
* which is the case the design is trying to avoid. Empty/omitted is allowed and means exactly
|
|
38
|
+
* that: submitting without answering leaves this question unanswered.
|
|
39
|
+
*/
|
|
40
|
+
defaults?: string[];
|
|
41
|
+
/** Accept values that are not among `options` (a typed-in answer). Omit → options only. */
|
|
42
|
+
allowFreeText?: boolean;
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* A question set awaiting a human. Identical in shape whether an `@Agent`'s configured intake or
|
|
46
|
+
* the model's `ask` tool authored it — `source` records which, for audit, not for control flow.
|
|
47
|
+
*
|
|
48
|
+
* `questions.length` is known when the request is written, which is what lets a client render
|
|
49
|
+
* "Question 1 of 3" without guessing whether a fourth is coming.
|
|
50
|
+
*/
|
|
51
|
+
interface ElicitationRequest {
|
|
52
|
+
/** The tool-call id this request is persisted under, and the signal it is answered through. */
|
|
53
|
+
id: string;
|
|
54
|
+
source: 'intake' | 'ask';
|
|
55
|
+
/** What the assistant says above the form. */
|
|
56
|
+
preamble?: string;
|
|
57
|
+
questions: ElicitationQuestion[];
|
|
58
|
+
}
|
|
59
|
+
/** What a human sent back for an {@link ElicitationRequest}. */
|
|
60
|
+
interface ElicitationReply {
|
|
61
|
+
/**
|
|
62
|
+
* questionId → chosen values. A question whose id is ABSENT takes the request's own `defaults` —
|
|
63
|
+
* that is what makes "just submit" mean "yes, your pre-picked answers". A present-but-empty array
|
|
64
|
+
* is an explicit "none of these" and does NOT fall back.
|
|
65
|
+
*/
|
|
66
|
+
answers: Record<string, string[]>;
|
|
67
|
+
/**
|
|
68
|
+
* The user declined to answer and told the agent to proceed on its own assumptions. Distinct from
|
|
69
|
+
* confirming the defaults even though the resulting values are the same: one is a decision the
|
|
70
|
+
* user made, the other is one they refused to make, and only the first is evidence of intent.
|
|
71
|
+
*/
|
|
72
|
+
skipped?: boolean;
|
|
73
|
+
/** Opaque ref of WHO answered, when it wasn't the run's own actor. */
|
|
74
|
+
answeredByRef?: string;
|
|
75
|
+
}
|
|
76
|
+
/** A settled elicitation: what the agent proceeds on, and how it got there. */
|
|
77
|
+
interface ElicitationOutcome {
|
|
78
|
+
/** One entry per question, in request order — always present, so a caller never re-applies defaults. */
|
|
79
|
+
answers: Record<string, string[]>;
|
|
80
|
+
skipped: boolean;
|
|
81
|
+
/** Question ids filled from the request's `defaults` rather than by the human. */
|
|
82
|
+
defaulted: string[];
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Read whatever the human channel delivered as an {@link ElicitationReply}.
|
|
86
|
+
*
|
|
87
|
+
* A question set is persisted as an `action` tool call in `pending_approval` — that is what puts it
|
|
88
|
+
* in the approvals inbox a deployment already has, instead of needing one of its own. The cost of
|
|
89
|
+
* that choice is that the thing which comes back may be a {@link Decision} someone pressed
|
|
90
|
+
* Approve/Reject on rather than a set of answers, and a `Decision` carries no `answers` at all.
|
|
91
|
+
*
|
|
92
|
+
* Approve means every question keeps its own pre-picked `defaults`, which is exactly what "just
|
|
93
|
+
* submit" already means on this surface; Reject is the same declining-to-answer a skip is. Neither
|
|
94
|
+
* reading is a guess — a yes/no channel cannot say more than that, and saying it here is what lets
|
|
95
|
+
* one inbox settle both kinds of pending work.
|
|
96
|
+
*
|
|
97
|
+
* Returns the reply UNCHANGED when it already carries answers, so the common path allocates nothing
|
|
98
|
+
* and a caller can identity-compare.
|
|
99
|
+
*/
|
|
100
|
+
declare function normalizeElicitationReply(reply: ElicitationReply | Decision): ElicitationReply;
|
|
101
|
+
/**
|
|
102
|
+
* Settle a reply against the request it answers: fill every unanswered question from its own
|
|
103
|
+
* `defaults`, drop submitted values that aren't on offer, and collapse a single-choice question to
|
|
104
|
+
* one value.
|
|
105
|
+
*
|
|
106
|
+
* PURE, and deliberately so. Both of its inputs are already journaled by the time the loop calls it
|
|
107
|
+
* — the request came from module config or from an `llm:<i>` checkpoint, the reply from the signal
|
|
108
|
+
* checkpoint — so every process replaying the turn reaches the same values without a checkpoint of
|
|
109
|
+
* its own. Resolving defaults in the HTTP layer instead would put them behind a store read that a
|
|
110
|
+
* replay would have to repeat.
|
|
111
|
+
*/
|
|
112
|
+
declare function resolveElicitation(request: ElicitationRequest, raw: ElicitationReply | Decision): ElicitationOutcome;
|
|
113
|
+
/**
|
|
114
|
+
* What a settled elicitation looks like to everyone downstream: the model reading it back as a tool
|
|
115
|
+
* result, the thread reader rendering it, the auditor asking what the agent was told to do. One
|
|
116
|
+
* shape for both surfaces — nothing here records which of them asked.
|
|
117
|
+
*/
|
|
118
|
+
interface ElicitationResult extends ElicitationOutcome {
|
|
119
|
+
/** The questions against the chosen LABELS, so a reader (and a model) can act on it. */
|
|
120
|
+
summary: string;
|
|
121
|
+
}
|
|
122
|
+
/** {@link resolveElicitation} plus its human-readable rendering. Pure, for the same reason. */
|
|
123
|
+
declare function settleElicitation(request: ElicitationRequest, reply: ElicitationReply | Decision): ElicitationResult;
|
|
124
|
+
/**
|
|
125
|
+
* The answers as the model reads them: the question's own prompt against the chosen options' LABELS,
|
|
126
|
+
* not their opaque `value`s — a model shown `{"scope":["b"]}` has been told nothing.
|
|
127
|
+
*/
|
|
128
|
+
declare function renderElicitationAnswers(request: ElicitationRequest, outcome: ElicitationOutcome): string;
|
|
129
|
+
/** The reserved tool name the model calls to ask the user something. */
|
|
130
|
+
declare const ASK_TOOL_NAME = "ask";
|
|
131
|
+
/** What the model must supply when it calls `ask`. */
|
|
132
|
+
interface AskToolInput {
|
|
133
|
+
preamble?: string;
|
|
134
|
+
questions: ElicitationQuestion[];
|
|
135
|
+
}
|
|
136
|
+
/** How many questions one `ask` may carry. A form the user has to scroll is a form they skip. */
|
|
137
|
+
declare const MAX_ASK_QUESTIONS = 5;
|
|
138
|
+
/**
|
|
139
|
+
* The `ask` tool's input schema, hand-written rather than borrowed from a validation library: core
|
|
140
|
+
* depends on no validator, and the schema has to carry a JSON Schema a provider can constrain
|
|
141
|
+
* generation against. It publishes one through the Standard JSON Schema extension
|
|
142
|
+
* (`~standard.jsonSchema.input`), which is the path the AI SDK adapter already recognises for
|
|
143
|
+
* Valibot / ArkType / Zod 4.
|
|
144
|
+
*/
|
|
145
|
+
declare const askInputSchema: StandardSchemaV1<unknown, AskToolInput>;
|
|
146
|
+
/**
|
|
147
|
+
* What the model is told the `ask` tool is for. Written to discourage the two failure modes that
|
|
148
|
+
* make a clarifying question worse than a guess: asking about something the conversation already
|
|
149
|
+
* settled, and asking without saying what you would have done.
|
|
150
|
+
*/
|
|
151
|
+
declare const ASK_TOOL_DESCRIPTION = "Ask the user to settle the scope of the work before you do it. Use it when a reasonable person would produce a materially different result depending on the answer \u2014 not to confirm something the conversation already says. Every question must pre-pick the answer you would choose, so the user can confirm instead of deciding. The user may decline, in which case you proceed on those pre-picked answers.";
|
|
152
|
+
/**
|
|
153
|
+
* The `ask` tool as the model sees it. NOT a `ToolSpec` and never registered: `ask` has no handler,
|
|
154
|
+
* because the loop settles it against a human instead of invoking anything. Keeping it out of the
|
|
155
|
+
* `ToolRegistry` is also what keeps the kind decision off a process-local lookup — see
|
|
156
|
+
* `claimToolCall`.
|
|
157
|
+
*/
|
|
158
|
+
declare function askToolDefinition(): ToolDefinition;
|
|
159
|
+
/** A question set an `@Agent` asks before it starts working. See `AgentLoopDeps.intake`. */
|
|
160
|
+
interface AgentIntake {
|
|
161
|
+
questions: ElicitationQuestion[];
|
|
162
|
+
/** What the assistant says above the form. Omit → {@link DEFAULT_INTAKE_PREAMBLE}. */
|
|
163
|
+
preamble?: string;
|
|
164
|
+
/**
|
|
165
|
+
* `'thread-start'` (default) asks once, on the first turn of a thread; `'every-turn'` asks before
|
|
166
|
+
* every turn. Both are decided from what `load:thread` recorded about the thread when the turn
|
|
167
|
+
* began, never from anything this process happens to know — by the time a replay reaches the
|
|
168
|
+
* question, the thread already holds the assistant message the first attempt wrote.
|
|
169
|
+
*/
|
|
170
|
+
when?: 'thread-start' | 'every-turn';
|
|
171
|
+
}
|
|
172
|
+
declare const DEFAULT_INTAKE_PREAMBLE = "A few questions before I start. I have pre-picked what I would choose, so confirming is enough.";
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* The ceiling on how much of a thread rides into a turn. Without one, `runAgentLoop` maps EVERY
|
|
176
|
+
* message the store returns into the model's messages, so a long-lived thread grows until the
|
|
177
|
+
* provider rejects the request — and every turn before that one pays for the whole transcript.
|
|
178
|
+
* `@dudousxd/nestjs-agent-core` ships `windowHistory` as the built-in; anything satisfying this SPI
|
|
179
|
+
* works. Wire it as `AgentLoopDeps.historyPolicy`, or via `AgentModule.forRoot({ history })` /
|
|
180
|
+
* `@Agent({ history })`.
|
|
181
|
+
*/
|
|
182
|
+
|
|
183
|
+
/** Whose history this is — enough for a policy to size the window per agent or per actor. */
|
|
184
|
+
interface HistoryPolicyContext {
|
|
185
|
+
threadId: string;
|
|
186
|
+
actor: Actor;
|
|
187
|
+
/** The agent running this turn. Undefined → the default agent. */
|
|
188
|
+
agentName?: string;
|
|
189
|
+
}
|
|
190
|
+
/** How a policy split the thread: what rides into the turn, and what the ceiling left out. */
|
|
191
|
+
interface HistorySelection {
|
|
192
|
+
/** Sent to the model, oldest-first. */
|
|
193
|
+
keep: ModelMessage[];
|
|
194
|
+
/** Left out, oldest-first. Folded into a leading summary when the policy implements `summarize`. */
|
|
195
|
+
drop: ModelMessage[];
|
|
196
|
+
}
|
|
197
|
+
/** A stand-in for the messages a window left out, plus what producing it cost. */
|
|
198
|
+
interface HistorySummary {
|
|
199
|
+
/** Prose the loop folds into the window as a leading `system` message. */
|
|
200
|
+
text: string;
|
|
201
|
+
/**
|
|
202
|
+
* What the summarizer spent, when it called a model. Recorded as a `history_summary` usage row, so
|
|
203
|
+
* a ceiling on context cost cannot itself become spend nothing accounts for. Omit for a summarizer
|
|
204
|
+
* that calls no model (a rollup of tool names, a digest the app already stored).
|
|
205
|
+
*/
|
|
206
|
+
usage?: MessageUsage;
|
|
207
|
+
/** Accounting label for the model that produced it; falls back to `AgentLoopDeps.modelId`. */
|
|
208
|
+
modelId?: string;
|
|
209
|
+
}
|
|
210
|
+
interface HistoryPolicy {
|
|
211
|
+
/**
|
|
212
|
+
* The most messages {@link select} can ever keep, where the ceiling can be stated as a row count.
|
|
213
|
+
*
|
|
214
|
+
* A hint the loop hands to the store, which then reads that many of the thread's newest rows
|
|
215
|
+
* rather than its whole transcript (see `ThreadTurnReader`). Declaring it is a PROMISE about
|
|
216
|
+
* `select`: that it keeps at most this many messages, and that they are the NEWEST ones — so a
|
|
217
|
+
* window of this size is indistinguishable, to `select`, from the full transcript. A policy that
|
|
218
|
+
* can keep more than this, or can keep something older than the newest `maxMessages`, must omit it
|
|
219
|
+
* rather than select over rows the store was never asked for.
|
|
220
|
+
*
|
|
221
|
+
* Omit where the ceiling is not a row count at all — a token budget alone cannot name one, since
|
|
222
|
+
* one message can be four tokens or forty thousand. Omitting costs only the bound on the READ; the
|
|
223
|
+
* prompt is identical either way.
|
|
224
|
+
*/
|
|
225
|
+
readonly maxMessages?: number;
|
|
226
|
+
/**
|
|
227
|
+
* Decide what the model sees.
|
|
228
|
+
*
|
|
229
|
+
* MUST be a pure function of `messages`. The loop calls it INSIDE the `load:thread` checkpoint and
|
|
230
|
+
* records its result there, so the ceiling bounds the journal as well as the prompt: what the
|
|
231
|
+
* checkpoint holds is the selection rather than the store's whole `ThreadDetail`, on a payload
|
|
232
|
+
* every replay re-reads. It still adds no position of its own — a policy cannot change the name or
|
|
233
|
+
* position of a single existing checkpoint.
|
|
234
|
+
*
|
|
235
|
+
* Read a clock, a feature flag or a database here and the resumed run windows differently from the
|
|
236
|
+
* one that suspended: the model gets a different prompt, and any step whose existence depends on
|
|
237
|
+
* the split lands at a position the history has no room for. Anything non-deterministic belongs in
|
|
238
|
+
* {@link summarize}, which has a checkpoint of its own.
|
|
239
|
+
*
|
|
240
|
+
* The newest message must always be in `keep` — dropping it leaves the turn with nothing to answer.
|
|
241
|
+
*/
|
|
242
|
+
select(messages: ModelMessage[], ctx: HistoryPolicyContext): HistorySelection;
|
|
243
|
+
/**
|
|
244
|
+
* Fold the dropped messages into prose the model reads in their place, prepended to the window as
|
|
245
|
+
* a `system` message. Optional — without it, dropped messages are simply gone, and `load:thread`
|
|
246
|
+
* does not record them either: a summarizer is the only thing that ever reads them back.
|
|
247
|
+
*
|
|
248
|
+
* Runs inside the loop's `history:summarize` checkpoint, so it may call a model or hit the
|
|
249
|
+
* network: the first attempt's result is journaled and every replay reads it back instead of
|
|
250
|
+
* re-summarizing. It runs once per RUN (not per model step), and only when `select` actually
|
|
251
|
+
* dropped something.
|
|
252
|
+
*/
|
|
253
|
+
summarize?(dropped: ModelMessage[], ctx: HistoryPolicyContext): Promise<HistorySummary>;
|
|
254
|
+
}
|
|
255
|
+
/**
|
|
256
|
+
* The window an agent asks for declaratively — `AgentModule.forRoot({ history })` and
|
|
257
|
+
* `@Agent({ history })`. Plain data, so it can live in a decorator's metadata; the NestJS layer
|
|
258
|
+
* turns it into a `windowHistory` policy. A consumer needing anything the window can't express
|
|
259
|
+
* supplies a {@link HistoryPolicy} instead.
|
|
260
|
+
*/
|
|
261
|
+
interface AgentHistoryWindow {
|
|
262
|
+
/** Keep at most this many of the newest messages. */
|
|
263
|
+
maxMessages?: number;
|
|
264
|
+
/** Keep the newest messages whose estimated tokens fit this budget. */
|
|
265
|
+
maxTokens?: number;
|
|
266
|
+
/** Fold what the window left out into a leading summary — one extra model call per run. */
|
|
267
|
+
summarize?: boolean;
|
|
268
|
+
}
|
|
269
|
+
|
|
4
270
|
/**
|
|
5
271
|
* Transient tool-error classification + the retry loop that wraps a tool's own invocation. A
|
|
6
272
|
* classified-transient error (a DB deadlock, a lock-wait timeout, a serialization failure) means
|
|
@@ -47,9 +313,10 @@ interface ToolTransientRetryNumbers {
|
|
|
47
313
|
declare function resolveToolTransientRetryNumbers(setting: ToolTransientRetrySetting | undefined): ToolTransientRetryNumbers | false;
|
|
48
314
|
interface InvokeWithTransientRetryOptions {
|
|
49
315
|
/**
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
*
|
|
316
|
+
* Widens what counts as a control-flow signal (durable suspend / continue-as-new) so a retry never
|
|
317
|
+
* swallows one — same rule the loop's tool catch applies. An ADDITION to the built-in marker check
|
|
318
|
+
* ({@link isControlFlowSignal}), which already covers every signal the durable runtimes raise;
|
|
319
|
+
* supply this only for a runner whose signals carry no marker.
|
|
53
320
|
*/
|
|
54
321
|
isControlFlowError?: (error: unknown) => boolean;
|
|
55
322
|
/**
|
|
@@ -76,13 +343,44 @@ interface Actor {
|
|
|
76
343
|
roles?: string[];
|
|
77
344
|
tenantRef?: string;
|
|
78
345
|
}
|
|
79
|
-
type ToolKind = 'read' | 'action' | 'agent';
|
|
346
|
+
type ToolKind = 'read' | 'action' | 'agent' | 'ask' | 'skill' | 'memory';
|
|
347
|
+
/**
|
|
348
|
+
* One agent->agent edge on {@link AgentDefinition.delegatesTo}. A bare name is the awaited
|
|
349
|
+
* delegation that has always existed; the object form is how an author says this one runs in the
|
|
350
|
+
* background.
|
|
351
|
+
*/
|
|
352
|
+
type AgentDelegation = string | {
|
|
353
|
+
agent: string;
|
|
354
|
+
detached?: boolean;
|
|
355
|
+
};
|
|
356
|
+
/**
|
|
357
|
+
* Where a detached sub-agent run posts its answer: the thread that delegated it, and the
|
|
358
|
+
* `agent`-kind tool call that started it. Carried on the child run's own {@link AgentRunInput},
|
|
359
|
+
* because by the time the child finishes the parent turn is over and nothing is holding the
|
|
360
|
+
* address.
|
|
361
|
+
*/
|
|
362
|
+
interface DetachedDelivery {
|
|
363
|
+
threadId: string;
|
|
364
|
+
toolCallId: string;
|
|
365
|
+
}
|
|
80
366
|
/**
|
|
81
367
|
* Declared shape of a tool.
|
|
82
368
|
* - `read` auto-executes.
|
|
83
369
|
* - `action` never auto-executes — requires HITL approval.
|
|
84
370
|
* - `agent` delegates to another named agent (durable: a child workflow; inline: a nested loop),
|
|
85
371
|
* handled at the loop level — NOT via a handler. Carries `targetAgent`.
|
|
372
|
+
* - `ask` puts a question set to the user and waits for the answers (see `elicitation.ts`).
|
|
373
|
+
* Handled at the loop level and never registered, so no `ToolSpec` carries this kind:
|
|
374
|
+
* the only tool that has it is the built-in `ask`, whose definition the loop supplies.
|
|
375
|
+
* - `skill` reads one of the procedures offered in the turn's `<skills>` catalog (see `skills.ts`).
|
|
376
|
+
* Auto-executes like a read and performs nothing, but is served by the LOOP from the
|
|
377
|
+
* catalog the journal holds rather than by a registered handler — so, like `ask`, no
|
|
378
|
+
* `ToolSpec` carries this kind.
|
|
379
|
+
* - `memory` records one fact about the actor for later turns (see `memory.ts`). Served by the LOOP
|
|
380
|
+
* and never registered, like `skill`. It WRITES, but auto-executes rather than asking
|
|
381
|
+
* for approval: an agent may only ever write the scope of the actor it is running for,
|
|
382
|
+
* so the blast radius of a bad one is the prompt of the person who was talking, and the
|
|
383
|
+
* remedy is the read-back that lets them delete it.
|
|
86
384
|
*/
|
|
87
385
|
interface ToolSpec {
|
|
88
386
|
name: string;
|
|
@@ -96,6 +394,20 @@ interface ToolSpec {
|
|
|
96
394
|
inputSchema: StandardSchemaV1;
|
|
97
395
|
/** For `kind: 'agent'` — the name of the agent to delegate to. */
|
|
98
396
|
targetAgent?: string;
|
|
397
|
+
/**
|
|
398
|
+
* For `kind: 'agent'` — start the delegation and let the calling turn END, instead of holding it
|
|
399
|
+
* open until the delegate answers. The call's result is a {@link DetachedDelegationReceipt}, and
|
|
400
|
+
* the answer arrives later as its own message in the same thread (see
|
|
401
|
+
* {@link AgentRunInput.deliverTo}).
|
|
402
|
+
*
|
|
403
|
+
* Authored per EDGE, never chosen by the model: a model that can decide to detach can decide to
|
|
404
|
+
* detach the one thing the user is sitting there waiting for, and it has no way to know which that
|
|
405
|
+
* is. The person wiring `A -> B` does.
|
|
406
|
+
*
|
|
407
|
+
* Settled into the call's `persist:toolcall` checkpoint alongside `targetAgent`, so every replay
|
|
408
|
+
* reads the branch back rather than re-deciding it against a registry that may have changed.
|
|
409
|
+
*/
|
|
410
|
+
detached?: boolean;
|
|
99
411
|
/** Roles allowed to invoke. Undefined → defaults applied by RolesPolicy (e.g. ADMIN-only). */
|
|
100
412
|
roles?: string[];
|
|
101
413
|
/**
|
|
@@ -176,7 +488,14 @@ interface MessageUsage {
|
|
|
176
488
|
*/
|
|
177
489
|
costUsd?: number | null;
|
|
178
490
|
}
|
|
179
|
-
|
|
491
|
+
/**
|
|
492
|
+
* What a usage row was spent ON. `chat` is a model step of the turn itself; `follow_ups` is the
|
|
493
|
+
* extra call that proposes follow-up questions; `history_summary` is the extra call a
|
|
494
|
+
* {@link import('./spi/history-policy.js').HistoryPolicy} makes to fold windowed-out messages into a
|
|
495
|
+
* summary — so bounding context cost never becomes spend nothing accounts for; `structured_output`
|
|
496
|
+
* is the formatting pass that restates a finished answer as `AgentLoopDeps.outputSchema` requires.
|
|
497
|
+
*/
|
|
498
|
+
type UsagePurpose = 'chat' | 'follow_ups' | 'history_summary' | 'structured_output';
|
|
180
499
|
interface QuotaState {
|
|
181
500
|
usedTokens: number;
|
|
182
501
|
limitTokens: number;
|
|
@@ -194,6 +513,12 @@ interface QuotaView {
|
|
|
194
513
|
withinLimit: boolean;
|
|
195
514
|
costUsd: number;
|
|
196
515
|
}
|
|
516
|
+
/**
|
|
517
|
+
* Anything a human sends back into a parked run: a {@link Decision} on an action tool, or an
|
|
518
|
+
* `ElicitationReply` answering a question set. Both travel the same `tool:<runId>:<toolCallId>`
|
|
519
|
+
* signal, so the runner that delivers them does not need to know which it is carrying.
|
|
520
|
+
*/
|
|
521
|
+
type HumanReply = Decision | ElicitationReply;
|
|
197
522
|
/** A human decision on a pending action tool call. */
|
|
198
523
|
interface Decision {
|
|
199
524
|
approved: boolean;
|
|
@@ -273,9 +598,19 @@ interface AgentRunInput {
|
|
|
273
598
|
agentName?: string;
|
|
274
599
|
/**
|
|
275
600
|
* How many agent→agent delegations deep this run already is (0 for a top-level turn). The runner
|
|
276
|
-
* increments it for each child run; the loop refuses to delegate past
|
|
601
|
+
* increments it for each child run; the loop refuses to delegate past its depth ceiling.
|
|
277
602
|
*/
|
|
278
603
|
delegationDepth?: number;
|
|
604
|
+
/**
|
|
605
|
+
* The named agents already on this delegation chain, root first — what {@link delegationDepth}
|
|
606
|
+
* counts, spelled out. The runner appends its own agent's name for each child it starts.
|
|
607
|
+
*
|
|
608
|
+
* A count can only say a chain is LONG. This says whether it is going in circles, and how often:
|
|
609
|
+
* an agent that appears here is one the chain has already passed through, so a delegation back to
|
|
610
|
+
* it is a cycle by inspection rather than by proxy. A run whose runner does not supply it falls
|
|
611
|
+
* back to the depth ceiling alone.
|
|
612
|
+
*/
|
|
613
|
+
delegationPath?: readonly string[];
|
|
279
614
|
/**
|
|
280
615
|
* When set, this run streams into ANOTHER run's sink instead of its own. A sub-agent run carries
|
|
281
616
|
* its top-level ancestor's runId here so its tokens (and its pending action-tool frames) land in
|
|
@@ -283,6 +618,18 @@ interface AgentRunInput {
|
|
|
283
618
|
* approve, a sub-agent's HITL action. Propagated unchanged down the delegation chain.
|
|
284
619
|
*/
|
|
285
620
|
sinkRunId?: string;
|
|
621
|
+
/**
|
|
622
|
+
* Set on a DETACHED sub-agent run: the thread and tool call this run answers into when it
|
|
623
|
+
* finishes. Its presence is also what makes a run detached from the inside — it has no ancestor
|
|
624
|
+
* sink to stream into, so nothing else distinguishes it from a top-level turn.
|
|
625
|
+
*/
|
|
626
|
+
deliverTo?: DetachedDelivery;
|
|
627
|
+
/**
|
|
628
|
+
* The run that started this one (a delegation's parent). Recorded with the run so a governance
|
|
629
|
+
* surface can roll a delegation's cost up to the turn that asked for it; a detached child is
|
|
630
|
+
* otherwise a row with nothing pointing at it.
|
|
631
|
+
*/
|
|
632
|
+
parentRunId?: string;
|
|
286
633
|
/**
|
|
287
634
|
* Re-run the last exchange instead of adding a new message: the loop truncates everything after
|
|
288
635
|
* the thread's last user message and re-answers it (no `userText` is appended). Used by a
|
|
@@ -305,10 +652,51 @@ interface AgentDefinition {
|
|
|
305
652
|
systemPrompt?: string | PromptBuilder;
|
|
306
653
|
/** Allow-list of tool names this agent may use (subset of all registered tools). */
|
|
307
654
|
tools?: string[];
|
|
308
|
-
/**
|
|
309
|
-
|
|
655
|
+
/**
|
|
656
|
+
* Other agents this agent may hand off to (auto-registered as `agent`-kind tools). A bare name
|
|
657
|
+
* is the awaited form; `{ agent, detached: true }` starts the delegate and lets this agent's turn
|
|
658
|
+
* finish without its answer — see {@link ToolSpec.detached}.
|
|
659
|
+
*/
|
|
660
|
+
delegatesTo?: AgentDelegation[];
|
|
310
661
|
modelId?: string;
|
|
311
662
|
maxSteps?: number;
|
|
663
|
+
/**
|
|
664
|
+
* How deep delegation may nest below this agent. Undefined → {@link MAX_DELEGATION_DEPTH}.
|
|
665
|
+
*
|
|
666
|
+
* Bounds the CHAIN, not the fan-out: how many agents a turn delegates to is the model's.
|
|
667
|
+
*/
|
|
668
|
+
maxDelegationDepth?: number;
|
|
669
|
+
/**
|
|
670
|
+
* How many times one agent may appear on a single delegation chain.
|
|
671
|
+
* Undefined → {@link DEFAULT_MAX_AGENT_APPEARANCES}.
|
|
672
|
+
*/
|
|
673
|
+
maxAgentAppearances?: number;
|
|
674
|
+
/**
|
|
675
|
+
* This agent's own ceiling on how much of a thread rides into its turn, overriding the
|
|
676
|
+
* module-wide one. A persona that reasons over a long back-and-forth and one that answers a single
|
|
677
|
+
* question from a page context want very different windows.
|
|
678
|
+
*/
|
|
679
|
+
history?: AgentHistoryWindow;
|
|
680
|
+
/**
|
|
681
|
+
* Constrain this agent's final answer to a schema. A live schema INSTANCE, so it is resolved from
|
|
682
|
+
* DI on whichever process runs the turn and never travels on `AgentRunInput` — a Standard Schema
|
|
683
|
+
* cannot survive the JSON hop into a durable workflow, which is why there is no per-request
|
|
684
|
+
* override on the HTTP surface.
|
|
685
|
+
*/
|
|
686
|
+
outputSchema?: StandardSchemaV1;
|
|
687
|
+
/**
|
|
688
|
+
* Extra model calls allowed to fix an answer that failed {@link outputSchema}. Undefined → 1.
|
|
689
|
+
*/
|
|
690
|
+
outputRepairAttempts?: number;
|
|
691
|
+
/**
|
|
692
|
+
* Questions this agent puts to the user BEFORE it starts working. Authored, so the turn pays no
|
|
693
|
+
* model call to produce them and a client knows the total up front. Undefined → no intake.
|
|
694
|
+
*/
|
|
695
|
+
intake?: AgentIntake;
|
|
696
|
+
/**
|
|
697
|
+
* Whether this agent is offered the built-in `ask` tool. Undefined → the module-wide setting.
|
|
698
|
+
*/
|
|
699
|
+
ask?: boolean;
|
|
312
700
|
}
|
|
313
701
|
/**
|
|
314
702
|
* The read-model the `GET agents` endpoint returns to a client — the safe public subset of an
|
|
@@ -352,6 +740,8 @@ interface StoredMessage {
|
|
|
352
740
|
attachments?: MessageAttachment[];
|
|
353
741
|
followUps?: string[];
|
|
354
742
|
usage?: MessageUsage;
|
|
743
|
+
/** The run (turn) that produced this message; absent on a row written before this was recorded. */
|
|
744
|
+
runId?: string;
|
|
355
745
|
createdAt: string;
|
|
356
746
|
}
|
|
357
747
|
interface ThreadDetail extends ThreadSummary {
|
|
@@ -370,6 +760,13 @@ interface LlmStepEnvelope {
|
|
|
370
760
|
messages: ModelMessage[];
|
|
371
761
|
/** The turn's actor — the handler re-derives tool definitions from it (definitionsFor). */
|
|
372
762
|
actor: Actor;
|
|
763
|
+
/**
|
|
764
|
+
* Hold this call's stream frames rather than writing them to the run's sink, and return them on
|
|
765
|
+
* the result. Set by the loop when an output processor has to see the whole answer before the
|
|
766
|
+
* subscriber does — the dispatched handler streams to a worker-side sink the loop cannot
|
|
767
|
+
* interpose on, so the instruction has to ride the envelope. Absent → stream live, as before.
|
|
768
|
+
*/
|
|
769
|
+
bufferOutput?: boolean;
|
|
373
770
|
}
|
|
374
771
|
/**
|
|
375
772
|
* The serializable subset of `AiToolCtx` — everything except `host` (re-attached handler-side
|
|
@@ -455,6 +852,23 @@ declare const AGENT_ATTACHMENT_STAGING: unique symbol;
|
|
|
455
852
|
* {@link import('./spi/approval-port.js').AgentApprovalPort}.
|
|
456
853
|
*/
|
|
457
854
|
declare const AGENT_APPROVAL_PORT: unique symbol;
|
|
855
|
+
/**
|
|
856
|
+
* The resolved `SkillsConfig` (provider + scope resolver + ceiling), or `undefined` where the host
|
|
857
|
+
* configured no skills. Bound once and read by BOTH the loop's deps and the listing endpoint, so
|
|
858
|
+
* what a user can invoke and what the model can reach are the same resolution.
|
|
859
|
+
*/
|
|
860
|
+
declare const AGENT_SKILLS: unique symbol;
|
|
861
|
+
/**
|
|
862
|
+
* The resolved `MemoryConfig` (provider + scope resolver + ceilings), or `undefined` where the host
|
|
863
|
+
* configured no memory. Bound once and read by BOTH the loop's deps and the read-back endpoint, so
|
|
864
|
+
* what a person is shown and what the model was shown are the same resolution.
|
|
865
|
+
*/
|
|
866
|
+
declare const AGENT_MEMORY: unique symbol;
|
|
867
|
+
/**
|
|
868
|
+
* A shared, mutable list `SkillDiscoveryService` fills with `@Skill`-decorated providers. Read
|
|
869
|
+
* lazily by the bound {@link AGENT_SKILLS} provider — discovery runs after DI has built it.
|
|
870
|
+
*/
|
|
871
|
+
declare const AGENT_SKILL_SOURCES: unique symbol;
|
|
458
872
|
|
|
459
873
|
/**
|
|
460
874
|
* Per-invocation context handed to a tool handler. Host-supplied bits are optional. Identity lives
|
|
@@ -555,6 +969,14 @@ interface ModelTurnArgs {
|
|
|
555
969
|
/** The model writes streamed text deltas here as it generates them. */
|
|
556
970
|
sink: SinkWriter;
|
|
557
971
|
abortSignal?: AbortSignal;
|
|
972
|
+
/**
|
|
973
|
+
* Constrain this turn's reply to a schema (a provider's JSON/response-format mode). Set only on
|
|
974
|
+
* the loop's structured-output formatting pass, which always sends `tools: []` — most providers
|
|
975
|
+
* refuse a response format and a tool set in the same request, and the ones that accept both stop
|
|
976
|
+
* calling tools. A provider that cannot constrain generation may ignore this: the loop validates
|
|
977
|
+
* the reply against the same schema either way, so ignoring it costs reliability, not safety.
|
|
978
|
+
*/
|
|
979
|
+
outputSchema?: StandardSchemaV1;
|
|
558
980
|
}
|
|
559
981
|
/** The outcome of ONE assistant turn. The loop — not the model — drives tool execution. */
|
|
560
982
|
interface ModelTurnResult {
|
|
@@ -574,6 +996,41 @@ interface ModelTurnResult {
|
|
|
574
996
|
* governance read-model uses it verbatim; otherwise it estimates from tokens × the pricing table.
|
|
575
997
|
*/
|
|
576
998
|
costUsd?: number;
|
|
999
|
+
/**
|
|
1000
|
+
* The reply already parsed, for a provider that constrained generation against
|
|
1001
|
+
* {@link ModelTurnArgs.outputSchema} and therefore has the value in hand. A fast path only — the
|
|
1002
|
+
* loop validates it against the schema regardless, so omitting it (and leaving the loop to read
|
|
1003
|
+
* the JSON out of `text`) is always correct.
|
|
1004
|
+
*/
|
|
1005
|
+
object?: unknown;
|
|
1006
|
+
}
|
|
1007
|
+
/**
|
|
1008
|
+
* A model turn whose live frames were HELD instead of streamed, because an output processor has to
|
|
1009
|
+
* see the whole answer before anything downstream does. Produced by whatever ran the model call —
|
|
1010
|
+
* the loop itself, or the dispatched-step handler when the envelope asked for it — and released to
|
|
1011
|
+
* the real sink by the loop once the gate has passed.
|
|
1012
|
+
*/
|
|
1013
|
+
interface BufferedModelTurnResult extends ModelTurnResult {
|
|
1014
|
+
/** The NDJSON stream lines the call produced, in write order. */
|
|
1015
|
+
bufferedFrames?: string[];
|
|
1016
|
+
/**
|
|
1017
|
+
* The transformed PREFIX an incremental gate already wrote to the run's sink while the call was
|
|
1018
|
+
* streaming. Its presence — not the chain configured on the process that reads it back — is what
|
|
1019
|
+
* tells the gate step which release this turn still owes, so a run that resumes under a
|
|
1020
|
+
* re-declared chain cannot flush the same text twice. Empty string when an incremental gate ran
|
|
1021
|
+
* and released nothing; absent when none ran.
|
|
1022
|
+
*/
|
|
1023
|
+
releasedText?: string;
|
|
1024
|
+
/**
|
|
1025
|
+
* The refusal an incremental gate reached from a prefix, carried on the RESULT rather than thrown
|
|
1026
|
+
* from the call. The loop raises it only after the turn's `persist:usage` and `quota:bump`
|
|
1027
|
+
* checkpoints — those tokens were genuinely spent, and a gate that hid its own cost would let a
|
|
1028
|
+
* mis-tuned chain burn a budget invisibly.
|
|
1029
|
+
*/
|
|
1030
|
+
gateRejection?: {
|
|
1031
|
+
processor: string;
|
|
1032
|
+
reason: string;
|
|
1033
|
+
};
|
|
577
1034
|
}
|
|
578
1035
|
/**
|
|
579
1036
|
* Thin wrapper over the actual LLM. The concrete impl (e.g. Vercel AI SDK `streamText`
|
|
@@ -644,9 +1101,41 @@ type AgentStreamEvent = {
|
|
|
644
1101
|
kind: 'tool-output-error';
|
|
645
1102
|
id: string;
|
|
646
1103
|
error: string;
|
|
1104
|
+
}
|
|
1105
|
+
/**
|
|
1106
|
+
* The run has put a question set to the user and is parked until someone answers it (or skips).
|
|
1107
|
+
* Written by the LOOP for both elicitation surfaces — the configured intake and the model's `ask`
|
|
1108
|
+
* tool — so a client renders one form either way rather than learning to recognise a tool name.
|
|
1109
|
+
* The matching `tool-output` frame, under the same `id`, carries the settled answers.
|
|
1110
|
+
*/
|
|
1111
|
+
| {
|
|
1112
|
+
kind: 'elicitation';
|
|
1113
|
+
id: string;
|
|
1114
|
+
request: ElicitationRequest;
|
|
1115
|
+
}
|
|
1116
|
+
/**
|
|
1117
|
+
* Someone stopped this run. The stream's LAST frame, written by the runner that settled the
|
|
1118
|
+
* cancel, immediately before a normal `end()` — never a `fail()`, because a cancel is not an
|
|
1119
|
+
* error and a client that retries on a failed stream must not retry this.
|
|
1120
|
+
*
|
|
1121
|
+
* A run that simply ends wrote everything it had; one that ends after this frame did not, and the
|
|
1122
|
+
* difference is the whole point: without it a reader cannot tell a truncated answer from a
|
|
1123
|
+
* complete one. Consumers that predate the frame ignore it and see the `end()` they always saw.
|
|
1124
|
+
*/
|
|
1125
|
+
| {
|
|
1126
|
+
kind: 'cancelled';
|
|
647
1127
|
};
|
|
648
1128
|
/** Encode one event as an NDJSON line (`{...}\n`) for {@link SinkWriter.write}. */
|
|
649
1129
|
declare function encodeStreamEvent(event: AgentStreamEvent): Uint8Array;
|
|
1130
|
+
/**
|
|
1131
|
+
* Read one NDJSON line back, or `null` when the line is not a stream event at all.
|
|
1132
|
+
*
|
|
1133
|
+
* `null` covers a genuinely opaque chunk, not just malformed JSON: the sink is a byte channel, so a
|
|
1134
|
+
* model provider is free to write anything into it and some write bare text. A caller that has to
|
|
1135
|
+
* CLASSIFY a chunk — the output gate, which may only forward what it can prove is not the answer —
|
|
1136
|
+
* treats an unreadable frame as unclassifiable rather than guessing.
|
|
1137
|
+
*/
|
|
1138
|
+
declare function decodeStreamEvent(line: string): AgentStreamEvent | null;
|
|
650
1139
|
|
|
651
1140
|
interface CreateThreadInput {
|
|
652
1141
|
actor: Actor;
|
|
@@ -665,6 +1154,14 @@ interface AppendMessageInput {
|
|
|
665
1154
|
attachments?: MessageAttachment[];
|
|
666
1155
|
followUps?: string[];
|
|
667
1156
|
usage?: MessageUsage;
|
|
1157
|
+
/**
|
|
1158
|
+
* The run (turn) that produced this message. Without it a consumer can only guess which turn a
|
|
1159
|
+
* message belongs to by comparing timestamps against the run's `startedAt`, and that guess breaks
|
|
1160
|
+
* the moment a turn is regenerated — the replaced answer is truncated away, so the times no longer
|
|
1161
|
+
* line up 1:1. Optional so a caller predating this (and a host that appends messages outside a
|
|
1162
|
+
* run) can omit it; the store persists it as `null` when absent.
|
|
1163
|
+
*/
|
|
1164
|
+
runId?: string;
|
|
668
1165
|
}
|
|
669
1166
|
interface RecordToolCallInput {
|
|
670
1167
|
toolCallId: string;
|
|
@@ -709,16 +1206,67 @@ interface RecordRunStartInput {
|
|
|
709
1206
|
threadId: string;
|
|
710
1207
|
actorRef: string;
|
|
711
1208
|
agentName?: string;
|
|
1209
|
+
/**
|
|
1210
|
+
* The run that started this one, for a delegation's child run. The parent->child edge exists in
|
|
1211
|
+
* the durable runtime's own journal, but only there: a governance surface reading run rows alone
|
|
1212
|
+
* cannot roll a delegation's cost up to the turn that asked for it, and a DETACHED child outlives
|
|
1213
|
+
* its parent's turn entirely, so nothing in the transcript pairs them either.
|
|
1214
|
+
*
|
|
1215
|
+
* Optional, and a store that persists nothing for it still works — it loses the tree, not the run.
|
|
1216
|
+
*/
|
|
1217
|
+
parentRunId?: string;
|
|
712
1218
|
/** sha256 hex of the run's resolved (pre-RAG) system prompt — identifies the prompt VERSION. */
|
|
713
1219
|
promptHash?: string;
|
|
714
1220
|
}
|
|
715
1221
|
interface RecordRunEndInput {
|
|
716
1222
|
runId: string;
|
|
717
|
-
|
|
1223
|
+
/**
|
|
1224
|
+
* `cancelled` is a THIRD terminal, not a flavour of `failed`: someone asked the run to stop and it
|
|
1225
|
+
* did, which is the control working. A consumer computing a failure rate over these rows has to be
|
|
1226
|
+
* able to leave it out — counting a user pressing Stop as an error pages whoever is on call for
|
|
1227
|
+
* model failures. It carries no `errorCode`/`errorMessage`, since there is nothing to diagnose.
|
|
1228
|
+
*/
|
|
1229
|
+
status: 'completed' | 'failed' | 'cancelled';
|
|
718
1230
|
durationMs?: number;
|
|
719
1231
|
errorCode?: string;
|
|
720
1232
|
errorMessage?: string;
|
|
721
1233
|
}
|
|
1234
|
+
/** Which thread, and how many of its newest messages, {@link ThreadTurnReader.loadThreadForTurn} reads. */
|
|
1235
|
+
interface ThreadTurnQuery {
|
|
1236
|
+
threadId: string;
|
|
1237
|
+
/** Omitted reads every message; `0` reads none. */
|
|
1238
|
+
messageLimit?: number;
|
|
1239
|
+
}
|
|
1240
|
+
/** What a turn reads off a thread — a bounded window, not the transcript. */
|
|
1241
|
+
interface ThreadTurnPage {
|
|
1242
|
+
title: string;
|
|
1243
|
+
defaultAgent: string | null;
|
|
1244
|
+
/** Whether the THREAD has ever been answered, not whether {@link messages} holds an answer. */
|
|
1245
|
+
hasAssistantMessage: boolean;
|
|
1246
|
+
/** Oldest first, carrying only the fields a model turn reads. */
|
|
1247
|
+
messages: StoredMessage[];
|
|
1248
|
+
}
|
|
1249
|
+
/**
|
|
1250
|
+
* A store that can hand a turn the WINDOW it is about to send, instead of the thread's transcript.
|
|
1251
|
+
*
|
|
1252
|
+
* {@link AgentStore.getThread} materializes every message row, every attachment and every tool
|
|
1253
|
+
* output a thread ever recorded, and the run then journals what it loaded — so a long thread pays
|
|
1254
|
+
* for its whole history on every turn and again on every replay, to send a prompt bounded to its
|
|
1255
|
+
* last few messages. This read is bounded by the database (`order by created_at desc limit ?`),
|
|
1256
|
+
* projected to the columns a model turn actually reads.
|
|
1257
|
+
*
|
|
1258
|
+
* `hasAssistantMessage` is answered over the WHOLE thread, never the page: it answers "has this
|
|
1259
|
+
* conversation been answered before?" — what a `thread-start` intake asks — and a thread whose window
|
|
1260
|
+
* happens to hold only the user's last questions has still been answered. `null` for a thread that is
|
|
1261
|
+
* unknown or soft-deleted, matching `getThread`.
|
|
1262
|
+
*
|
|
1263
|
+
* Probed STRUCTURALLY rather than declared on {@link AgentStore}, the same way `defaultAgentForThread`
|
|
1264
|
+
* is: it is an optimization a store either offers or does not, and one that predates it still answers
|
|
1265
|
+
* correctly through the full read.
|
|
1266
|
+
*/
|
|
1267
|
+
interface ThreadTurnReader {
|
|
1268
|
+
loadThreadForTurn(query: ThreadTurnQuery): Promise<ThreadTurnPage | null>;
|
|
1269
|
+
}
|
|
722
1270
|
/** ORM-agnostic persistence. Refs are string ids; adapters may add real relations. */
|
|
723
1271
|
interface AgentStore {
|
|
724
1272
|
createThread(input: CreateThreadInput): Promise<ThreadSummary>;
|
|
@@ -769,11 +1317,15 @@ interface AgentStore {
|
|
|
769
1317
|
*/
|
|
770
1318
|
ownerOfToolCall(toolCallId: string): Promise<string | null>;
|
|
771
1319
|
/**
|
|
772
|
-
* The
|
|
773
|
-
*
|
|
774
|
-
*
|
|
1320
|
+
* The run awaiting a decision on `toolCallId`: the call's OWN `runId` when the row carries one,
|
|
1321
|
+
* else the thread's `activeStreamId`. Both HITL approve/reject and an elicitation answer route
|
|
1322
|
+
* through this, derived server-side from the tool call alone — so a decision reaches the exact run
|
|
775
1323
|
* awaiting it, including a sub-agent's own child run, which the client never sees and could not
|
|
776
1324
|
* name. No client-supplied runId is trusted (or needed).
|
|
1325
|
+
*
|
|
1326
|
+
* The row's own runId comes FIRST because `activeStreamId` names whichever run is streaming the
|
|
1327
|
+
* thread right now, and that is only the same run while a thread holds exactly one. The fallback
|
|
1328
|
+
* is for rows written before tool calls recorded a runId, which have nothing else to answer with.
|
|
777
1329
|
*/
|
|
778
1330
|
runForToolCall(toolCallId: string): Promise<string | null>;
|
|
779
1331
|
/**
|
|
@@ -783,9 +1335,43 @@ interface AgentStore {
|
|
|
783
1335
|
*/
|
|
784
1336
|
ownerOfActiveStream(runId: string): Promise<string | null>;
|
|
785
1337
|
appendMessage(input: AppendMessageInput): Promise<StoredMessage>;
|
|
1338
|
+
/**
|
|
1339
|
+
* Attach a turn's settled tool RESULTS to a message that was already appended, replacing whatever
|
|
1340
|
+
* it held. A message's tool calls are known when it is written and their outputs are not, but a
|
|
1341
|
+
* thread reader pairs the two off THAT MESSAGE — so an output that only ever reaches the tool-call
|
|
1342
|
+
* table leaves every call on a reopened thread looking like a tool still running.
|
|
1343
|
+
*
|
|
1344
|
+
* Required rather than optional: a store that silently declines this renders a finished turn as a
|
|
1345
|
+
* permanently in-flight one, with nothing logged and nothing to notice. A missing method should
|
|
1346
|
+
* fail to compile instead.
|
|
1347
|
+
*/
|
|
1348
|
+
setMessageToolResults(messageId: string, results: ToolResult[]): Promise<void>;
|
|
786
1349
|
truncateFrom(threadId: string, messageId: string): Promise<void>;
|
|
787
1350
|
recordToolCall(input: RecordToolCallInput): Promise<void>;
|
|
788
1351
|
updateToolCall(input: UpdateToolCallInput): Promise<void>;
|
|
1352
|
+
/**
|
|
1353
|
+
* OPTIONAL: of `mediaIds`, the ones a message that still exists — in a thread owned by
|
|
1354
|
+
* `actorRef` — still carries as an attachment. The inverse of
|
|
1355
|
+
* {@link import('./attachment-staging.js').AttachmentStagingStore.list}: the host can enumerate
|
|
1356
|
+
* the media it staged but cannot see a transcript, and this side sees every transcript but never
|
|
1357
|
+
* holds the bytes, so neither can decide alone what is safe to delete.
|
|
1358
|
+
*
|
|
1359
|
+
* DERIVED, not tracked. A reference is not permanent: `truncateFrom` deletes messages — which is
|
|
1360
|
+
* exactly what regenerating a turn does — so media that was referenced becomes unreferenced
|
|
1361
|
+
* again. A flag set when a message is sent would never be unset by that delete, and the bytes
|
|
1362
|
+
* would be pinned for ever with nothing pointing at them. Answering from the surviving message
|
|
1363
|
+
* rows on every call is the only form of this that stays true after a truncation.
|
|
1364
|
+
*
|
|
1365
|
+
* Scoped to one actor, like every other read on this surface: media referenced only by ANOTHER
|
|
1366
|
+
* actor's thread is reported unreferenced here, so this can never be turned into a probe for what
|
|
1367
|
+
* exists in someone else's conversation. A host pairs it with its own per-actor inventory, so the
|
|
1368
|
+
* candidate ids are already the caller's own.
|
|
1369
|
+
*
|
|
1370
|
+
* Returns each id at most once, in the order asked. Absent on a store that predates this — a
|
|
1371
|
+
* caller must treat the absence as "cannot answer" and collect NOTHING, never as "nothing is
|
|
1372
|
+
* referenced", which would delete every attachment the actor ever sent.
|
|
1373
|
+
*/
|
|
1374
|
+
referencedMediaIds?(actorRef: string, mediaIds: readonly string[]): Promise<string[]>;
|
|
789
1375
|
recordUsage(input: RecordUsageInput): Promise<void>;
|
|
790
1376
|
/**
|
|
791
1377
|
* The actor's spend for `day` (UTC): total tokens plus the summed provider-reported USD cost.
|
|
@@ -879,6 +1465,163 @@ interface Retriever {
|
|
|
879
1465
|
retrieve(query: string, options?: RetrieveOptions): Promise<Passage[]>;
|
|
880
1466
|
}
|
|
881
1467
|
|
|
1468
|
+
/**
|
|
1469
|
+
* The two seams that see a turn's traffic to and from the model: an {@link InputProcessor} rewrites
|
|
1470
|
+
* the prompt on its way out, an {@link OutputProcessor} inspects the answer on its way back and may
|
|
1471
|
+
* redact, replace or refuse it. Wire them as `AgentLoopDeps.inputProcessors` /
|
|
1472
|
+
* `outputProcessors`, or via `AgentModule.forRoot({ inputProcessors, outputProcessors })`.
|
|
1473
|
+
*
|
|
1474
|
+
* WHAT THESE ARE NOT: a second way to decide what enters the context. `HistoryPolicy` owns
|
|
1475
|
+
* SELECTION — which of the thread's messages ride into the turn, and what stands in for the rest.
|
|
1476
|
+
* Processors run on whatever selection produced and own TRANSFORMATION — what those messages say.
|
|
1477
|
+
* The distinction is load-bearing rather than stylistic: selection is contractually pure and is what
|
|
1478
|
+
* the `load:thread` checkpoint records, so it bounds the journal as well as the prompt, while a
|
|
1479
|
+
* processor is allowed to call a model, runs in a checkpoint of its own per step, and rewrites a
|
|
1480
|
+
* DERIVED prompt that never becomes the thread's own memory of what was said. Splitting the same
|
|
1481
|
+
* decision across both means neither can be reasoned about alone, and the cheap one stops being the
|
|
1482
|
+
* whole answer to "why did this turn cost that much".
|
|
1483
|
+
*/
|
|
1484
|
+
|
|
1485
|
+
/** Which turn, and which model step of it, a processor is looking at. */
|
|
1486
|
+
interface ProcessorContext {
|
|
1487
|
+
threadId: string;
|
|
1488
|
+
actor: Actor;
|
|
1489
|
+
/** The agent running this turn. Undefined → the default agent. */
|
|
1490
|
+
agentName?: string;
|
|
1491
|
+
/** 0-based model step within the run — the same index the `llm:<step>` checkpoint carries. */
|
|
1492
|
+
step: number;
|
|
1493
|
+
}
|
|
1494
|
+
/** Everything the model is about to be sent, as the previous processor in the chain left it. */
|
|
1495
|
+
interface ProcessedPrompt {
|
|
1496
|
+
/** The composed system prompt (agent base + contributors + any injected retrieval block). */
|
|
1497
|
+
system: string;
|
|
1498
|
+
/** The turn's messages, oldest-first, already through the history ceiling. */
|
|
1499
|
+
messages: ModelMessage[];
|
|
1500
|
+
}
|
|
1501
|
+
/**
|
|
1502
|
+
* Rewrites the prompt before each model call of a turn — masking identifiers, stamping a policy
|
|
1503
|
+
* preamble, collapsing an oversized tool result. Runs on EVERY step, not once per run, because the
|
|
1504
|
+
* transcript grows between steps: a redactor that only saw the opening prompt would wave through
|
|
1505
|
+
* whatever a tool result carried back.
|
|
1506
|
+
*
|
|
1507
|
+
* Runs inside the loop's `process:input:<step>` checkpoint and its result is journaled, so a
|
|
1508
|
+
* processor may call a model or hit the network — a resumed run reads back the prompt the suspended
|
|
1509
|
+
* attempt built rather than composing a different one.
|
|
1510
|
+
*/
|
|
1511
|
+
interface InputProcessor {
|
|
1512
|
+
/** Identifies this processor in a failure. Keep it stable — it is user-visible on an error. */
|
|
1513
|
+
readonly name: string;
|
|
1514
|
+
process(prompt: ProcessedPrompt, ctx: ProcessorContext): ProcessedPrompt | Promise<ProcessedPrompt>;
|
|
1515
|
+
}
|
|
1516
|
+
/** One model step's answer, as the previous processor in the chain left it. */
|
|
1517
|
+
interface ModelAnswer {
|
|
1518
|
+
/** The assembled assistant text for this step. */
|
|
1519
|
+
text: string;
|
|
1520
|
+
/** The tool calls the same step asked for. Read-only context — a processor cannot change them. */
|
|
1521
|
+
toolCalls: readonly ToolCallRequest[];
|
|
1522
|
+
}
|
|
1523
|
+
/**
|
|
1524
|
+
* What an {@link OutputProcessor} decided. `pass` hands the text to the next processor unchanged;
|
|
1525
|
+
* `replace` hands it on rewritten (a redaction is a replacement); `reject` ends the chain AND the
|
|
1526
|
+
* run — the text is never streamed, never persisted, and the caller gets an
|
|
1527
|
+
* {@link OutputRejectedError} rather than an answer.
|
|
1528
|
+
*/
|
|
1529
|
+
type OutputVerdict = {
|
|
1530
|
+
action: 'pass';
|
|
1531
|
+
} | {
|
|
1532
|
+
action: 'replace';
|
|
1533
|
+
text: string;
|
|
1534
|
+
} | {
|
|
1535
|
+
action: 'reject';
|
|
1536
|
+
reason: string;
|
|
1537
|
+
};
|
|
1538
|
+
/**
|
|
1539
|
+
* Characters an incremental gate keeps holding at the end of the transformed answer, when a
|
|
1540
|
+
* processor declares {@link IncrementalGating} without naming its own window. Wide enough for the
|
|
1541
|
+
* patterns a redactor is usually written against — an SSN, an email address, a card number — and
|
|
1542
|
+
* deliberately not wider: the window IS the answer's minimum latency tail, since those characters
|
|
1543
|
+
* are only released once the whole-answer pass runs.
|
|
1544
|
+
*/
|
|
1545
|
+
declare const DEFAULT_INCREMENTAL_LOOKBACK_CHARS = 64;
|
|
1546
|
+
/**
|
|
1547
|
+
* A processor's statement that its verdict on a PREFIX of an answer is worth acting on, which is
|
|
1548
|
+
* what lets the loop release that prefix to the reader instead of holding the whole answer.
|
|
1549
|
+
*
|
|
1550
|
+
* Declaring it is a promise about every prefix `P` of the final answer, and the loop takes it at
|
|
1551
|
+
* face value on the streaming path:
|
|
1552
|
+
*
|
|
1553
|
+
* 1. REJECTION IS PREFIX-DECIDABLE. The refusal this processor would return for the whole answer is
|
|
1554
|
+
* already returned for the first prefix that contains the reason. A refusal that only emerges
|
|
1555
|
+
* from the complete answer still fails the run, but by then the reader has seen a prefix — there
|
|
1556
|
+
* is no un-sending bytes, and that is the cost of opting in.
|
|
1557
|
+
* 2. REPLACEMENT IS PREFIX-STABLE UP TO THE WINDOW. Once a character of `process(P)`'s output is
|
|
1558
|
+
* more than {@link lookbackChars} from the end of that output, it never changes as `P` grows.
|
|
1559
|
+
*
|
|
1560
|
+
* A broken second promise is caught, not tolerated: the whole-answer pass stays authoritative, and
|
|
1561
|
+
* the loop raises a `ProcessorFailedError` when its result is not an extension of what the chain
|
|
1562
|
+
* already released. So a window too short for a pattern fails loudly rather than streaming the text
|
|
1563
|
+
* it was supposed to redact.
|
|
1564
|
+
*/
|
|
1565
|
+
interface IncrementalGating {
|
|
1566
|
+
/**
|
|
1567
|
+
* How many characters of this processor's own output stay held back. Undefined →
|
|
1568
|
+
* {@link DEFAULT_INCREMENTAL_LOOKBACK_CHARS}. Set it to the length of the longest pattern the
|
|
1569
|
+
* processor can act on — anything shorter is a run that fails on the pattern it was written for.
|
|
1570
|
+
*/
|
|
1571
|
+
readonly lookbackChars?: number;
|
|
1572
|
+
}
|
|
1573
|
+
/**
|
|
1574
|
+
* Inspects each model step's answer before anything downstream sees it — before it reaches the live
|
|
1575
|
+
* stream, before it is persisted, before it becomes the next step's context.
|
|
1576
|
+
*
|
|
1577
|
+
* Gating an answer and streaming it as it is generated are mutually exclusive, so registering ANY
|
|
1578
|
+
* output processor switches the turn's model call off the run's sink: nothing reaches the
|
|
1579
|
+
* subscriber until this chain has ruled on it — see `AgentLoopDeps.outputProcessors`. How much of
|
|
1580
|
+
* that costs the reader is what {@link incremental} decides.
|
|
1581
|
+
*
|
|
1582
|
+
* Runs inside the loop's `process:output:<step>` checkpoint (which also performs the release), so a
|
|
1583
|
+
* processor may call a model — a moderation pass is the motivating case — and a replay reads the
|
|
1584
|
+
* verdict back instead of re-deciding it.
|
|
1585
|
+
*/
|
|
1586
|
+
interface OutputProcessor {
|
|
1587
|
+
/** Identifies this processor in a rejection or a failure. User-visible; keep it stable. */
|
|
1588
|
+
readonly name: string;
|
|
1589
|
+
/**
|
|
1590
|
+
* Opt this processor into gating a PREFIX, so the turn keeps streaming — see
|
|
1591
|
+
* {@link IncrementalGating} for what declaring it promises. Undefined means the whole answer,
|
|
1592
|
+
* which is what a processor written against the complete text needs and therefore the only safe
|
|
1593
|
+
* default; a chain is incremental only when EVERY member of it declares this, so a neighbour can
|
|
1594
|
+
* never downgrade what another author was given.
|
|
1595
|
+
*/
|
|
1596
|
+
readonly incremental?: IncrementalGating;
|
|
1597
|
+
process(answer: ModelAnswer, ctx: ProcessorContext): OutputVerdict | Promise<OutputVerdict>;
|
|
1598
|
+
}
|
|
1599
|
+
/**
|
|
1600
|
+
* The run ended because an output processor refused the answer — NOT because the model failed. The
|
|
1601
|
+
* two must stay distinguishable: a model failure is worth retrying and worth paging someone about,
|
|
1602
|
+
* a refusal is the control doing its job. Both runners map this to the `output_rejected` stream
|
|
1603
|
+
* error code.
|
|
1604
|
+
*/
|
|
1605
|
+
declare class OutputRejectedError extends Error {
|
|
1606
|
+
/** {@link OutputProcessor.name} of the processor that refused. */
|
|
1607
|
+
readonly processor: string;
|
|
1608
|
+
/** The reason it gave, verbatim. */
|
|
1609
|
+
readonly reason: string;
|
|
1610
|
+
constructor(processor: string, reason: string);
|
|
1611
|
+
}
|
|
1612
|
+
/**
|
|
1613
|
+
* A processor threw. Wrapped so it can never be mistaken for the model call failing — the loop's
|
|
1614
|
+
* only other source of failure at that point in the turn — and so the failure names the processor
|
|
1615
|
+
* that produced it instead of surfacing a bare `TypeError` from someone else's code.
|
|
1616
|
+
*/
|
|
1617
|
+
declare class ProcessorFailedError extends Error {
|
|
1618
|
+
/** Which seam it was on, so a reader knows whether the prompt or the answer was in flight. */
|
|
1619
|
+
readonly phase: 'input' | 'output';
|
|
1620
|
+
/** The processor's `name`. */
|
|
1621
|
+
readonly processor: string;
|
|
1622
|
+
constructor(phase: 'input' | 'output', processor: string, cause: unknown);
|
|
1623
|
+
}
|
|
1624
|
+
|
|
882
1625
|
/**
|
|
883
1626
|
* Turns text into embedding vectors — the sibling of {@link import('./model-provider.js').ModelProvider}
|
|
884
1627
|
* for the retrieval side. Batched (`texts` → one vector each, same order) so ingestion can embed many
|
|
@@ -919,8 +1662,12 @@ interface AgentRunner {
|
|
|
919
1662
|
start(input: AgentRunInput): Promise<{
|
|
920
1663
|
runId: string;
|
|
921
1664
|
}>;
|
|
922
|
-
/**
|
|
923
|
-
|
|
1665
|
+
/**
|
|
1666
|
+
* Deliver a human's reply to a parked tool call — a {@link import('../types.js').Decision} on an
|
|
1667
|
+
* action tool, or an `ElicitationReply` answering a question set. One channel for both, because
|
|
1668
|
+
* both park the run the same way and a runner has no reason to tell them apart.
|
|
1669
|
+
*/
|
|
1670
|
+
signal(runId: string, toolCallId: string, reply: HumanReply): Promise<void>;
|
|
924
1671
|
cancel(runId: string): Promise<void>;
|
|
925
1672
|
}
|
|
926
1673
|
|
|
@@ -1048,7 +1795,11 @@ interface RecentRunRow {
|
|
|
1048
1795
|
threadId: string;
|
|
1049
1796
|
actorRef: string;
|
|
1050
1797
|
agentName: string | null;
|
|
1051
|
-
/**
|
|
1798
|
+
/**
|
|
1799
|
+
* `'running'` | `'completed'` | `'failed'` | `'cancelled'`. `cancelled` is a terminal of its own —
|
|
1800
|
+
* someone asked the run to stop and it did — so a consumer computing a failure rate over these
|
|
1801
|
+
* rows has to be able to leave it out rather than fold it into `failed`.
|
|
1802
|
+
*/
|
|
1052
1803
|
status: string;
|
|
1053
1804
|
durationMs: number | null;
|
|
1054
1805
|
errorCode: string | null;
|
|
@@ -1058,6 +1809,14 @@ interface RecentRunRow {
|
|
|
1058
1809
|
startedAt: string;
|
|
1059
1810
|
/** sha256 hex of the run's resolved (pre-RAG) system prompt; `null` for a run recorded before this shipped. */
|
|
1060
1811
|
promptHash: string | null;
|
|
1812
|
+
/**
|
|
1813
|
+
* The run that delegated this one; `null` for a turn nobody delegated, and for any run recorded
|
|
1814
|
+
* before the column existed. Pairing a child with its parent is what lets a console draw the
|
|
1815
|
+
* delegation tree and roll a child's cost up to the turn that asked for it — and for a DETACHED
|
|
1816
|
+
* child it is the only pairing there is, since it outlives its parent's turn and the transcript
|
|
1817
|
+
* holds no other link.
|
|
1818
|
+
*/
|
|
1819
|
+
parentRunId: string | null;
|
|
1061
1820
|
}
|
|
1062
1821
|
/** One tool call awaiting a HITL decision, for the cross-thread approvals inbox. */
|
|
1063
1822
|
interface PendingApprovalRow {
|
|
@@ -1335,13 +2094,57 @@ interface StageAttachmentInput {
|
|
|
1335
2094
|
sizeBytes: number;
|
|
1336
2095
|
actor: Actor;
|
|
1337
2096
|
}
|
|
2097
|
+
/**
|
|
2098
|
+
* All a chat turn may say about a file it attaches: the id of something already staged. Everything
|
|
2099
|
+
* else about the attachment (url, content type, name) is read back from the staging store, never
|
|
2100
|
+
* from the request — see {@link AttachmentStagingStore.resolve}.
|
|
2101
|
+
*/
|
|
2102
|
+
interface AttachmentRef {
|
|
2103
|
+
mediaId: string;
|
|
2104
|
+
}
|
|
2105
|
+
/** Input to {@link AttachmentStagingStore.resolve} — the claimed id plus who is claiming it. */
|
|
2106
|
+
interface ResolveAttachmentInput {
|
|
2107
|
+
mediaId: string;
|
|
2108
|
+
actor: Actor;
|
|
2109
|
+
}
|
|
2110
|
+
/**
|
|
2111
|
+
* One entry in an actor's staged-media inventory: enough to show the file and to decide whether it
|
|
2112
|
+
* has aged out, and nothing more.
|
|
2113
|
+
*
|
|
2114
|
+
* Carries no url, unlike {@link MessageAttachment}. A url is minted per turn by
|
|
2115
|
+
* {@link AttachmentStagingStore.resolve} precisely so it can be short-lived; a listing that handed
|
|
2116
|
+
* one back would mint a fetchable url for every entry on the page, for a caller that asked to see
|
|
2117
|
+
* a file list. Whoever needs the bytes goes through `resolve` and is checked there.
|
|
2118
|
+
*/
|
|
2119
|
+
interface StagedAttachment {
|
|
2120
|
+
mediaId: string;
|
|
2121
|
+
/** Original filename, as `stage` received it. */
|
|
2122
|
+
name: string;
|
|
2123
|
+
contentType: string;
|
|
2124
|
+
sizeBytes: number;
|
|
2125
|
+
/** ISO-8601 UTC instant the bytes were staged — what an age threshold is measured against. */
|
|
2126
|
+
createdAt: string;
|
|
2127
|
+
}
|
|
2128
|
+
/** Input to {@link AttachmentStagingStore.list} — whose inventory, and how much of it. */
|
|
2129
|
+
interface ListStagedAttachmentsInput {
|
|
2130
|
+
actor: Actor;
|
|
2131
|
+
/**
|
|
2132
|
+
* Only media staged strictly before this ISO-8601 UTC instant. This is how a caller expresses
|
|
2133
|
+
* "old enough to be worth looking at": freshly staged media belongs to an upload the user has
|
|
2134
|
+
* not sent yet, which is in flight rather than garbage. The instant comes from the caller
|
|
2135
|
+
* because how long a composer may sit open is the host's knowledge, not this library's.
|
|
2136
|
+
*/
|
|
2137
|
+
stagedBefore?: string;
|
|
2138
|
+
/** Cap on entries returned, newest first. */
|
|
2139
|
+
limit?: number;
|
|
2140
|
+
}
|
|
1338
2141
|
/**
|
|
1339
2142
|
* Optional upload-side seam for message attachments (an image/PDF a user attaches to a chat
|
|
1340
2143
|
* message before the model ever sees it). The lib never fetches bytes itself — {@link MessageAttachment.url}
|
|
1341
2144
|
* must already be reachable by the model provider — so something has to turn an uploaded file into
|
|
1342
2145
|
* that URL first. A store adapter (or a thin wrapper over the host's own media pipeline) implements
|
|
1343
2146
|
* this; consumers inject via `AGENT_ATTACHMENT_STAGING`. Unbound, the optional `POST /agent/attachments`
|
|
1344
|
-
* upload controller is never mounted.
|
|
2147
|
+
* upload controller is never mounted, and a chat turn cannot carry attachments at all.
|
|
1345
2148
|
*/
|
|
1346
2149
|
interface AttachmentStagingStore {
|
|
1347
2150
|
/**
|
|
@@ -1350,10 +2153,44 @@ interface AttachmentStagingStore {
|
|
|
1350
2153
|
* returned url must be reachable by the model provider.
|
|
1351
2154
|
*/
|
|
1352
2155
|
stage(input: StageAttachmentInput): Promise<MessageAttachment>;
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
2156
|
+
/**
|
|
2157
|
+
* Turn a `mediaId` a chat turn claims back into the attachment to send with it, or `null` when
|
|
2158
|
+
* the id is unknown OR is not this actor's. The two cases are deliberately indistinguishable, so
|
|
2159
|
+
* the chat endpoint cannot be used to probe which media ids exist.
|
|
2160
|
+
*
|
|
2161
|
+
* SECURITY: this is the ONLY source of a url the model provider will be asked to fetch. The chat
|
|
2162
|
+
* endpoint discards every other field a client sends and rebuilds each attachment from here — an
|
|
2163
|
+
* implementation that echoes back a url taken from the request reopens the SSRF this seam exists
|
|
2164
|
+
* to close. Called once per turn, so a short-lived presigned url is minted fresh rather than
|
|
2165
|
+
* replayed stale.
|
|
2166
|
+
*/
|
|
2167
|
+
resolve(input: ResolveAttachmentInput): Promise<MessageAttachment | null>;
|
|
2168
|
+
/**
|
|
2169
|
+
* OPTIONAL: the actor's staged media, newest first. The host staged these bytes, so only the host
|
|
2170
|
+
* can enumerate them — this library holds references, never an inventory.
|
|
2171
|
+
*
|
|
2172
|
+
* Pairs with {@link import('./agent-store.js').AgentStore.referencedMediaIds}, which answers the
|
|
2173
|
+
* half only the library can: of these ids, which a live message still carries. Together they are
|
|
2174
|
+
* a sweep — inventory minus references, restricted to entries old enough not to be an upload in
|
|
2175
|
+
* flight. Deleting whatever is left is the host's call and the host's alone; this library never
|
|
2176
|
+
* removes a host's bytes.
|
|
2177
|
+
*
|
|
2178
|
+
* Scoped to `input.actor` without exception. An implementation that ignores it turns a file list
|
|
2179
|
+
* into a way to read someone else's documents.
|
|
2180
|
+
*
|
|
2181
|
+
* `stagedBefore` and `limit` are the store's to apply — but a caller on a delete path must not
|
|
2182
|
+
* assume they were, since getting that wrong deletes files. `AgentService.collectableAttachments`
|
|
2183
|
+
* re-applies the age cut on the results for that reason.
|
|
2184
|
+
*
|
|
2185
|
+
* Absent on a store that predates this: there is no inventory to fall back on, so listing and
|
|
2186
|
+
* collection are simply unavailable (the read surface answers 501) rather than quietly empty —
|
|
2187
|
+
* an empty inventory and an unanswerable one look identical and mean opposite things.
|
|
2188
|
+
*/
|
|
2189
|
+
list?(input: ListStagedAttachmentsInput): Promise<StagedAttachment[]>;
|
|
2190
|
+
}
|
|
2191
|
+
|
|
2192
|
+
/**
|
|
2193
|
+
* Console-side HITL decisions. Implemented by the agent runtime (the nestjs package binds it to
|
|
1357
2194
|
* the same signal path chat approvals use); the dashboard injects it OPTIONALLY — absent = the
|
|
1358
2195
|
* approvals inbox renders read-only.
|
|
1359
2196
|
*/
|
|
@@ -1450,7 +2287,7 @@ declare function dayBoundsUtc(range: GovernanceRange): {
|
|
|
1450
2287
|
*/
|
|
1451
2288
|
declare function isToolEnabled(spec: ToolSpec, handler?: ToolHandler): Promise<boolean>;
|
|
1452
2289
|
/**
|
|
1453
|
-
*
|
|
2290
|
+
* Second filter layer: drop tools this deployment has turned off, before anyone asks who may call
|
|
1454
2291
|
* them. A disabled tool is absent, not forbidden — the difference matters, because "forbidden"
|
|
1455
2292
|
* tells the model (and the user reading a refusal) that the capability exists.
|
|
1456
2293
|
*/
|
|
@@ -1464,18 +2301,1011 @@ declare function filterToolsByEnabled<T extends {
|
|
|
1464
2301
|
*/
|
|
1465
2302
|
declare function canActorUseTool(actor: Actor, handler?: ToolHandler): Promise<boolean>;
|
|
1466
2303
|
/**
|
|
1467
|
-
*
|
|
2304
|
+
* Fourth filter layer: drop tools whose own `canUse` refuses this actor. Runs after the app-wide
|
|
1468
2305
|
* `RolesPolicy`, and is additive to it — a tool can narrow who reaches it, never widen.
|
|
1469
2306
|
*/
|
|
1470
2307
|
declare function filterToolsByCanUse<T extends {
|
|
1471
2308
|
spec: ToolSpec;
|
|
1472
2309
|
handler?: ToolHandler;
|
|
1473
2310
|
}>(entries: T[], actor: Actor): Promise<T[]>;
|
|
1474
|
-
/**
|
|
2311
|
+
/** Third filter layer: drop tools the actor's role may not invoke. `can` may be async (authz). */
|
|
1475
2312
|
declare function filterToolsByRole(tools: ToolSpec[], actor: Actor, policy: RolesPolicy): Promise<ToolSpec[]>;
|
|
1476
|
-
/**
|
|
2313
|
+
/**
|
|
2314
|
+
* First filter layer: if the agent pins an allow-list, keep only those tool names. Pure and
|
|
2315
|
+
* synchronous, which is why it runs ahead of the three gates that may do I/O — see
|
|
2316
|
+
* `ToolRegistry.definitionsFor`.
|
|
2317
|
+
*/
|
|
1477
2318
|
declare function filterToolsByAllowList(tools: ToolSpec[], allowedTools: string[] | undefined): ToolSpec[];
|
|
1478
2319
|
|
|
2320
|
+
/**
|
|
2321
|
+
* The built-in {@link HistoryPolicy}: keep the newest messages that fit a count and/or a token
|
|
2322
|
+
* budget, and (optionally) fold the rest into a summary. See `./spi/history-policy.ts` for the seam
|
|
2323
|
+
* itself and the determinism contract `select` has to hold to.
|
|
2324
|
+
*/
|
|
2325
|
+
|
|
2326
|
+
/**
|
|
2327
|
+
* Rough token count for one message: ~4 characters per token over its content and its serialized
|
|
2328
|
+
* tool calls/results, plus a small per-message envelope allowance for the role and part framing.
|
|
2329
|
+
*
|
|
2330
|
+
* A heuristic on purpose. A real tokenizer is model-specific and would drag a provider dependency
|
|
2331
|
+
* into core, while a budget only has to be approximately right to keep a thread clear of the
|
|
2332
|
+
* provider's hard limit — and it must be a pure function, because it runs inside `select`. Pass
|
|
2333
|
+
* `estimate` to {@link windowHistory} to substitute a real one.
|
|
2334
|
+
*/
|
|
2335
|
+
declare function estimateMessageTokens(message: ModelMessage): number;
|
|
2336
|
+
interface WindowHistoryOptions {
|
|
2337
|
+
/** Keep at most this many of the newest messages. Omit → no count limit. */
|
|
2338
|
+
maxMessages?: number;
|
|
2339
|
+
/** Keep the newest messages whose estimated tokens fit this budget. Omit → no token limit. */
|
|
2340
|
+
maxTokens?: number;
|
|
2341
|
+
/** Substitute for {@link estimateMessageTokens}. Must be pure — it runs inside `select`. */
|
|
2342
|
+
estimate?: (message: ModelMessage) => number;
|
|
2343
|
+
/** Folds the dropped messages into a leading summary. Omit → they are simply gone. */
|
|
2344
|
+
summarize?: HistoryPolicy['summarize'];
|
|
2345
|
+
}
|
|
2346
|
+
/**
|
|
2347
|
+
* Keep the newest messages that fit. Both limits apply when both are set — whichever cuts more wins.
|
|
2348
|
+
* Neither set is a policy that keeps everything, which is what an unconfigured loop already does.
|
|
2349
|
+
*
|
|
2350
|
+
* A naive slice is safe here because a tool exchange is ONE message: the loop persists a call's
|
|
2351
|
+
* results onto the same assistant message that made them (`toolCalls` + `toolResults`), and each
|
|
2352
|
+
* model adapter expands that into the assistant/tool pair the provider wants. So a cut can't orphan
|
|
2353
|
+
* a tool result from its call, the way it could against a provider-shaped transcript.
|
|
2354
|
+
*/
|
|
2355
|
+
declare function windowHistory(options: WindowHistoryOptions): HistoryPolicy;
|
|
2356
|
+
/**
|
|
2357
|
+
* What a summarizer is told to produce when none is supplied. Written for a reader who will continue
|
|
2358
|
+
* the conversation without seeing the messages themselves, so it asks for the parts a later turn
|
|
2359
|
+
* still has to act on rather than a readable recap.
|
|
2360
|
+
*/
|
|
2361
|
+
declare const DEFAULT_HISTORY_SUMMARY_INSTRUCTION = "Summarize this conversation for an assistant that will continue it without seeing these messages. Preserve decisions made, facts and constraints the user stated, identifiers and names mentioned, and anything left unresolved. Omit pleasantries. Reply with the summary only \u2014 no preamble, no headings.";
|
|
2362
|
+
/**
|
|
2363
|
+
* A {@link HistoryPolicy.summarize} backed by one extra, non-streamed model call — what
|
|
2364
|
+
* `AgentModule.forRoot({ history: { summarize: true } })` binds. Writes to a discarding sink so the
|
|
2365
|
+
* summary's tokens never reach the user's live stream, and reports its usage so the call lands in
|
|
2366
|
+
* the spend read-model like any other.
|
|
2367
|
+
*/
|
|
2368
|
+
declare function summarizeWithModel(model: ModelProvider, instruction?: string): NonNullable<HistoryPolicy['summarize']>;
|
|
2369
|
+
|
|
2370
|
+
/**
|
|
2371
|
+
* Running the processor chains, and the stream buffering an output gate requires. See
|
|
2372
|
+
* `./spi/processors.ts` for the seams themselves and the boundary against `HistoryPolicy`.
|
|
2373
|
+
*/
|
|
2374
|
+
|
|
2375
|
+
/**
|
|
2376
|
+
* Holds a model call's stream frames instead of letting them reach the subscriber. `end`/`fail` are
|
|
2377
|
+
* swallowed for the same reason `childSinkWriter` swallows them: the loop owns the run's stream
|
|
2378
|
+
* lifecycle across however many steps the turn takes, and the model call is one step of it.
|
|
2379
|
+
*
|
|
2380
|
+
* Frames are kept as the decoded NDJSON LINES rather than the raw `Uint8Array`s so the buffer can
|
|
2381
|
+
* ride a durable checkpoint — the gate has to survive a suspend between the model call and the
|
|
2382
|
+
* verdict, and bytes do not round-trip through JSON.
|
|
2383
|
+
*/
|
|
2384
|
+
interface FrameBuffer {
|
|
2385
|
+
writer: SinkWriter;
|
|
2386
|
+
/** Everything written so far, one entry per NDJSON line, in write order. */
|
|
2387
|
+
frames(): string[];
|
|
2388
|
+
}
|
|
2389
|
+
declare function createFrameBuffer(): FrameBuffer;
|
|
2390
|
+
/**
|
|
2391
|
+
* The chunks a passed gate releases to the live stream: the held frames, with every text frame
|
|
2392
|
+
* collapsed into ONE carrying `text` — the answer as the chain left it, which is the only version
|
|
2393
|
+
* anything downstream is allowed to see.
|
|
2394
|
+
*
|
|
2395
|
+
* A frame that does not decode is dropped along with them. It cannot be forwarded, because a gate
|
|
2396
|
+
* that forwards bytes it cannot classify is not a gate: a provider writing bare text into the sink
|
|
2397
|
+
* (rather than the `AgentStreamEvent` vocabulary) would otherwise stream the ungated answer past
|
|
2398
|
+
* the processor that was supposed to hold it. That is the cost of an output gate for a provider
|
|
2399
|
+
* outside the encoded vocabulary, and it is stated in `AgentLoopDeps.outputProcessors`.
|
|
2400
|
+
*/
|
|
2401
|
+
declare function releaseGatedFrames(frames: readonly string[], text: string): Uint8Array[];
|
|
2402
|
+
/**
|
|
2403
|
+
* Fold the prompt through each processor in order, every one seeing what the previous one produced.
|
|
2404
|
+
* A throw is wrapped so it cannot read as the model call failing — see {@link ProcessorFailedError}.
|
|
2405
|
+
*/
|
|
2406
|
+
declare function runInputProcessors(processors: readonly InputProcessor[], prompt: ProcessedPrompt, ctx: ProcessorContext): Promise<ProcessedPrompt>;
|
|
2407
|
+
/** A settled output chain: the answer as the chain left it, plus who refused it if anyone did. */
|
|
2408
|
+
interface OutputGateResult {
|
|
2409
|
+
/** The text every downstream consumer sees — the stream, the persisted message, the next step. */
|
|
2410
|
+
text: string;
|
|
2411
|
+
/** Set only on a refusal; the run ends with an `OutputRejectedError` naming these. */
|
|
2412
|
+
rejection?: {
|
|
2413
|
+
processor: string;
|
|
2414
|
+
reason: string;
|
|
2415
|
+
};
|
|
2416
|
+
}
|
|
2417
|
+
/**
|
|
2418
|
+
* Fold the answer through each processor in order. The FIRST rejection ends the chain: a later
|
|
2419
|
+
* processor has nothing to add about text that is never going anywhere, and running it would bill
|
|
2420
|
+
* a moderation call for a turn already refused.
|
|
2421
|
+
*/
|
|
2422
|
+
declare function runOutputProcessors(processors: readonly OutputProcessor[], answer: ModelAnswer, ctx: ProcessorContext): Promise<OutputGateResult>;
|
|
2423
|
+
/**
|
|
2424
|
+
* Put each suggestion through the output chain on its own, and keep what survives.
|
|
2425
|
+
*
|
|
2426
|
+
* A follow-up is model-generated text that gets persisted and rendered, so the gate has to see it.
|
|
2427
|
+
* What it does NOT get is the chain's usual answer to a refusal — ending the run. The suggestions
|
|
2428
|
+
* are produced AFTER the turn's answer has already cleared the same chain, and retracting a cleared
|
|
2429
|
+
* answer because a question nobody asked for was refused would make a run's outcome depend on a
|
|
2430
|
+
* by-product. Dropping the suggestion removes it just as completely, which is what the refusal was
|
|
2431
|
+
* for. A `replace` is honoured; a processor that empties one drops it too, since an empty suggestion
|
|
2432
|
+
* is nothing to render.
|
|
2433
|
+
*
|
|
2434
|
+
* One suggestion at a time rather than the joined list, so a refusal is confined to the one that
|
|
2435
|
+
* earned it and a `replace` cannot smear across a neighbour.
|
|
2436
|
+
*/
|
|
2437
|
+
declare function gateFollowUps(processors: readonly OutputProcessor[], followUps: readonly string[], ctx: ProcessorContext): Promise<string[]>;
|
|
2438
|
+
/**
|
|
2439
|
+
* How much of a turn's stream an output chain costs the reader.
|
|
2440
|
+
*
|
|
2441
|
+
* `off` — nothing registered, the model writes straight to the run's sink.
|
|
2442
|
+
* `whole` — at least one processor needs the complete answer, so every frame is held until the
|
|
2443
|
+
* chain has passed and the answer arrives as one `text` frame.
|
|
2444
|
+
* `incremental` — every processor declared {@link IncrementalGating}, so a lookback-bounded prefix
|
|
2445
|
+
* is released while the call streams.
|
|
2446
|
+
*/
|
|
2447
|
+
type OutputGateMode = 'off' | 'whole' | 'incremental';
|
|
2448
|
+
/**
|
|
2449
|
+
* ALL or nothing: one undeclared processor puts the whole chain on the whole-answer path. Its
|
|
2450
|
+
* author wrote `process` against the complete text, and a chain that ran it on a prefix because a
|
|
2451
|
+
* neighbour opted in would be handing it an input it never agreed to read.
|
|
2452
|
+
*/
|
|
2453
|
+
declare function resolveOutputGateMode(processors: readonly OutputProcessor[]): OutputGateMode;
|
|
2454
|
+
/**
|
|
2455
|
+
* The chain's window is the WIDEST any member asked for: a release safe for the shortest-sighted
|
|
2456
|
+
* processor is not safe for the one that matches longer patterns, and the gate makes one release
|
|
2457
|
+
* decision for all of them.
|
|
2458
|
+
*
|
|
2459
|
+
* An undeclared processor contributes nothing rather than the default — it has no window, because a
|
|
2460
|
+
* chain containing one never reaches the incremental path at all.
|
|
2461
|
+
*/
|
|
2462
|
+
declare function resolveGateLookback(processors: readonly OutputProcessor[]): number;
|
|
2463
|
+
/** A live gate over one model call: the sink the model writes to, plus what it let through. */
|
|
2464
|
+
interface IncrementalGate {
|
|
2465
|
+
/** Hand this to the model instead of the run's writer. */
|
|
2466
|
+
writer: SinkWriter;
|
|
2467
|
+
/** Resolves once every chunk handed to {@link writer} has been ruled on. */
|
|
2468
|
+
settled(): Promise<void>;
|
|
2469
|
+
/** The transformed prefix already written to the run's sink. */
|
|
2470
|
+
released(): string;
|
|
2471
|
+
/** Set once a prefix was refused; nothing further was released after that. */
|
|
2472
|
+
rejection(): OutputGateResult['rejection'];
|
|
2473
|
+
}
|
|
2474
|
+
/**
|
|
2475
|
+
* Releases a model call's answer to `writer` as it arrives, holding back the last `lookbackChars`
|
|
2476
|
+
* characters of what the chain produces so a processor can still change them.
|
|
2477
|
+
*
|
|
2478
|
+
* The chain runs over the whole PREFIX accumulated so far rather than over each new chunk, so a
|
|
2479
|
+
* processor always sees well-formed text and never has to reassemble a pattern split across
|
|
2480
|
+
* frames — which is what makes {@link IncrementalGating}'s promise a claim about prefixes. It runs
|
|
2481
|
+
* once per streamed `text` frame.
|
|
2482
|
+
*
|
|
2483
|
+
* Only an EXTENSION of what the reader already has is ever written: a chain whose output stops
|
|
2484
|
+
* agreeing with its earlier output has broken its promise, and the loop's whole-answer pass reports
|
|
2485
|
+
* that rather than this gate papering over it by re-sending a different answer.
|
|
2486
|
+
*
|
|
2487
|
+
* Frames that are not text are forwarded live, in arrival order. They are not the answer, so the
|
|
2488
|
+
* gate does not own them — but it does drop any frame it cannot decode, for the same reason
|
|
2489
|
+
* {@link releaseGatedFrames} does: a gate that forwards bytes it cannot classify is not a gate.
|
|
2490
|
+
*/
|
|
2491
|
+
declare function createIncrementalGate(options: {
|
|
2492
|
+
processors: readonly OutputProcessor[];
|
|
2493
|
+
ctx: ProcessorContext;
|
|
2494
|
+
lookbackChars: number;
|
|
2495
|
+
writer: SinkWriter;
|
|
2496
|
+
}): IncrementalGate;
|
|
2497
|
+
/**
|
|
2498
|
+
* The tail an incremental gate still owes the reader once the whole-answer pass has settled: the
|
|
2499
|
+
* authoritative text minus the prefix already released.
|
|
2500
|
+
*
|
|
2501
|
+
* Throws when the settled answer is not an extension of that prefix. That check is the reason the
|
|
2502
|
+
* final pass stays authoritative for both the stream and the store rather than the stream being
|
|
2503
|
+
* stitched together from per-prefix results: agreement between what was streamed and what was
|
|
2504
|
+
* stored becomes structural instead of assumed.
|
|
2505
|
+
*/
|
|
2506
|
+
declare function gateTail(processors: readonly OutputProcessor[], released: string, text: string): string;
|
|
2507
|
+
|
|
2508
|
+
/**
|
|
2509
|
+
* Constraining a turn's ANSWER to a schema. See `AgentLoopDeps.outputSchema` for where this sits in
|
|
2510
|
+
* the loop and why it is a separate model call rather than a constraint on the turn's own calls.
|
|
2511
|
+
*/
|
|
2512
|
+
|
|
2513
|
+
/**
|
|
2514
|
+
* The turn produced an answer that does not satisfy `outputSchema`, and the bounded repair attempts
|
|
2515
|
+
* did not fix it. A DEFINED outcome of asking for structured output, not a crash: it carries the
|
|
2516
|
+
* validation issues and the text that failed them, so a caller can log what the model actually said
|
|
2517
|
+
* instead of guessing from a parse error. Both runners map it to the `structured_output_invalid`
|
|
2518
|
+
* stream error code.
|
|
2519
|
+
*/
|
|
2520
|
+
declare class StructuredOutputError extends Error {
|
|
2521
|
+
/** Why it failed the schema. Empty only when the text was not JSON at all. */
|
|
2522
|
+
readonly issues: readonly StandardSchemaV1.Issue[];
|
|
2523
|
+
/** The last text the model produced, verbatim. */
|
|
2524
|
+
readonly text: string;
|
|
2525
|
+
/** How many model calls were spent trying (1 = the formatting pass, no repairs). */
|
|
2526
|
+
readonly attempts: number;
|
|
2527
|
+
constructor(issues: readonly StandardSchemaV1.Issue[], text: string, attempts: number);
|
|
2528
|
+
}
|
|
2529
|
+
/** A validated answer, or the issues that stopped it being one. */
|
|
2530
|
+
type StructuredOutcome<T> = {
|
|
2531
|
+
ok: true;
|
|
2532
|
+
value: T;
|
|
2533
|
+
} | {
|
|
2534
|
+
ok: false;
|
|
2535
|
+
issues: readonly StandardSchemaV1.Issue[];
|
|
2536
|
+
};
|
|
2537
|
+
/**
|
|
2538
|
+
* Pull a JSON value out of a model reply, tolerating the code fences and lead-in prose a provider
|
|
2539
|
+
* that cannot constrain its own generation still emits. `undefined` when there is no JSON in there
|
|
2540
|
+
* at all — distinct from a JSON `null`, which is a value the schema may well accept.
|
|
2541
|
+
*/
|
|
2542
|
+
declare function extractJson(text: string): unknown;
|
|
2543
|
+
/**
|
|
2544
|
+
* Validate one attempt. The provider's own parsed `object` is preferred when it reported one (a
|
|
2545
|
+
* provider that constrained generation already did the parse), but it is validated all the same —
|
|
2546
|
+
* "the provider says it matched" is not the same claim as "it matches", and a provider that ignored
|
|
2547
|
+
* `outputSchema` entirely must fail here rather than downstream.
|
|
2548
|
+
*/
|
|
2549
|
+
declare function validateStructured<T>(schema: StandardSchemaV1<unknown, T>, text: string, reported: unknown): Promise<StructuredOutcome<T>>;
|
|
2550
|
+
/**
|
|
2551
|
+
* What the formatting pass is told to do. Written for a model that is being shown a finished answer
|
|
2552
|
+
* and asked to restate it — never to extend or improve it, which would make the structured result
|
|
2553
|
+
* disagree with the prose the user already read.
|
|
2554
|
+
*/
|
|
2555
|
+
declare const DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION = "Restate the assistant's final answer as a single JSON value matching the required schema. Use only information already present in the conversation \u2014 add nothing, and answer nothing that was not asked. Reply with ONLY the JSON: no prose, no explanation, no code fences.";
|
|
2556
|
+
/** Appends the previous attempt's validation issues, so a repair call knows what to fix. */
|
|
2557
|
+
declare function repairInstruction(instruction: string, issues: readonly StandardSchemaV1.Issue[]): string;
|
|
2558
|
+
|
|
2559
|
+
/**
|
|
2560
|
+
* Skills: an authored procedure the model pulls in when a task calls for it, instead of every
|
|
2561
|
+
* instruction living in the system prompt.
|
|
2562
|
+
*
|
|
2563
|
+
* WHAT A SKILL IS NOT: an agent. An `@Agent` is WHO is answering — its persona, its tools, its
|
|
2564
|
+
* history ceiling, its output schema. A skill is HOW one particular task is done, and any agent may
|
|
2565
|
+
* pull one in. That is why a skill carries no model, no tool list and no schema: the moment it did,
|
|
2566
|
+
* the two would be the same thing wearing different names, and a consumer would have to choose
|
|
2567
|
+
* between them for reasons nobody could state.
|
|
2568
|
+
*
|
|
2569
|
+
* WHAT IT COSTS THE PROMPT: one line per skill. The catalog block below carries names, scopes and
|
|
2570
|
+
* descriptions only; a BODY reaches the model as a tool result, on the transcript, where the
|
|
2571
|
+
* `HistoryPolicy` ceiling already governs it. So skills are not a fourth thing competing for the
|
|
2572
|
+
* system block with the agent's own prompt, its contributors and injected retrieval — see
|
|
2573
|
+
* `AgentLoopDeps.skills` for how the four compose.
|
|
2574
|
+
*/
|
|
2575
|
+
|
|
2576
|
+
/** How a skill is identified in the catalog, to the model and to a `/`-autocomplete alike. */
|
|
2577
|
+
interface SkillSummary {
|
|
2578
|
+
/** Unique within a scope. The handle the model passes to the `skill` tool. */
|
|
2579
|
+
name: string;
|
|
2580
|
+
/** One line: what task this covers, and therefore when to load it. Read by the MODEL to choose. */
|
|
2581
|
+
description: string;
|
|
2582
|
+
/** The opaque scope token this skill is published at — see {@link ScopeResolver}. */
|
|
2583
|
+
scope: string;
|
|
2584
|
+
}
|
|
2585
|
+
/** A skill with the instructions themselves. Only ever materialized when something loads it. */
|
|
2586
|
+
interface Skill extends SkillSummary {
|
|
2587
|
+
body: string;
|
|
2588
|
+
}
|
|
2589
|
+
/** Whose turn is asking, and therefore which scopes apply. */
|
|
2590
|
+
interface ScopeContext {
|
|
2591
|
+
actor: Actor;
|
|
2592
|
+
threadId: string;
|
|
2593
|
+
/** The agent running this turn. Undefined → the default agent. */
|
|
2594
|
+
agentName?: string;
|
|
2595
|
+
pageContext?: PageContext;
|
|
2596
|
+
}
|
|
2597
|
+
/**
|
|
2598
|
+
* The skills-facing name for {@link ScopeContext}, so a `SkillProvider`'s signature reads in its own
|
|
2599
|
+
* vocabulary. Deliberately the SAME type rather than a parallel one: skills and memory resolve
|
|
2600
|
+
* scopes through one `ScopeResolver`, and a deployment that could answer "which scopes does this
|
|
2601
|
+
* actor have" twice would eventually answer it differently.
|
|
2602
|
+
*/
|
|
2603
|
+
type SkillContext = ScopeContext;
|
|
2604
|
+
/** The scope token every deployment has: skills nobody narrowed. */
|
|
2605
|
+
declare const GLOBAL_SCOPE = "global";
|
|
2606
|
+
/** The token for one actor's own skills. */
|
|
2607
|
+
declare function actorScope(actor: Actor): string;
|
|
2608
|
+
/** The token for a tenant's skills — `Actor.tenantRef` is the only tenant key this library knows. */
|
|
2609
|
+
declare function tenantScope(tenantRef: string): string;
|
|
2610
|
+
/**
|
|
2611
|
+
* Which scopes a turn may draw skills from, MOST SPECIFIC FIRST. Precedence is the order: a skill
|
|
2612
|
+
* from an earlier token outranks a same-named one from a later token, and the model is shown which
|
|
2613
|
+
* of them won.
|
|
2614
|
+
*
|
|
2615
|
+
* A HOST-SUPPLIED FUNCTION rather than an enum this library owns, because the axes a deployment
|
|
2616
|
+
* scopes by are the deployment's own. `Actor` gives an id and a tenant; it does not give a sector, a
|
|
2617
|
+
* squadron, a base, a shift — and every one of those is a real axis in some consumer. An enum here
|
|
2618
|
+
* would make each of them a schema change in a library that has no business knowing they exist,
|
|
2619
|
+
* while a token is a string a host mints for itself. Return `['sector:logistics', 'tenant:base-7',
|
|
2620
|
+
* 'global']` and precedence follows, with nothing in this package edited.
|
|
2621
|
+
*
|
|
2622
|
+
* MUST be a pure function of its context. It runs inside the `skills:catalog` checkpoint and its
|
|
2623
|
+
* result is journaled, so a resumed run reads back the scopes the first attempt resolved rather than
|
|
2624
|
+
* asking a membership table that may have changed since — which would otherwise let a run's prompt
|
|
2625
|
+
* differ from the one its journal records. Read a database here at your peril; read it in the
|
|
2626
|
+
* provider, whose answer is journaled at the same position.
|
|
2627
|
+
*/
|
|
2628
|
+
interface ScopeResolver {
|
|
2629
|
+
resolve(ctx: SkillContext): string[] | Promise<string[]>;
|
|
2630
|
+
}
|
|
2631
|
+
/**
|
|
2632
|
+
* The scopes derivable from an {@link Actor} alone — the common case, so a consumer wiring skills
|
|
2633
|
+
* for the first time supplies no resolver at all: the actor's own, their tenant's (when they have
|
|
2634
|
+
* one), and the deployment's. A host adding an axis of its own replaces this rather than extending
|
|
2635
|
+
* it, since the ORDER is the precedence and only the host knows where its axis belongs.
|
|
2636
|
+
*/
|
|
2637
|
+
declare const defaultScopeResolver: ScopeResolver;
|
|
2638
|
+
/**
|
|
2639
|
+
* Where skills come from. Deliberately two calls rather than one: `list` is asked on EVERY turn and
|
|
2640
|
+
* must stay cheap, while a body is read only when the model decides it needs that procedure — the
|
|
2641
|
+
* whole point of the surface. A provider over a table selects name/description/scope for the first
|
|
2642
|
+
* and one row for the second.
|
|
2643
|
+
*
|
|
2644
|
+
* This library owns no skill table. A skill's real scoping axes, its authoring UI and its audit
|
|
2645
|
+
* trail are all the host's, and a host that related its own `Sector` entity into a table this
|
|
2646
|
+
* package created at boot would be writing migrations against a schema the boot-time heal also
|
|
2647
|
+
* edits. Tokens split it the other way round: the library owns the contract (what a scope means, how
|
|
2648
|
+
* precedence works, what is journaled), the host owns the rows.
|
|
2649
|
+
*/
|
|
2650
|
+
/** Arguments to {@link SkillProvider.list}. */
|
|
2651
|
+
interface ListSkillsInput {
|
|
2652
|
+
scopes: readonly string[];
|
|
2653
|
+
ctx: SkillContext;
|
|
2654
|
+
}
|
|
2655
|
+
/**
|
|
2656
|
+
* Arguments to {@link SkillProvider.load}. An object rather than positional arguments because
|
|
2657
|
+
* `name` and `scope` are both strings: transposed, a positional call compiles clean, returns `null`,
|
|
2658
|
+
* and the skill silently fails to load.
|
|
2659
|
+
*/
|
|
2660
|
+
interface LoadSkillInput {
|
|
2661
|
+
name: string;
|
|
2662
|
+
scope: string;
|
|
2663
|
+
ctx: SkillContext;
|
|
2664
|
+
}
|
|
2665
|
+
interface SkillProvider {
|
|
2666
|
+
/**
|
|
2667
|
+
* Every skill visible at `scopes`, in any order — this library sorts and resolves precedence. A
|
|
2668
|
+
* provider MAY return skills outside `scopes`; they are dropped rather than trusted, so a filter
|
|
2669
|
+
* bug in a host cannot widen what an actor is offered.
|
|
2670
|
+
*/
|
|
2671
|
+
list(input: ListSkillsInput): SkillSummary[] | Promise<SkillSummary[]>;
|
|
2672
|
+
/**
|
|
2673
|
+
* The instructions for one skill, or `null` when it is gone. Called only from inside the loading
|
|
2674
|
+
* checkpoint, so the body it returns is journaled and every replay reads THAT text back — an
|
|
2675
|
+
* edited skill never rewrites the prompt of a run already in flight.
|
|
2676
|
+
*/
|
|
2677
|
+
load(input: LoadSkillInput): string | null | Promise<string | null>;
|
|
2678
|
+
}
|
|
2679
|
+
/** A source of skills that is a fixed list — `@Skill()`-decorated classes, or plain config. */
|
|
2680
|
+
declare function staticSkillProvider(skills: readonly Skill[]): SkillProvider;
|
|
2681
|
+
/**
|
|
2682
|
+
* Read several sources as one. The order is the tie-break and nothing else: precedence between
|
|
2683
|
+
* skills is by SCOPE, so two sources only ever compete when they publish the same name at the same
|
|
2684
|
+
* scope, and then the earlier source wins. Wire the host's own provider first — a row it can edit is
|
|
2685
|
+
* a better answer than a body that needs a deploy to change.
|
|
2686
|
+
*/
|
|
2687
|
+
declare function compositeSkillProvider(providers: readonly SkillProvider[]): SkillProvider;
|
|
2688
|
+
/**
|
|
2689
|
+
* One skill as it is offered — to the model in the catalog block, and to a client over
|
|
2690
|
+
* `GET /agent/skills`. Both read the same list, built by the same call, so what a user can type
|
|
2691
|
+
* after a `/` and what the model can reach cannot drift apart.
|
|
2692
|
+
*/
|
|
2693
|
+
interface SkillCatalogEntry {
|
|
2694
|
+
name: string;
|
|
2695
|
+
description: string;
|
|
2696
|
+
/** The scope token it resolved from — the provenance a user is entitled to see. */
|
|
2697
|
+
scope: string;
|
|
2698
|
+
/**
|
|
2699
|
+
* Scope tokens of same-named skills this one outranks, widest last. Present ONLY when something
|
|
2700
|
+
* was shadowed, so a reader can tell "there is no org default" from "there is one and yours wins".
|
|
2701
|
+
* The model is shown it for the same reason: a silent override is indistinguishable from an
|
|
2702
|
+
* instruction nobody wrote, and the user can only be told "your setting differs from the org
|
|
2703
|
+
* default" by something that knows both existed.
|
|
2704
|
+
*/
|
|
2705
|
+
shadows?: string[];
|
|
2706
|
+
}
|
|
2707
|
+
/** What a turn resolved — journaled whole, so the prompt is reconstructible from the journal alone. */
|
|
2708
|
+
interface SkillOffer {
|
|
2709
|
+
/** The scope tokens this turn drew from, most specific first. */
|
|
2710
|
+
scopes: string[];
|
|
2711
|
+
/** The skills offered, most specific first then by name. */
|
|
2712
|
+
entries: SkillCatalogEntry[];
|
|
2713
|
+
/** Applicable skills `maxSkills` left out. Non-zero means the catalog is not the whole truth. */
|
|
2714
|
+
omitted: number;
|
|
2715
|
+
}
|
|
2716
|
+
/**
|
|
2717
|
+
* How many skills a catalog offers before it starts leaving some out. A ceiling on the SYSTEM
|
|
2718
|
+
* PROMPT's share, not on how many a deployment may have: the block costs one line each, and a
|
|
2719
|
+
* hundred lines of "here is something you could load" crowds out the agent's own instructions while
|
|
2720
|
+
* making the choice harder rather than easier.
|
|
2721
|
+
*/
|
|
2722
|
+
declare const DEFAULT_MAX_SKILLS = 20;
|
|
2723
|
+
/** How a turn reaches its skills. See `AgentLoopDeps.skills`. */
|
|
2724
|
+
interface SkillsConfig {
|
|
2725
|
+
provider: SkillProvider;
|
|
2726
|
+
/** Undefined → {@link defaultScopeResolver}: the actor's own, their tenant's, the deployment's. */
|
|
2727
|
+
scopes?: ScopeResolver;
|
|
2728
|
+
/** Undefined → {@link DEFAULT_MAX_SKILLS}. */
|
|
2729
|
+
maxSkills?: number;
|
|
2730
|
+
}
|
|
2731
|
+
/**
|
|
2732
|
+
* Resolve a provider's skills against an ordered scope list: most specific wins, and the loser's
|
|
2733
|
+
* scope is recorded rather than discarded.
|
|
2734
|
+
*
|
|
2735
|
+
* Pure, and separately exported, because two callers must reach the identical answer — the loop, so
|
|
2736
|
+
* the model is offered it, and the listing endpoint, so a user is. A second implementation of
|
|
2737
|
+
* "which skills apply" is a second answer, and the two diverge the first time either is edited.
|
|
2738
|
+
*/
|
|
2739
|
+
declare function resolveSkillCatalog(summaries: readonly SkillSummary[], scopes: readonly string[], maxSkills?: number): Omit<SkillOffer, 'scopes'>;
|
|
2740
|
+
/** Resolve the scopes and the catalog for one turn. Called INSIDE the loop's `skills:catalog` step. */
|
|
2741
|
+
declare function offerSkills(config: SkillsConfig, ctx: SkillContext): Promise<SkillOffer>;
|
|
2742
|
+
/**
|
|
2743
|
+
* The catalog as the model reads it. One line per skill — name, scope, description, and what it
|
|
2744
|
+
* overrides — because the entire claim of progressive disclosure is that choosing what to read costs
|
|
2745
|
+
* far less than reading everything.
|
|
2746
|
+
*/
|
|
2747
|
+
declare function buildSkillsBlock(entries: readonly SkillCatalogEntry[]): string;
|
|
2748
|
+
/** The reserved name of the built-in skill-loading tool. */
|
|
2749
|
+
declare const SKILL_TOOL_NAME = "skill";
|
|
2750
|
+
/** What the model passes to the `skill` tool. */
|
|
2751
|
+
interface SkillToolInput {
|
|
2752
|
+
name: string;
|
|
2753
|
+
}
|
|
2754
|
+
/**
|
|
2755
|
+
* The `skill` tool's input schema. Hand-written for the same reason `askInputSchema` is: core
|
|
2756
|
+
* depends on no validator and the schema has to publish a JSON Schema a provider can constrain
|
|
2757
|
+
* generation against.
|
|
2758
|
+
*
|
|
2759
|
+
* The available names are deliberately NOT an enum here. A per-turn enum would make the tool's
|
|
2760
|
+
* definition depend on the catalog — and the dispatched llm step re-derives the tool list on
|
|
2761
|
+
* whichever worker serves it, from ITS OWN provider, which is exactly the process-local lookup this
|
|
2762
|
+
* design keeps off the turn's decisions. The names live in the prompt, where the journal holds them;
|
|
2763
|
+
* a name that is not in the catalog comes back as an ordinary tool failure listing the ones that are.
|
|
2764
|
+
*/
|
|
2765
|
+
declare const skillInputSchema: StandardSchemaV1<unknown, SkillToolInput>;
|
|
2766
|
+
declare const SKILL_TOOL_DESCRIPTION = "Read one of the procedures listed in <skills>. Call it before doing a task a listed skill covers, and follow what it returns. It performs nothing and changes nothing \u2014 it only gives you instructions you do not yet have.";
|
|
2767
|
+
/**
|
|
2768
|
+
* The `skill` tool as the model sees it. NOT a `ToolSpec` and never registered, exactly like `ask`:
|
|
2769
|
+
* it has no handler, because the loop serves it from the catalog the journal holds. Keeping it out
|
|
2770
|
+
* of the `ToolRegistry` is also what keeps its kind off a process-local lookup — see `claimToolCall`.
|
|
2771
|
+
*/
|
|
2772
|
+
declare function skillToolDefinition(): ToolDefinition;
|
|
2773
|
+
/**
|
|
2774
|
+
* Append the built-in `skill` definition to a turn's tool list. Exported because the dispatched llm
|
|
2775
|
+
* step re-derives the tool list on a worker and has to reach the same list the loop would have.
|
|
2776
|
+
*/
|
|
2777
|
+
declare function withSkillTool({ tools, enabled, }: {
|
|
2778
|
+
tools: ToolDefinition[];
|
|
2779
|
+
enabled: boolean;
|
|
2780
|
+
}): ToolDefinition[];
|
|
2781
|
+
/** What a `skill` call resolved to — the body, or why it did not. */
|
|
2782
|
+
type SkillLoadOutcome = {
|
|
2783
|
+
ok: true;
|
|
2784
|
+
skill: Skill;
|
|
2785
|
+
shadows?: string[];
|
|
2786
|
+
} | {
|
|
2787
|
+
ok: false;
|
|
2788
|
+
error: string;
|
|
2789
|
+
};
|
|
2790
|
+
/**
|
|
2791
|
+
* Serve one `skill` call against the catalog THIS TURN was offered.
|
|
2792
|
+
*
|
|
2793
|
+
* The catalog is the authorization boundary, not just a menu: a name the turn was never offered is
|
|
2794
|
+
* refused here, so a model that invents one (or repeats one it saw in another actor's thread) cannot
|
|
2795
|
+
* reach a body through a provider that would happily serve it. Because the catalog came out of the
|
|
2796
|
+
* `skills:catalog` checkpoint, that boundary is the one the journal records rather than the one this
|
|
2797
|
+
* process's provider currently believes in.
|
|
2798
|
+
*/
|
|
2799
|
+
declare function loadSkill(config: SkillsConfig, offer: SkillOffer, name: string, ctx: SkillContext): Promise<SkillLoadOutcome>;
|
|
2800
|
+
/** Who is trying to write a skill. */
|
|
2801
|
+
interface SkillAuthor {
|
|
2802
|
+
/**
|
|
2803
|
+
* `'human'` is a person acting through a UI; `'agent'` is anything else — a tool, a turn, a
|
|
2804
|
+
* batch job. The distinction is the whole of the rule below, so it is not inferable and has to
|
|
2805
|
+
* be stated by the caller.
|
|
2806
|
+
*/
|
|
2807
|
+
kind: 'human' | 'agent';
|
|
2808
|
+
actorRef?: string;
|
|
2809
|
+
}
|
|
2810
|
+
/** A request to author a skill at a scope. */
|
|
2811
|
+
interface SkillWriteRequest {
|
|
2812
|
+
/** The scope token the skill would be published at. */
|
|
2813
|
+
scope: string;
|
|
2814
|
+
actor: Actor;
|
|
2815
|
+
/** The actor's resolved scopes, most specific first — the same list a turn draws from. */
|
|
2816
|
+
scopes: readonly string[];
|
|
2817
|
+
author: SkillAuthor;
|
|
2818
|
+
/**
|
|
2819
|
+
* The host's own answer to "may this person administer that scope" — a sector lead editing their
|
|
2820
|
+
* sector's skills, an operator editing the deployment's. This library cannot know it: it has no
|
|
2821
|
+
* notion of who administers `sector:logistics`, and inventing one would be a second, weaker
|
|
2822
|
+
* authorization model next to the host's real one. Undefined/false → a wider scope is refused.
|
|
2823
|
+
*/
|
|
2824
|
+
elevated?: boolean;
|
|
2825
|
+
}
|
|
2826
|
+
type SkillWriteVerdict = {
|
|
2827
|
+
allowed: true;
|
|
2828
|
+
} | {
|
|
2829
|
+
allowed: false;
|
|
2830
|
+
reason: string;
|
|
2831
|
+
};
|
|
2832
|
+
/**
|
|
2833
|
+
* May this author publish a skill at this scope?
|
|
2834
|
+
*
|
|
2835
|
+
* A skill's body is instructions the model follows, which makes writing one at a scope an edit to
|
|
2836
|
+
* everyone in that scope's system prompt. The rules follow from that, and only the third is
|
|
2837
|
+
* interesting:
|
|
2838
|
+
*
|
|
2839
|
+
* 1. You may only write into a scope you are yourself in. Writing into a scope you are not in is
|
|
2840
|
+
* authoring for people you have no relationship with.
|
|
2841
|
+
* 2. Your OWN scope is yours. A per-actor skill affects exactly one prompt — the author's.
|
|
2842
|
+
* 3. NOTHING BUT A HUMAN MAY WRITE ABOVE ITS OWN SCOPE, whatever `elevated` says. An agent that can
|
|
2843
|
+
* write a `tenant:` skill is an agent whose prompt anyone in the tenant can edit by talking to
|
|
2844
|
+
* it: the user asks for something, the turn writes the instruction, and every later turn for
|
|
2845
|
+
* every other user follows it. That is prompt injection with persistence, and no elevation a
|
|
2846
|
+
* host could grant makes it a different shape. An agent that has genuinely learned something
|
|
2847
|
+
* worth sharing proposes it; a person publishes it.
|
|
2848
|
+
* 4. A human writing above their own scope needs the host to say so (`elevated`).
|
|
2849
|
+
*/
|
|
2850
|
+
declare function skillWriteVerdict(request: SkillWriteRequest): SkillWriteVerdict;
|
|
2851
|
+
|
|
2852
|
+
/**
|
|
2853
|
+
* Memory: what the assistant has concluded about a person or an organisation, carried across turns
|
|
2854
|
+
* and across threads.
|
|
2855
|
+
*
|
|
2856
|
+
* HOW A READER TELLS IT FROM RAG. Retrieval answers "what do the documents say"; memory answers
|
|
2857
|
+
* "what did I decide about you". A passage is content a person authored and can correct at its
|
|
2858
|
+
* source, and it is cited. A memory has no source to go and fix: it is the agent's own inference,
|
|
2859
|
+
* which is why every record carries an {@link MemoryOrigin} (who wrote it, in which conversation),
|
|
2860
|
+
* why the block tells the model to treat it as fallible, and why `forget` is a REQUIRED method on
|
|
2861
|
+
* the provider rather than an optional one. A wrong document is a content problem. A wrong memory is
|
|
2862
|
+
* the assistant being confidently wrong about someone with nobody aware it is there, so the whole
|
|
2863
|
+
* design is arranged around making it visible and removable.
|
|
2864
|
+
*
|
|
2865
|
+
* WHAT IT COSTS THE PROMPT. One line per memory, capped by `maxMemories`, and each line capped by
|
|
2866
|
+
* `maxFactChars` at WRITE time — so the block a turn can carry is the product of two numbers an
|
|
2867
|
+
* operator sets, rather than however much the model felt like writing down. Unlike a skill, a memory
|
|
2868
|
+
* has no body/catalog split: a fact that cannot be stated in a line is not a memory, it is a
|
|
2869
|
+
* document, and documents are retrieval's job.
|
|
2870
|
+
*
|
|
2871
|
+
* WHAT IT IS NOT: recall over a TRANSCRIPT. Searching what was said earlier is retrieval, and this
|
|
2872
|
+
* library already has `Retriever`/`Reranker` for that.
|
|
2873
|
+
*
|
|
2874
|
+
* RECALL OVER THE MEMORIES THEMSELVES IS A DIFFERENT QUESTION, and it is answered here. The prompt
|
|
2875
|
+
* budget must be bounded; the store has no reason to be. A person accumulates preferences over
|
|
2876
|
+
* months and an organisation publishes facts for everyone in it, so the applicable set outgrows
|
|
2877
|
+
* `maxMemories` quickly — and once it does, WHICH of them the block carries is a decision somebody
|
|
2878
|
+
* has to make. Making it by scope starves the widest scopes first, which is precisely backwards: the
|
|
2879
|
+
* facts that apply to the most people are the ones nobody sees. So a provider MAY supply
|
|
2880
|
+
* {@link MemoryProvider.search}, and where it does, the block is filled by relevance to the turn
|
|
2881
|
+
* (see {@link resolveMemoryDigest}) with scope still a hard filter and never a ranking signal.
|
|
2882
|
+
* `maxMemories` then means "how many matter right now" rather than "how many a person may have".
|
|
2883
|
+
*/
|
|
2884
|
+
|
|
2885
|
+
/** Who wrote a memory, and out of what. Kept so a person reading it back can ask "says who?". */
|
|
2886
|
+
interface MemoryOrigin {
|
|
2887
|
+
/** `'agent'` — a turn concluded it. `'human'` — a person wrote it in the host's own console. */
|
|
2888
|
+
author: 'agent' | 'human';
|
|
2889
|
+
/**
|
|
2890
|
+
* The conversation the conclusion was drawn in. MAY name a thread whose messages the history
|
|
2891
|
+
* ceiling has since dropped: a memory deliberately outlives its source, so this is a pointer that
|
|
2892
|
+
* is allowed to dangle, and a read-back that cannot resolve it says so rather than hiding the row.
|
|
2893
|
+
*/
|
|
2894
|
+
threadId?: string;
|
|
2895
|
+
runId?: string;
|
|
2896
|
+
actorRef?: string;
|
|
2897
|
+
}
|
|
2898
|
+
/** One thing the assistant believes, at one scope. */
|
|
2899
|
+
interface MemoryRecord {
|
|
2900
|
+
/** The host's row id — what `forget` takes and what a read-back offers a delete button for. */
|
|
2901
|
+
id: string;
|
|
2902
|
+
/**
|
|
2903
|
+
* What the fact is ABOUT. The handle that makes a conflict mechanically detectable: two memories
|
|
2904
|
+
* sharing a key at different scopes are one question answered twice, and the narrower wins. A
|
|
2905
|
+
* keyless store could only ever hope its facts did not contradict each other.
|
|
2906
|
+
*/
|
|
2907
|
+
key: string;
|
|
2908
|
+
/** The fact itself, as the model reads it. */
|
|
2909
|
+
text: string;
|
|
2910
|
+
/** The opaque scope token it is held at — see `ScopeResolver` in `skills.ts`. */
|
|
2911
|
+
scope: string;
|
|
2912
|
+
origin: MemoryOrigin;
|
|
2913
|
+
/** ISO-8601, so the ceiling can keep the newest with a string comparison it can do purely. */
|
|
2914
|
+
updatedAt: string;
|
|
2915
|
+
/**
|
|
2916
|
+
* Always-on: this fact is in the block whether or not the turn is about it. The difference between
|
|
2917
|
+
* working memory and recall, drawn per record — "they report on the calendar year" must not depend
|
|
2918
|
+
* on the turn mentioning dates, while "they prefer the shorter runway" can wait until it comes up.
|
|
2919
|
+
*
|
|
2920
|
+
* A PROPERTY OF THE ROW, NOT OF A WRITE. This library reads it and never sets it, the same way it
|
|
2921
|
+
* never mints an `id`: an agent deciding its own conclusions are always-on is an agent deciding
|
|
2922
|
+
* how much of every future prompt it gets, so pinning is an operator's act in the host's console.
|
|
2923
|
+
* That is also why {@link StoreMemoryInput} carries no `pinned` — the `remember` tool has no
|
|
2924
|
+
* parameter to refuse. A host whose upsert drops the flag has silently unpinned the row; preserve
|
|
2925
|
+
* it across an upsert on (`scope`, `key`).
|
|
2926
|
+
*/
|
|
2927
|
+
pinned?: boolean;
|
|
2928
|
+
}
|
|
2929
|
+
/** A same-key memory a narrower scope outranked, with the value it holds. */
|
|
2930
|
+
interface OverriddenMemory {
|
|
2931
|
+
scope: string;
|
|
2932
|
+
text: string;
|
|
2933
|
+
/**
|
|
2934
|
+
* Who asserted the beaten value. Travels because precedence is blind to it: an agent's own
|
|
2935
|
+
* inference at `actor:` outranks an administrator's published policy at `global`, and an agent
|
|
2936
|
+
* that knew only that a wider value existed could not tell the two apart — one is a stale guess,
|
|
2937
|
+
* the other is what the organisation decided.
|
|
2938
|
+
*/
|
|
2939
|
+
author: MemoryOrigin['author'];
|
|
2940
|
+
}
|
|
2941
|
+
/** One memory as the model and a read-back both meet it. */
|
|
2942
|
+
interface MemoryDigestEntry extends MemoryRecord {
|
|
2943
|
+
/**
|
|
2944
|
+
* Same-key memories this one outranks, widest last. Present ONLY where something was outranked.
|
|
2945
|
+
*
|
|
2946
|
+
* Carries the beaten TEXT, which is where memory departs from a skill's `shadows`. A skill only
|
|
2947
|
+
* has to say which scope it beat — the agent follows one procedure either way. A memory is a
|
|
2948
|
+
* VALUE, and an agent that knows only that a wider value existed cannot tell the user what the
|
|
2949
|
+
* difference is; it can only choose, silently, which is the thing this is meant to prevent.
|
|
2950
|
+
*/
|
|
2951
|
+
overrides?: OverriddenMemory[];
|
|
2952
|
+
}
|
|
2953
|
+
/** What a turn resolved — journaled whole, so its prompt is reconstructible from its journal alone. */
|
|
2954
|
+
interface MemoryDigest {
|
|
2955
|
+
/** The scope tokens this turn drew from, most specific first. */
|
|
2956
|
+
scopes: string[];
|
|
2957
|
+
/** The memories in the block, most specific first then by key. */
|
|
2958
|
+
entries: MemoryDigestEntry[];
|
|
2959
|
+
/**
|
|
2960
|
+
* Applicable memories `maxMemories` left out. Non-zero means the block is not the whole truth —
|
|
2961
|
+
* and under {@link recalled} it counts what the SEARCH offered and the ceiling dropped, not what
|
|
2962
|
+
* the store holds, because nothing asked the store for a total.
|
|
2963
|
+
*/
|
|
2964
|
+
omitted: number;
|
|
2965
|
+
/**
|
|
2966
|
+
* Of {@link omitted}, how many were ALWAYS-ON. Reported separately because it means something else
|
|
2967
|
+
* entirely: ordinary omission is the budget doing its job, while an always-on memory the ceiling
|
|
2968
|
+
* dropped is a deployment's own standing policies having silently stopped reaching any prompt.
|
|
2969
|
+
* Non-zero is a misconfiguration — more was pinned than the block holds — and the fix is to unpin
|
|
2970
|
+
* something or raise `maxMemories`, not to wait for it to come back.
|
|
2971
|
+
*/
|
|
2972
|
+
pinnedOmitted: number;
|
|
2973
|
+
/**
|
|
2974
|
+
* Whether the entries were selected by relevance to this turn rather than read whole. The model is
|
|
2975
|
+
* told (see {@link buildMemoryBlock}), because a partial set read as a whole one turns an absence
|
|
2976
|
+
* into evidence: "you never told me that" about a fact it simply was not shown.
|
|
2977
|
+
*
|
|
2978
|
+
* Optional so a digest journaled before recall existed reads back as the plain scoped read it was.
|
|
2979
|
+
*/
|
|
2980
|
+
recalled?: boolean;
|
|
2981
|
+
}
|
|
2982
|
+
/** What a write asks the host to store. `id` and `updatedAt` are the host's to mint. */
|
|
2983
|
+
interface MemoryFact {
|
|
2984
|
+
key: string;
|
|
2985
|
+
text: string;
|
|
2986
|
+
scope: string;
|
|
2987
|
+
origin: MemoryOrigin;
|
|
2988
|
+
}
|
|
2989
|
+
/** Arguments to {@link MemoryProvider.list}. */
|
|
2990
|
+
interface ListMemoriesInput {
|
|
2991
|
+
/** The scope tokens that apply to this turn, most specific first. */
|
|
2992
|
+
scopes: readonly string[];
|
|
2993
|
+
ctx: ScopeContext;
|
|
2994
|
+
}
|
|
2995
|
+
/**
|
|
2996
|
+
* Arguments to {@link MemoryProvider.search}.
|
|
2997
|
+
*
|
|
2998
|
+
* `scopes` GATES, AND IT GATES FIRST. A search that ranks before it filters is a cross-tenant leak
|
|
2999
|
+
* wearing a relevance score: the nearest neighbour to "what is our rollback policy" is another
|
|
3000
|
+
* base's rollback policy. Filter in the query, not after it. Records returned outside `scopes` are
|
|
3001
|
+
* dropped rather than trusted, so a mistake here costs throughput rather than privacy — but the
|
|
3002
|
+
* drop is a backstop, not the boundary.
|
|
3003
|
+
*/
|
|
3004
|
+
interface SearchMemoriesInput {
|
|
3005
|
+
/** The scope tokens that apply to this turn, most specific first. A HARD FILTER. */
|
|
3006
|
+
scopes: readonly string[];
|
|
3007
|
+
/** What this turn is about. Never blank — a blank query is served by `list` instead. */
|
|
3008
|
+
query: string;
|
|
3009
|
+
/**
|
|
3010
|
+
* At most this many distinct KEYS may be ranked in, which is the block's own unit: precedence
|
|
3011
|
+
* resolves a key to one line, so `limit` keys is `limit` lines. Pinned records are returned in
|
|
3012
|
+
* addition and do not count against it.
|
|
3013
|
+
*/
|
|
3014
|
+
limit: number;
|
|
3015
|
+
ctx: ScopeContext;
|
|
3016
|
+
}
|
|
3017
|
+
/** Arguments to {@link MemoryProvider.forget}. */
|
|
3018
|
+
interface ForgetMemoryInput {
|
|
3019
|
+
id: string;
|
|
3020
|
+
ctx: ScopeContext;
|
|
3021
|
+
}
|
|
3022
|
+
/**
|
|
3023
|
+
* Arguments to {@link MemoryProvider.write}. An object rather than positional arguments because
|
|
3024
|
+
* `key`, `text` and `scope` are all strings: transposed, a positional call compiles clean and writes
|
|
3025
|
+
* a fact whose key is its value, at a scope nobody meant.
|
|
3026
|
+
*/
|
|
3027
|
+
interface StoreMemoryInput extends MemoryFact {
|
|
3028
|
+
ctx: ScopeContext;
|
|
3029
|
+
}
|
|
3030
|
+
/**
|
|
3031
|
+
* Where memories live. The host owns the rows, for the same reason it owns skill rows: the axes a
|
|
3032
|
+
* deployment scopes by, the console that edits them and the audit trail are all the host's, and a
|
|
3033
|
+
* table this package created at boot would put a consumer's migrations in the path of the schema
|
|
3034
|
+
* heal that manages the `agent_*` tables.
|
|
3035
|
+
*
|
|
3036
|
+
* `forget` is REQUIRED, unlike `write`. A deployment may reasonably populate memory from its own
|
|
3037
|
+
* pipeline and offer the agent no write tool; no deployment may reasonably hold conclusions about a
|
|
3038
|
+
* person that the person cannot have deleted. Making it optional would make forgetting a wiring
|
|
3039
|
+
* choice, and it is not one.
|
|
3040
|
+
*/
|
|
3041
|
+
interface MemoryProvider {
|
|
3042
|
+
/**
|
|
3043
|
+
* Every memory visible at `scopes`, in any order — this library resolves precedence and the
|
|
3044
|
+
* ceiling. Asked on EVERY turn, so keep it cheap. A provider MAY return records outside `scopes`;
|
|
3045
|
+
* they are dropped rather than trusted, so a filter bug in a host cannot widen what an actor sees.
|
|
3046
|
+
*/
|
|
3047
|
+
list(input: ListMemoriesInput): MemoryRecord[] | Promise<MemoryRecord[]>;
|
|
3048
|
+
/** Delete one record by id. `false` where there was nothing by that id to delete. */
|
|
3049
|
+
forget(input: ForgetMemoryInput): boolean | Promise<boolean>;
|
|
3050
|
+
/**
|
|
3051
|
+
* Upsert on (`scope`, `key`) and return the stored record. Omit to serve memory read-only — the
|
|
3052
|
+
* `remember` tool is then never offered, and a turn's tool list is the same on every pod.
|
|
3053
|
+
*/
|
|
3054
|
+
write?(input: StoreMemoryInput): MemoryRecord | Promise<MemoryRecord>;
|
|
3055
|
+
/**
|
|
3056
|
+
* The memories worth putting in front of a turn about `query`, MOST RELEVANT FIRST. Omit and every
|
|
3057
|
+
* turn is served by {@link list} — a deployment with twenty memories must not have to stand up an
|
|
3058
|
+
* index to keep working, and one with two thousand uses whatever it already runs (pgvector, a
|
|
3059
|
+
* full-text index, a hybrid service). The host owns the index for the same reason it owns the
|
|
3060
|
+
* rows.
|
|
3061
|
+
*
|
|
3062
|
+
* THREE CLAUSES, and the third is the one that is easy to miss:
|
|
3063
|
+
*
|
|
3064
|
+
* 1. Filter to `scopes` before ranking — see {@link SearchMemoriesInput}.
|
|
3065
|
+
* 2. Rank the KEYS visible at those scopes and take the best `limit` of them.
|
|
3066
|
+
* 3. Return EVERY record sharing a returned key, plus every `pinned` record at those scopes.
|
|
3067
|
+
*
|
|
3068
|
+
* Clause three is what keeps precedence intact. Precedence resolves a conflict between two records
|
|
3069
|
+
* at one key, narrower winning and the beaten value riding along; a search that returned the
|
|
3070
|
+
* `global` half of a conflict and not the `actor:` half would render the org default as the answer
|
|
3071
|
+
* — the exact failure memory exists to prevent, and one nothing downstream can detect. It is one
|
|
3072
|
+
* query either way:
|
|
3073
|
+
*
|
|
3074
|
+
* ```sql
|
|
3075
|
+
* SELECT * FROM agent_memory
|
|
3076
|
+
* WHERE scope IN (:scopes)
|
|
3077
|
+
* AND (pinned OR key IN (SELECT key FROM agent_memory
|
|
3078
|
+
* WHERE scope IN (:scopes)
|
|
3079
|
+
* ORDER BY embedding <=> :queryVector
|
|
3080
|
+
* LIMIT :limit))
|
|
3081
|
+
* ```
|
|
3082
|
+
*/
|
|
3083
|
+
search?(input: SearchMemoriesInput): MemoryRecord[] | Promise<MemoryRecord[]>;
|
|
3084
|
+
}
|
|
3085
|
+
/**
|
|
3086
|
+
* How many memories the block carries. The same ceiling shape as `maxSkills`, and for the same
|
|
3087
|
+
* reason: a prompt that grows with how much the agent has written down is a prompt whose cost nobody
|
|
3088
|
+
* set. A budget on the PROMPT and never on the store — with a provider that can
|
|
3089
|
+
* {@link MemoryProvider.search}, it is how many matter right now rather than how many a person may
|
|
3090
|
+
* have.
|
|
3091
|
+
*/
|
|
3092
|
+
declare const DEFAULT_MAX_MEMORIES = 20;
|
|
3093
|
+
/**
|
|
3094
|
+
* How long one memory may be, enforced when it is WRITTEN rather than when it is rendered. Enforcing
|
|
3095
|
+
* it at render time would make the prompt disagree with the store; enforcing it at write time makes
|
|
3096
|
+
* the block's whole ceiling a multiplication an operator can do — `maxMemories × maxFactChars` — and
|
|
3097
|
+
* pushes back on the model at the moment it is writing an essay instead of a fact.
|
|
3098
|
+
*/
|
|
3099
|
+
declare const DEFAULT_MAX_FACT_CHARS = 240;
|
|
3100
|
+
/** How a turn reaches its memory. See `AgentLoopDeps.memory`. */
|
|
3101
|
+
interface MemoryConfig {
|
|
3102
|
+
provider: MemoryProvider;
|
|
3103
|
+
/** Undefined → `defaultScopeResolver`: the actor's own, their tenant's, the deployment's. */
|
|
3104
|
+
scopes?: ScopeResolver;
|
|
3105
|
+
/** Undefined → {@link DEFAULT_MAX_MEMORIES}. Always-on memories are taken from it first. */
|
|
3106
|
+
maxMemories?: number;
|
|
3107
|
+
/** Undefined → {@link DEFAULT_MAX_FACT_CHARS}. */
|
|
3108
|
+
maxFactChars?: number;
|
|
3109
|
+
}
|
|
3110
|
+
/**
|
|
3111
|
+
* Resolve a provider's records against an ordered scope list: most specific wins a key, and the
|
|
3112
|
+
* losers' values are carried rather than discarded.
|
|
3113
|
+
*
|
|
3114
|
+
* Pure, and separately exported, because the loop and the read-back endpoint must reach the same
|
|
3115
|
+
* answer — a person has to be shown what the model was shown, or "see what it believes about you"
|
|
3116
|
+
* means nothing.
|
|
3117
|
+
*/
|
|
3118
|
+
/** Arguments to {@link resolveMemoryDigest}. */
|
|
3119
|
+
interface ResolveMemoryDigestInput {
|
|
3120
|
+
records: readonly MemoryRecord[];
|
|
3121
|
+
/** The scope tokens that apply, most specific first — the precedence order. */
|
|
3122
|
+
scopes: readonly string[];
|
|
3123
|
+
/** Undefined → {@link DEFAULT_MAX_MEMORIES}. */
|
|
3124
|
+
maxMemories?: number;
|
|
3125
|
+
/**
|
|
3126
|
+
* Whether `records` arrived most-relevant-first, from {@link MemoryProvider.search}. A key's rank
|
|
3127
|
+
* is its best-placed record's. Undefined/false → the ceiling selects by scope and recency, which
|
|
3128
|
+
* is the right order only while nothing is being starved.
|
|
3129
|
+
*/
|
|
3130
|
+
ranked?: boolean;
|
|
3131
|
+
}
|
|
3132
|
+
declare function resolveMemoryDigest({ records, scopes, maxMemories, ranked, }: ResolveMemoryDigestInput): Omit<MemoryDigest, 'scopes' | 'recalled'>;
|
|
3133
|
+
/** Arguments to {@link offerMemories}. */
|
|
3134
|
+
interface OfferMemoriesInput {
|
|
3135
|
+
config: MemoryConfig;
|
|
3136
|
+
ctx: ScopeContext;
|
|
3137
|
+
/**
|
|
3138
|
+
* What this turn is about — the loop passes the user's own message, and nothing else. Given, and
|
|
3139
|
+
* with a provider that has an index, the block is filled by relevance to it. Absent, the scope is
|
|
3140
|
+
* read whole: that is what the read-back endpoint wants, since a person must be shown every belief
|
|
3141
|
+
* held about them rather than the slice one turn happened to need.
|
|
3142
|
+
*/
|
|
3143
|
+
query?: string;
|
|
3144
|
+
}
|
|
3145
|
+
/**
|
|
3146
|
+
* Resolve the scopes and the digest for one turn. Called INSIDE the loop's `memory:digest` step, and
|
|
3147
|
+
* that placement is the whole of its determinism: a search is the most re-derivable thing in this
|
|
3148
|
+
* library — the index moves, a neighbour is written, embeddings are recomputed — so its result has
|
|
3149
|
+
* to be part of the checkpoint every replay reads back, never something a replaying process asks
|
|
3150
|
+
* again.
|
|
3151
|
+
*/
|
|
3152
|
+
declare function offerMemories({ config, ctx, query, }: OfferMemoriesInput): Promise<MemoryDigest>;
|
|
3153
|
+
/**
|
|
3154
|
+
* The block as the model reads it.
|
|
3155
|
+
*
|
|
3156
|
+
* THREE THINGS IT HAS TO DO THAT A SKILLS CATALOG DOES NOT.
|
|
3157
|
+
*
|
|
3158
|
+
* It must frame each note by WHO ASSERTED IT, which is why there are two sections rather than one
|
|
3159
|
+
* sentence over the whole list. An unlabelled fact in a system prompt reads with the authority of an
|
|
3160
|
+
* instruction, and the hazard of an agent's own inference is exactly that authority — so those are
|
|
3161
|
+
* hedged. But a memory an administrator published for a whole organisation is not the agent's guess,
|
|
3162
|
+
* and telling the model to "prefer what the user says now" about it hands any user an override of
|
|
3163
|
+
* their organisation's policy by asserting the opposite. `MemoryOrigin.author` is the axis, and it
|
|
3164
|
+
* is the only one that matters here: scope says who a fact applies to, not who decided it.
|
|
3165
|
+
*
|
|
3166
|
+
* Where a narrower scope won, it must print the value it beat, so the model can tell the user their
|
|
3167
|
+
* setting differs from the org's instead of quietly applying one of them.
|
|
3168
|
+
*
|
|
3169
|
+
* And a block that is a SELECTION must say so. A model reading a partial set as a whole one turns an
|
|
3170
|
+
* absence into evidence — "you never told me that" about a fact it simply was not shown.
|
|
3171
|
+
*/
|
|
3172
|
+
/** Arguments to {@link buildMemoryBlock}. */
|
|
3173
|
+
interface BuildMemoryBlockInput {
|
|
3174
|
+
entries: readonly MemoryDigestEntry[];
|
|
3175
|
+
/** Whether this deployment offers `remember` — the block names the tool only where it exists. */
|
|
3176
|
+
writable: boolean;
|
|
3177
|
+
/** Whether applicable memories are missing from `entries` — the ceiling bit, or recall selected. */
|
|
3178
|
+
partial: boolean;
|
|
3179
|
+
}
|
|
3180
|
+
declare function buildMemoryBlock({ entries, writable, partial }: BuildMemoryBlockInput): string;
|
|
3181
|
+
/** The reserved name of the built-in memory-writing tool. */
|
|
3182
|
+
declare const REMEMBER_TOOL_NAME = "remember";
|
|
3183
|
+
/** What the model passes to the `remember` tool. */
|
|
3184
|
+
interface RememberToolInput {
|
|
3185
|
+
key: string;
|
|
3186
|
+
fact: string;
|
|
3187
|
+
}
|
|
3188
|
+
/**
|
|
3189
|
+
* The `remember` tool's input schema. Hand-written for the same reason `askInputSchema` and
|
|
3190
|
+
* `skillInputSchema` are: core depends on no validator and has to publish a JSON Schema a provider
|
|
3191
|
+
* can constrain generation against.
|
|
3192
|
+
*
|
|
3193
|
+
* THERE IS NO SCOPE PARAMETER, and that is the enforcement rather than a simplification. An agent
|
|
3194
|
+
* may only ever write the actor it is running for (see {@link memoryWriteVerdict}), so a scope
|
|
3195
|
+
* argument could only ever be a request that gets refused — and a refusable request is one a model
|
|
3196
|
+
* will keep making, and one a future reader will be tempted to make grantable.
|
|
3197
|
+
*/
|
|
3198
|
+
declare const rememberInputSchema: StandardSchemaV1<unknown, RememberToolInput>;
|
|
3199
|
+
declare const REMEMBER_TOOL_DESCRIPTION = "Record one durable fact about this user under a key, so later conversations start knowing it. For things that will still be true another day \u2014 how they work, what their organisation requires, a correction they made. Not for what this conversation is about, and not for anything you were not told or could not reasonably infer. The user can read and delete everything you record here.";
|
|
3200
|
+
/**
|
|
3201
|
+
* The `remember` tool as the model sees it. NOT a `ToolSpec` and never registered, exactly like
|
|
3202
|
+
* `ask` and `skill`: it has no handler, because the loop serves it against the digest the journal
|
|
3203
|
+
* holds. Keeping it out of the `ToolRegistry` is also what keeps its kind off a process-local
|
|
3204
|
+
* lookup — see `claimToolCall`.
|
|
3205
|
+
*/
|
|
3206
|
+
declare function rememberToolDefinition(): ToolDefinition;
|
|
3207
|
+
/**
|
|
3208
|
+
* Append the built-in `remember` definition to a turn's tool list. Exported because the dispatched
|
|
3209
|
+
* llm step re-derives the tool list on a worker and has to reach the same list the loop would have.
|
|
3210
|
+
*/
|
|
3211
|
+
/** Arguments to {@link withMemoryTool}. */
|
|
3212
|
+
interface WithMemoryToolInput {
|
|
3213
|
+
tools: ToolDefinition[];
|
|
3214
|
+
/** Whether this deployment's provider can write at all — module config, uniform across its pods. */
|
|
3215
|
+
enabled: boolean;
|
|
3216
|
+
}
|
|
3217
|
+
declare function withMemoryTool({ tools, enabled }: WithMemoryToolInput): ToolDefinition[];
|
|
3218
|
+
/** Who is trying to write a memory. */
|
|
3219
|
+
interface MemoryAuthor {
|
|
3220
|
+
/**
|
|
3221
|
+
* `'human'` is a person acting through a UI; `'agent'` is anything else — a turn, a tool, a batch
|
|
3222
|
+
* job. The distinction is the whole of rule three below, so it is not inferable and has to be
|
|
3223
|
+
* stated by the caller.
|
|
3224
|
+
*/
|
|
3225
|
+
kind: 'human' | 'agent';
|
|
3226
|
+
actorRef?: string;
|
|
3227
|
+
}
|
|
3228
|
+
/** A request to write a memory at a scope. */
|
|
3229
|
+
interface MemoryWriteRequest {
|
|
3230
|
+
scope: string;
|
|
3231
|
+
actor: Actor;
|
|
3232
|
+
/** The actor's resolved scopes, most specific first — the same list the turn drew its digest from. */
|
|
3233
|
+
scopes: readonly string[];
|
|
3234
|
+
author: MemoryAuthor;
|
|
3235
|
+
/**
|
|
3236
|
+
* The host's own answer to "may this person administer that scope". This library cannot know it,
|
|
3237
|
+
* and inventing an answer would be a second, weaker authorization model next to the host's real
|
|
3238
|
+
* one. Undefined/false → a wider scope is refused.
|
|
3239
|
+
*/
|
|
3240
|
+
elevated?: boolean;
|
|
3241
|
+
}
|
|
3242
|
+
type MemoryVerdict = {
|
|
3243
|
+
allowed: true;
|
|
3244
|
+
} | {
|
|
3245
|
+
allowed: false;
|
|
3246
|
+
reason: string;
|
|
3247
|
+
};
|
|
3248
|
+
/**
|
|
3249
|
+
* May this author write a memory at this scope?
|
|
3250
|
+
*
|
|
3251
|
+
* The same four rules as `skillWriteVerdict`, deliberately duplicated rather than shared: they are
|
|
3252
|
+
* the same rules about two different things, and folding them into one function would mean a future
|
|
3253
|
+
* change to how skills are authored silently changing who may edit what the assistant believes about
|
|
3254
|
+
* a person. Rule three is the one that matters, and it bites harder here than it does for skills:
|
|
3255
|
+
*
|
|
3256
|
+
* 1. You may only write into a scope you are yourself in.
|
|
3257
|
+
* 2. Your OWN scope is yours. A memory at `actor:<you>` affects exactly one prompt — your own.
|
|
3258
|
+
* 3. NOTHING BUT A HUMAN MAY WRITE ABOVE ITS OWN SCOPE, whatever `elevated` says. A skill an agent
|
|
3259
|
+
* could publish at `tenant:` is a procedure anyone in the tenant can edit by talking to the
|
|
3260
|
+
* assistant; a MEMORY it could publish there is a fact everyone in the tenant is then answered
|
|
3261
|
+
* from, with no document to inspect and nobody aware it was written. An agent that has genuinely
|
|
3262
|
+
* learned something the organisation should know proposes it; a person publishes it.
|
|
3263
|
+
* 4. A human writing above their own scope needs the host to say so (`elevated`).
|
|
3264
|
+
*/
|
|
3265
|
+
declare function memoryWriteVerdict(request: MemoryWriteRequest): MemoryVerdict;
|
|
3266
|
+
/**
|
|
3267
|
+
* May this actor delete this memory?
|
|
3268
|
+
*
|
|
3269
|
+
* Narrower than the write rule on purpose, and it takes no `elevated` flag. Deleting what the
|
|
3270
|
+
* assistant believes about YOU needs no permission from anyone — that is the point of the read-back
|
|
3271
|
+
* — so the endpoint that serves it must not be able to ask a host a question it might answer no to.
|
|
3272
|
+
* Deleting what it believes about a tenant is an administrative act on shared state, which belongs
|
|
3273
|
+
* in the host's console with the same elevation a write there needs.
|
|
3274
|
+
*/
|
|
3275
|
+
/** A request to delete one memory. */
|
|
3276
|
+
interface MemoryForgetRequest {
|
|
3277
|
+
record: Pick<MemoryRecord, 'scope'>;
|
|
3278
|
+
actor: Actor;
|
|
3279
|
+
}
|
|
3280
|
+
declare function memoryForgetVerdict({ record, actor }: MemoryForgetRequest): MemoryVerdict;
|
|
3281
|
+
/** What a `remember` call resolved to — the stored record, or why nothing was stored. */
|
|
3282
|
+
type MemoryWriteOutcome = {
|
|
3283
|
+
ok: true;
|
|
3284
|
+
record: MemoryRecord;
|
|
3285
|
+
} | {
|
|
3286
|
+
ok: false;
|
|
3287
|
+
error: string;
|
|
3288
|
+
};
|
|
3289
|
+
/**
|
|
3290
|
+
* Serve one `remember` call against the digest THIS TURN resolved.
|
|
3291
|
+
*
|
|
3292
|
+
* The digest's `scopes` are the authorization boundary, not the resolver: they came out of the
|
|
3293
|
+
* `memory:digest` checkpoint, so a write is checked against the scopes the run recorded rather than
|
|
3294
|
+
* the ones a replaying process's resolver would produce now. A membership table edited mid-run
|
|
3295
|
+
* cannot retroactively widen — or narrow — what a turn already in flight is allowed to write.
|
|
3296
|
+
*/
|
|
3297
|
+
/** Arguments to {@link writeMemory}. */
|
|
3298
|
+
interface WriteMemoryInput {
|
|
3299
|
+
config: MemoryConfig;
|
|
3300
|
+
/** The digest THIS TURN journaled — the authorization boundary, never a fresh resolution. */
|
|
3301
|
+
digest: MemoryDigest;
|
|
3302
|
+
/** The `remember` call as the model made it. */
|
|
3303
|
+
call: RememberToolInput;
|
|
3304
|
+
ctx: ScopeContext;
|
|
3305
|
+
runId: string;
|
|
3306
|
+
}
|
|
3307
|
+
declare function writeMemory({ config, digest, call, ctx, runId, }: WriteMemoryInput): Promise<MemoryWriteOutcome>;
|
|
3308
|
+
|
|
1479
3309
|
/** Holds the named agent definitions registered via `AgentModule.forFeature([...])`. */
|
|
1480
3310
|
declare class AgentRegistry {
|
|
1481
3311
|
private readonly definitions;
|
|
@@ -1485,6 +3315,121 @@ declare class AgentRegistry {
|
|
|
1485
3315
|
list(): AgentDefinition[];
|
|
1486
3316
|
}
|
|
1487
3317
|
|
|
3318
|
+
/**
|
|
3319
|
+
* Is this the durable runtime refusing a checkpoint position, rather than anything the agent did?
|
|
3320
|
+
*
|
|
3321
|
+
* Matched by NAME because this package cannot import either class: the in-process engine throws
|
|
3322
|
+
* `@dudousxd/nestjs-durable-core`'s `NonDeterminismError` and the thin BullMQ worker throws
|
|
3323
|
+
* `@dudousxd/nestjs-durable-worker`'s `NondeterminismError` — the same contract under two spellings
|
|
3324
|
+
* in two packages, neither of which core depends on. Same cross-runtime reasoning as the
|
|
3325
|
+
* `isControlFlowError` hook, which exists for exactly this reason on the suspend path.
|
|
3326
|
+
*
|
|
3327
|
+
* Suffix rather than equality, because a runtime is free to qualify the name: the sibling Adonis
|
|
3328
|
+
* stack raises a `WorkflowNondeterminismError` from its remote-replay path, which an equality check
|
|
3329
|
+
* would wave through as an ordinary tool failure. `replay-integrity.contract.spec.ts` pins this
|
|
3330
|
+
* against the real exported classes, so a rename upstream fails a test instead of quietly
|
|
3331
|
+
* disarming the guard.
|
|
3332
|
+
*
|
|
3333
|
+
* Callers must let these through untouched. Every `catch` in a workflow body reacts by writing more
|
|
3334
|
+
* checkpoints (a toolfail, a run-end, a deactivate), and on a journal that has already diverged each
|
|
3335
|
+
* of those asks for a position the history cannot supply — so the recovery attempt raises its own
|
|
3336
|
+
* refusal, and THAT is the error the operator reads: a message pointing at the wrong seq, naming
|
|
3337
|
+
* checkpoints from the recovery path rather than the two that actually disagreed.
|
|
3338
|
+
*/
|
|
3339
|
+
declare function isReplayIntegrityError(error: unknown): boolean;
|
|
3340
|
+
|
|
3341
|
+
/**
|
|
3342
|
+
* Is this the runner unwinding the turn (a durable suspend / continue-as-new) rather than something
|
|
3343
|
+
* the agent did? Callers must rethrow it untouched: every `catch` in the loop reacts by writing more
|
|
3344
|
+
* checkpoints, and a suspend recorded as a tool failure leaves a journal the resumed replay cannot
|
|
3345
|
+
* line up with.
|
|
3346
|
+
*
|
|
3347
|
+
* Detected from the error itself so a host that never wired `AgentLoopHooks.isControlFlowError` still
|
|
3348
|
+
* gets its suspends through — that hook is an override for a runner whose signals carry no marker,
|
|
3349
|
+
* not the only line of defence. Checked as a stamped property, NOT `instanceof`: the classes differ
|
|
3350
|
+
* per runtime, which is why the marker exists.
|
|
3351
|
+
*/
|
|
3352
|
+
declare function isControlFlowSignal(error: unknown): boolean;
|
|
3353
|
+
|
|
3354
|
+
/** An agent->agent edge with its awaited-or-not resolved, whichever form it was authored in. */
|
|
3355
|
+
interface ResolvedDelegation {
|
|
3356
|
+
agent: string;
|
|
3357
|
+
detached: boolean;
|
|
3358
|
+
}
|
|
3359
|
+
/** Read one {@link AgentDelegation} entry in either of its forms. */
|
|
3360
|
+
declare function normalizeDelegation(entry: AgentDelegation): ResolvedDelegation;
|
|
3361
|
+
/**
|
|
3362
|
+
* What a detached delegation hands the model in place of an answer. Its `status` is the whole point:
|
|
3363
|
+
* a model that is given an object shaped like a result will report one, and this run has none yet.
|
|
3364
|
+
*/
|
|
3365
|
+
interface DetachedDelegationReceipt {
|
|
3366
|
+
detached: true;
|
|
3367
|
+
status: 'started';
|
|
3368
|
+
/** The agent now working on it. */
|
|
3369
|
+
agent: string;
|
|
3370
|
+
/** The run doing the work — what a client subscribes to, and what stamps the delivered message. */
|
|
3371
|
+
runId: string;
|
|
3372
|
+
/**
|
|
3373
|
+
* The same fact in the only vocabulary the model reliably acts on: prose in a tool result. The
|
|
3374
|
+
* structured fields above are for a client; this is for the turn that has to explain itself to a
|
|
3375
|
+
* user without claiming a result it does not hold.
|
|
3376
|
+
*/
|
|
3377
|
+
note: string;
|
|
3378
|
+
}
|
|
3379
|
+
/** How a detached delegation ended, written back onto its tool-call row once the run settles. */
|
|
3380
|
+
interface DetachedDelegationOutcome {
|
|
3381
|
+
detached: true;
|
|
3382
|
+
status: 'delivered' | 'failed' | 'cancelled';
|
|
3383
|
+
agent: string;
|
|
3384
|
+
runId: string;
|
|
3385
|
+
/** The delivered answer. Present on `delivered` only. */
|
|
3386
|
+
text?: string;
|
|
3387
|
+
/** Why it did not deliver. Present on `failed` only. */
|
|
3388
|
+
error?: string;
|
|
3389
|
+
}
|
|
3390
|
+
/** The receipt an `agent`-kind call returns the moment its delegate is under way. */
|
|
3391
|
+
declare function detachedStarted(args: {
|
|
3392
|
+
agent: string;
|
|
3393
|
+
runId: string;
|
|
3394
|
+
}): DetachedDelegationReceipt;
|
|
3395
|
+
/** The outcome written onto the delegating tool call when a detached run finishes its answer. */
|
|
3396
|
+
declare function detachedDelivered(args: {
|
|
3397
|
+
agent: string;
|
|
3398
|
+
runId: string;
|
|
3399
|
+
text: string;
|
|
3400
|
+
}): DetachedDelegationOutcome;
|
|
3401
|
+
/**
|
|
3402
|
+
* The outcome written onto the delegating tool call when a detached run never produced an answer.
|
|
3403
|
+
* Written by the RUNNER, not the loop: a run that crashed or was stopped cannot record its own
|
|
3404
|
+
* ending, and a delegation card that stays "started" for ever is the one state a reader cannot act
|
|
3405
|
+
* on.
|
|
3406
|
+
*/
|
|
3407
|
+
declare function detachedUnsettled(args: {
|
|
3408
|
+
agent: string;
|
|
3409
|
+
runId: string;
|
|
3410
|
+
status: 'failed' | 'cancelled';
|
|
3411
|
+
error?: string;
|
|
3412
|
+
}): DetachedDelegationOutcome;
|
|
3413
|
+
/**
|
|
3414
|
+
* Settle the delegation a detached run was started for when that run ends with no answer.
|
|
3415
|
+
*
|
|
3416
|
+
* Two writes, because a reader needs both: the tool-call row so a governance surface stops counting
|
|
3417
|
+
* it as in flight, and a MESSAGE in the delegating thread so the person who asked finds out. A
|
|
3418
|
+
* delivered answer already arrives as a message; without this, the failure is the one outcome that
|
|
3419
|
+
* silently never does, and the conversation shows "started" for ever.
|
|
3420
|
+
*
|
|
3421
|
+
* Called by the RUNNER, never the loop — a run that crashed or was stopped cannot record its own
|
|
3422
|
+
* ending. Skipped entirely when the thread is gone, for the same reason delivery is.
|
|
3423
|
+
*/
|
|
3424
|
+
declare function settleUnsettledDelegation(args: {
|
|
3425
|
+
store: AgentStore;
|
|
3426
|
+
delivery: DetachedDelivery;
|
|
3427
|
+
agent: string;
|
|
3428
|
+
runId: string;
|
|
3429
|
+
status: 'failed' | 'cancelled';
|
|
3430
|
+
error?: string;
|
|
3431
|
+
}): Promise<void>;
|
|
3432
|
+
|
|
1488
3433
|
/** Thrown when an actor invokes a tool their role is not allowed. */
|
|
1489
3434
|
declare class ToolForbiddenError extends Error {
|
|
1490
3435
|
readonly toolName: string;
|
|
@@ -1529,10 +3474,14 @@ declare class ToolRegistry {
|
|
|
1529
3474
|
allSpecs(): ToolSpec[];
|
|
1530
3475
|
/**
|
|
1531
3476
|
* The tools to offer the model for this actor+agent, after the four filter layers: what this
|
|
1532
|
-
* deployment has enabled, what this actor's role allows, what each
|
|
1533
|
-
*
|
|
3477
|
+
* agent pinned, what this deployment has enabled, what this actor's role allows, and what each
|
|
3478
|
+
* tool's own `canUse` allows this actor.
|
|
1534
3479
|
*
|
|
1535
3480
|
* Every layer only ever removes tools, so no arrangement of them can widen what a turn reaches.
|
|
3481
|
+
* The agent's allow-list therefore goes FIRST, even though it is the narrowest statement: it is a
|
|
3482
|
+
* pure name-set match, while each of the three below it may be a round trip — an authz service, a
|
|
3483
|
+
* feature-flag store, an MCP server — and this runs once per model step. Asking those about a
|
|
3484
|
+
* tool the allow-list has already excluded is a call whose answer nothing reads.
|
|
1536
3485
|
*/
|
|
1537
3486
|
definitionsFor(actor: Actor, policy: RolesPolicy, allowedTools?: string[]): Promise<ToolDefinition[]>;
|
|
1538
3487
|
/**
|
|
@@ -1549,7 +3498,7 @@ declare class DefaultRolesPolicy implements RolesPolicy {
|
|
|
1549
3498
|
can(actor: Actor, tool: ToolSpec): boolean;
|
|
1550
3499
|
}
|
|
1551
3500
|
|
|
1552
|
-
interface AgentLoopDeps {
|
|
3501
|
+
interface AgentLoopDeps<TOutput = unknown> {
|
|
1553
3502
|
model: ModelProvider;
|
|
1554
3503
|
store: AgentStore;
|
|
1555
3504
|
registry: ToolRegistry;
|
|
@@ -1571,6 +3520,26 @@ interface AgentLoopDeps {
|
|
|
1571
3520
|
*/
|
|
1572
3521
|
promptContributors?: PromptContributor[];
|
|
1573
3522
|
maxSteps?: number;
|
|
3523
|
+
/**
|
|
3524
|
+
* How deep agent→agent delegation may nest before the loop refuses further hops.
|
|
3525
|
+
* Defaults to {@link MAX_DELEGATION_DEPTH}.
|
|
3526
|
+
*
|
|
3527
|
+
* It bounds NESTING, never fan-out: how many agents a turn delegates to is the model's, one tool
|
|
3528
|
+
* call each, and nothing here caps that. What it guards is a hop the model cannot see — a
|
|
3529
|
+
* `delegatesTo` cycle (A→B→A), where each agent is making one reasonable call and the recursion
|
|
3530
|
+
* is a property of the wiring rather than of any decision.
|
|
3531
|
+
*/
|
|
3532
|
+
maxDelegationDepth?: number;
|
|
3533
|
+
/**
|
|
3534
|
+
* How many times one agent may appear on a single delegation chain.
|
|
3535
|
+
* Defaults to {@link DEFAULT_MAX_AGENT_APPEARANCES}.
|
|
3536
|
+
*
|
|
3537
|
+
* This is the guard the depth ceiling was a proxy for, made exact: the loop compares the target
|
|
3538
|
+
* against {@link AgentRunInput.delegationPath} and knows whether the chain has been here before,
|
|
3539
|
+
* and how often. A chain of eight DISTINCT agents is long, not looping, and no longer refused for
|
|
3540
|
+
* resembling one.
|
|
3541
|
+
*/
|
|
3542
|
+
maxAgentAppearances?: number;
|
|
1574
3543
|
/** Optional host handle threaded to tool ctx (e.g. an ORM EntityManager). */
|
|
1575
3544
|
host?: unknown;
|
|
1576
3545
|
/** Agent-level tool allow-list. Undefined → all tools (after role filtering). */
|
|
@@ -1611,13 +3580,174 @@ interface AgentLoopDeps {
|
|
|
1611
3580
|
* message/step) and reused for every step's estimate. Undefined → `costUsd` is always `null`.
|
|
1612
3581
|
*/
|
|
1613
3582
|
pricingStore?: AgentPricingStore;
|
|
3583
|
+
/**
|
|
3584
|
+
* Bounds how much of the thread rides into the turn. Undefined → the WHOLE thread, every message
|
|
3585
|
+
* the store holds, which is unbounded: a long-lived thread eventually exceeds the provider's
|
|
3586
|
+
* context limit, and pays for the full transcript on every turn up to that point. See
|
|
3587
|
+
* `windowHistory` for the built-in.
|
|
3588
|
+
*/
|
|
3589
|
+
historyPolicy?: HistoryPolicy;
|
|
3590
|
+
/**
|
|
3591
|
+
* Rewrites the prompt before EACH model call of the turn, in order (see {@link InputProcessor}).
|
|
3592
|
+
* Transformation only — `historyPolicy` owns which messages are there in the first place. Adds
|
|
3593
|
+
* one `process:input:<step>` checkpoint per step; empty/undefined adds none.
|
|
3594
|
+
*/
|
|
3595
|
+
inputProcessors?: InputProcessor[];
|
|
3596
|
+
/**
|
|
3597
|
+
* Inspects each model step's answer before the stream, the store or the next step sees it, and
|
|
3598
|
+
* may redact, replace or refuse it (see {@link OutputProcessor}).
|
|
3599
|
+
*
|
|
3600
|
+
* REGISTERING ONE TAKES THE MODEL CALL OFF THE RUN'S SINK for the turn — a gate cannot run after
|
|
3601
|
+
* the answer has already reached the reader. What the chain costs the subscriber depends on what
|
|
3602
|
+
* it declares:
|
|
3603
|
+
*
|
|
3604
|
+
* - Any processor WITHOUT `incremental` → the whole answer is buffered and released as one `text`
|
|
3605
|
+
* frame once the chain has passed. No token-by-token text, and (for a provider that writes bytes
|
|
3606
|
+
* outside the `AgentStreamEvent` vocabulary) no frame the gate cannot classify.
|
|
3607
|
+
* - EVERY processor with `incremental` → the chain runs over the growing prefix and releases it as
|
|
3608
|
+
* it arrives, holding back the widest `lookbackChars` any of them asked for. The whole-answer
|
|
3609
|
+
* pass still runs and is still authoritative for the stream and the store.
|
|
3610
|
+
*
|
|
3611
|
+
* Both add one `process:output:<step>` checkpoint per step, in the same position, so a chain can
|
|
3612
|
+
* change its declaration without moving a checkpoint. Empty/undefined adds none, and the turn
|
|
3613
|
+
* streams exactly as it always did.
|
|
3614
|
+
*/
|
|
3615
|
+
outputProcessors?: OutputProcessor[];
|
|
3616
|
+
/**
|
|
3617
|
+
* Constrain the turn's answer to a schema, returned validated as `object` on the loop's result and
|
|
3618
|
+
* recorded on the assistant message as a synthetic `structured_output` tool call (the same shape
|
|
3619
|
+
* inject-mode retrieval uses, so no store gains a column for it).
|
|
3620
|
+
*
|
|
3621
|
+
* HOW IT COMPOSES WITH TOOL CALLING: as a separate formatting pass, ALWAYS. The turn runs its
|
|
3622
|
+
* model→tools iteration exactly as it would without a schema; once a step comes back with no tool
|
|
3623
|
+
* calls, one extra non-streamed call (`structured:<step>`, `tools: []`, `outputSchema` set)
|
|
3624
|
+
* restates that answer as the schema. Most providers cannot serve a response format and a tool set
|
|
3625
|
+
* in one request, and skipping the pass for an agent that happens to have no tools would make the
|
|
3626
|
+
* checkpoint sequence depend on a tool-registry lookup — the registry of whichever process is
|
|
3627
|
+
* replaying — which is exactly how a run ends up asking for a position its history has no room
|
|
3628
|
+
* for. So the pass is unconditional, and it costs one model call per turn (recorded as
|
|
3629
|
+
* `structured_output` usage).
|
|
3630
|
+
*/
|
|
3631
|
+
outputSchema?: StandardSchemaV1<unknown, TOutput>;
|
|
3632
|
+
/** Overrides the formatting pass's system prompt. Undefined → `DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION`. */
|
|
3633
|
+
outputInstruction?: string;
|
|
3634
|
+
/**
|
|
3635
|
+
* How many extra model calls may try to fix an answer that failed `outputSchema`, each shown the
|
|
3636
|
+
* previous attempt's validation issues. Undefined → 1; `0` → fail on the first invalid reply.
|
|
3637
|
+
* Bounded because a model that cannot satisfy a schema usually cannot satisfy it on the fourth
|
|
3638
|
+
* try either, and every attempt is billed.
|
|
3639
|
+
*/
|
|
3640
|
+
outputRepairAttempts?: number;
|
|
3641
|
+
/**
|
|
3642
|
+
* Show the formatting pass the turn's whole transcript instead of just the question and the
|
|
3643
|
+
* answer. For an agent whose answer cannot be restated from its own words — one that reports on
|
|
3644
|
+
* rows a tool returned and names only their total in the prose, say.
|
|
3645
|
+
*
|
|
3646
|
+
* OFF by default because the pass is a translation, and a translation needs the thing being
|
|
3647
|
+
* translated. The transcript it would otherwise carry is the turn's entire prompt a second time,
|
|
3648
|
+
* at no discount: the pass swaps the system block for the schema instruction, and the system block
|
|
3649
|
+
* is the prompt cache's prefix, so nothing of the first call's cache survives into it.
|
|
3650
|
+
*/
|
|
3651
|
+
outputFromTranscript?: boolean;
|
|
3652
|
+
/**
|
|
3653
|
+
* A question set the agent puts to the user BEFORE it starts working — collecting the scope, as
|
|
3654
|
+
* against `awaitApproval`, which sanctions work already proposed. The questions are AUTHORED, so
|
|
3655
|
+
* the turn spends no model call producing them and a client knows the total ("Question 1 of 3")
|
|
3656
|
+
* the moment the form appears.
|
|
3657
|
+
*
|
|
3658
|
+
* The turn parks on the answers exactly as it parks on an approval, and the request persists as
|
|
3659
|
+
* the same tool-call row the model's `ask` writes — see {@link ask}. Undefined → no intake, and a
|
|
3660
|
+
* turn's checkpoint sequence is byte-identical to one that never had the option.
|
|
3661
|
+
*/
|
|
3662
|
+
intake?: AgentIntake;
|
|
3663
|
+
/**
|
|
3664
|
+
* Offer the model the built-in `ask` tool, so it can put its own question set to the user when it
|
|
3665
|
+
* judges the scope is missing — the same surface as {@link intake}, minus the "known in advance".
|
|
3666
|
+
*
|
|
3667
|
+
* `ask` is NOT a registered tool: it has no handler (the loop settles it against a human), and
|
|
3668
|
+
* keeping it out of the `ToolRegistry` is what keeps its kind out of a process-local lookup. The
|
|
3669
|
+
* loop appends its definition to the turn's tool list from THIS flag, which is module config and
|
|
3670
|
+
* therefore uniform across a deployment. Undefined/false → the model never sees it.
|
|
3671
|
+
*/
|
|
3672
|
+
ask?: boolean;
|
|
3673
|
+
/**
|
|
3674
|
+
* Authored procedures the model may pull in when a task calls for one, resolved per turn against
|
|
3675
|
+
* the actor's scopes — see `skills.ts`. Undefined → no catalog block, no `skill` tool, and a
|
|
3676
|
+
* turn's checkpoint sequence is byte-identical to one that never had the option.
|
|
3677
|
+
*
|
|
3678
|
+
* HOW IT COMPOSES WITH THE OTHER FOUR THINGS THAT WRITE THE PROMPT. The system block is assembled
|
|
3679
|
+
* in one fixed order — the agent's base prompt, then each `promptContributors` section, then
|
|
3680
|
+
* {@link memory}, then the injected retrieval block, then the skills CATALOG — and skills are the
|
|
3681
|
+
* cheapest of the five by construction: the catalog is one line per skill and nothing else. The
|
|
3682
|
+
* instructions themselves arrive as a tool RESULT, on the transcript, which means the budget they
|
|
3683
|
+
* draw on is the one `historyPolicy` already governs rather than a private allowance of their own.
|
|
3684
|
+
*
|
|
3685
|
+
* A BODY IS NOT LIKE A MEMORY, which does ride the system block. A body is read BECAUSE the model
|
|
3686
|
+
* went and asked for it, so the transcript — what this conversation happened to pull in — is where
|
|
3687
|
+
* it belongs. A memory is worthless unless it is in front of the model on the turn nobody thought
|
|
3688
|
+
* to look for it, and it is affordable there only because it has no body to carry.
|
|
3689
|
+
*/
|
|
3690
|
+
skills?: SkillsConfig;
|
|
3691
|
+
/**
|
|
3692
|
+
* What the assistant has previously concluded about the actor and their organisation, resolved per
|
|
3693
|
+
* turn against the same scope tokens skills use — see `memory.ts`. Undefined → no memory block, no
|
|
3694
|
+
* `remember` tool, and a turn's checkpoint sequence is byte-identical to one that never had the
|
|
3695
|
+
* option.
|
|
3696
|
+
*
|
|
3697
|
+
* WHERE IT SITS IN THE PROMPT. The system block is assembled most-durable-first: the agent's base
|
|
3698
|
+
* prompt (the same for everyone, every turn), then `promptContributors`, then MEMORY (the same for
|
|
3699
|
+
* this person, every turn), then the injected retrieval block (this question only), then the
|
|
3700
|
+
* skills catalog (a menu rather than an instruction, so a reader meets instructions before
|
|
3701
|
+
* options).
|
|
3702
|
+
*
|
|
3703
|
+
* WHY IT IS IN THE SYSTEM BLOCK AT ALL, when a skill's body deliberately is not. A body is read
|
|
3704
|
+
* BECAUSE the model went and asked for it; a memory is worthless unless it is in front of the
|
|
3705
|
+
* model on the turn nobody thought to look for it — "they report in nautical miles" only works
|
|
3706
|
+
* unprompted. What makes that affordable is that a memory has no body: `maxMemories` lines, each
|
|
3707
|
+
* capped at `maxFactChars` when it is written, so the block's ceiling is a product of two numbers
|
|
3708
|
+
* an operator set rather than however much the model felt like writing down.
|
|
3709
|
+
*/
|
|
3710
|
+
memory?: MemoryConfig;
|
|
1614
3711
|
}
|
|
3712
|
+
/** One task's outcome under {@link AgentLoopHooks.parallel}, reported instead of thrown. */
|
|
3713
|
+
type SettledTask<T> = {
|
|
3714
|
+
ok: true;
|
|
3715
|
+
value: T;
|
|
3716
|
+
} | {
|
|
3717
|
+
ok: false;
|
|
3718
|
+
error: unknown;
|
|
3719
|
+
};
|
|
3720
|
+
/**
|
|
3721
|
+
* The {@link AgentLoopHooks.parallel} implementation for a runner whose checkpoint positions are
|
|
3722
|
+
* handed out on the CALL (both durable primitives are: `ctx.localStep` and `ctx.step` take their
|
|
3723
|
+
* position before their first `await`). Every task is invoked here, synchronously and in list
|
|
3724
|
+
* order, before any of them is awaited — which is what fixes the block of positions the tasks
|
|
3725
|
+
* occupy, whatever order they then settle in. Nothing rejects: the caller decides what an
|
|
3726
|
+
* individual failure means.
|
|
3727
|
+
*/
|
|
3728
|
+
declare function settleAll<T>(tasks: readonly (() => Promise<T>)[]): Promise<SettledTask<T>[]>;
|
|
1615
3729
|
interface AgentLoopHooks {
|
|
1616
3730
|
runId: string;
|
|
1617
3731
|
/** A writer for this run's live token stream (data plane). */
|
|
1618
3732
|
openSink(): SinkWriter | Promise<SinkWriter>;
|
|
1619
3733
|
/** HITL gate for an action tool. Inline resolves a pending promise; durable awaits a signal. */
|
|
1620
3734
|
awaitApproval(call: ToolCallRequest, ctx: AiToolCtx): Promise<Decision>;
|
|
3735
|
+
/**
|
|
3736
|
+
* Park the run on a question set and resolve with what the human sent back. The SAME wait
|
|
3737
|
+
* `awaitApproval` is — the durable runner maps both to `ctx.waitForSignal` on
|
|
3738
|
+
* `tool:<runId>:<callId>`, so an answer and an approval reach a parked run through one path.
|
|
3739
|
+
*
|
|
3740
|
+
* Optional, and its absence changes no checkpoint: the loop falls back to `awaitApproval` and
|
|
3741
|
+
* reads the decision as "the user confirmed the pre-picked answers" (approved) or "the user
|
|
3742
|
+
* skipped" (rejected). That is the honest reduction of a yes/no channel, and it means a host that
|
|
3743
|
+
* only ever implemented approval still runs an elicitation to completion instead of hanging.
|
|
3744
|
+
*
|
|
3745
|
+
* Declared as `HumanReply` rather than `ElicitationReply` because that is what the channel really
|
|
3746
|
+
* carries: a question set is parked as a `pending_approval` action, so a `Decision` can arrive on
|
|
3747
|
+
* this wait from the approvals inbox even where the host implements it. The loop reduces one to
|
|
3748
|
+
* the other — see {@link normalizeElicitationReply} — so an implementer never has to.
|
|
3749
|
+
*/
|
|
3750
|
+
awaitAnswers?(request: ElicitationRequest, ctx: AiToolCtx): Promise<HumanReply>;
|
|
1621
3751
|
/**
|
|
1622
3752
|
* Run another named agent and return its answer. Provided only when the host wired multi-agent
|
|
1623
3753
|
* support (durable → child workflow, inline → nested loop). Exposed to tools as `ctx.runAgent`.
|
|
@@ -1625,6 +3755,27 @@ interface AgentLoopHooks {
|
|
|
1625
3755
|
runAgent?(agentName: string, task: string): Promise<{
|
|
1626
3756
|
text: string;
|
|
1627
3757
|
}>;
|
|
3758
|
+
/**
|
|
3759
|
+
* Start another named agent and return its run id WITHOUT waiting for it, so the calling turn can
|
|
3760
|
+
* finish while the delegate is still working. The durable runner maps this to `ctx.startChild`
|
|
3761
|
+
* (checkpointed as `spawn:<id>`; no suspend, unlike the `ctx.child` behind {@link runAgent}); the
|
|
3762
|
+
* inline runner to a nested loop nobody awaits.
|
|
3763
|
+
*
|
|
3764
|
+
* `toolCallId` is the delegation's own call, which the started run carries as its delivery
|
|
3765
|
+
* address: it posts its answer back into the calling thread against that row.
|
|
3766
|
+
*
|
|
3767
|
+
* Absent -> a delegation the journal declares detached is AWAITED instead. The loop writes the
|
|
3768
|
+
* same checkpoint names either way (see {@link delegateToolCall}), so a runner that cannot detach
|
|
3769
|
+
* still answers the user; only the runner's own positions differ, and a given runner always makes
|
|
3770
|
+
* the same choice for the same call.
|
|
3771
|
+
*/
|
|
3772
|
+
startAgent?(args: {
|
|
3773
|
+
agentName: string;
|
|
3774
|
+
task: string;
|
|
3775
|
+
toolCallId: string;
|
|
3776
|
+
}): Promise<{
|
|
3777
|
+
runId: string;
|
|
3778
|
+
}>;
|
|
1628
3779
|
/**
|
|
1629
3780
|
* Checkpoint wrapper. Inline = call fn directly; durable = ctx.localStep(name, fn) — the
|
|
1630
3781
|
* in-process checkpointed primitive (`name` is a checkpoint identity, not a worker group).
|
|
@@ -1635,10 +3786,12 @@ interface AgentLoopHooks {
|
|
|
1635
3786
|
/**
|
|
1636
3787
|
* Dispatch the model turn as a routed remote step. When present the loop uses it INSTEAD of
|
|
1637
3788
|
* running the model inline under hooks.step. Must resolve to exactly what deps.model.runTurn
|
|
1638
|
-
* resolves
|
|
1639
|
-
*
|
|
3789
|
+
* resolves, plus `bufferedFrames` when (and only when) the envelope set `bufferOutput` — the
|
|
3790
|
+
* handler streams to a worker-side sink the loop cannot wrap, so an output gate depends on it
|
|
3791
|
+
* honouring that flag. The durable runner enriches the envelope with sink routing
|
|
3792
|
+
* (sinkRunId/childSink) on its side — core stays sink-topology-agnostic.
|
|
1640
3793
|
*/
|
|
1641
|
-
dispatchLlm?(index: number, envelope: LlmStepEnvelope): Promise<
|
|
3794
|
+
dispatchLlm?(index: number, envelope: LlmStepEnvelope): Promise<BufferedModelTurnResult>;
|
|
1642
3795
|
/**
|
|
1643
3796
|
* Dispatch one tool execution as a routed remote step. Replaces ONLY the registry.invoke call
|
|
1644
3797
|
* (+ its timeout, applied handler-side); all persist steps around it stay local.
|
|
@@ -1646,14 +3799,83 @@ interface AgentLoopHooks {
|
|
|
1646
3799
|
dispatchTool?(call: ToolCallRequest, envelope: ToolStepEnvelope): Promise<unknown>;
|
|
1647
3800
|
/**
|
|
1648
3801
|
* Recognizes the runner's control-flow signals (durable suspend / continue-as-new) so the loop
|
|
1649
|
-
* rethrows them untouched instead of recording a failure.
|
|
1650
|
-
*
|
|
3802
|
+
* rethrows them untouched instead of recording a failure. An OVERRIDE, not the gate: the loop
|
|
3803
|
+
* already recognizes any signal carrying the durable runtime's `Symbol.for` marker on its own (see
|
|
3804
|
+
* {@link isControlFlowSignal}), so a host that omits this — an inline runner with no control-flow
|
|
3805
|
+
* exceptions, or one that simply forgot — never turns a suspend into a persisted tool failure.
|
|
3806
|
+
* Supply it for a runner whose signals carry no marker.
|
|
1651
3807
|
*/
|
|
1652
3808
|
isControlFlowError?(error: unknown): boolean;
|
|
3809
|
+
/**
|
|
3810
|
+
* Run `tasks` concurrently, resolving once EVERY one has settled — one outcome per task, in INPUT
|
|
3811
|
+
* order, never rejecting. Supplying it is a statement about the runner's checkpointing: each task
|
|
3812
|
+
* MUST be invoked synchronously, in list order, before any is awaited, so a runner that hands out
|
|
3813
|
+
* positions on the call assigns them in that order regardless of which task finishes first.
|
|
3814
|
+
* {@link settleAll} is exactly that, and is what both bundled runners pass.
|
|
3815
|
+
*
|
|
3816
|
+
* Waiting for ALL of them is the other half of the contract. A durable runner unwinds a turn by
|
|
3817
|
+
* THROWING (a dispatched step suspends), and a sibling abandoned part-way through its own dispatch
|
|
3818
|
+
* is a tool nobody ever runs.
|
|
3819
|
+
*
|
|
3820
|
+
* Absent → the loop runs a turn's tool calls one at a time. That is the honest answer for a runner
|
|
3821
|
+
* whose positions are assigned anywhere other than the call, and it is the only behaviour that
|
|
3822
|
+
* existed before, so nothing gains concurrency without saying so.
|
|
3823
|
+
*/
|
|
3824
|
+
parallel?<T>(tasks: readonly (() => Promise<T>)[]): Promise<SettledTask<T>[]>;
|
|
3825
|
+
/**
|
|
3826
|
+
* Has someone asked this run to stop?
|
|
3827
|
+
*
|
|
3828
|
+
* Answered at the points where stopping is safe and cheap — between steps, before the next model
|
|
3829
|
+
* call, before the turn's tools are dispatched — and NEVER consulted anywhere else. Two properties
|
|
3830
|
+
* make it safe to ask a live question inside a replayed loop body:
|
|
3831
|
+
*
|
|
3832
|
+
* 1. THE ANSWER IS JOURNALED. The loop only ever calls this inside a checkpoint, so the first
|
|
3833
|
+
* process to reach a given position writes the answer there and every later replay reads it
|
|
3834
|
+
* back. A cancel that arrives between two replays is therefore seen at the first position the
|
|
3835
|
+
* history does NOT yet hold, and cannot change the branch a replayed position already took.
|
|
3836
|
+
* 2. THE POSITIONS ARE PATCHED IN. They exist only for a run that {@link patched} admits to the
|
|
3837
|
+
* `agent:cancellation` shape, so a run already in flight when a deployment gained this keeps
|
|
3838
|
+
* replaying against the sequence it recorded.
|
|
3839
|
+
*
|
|
3840
|
+
* Undefined → the run cannot be cancelled and takes not one extra checkpoint, which is the honest
|
|
3841
|
+
* answer for a host with nowhere to record the request.
|
|
3842
|
+
*
|
|
3843
|
+
* WHAT IT CANNOT REACH: a turn parked on a human (`awaitApproval`/`awaitAnswers`) is suspended
|
|
3844
|
+
* INSIDE a position the journal already holds, so no observation of any kind fires there. Stopping
|
|
3845
|
+
* a parked run is the runner's job, through whatever hard cancel its runtime has.
|
|
3846
|
+
*/
|
|
3847
|
+
cancelled?(): Promise<boolean>;
|
|
3848
|
+
/**
|
|
3849
|
+
* Does this run take the loop shape guarded by `id`? A runner replaying against recorded
|
|
3850
|
+
* checkpoints answers `false` for a run that started before the shape changed, so that run keeps
|
|
3851
|
+
* replaying the shape its history holds (the durable runtime's `ctx.patched`, which consumes a
|
|
3852
|
+
* position for a new run and gives it back to an old one). Absent → `true`: a runner that records
|
|
3853
|
+
* no positions has no older shape to preserve.
|
|
3854
|
+
*/
|
|
3855
|
+
patched?(id: string): Promise<boolean>;
|
|
1653
3856
|
}
|
|
1654
3857
|
declare class QuotaExceededError extends Error {
|
|
1655
3858
|
constructor();
|
|
1656
3859
|
}
|
|
3860
|
+
/**
|
|
3861
|
+
* Someone asked this run to stop, and it did. NOT a failure — it is the outcome a user pressing Stop
|
|
3862
|
+
* is entitled to, and a consumer whose reliability numbers count `failed` runs must be able to leave
|
|
3863
|
+
* it out. Thrown by the loop at the point it observed the cancel, and settled by the runner, which
|
|
3864
|
+
* records the run `cancelled` and ends the stream with a `cancelled` frame rather than failing it.
|
|
3865
|
+
*
|
|
3866
|
+
* Carries no message detail on purpose: there is nothing to diagnose, and a cancel reads the same
|
|
3867
|
+
* whether it came from a Stop button, an operator console, or a deployment draining.
|
|
3868
|
+
*/
|
|
3869
|
+
declare class RunCancelledError extends Error {
|
|
3870
|
+
constructor();
|
|
3871
|
+
}
|
|
3872
|
+
/**
|
|
3873
|
+
* The stream error code a failed run surfaces to its subscriber. A refusal by an output processor
|
|
3874
|
+
* and an answer that never satisfied `outputSchema` are the CONTROLS working, not the model
|
|
3875
|
+
* breaking: a client that retries on `run_failed` must not retry either of them, and neither should
|
|
3876
|
+
* page whoever is on call for model failures.
|
|
3877
|
+
*/
|
|
3878
|
+
declare function agentFailureCode(error: unknown): string;
|
|
1657
3879
|
/** Reject if `work` doesn't settle within `ms`. The underlying work is left to finish on its own. */
|
|
1658
3880
|
declare function withToolTimeout<T>(work: Promise<T>, ms: number, toolName: string): Promise<T>;
|
|
1659
3881
|
/**
|
|
@@ -1672,14 +3894,26 @@ declare function traceToolExecution<T>(runId: string, call: {
|
|
|
1672
3894
|
toolName: string;
|
|
1673
3895
|
toolType: 'read' | 'action';
|
|
1674
3896
|
}, run: () => Promise<T>): Promise<T>;
|
|
3897
|
+
/**
|
|
3898
|
+
* Append the built-in `ask` definition to a turn's tool list. Exported because the dispatched llm
|
|
3899
|
+
* step re-derives the tool list on a worker and has to reach the same list the loop would have.
|
|
3900
|
+
*/
|
|
3901
|
+
declare function withAskTool({ tools, ask, }: {
|
|
3902
|
+
tools: ToolDefinition[];
|
|
3903
|
+
ask: boolean | undefined;
|
|
3904
|
+
}): ToolDefinition[];
|
|
3905
|
+
/** What a turn answered with: the assistant text, plus the validated `outputSchema` value if any. */
|
|
3906
|
+
interface AgentLoopResult<TOutput = unknown> {
|
|
3907
|
+
text: string;
|
|
3908
|
+
/** Present only when `AgentLoopDeps.outputSchema` was set — the validated structured answer. */
|
|
3909
|
+
object?: TOutput;
|
|
3910
|
+
}
|
|
1675
3911
|
/**
|
|
1676
3912
|
* The provider-agnostic agent turn, reused by both the inline and durable runners.
|
|
1677
3913
|
* It drives the model→tools→model iteration; the runner supplies the `step`/`awaitApproval`
|
|
1678
3914
|
* hooks that make the same loop body either in-process or a replay-safe durable workflow.
|
|
1679
3915
|
*/
|
|
1680
|
-
declare function runAgentLoop(deps: AgentLoopDeps
|
|
1681
|
-
text: string;
|
|
1682
|
-
}>;
|
|
3916
|
+
declare function runAgentLoop<TOutput = unknown>(deps: AgentLoopDeps<TOutput>, input: AgentRunInput, hooks: AgentLoopHooks): Promise<AgentLoopResult<TOutput>>;
|
|
1683
3917
|
|
|
1684
3918
|
/** Payloads carried on each `aviary:agent:*` channel. */
|
|
1685
3919
|
interface AgentRunStarted {
|
|
@@ -1724,6 +3958,8 @@ interface AgentDelegated {
|
|
|
1724
3958
|
runId: string;
|
|
1725
3959
|
fromAgent?: string;
|
|
1726
3960
|
toAgent: string;
|
|
3961
|
+
/** The delegate was STARTED, not awaited — this run's turn ended without its answer. */
|
|
3962
|
+
detached?: boolean;
|
|
1727
3963
|
}
|
|
1728
3964
|
interface AgentRetrieved {
|
|
1729
3965
|
runId: string;
|
|
@@ -1731,6 +3967,64 @@ interface AgentRetrieved {
|
|
|
1731
3967
|
/** How many passages the retriever returned. */
|
|
1732
3968
|
count: number;
|
|
1733
3969
|
}
|
|
3970
|
+
/**
|
|
3971
|
+
* The skills a turn was offered, published once per run at the `skills:catalog` checkpoint. Metadata
|
|
3972
|
+
* only — counts and the block's size, never a skill's name or its body.
|
|
3973
|
+
*/
|
|
3974
|
+
interface AgentSkillsResolved {
|
|
3975
|
+
runId: string;
|
|
3976
|
+
/** How many scope tokens the resolver returned for this turn. */
|
|
3977
|
+
scopes: number;
|
|
3978
|
+
/** Skills in the catalog block the model was shown. */
|
|
3979
|
+
offered: number;
|
|
3980
|
+
/** Applicable skills the `maxSkills` ceiling left out — non-zero means the catalog is partial. */
|
|
3981
|
+
omitted: number;
|
|
3982
|
+
/**
|
|
3983
|
+
* Characters the catalog block added to the system prompt. The WHOLE of what skills cost it: a
|
|
3984
|
+
* skill's body never enters the system block, it arrives as a tool result on the transcript.
|
|
3985
|
+
*/
|
|
3986
|
+
promptChars: number;
|
|
3987
|
+
}
|
|
3988
|
+
/**
|
|
3989
|
+
* The memories a turn was shown, published once per run at the `memory:digest` checkpoint. Metadata
|
|
3990
|
+
* only — counts and the block's size, never a memory's key or its text. A key is as revealing as a
|
|
3991
|
+
* fact (`diagnosis`, `clearance`), so it does not travel on a diagnostics channel either.
|
|
3992
|
+
*/
|
|
3993
|
+
interface AgentMemoryResolved {
|
|
3994
|
+
runId: string;
|
|
3995
|
+
/** How many scope tokens the resolver returned for this turn. */
|
|
3996
|
+
scopes: number;
|
|
3997
|
+
/** Memories in the block the model was shown. */
|
|
3998
|
+
offered: number;
|
|
3999
|
+
/** Applicable memories the `maxMemories` ceiling left out — non-zero means the block is partial. */
|
|
4000
|
+
omitted: number;
|
|
4001
|
+
/**
|
|
4002
|
+
* Of `omitted`, how many were ALWAYS-ON. The one to alert on: ordinary omission is the budget
|
|
4003
|
+
* working, while a dropped always-on memory is a deployment's own standing policies having stopped
|
|
4004
|
+
* reaching any prompt — more was pinned than the block holds.
|
|
4005
|
+
*/
|
|
4006
|
+
pinnedOmitted: number;
|
|
4007
|
+
/**
|
|
4008
|
+
* Whether the block was selected by relevance to this turn rather than read whole. Distinguishes
|
|
4009
|
+
* "this deployment holds few memories" from "this turn drew twenty out of two thousand", which
|
|
4010
|
+
* `offered` alone reports identically.
|
|
4011
|
+
*/
|
|
4012
|
+
recalled: boolean;
|
|
4013
|
+
/** Characters the memory block added to the system prompt. The WHOLE of what memory cost it. */
|
|
4014
|
+
promptChars: number;
|
|
4015
|
+
}
|
|
4016
|
+
/**
|
|
4017
|
+
* A turn writing one memory — the event an operator watches to see the agent's own write volume,
|
|
4018
|
+
* which is the risk surface memory has and retrieval does not. The scope token travels (it is what
|
|
4019
|
+
* says whose prompt just changed, and `run.started` already carries an actor id); the key and the
|
|
4020
|
+
* text do not.
|
|
4021
|
+
*/
|
|
4022
|
+
interface AgentMemoryWritten {
|
|
4023
|
+
runId: string;
|
|
4024
|
+
scope: string;
|
|
4025
|
+
/** Length of the stored fact in characters — never the fact itself. */
|
|
4026
|
+
chars: number;
|
|
4027
|
+
}
|
|
1734
4028
|
/**
|
|
1735
4029
|
* A transient-classified tool error being retried in place (no new checkpoint) — see
|
|
1736
4030
|
* `invokeWithTransientRetry`. Emitted once per retry (not for the final, non-retried outcome).
|
|
@@ -1771,6 +4065,17 @@ interface AgentFollowUpsSpan {
|
|
|
1771
4065
|
/** How many follow-up questions were requested. */
|
|
1772
4066
|
count: number;
|
|
1773
4067
|
}
|
|
4068
|
+
/**
|
|
4069
|
+
* START payload of an `aviary:agent:structured-output:*` span — the formatting pass that restates a
|
|
4070
|
+
* finished answer as `AgentLoopDeps.outputSchema`. `attempt` is 0 for the pass itself and counts up
|
|
4071
|
+
* for each bounded repair, so a run that needed three tries is visible as three spans.
|
|
4072
|
+
*/
|
|
4073
|
+
interface AgentStructuredOutputSpan {
|
|
4074
|
+
runId: string;
|
|
4075
|
+
/** Zero-based model-call index of the final turn whose answer is being restated. */
|
|
4076
|
+
step: number;
|
|
4077
|
+
attempt: number;
|
|
4078
|
+
}
|
|
1774
4079
|
/** Declaration-merge so `emit('agent', ...)`, `trace('agent', ...)` and telescope infer the agent payloads. */
|
|
1775
4080
|
declare module '@dudousxd/nestjs-diagnostics' {
|
|
1776
4081
|
interface ChannelRegistry {
|
|
@@ -1783,11 +4088,15 @@ declare module '@dudousxd/nestjs-diagnostics' {
|
|
|
1783
4088
|
'run.failed': AgentRunFailed;
|
|
1784
4089
|
delegated: AgentDelegated;
|
|
1785
4090
|
retrieved: AgentRetrieved;
|
|
4091
|
+
'skills.resolved': AgentSkillsResolved;
|
|
4092
|
+
'memory.resolved': AgentMemoryResolved;
|
|
4093
|
+
'memory.written': AgentMemoryWritten;
|
|
1786
4094
|
'tool.retry': AgentToolRetry;
|
|
1787
4095
|
'llm.turn': AgentLlmTurnSpan;
|
|
1788
4096
|
'tool.execution': AgentToolExecutionSpan;
|
|
1789
4097
|
retrieval: AgentRetrievalSpan;
|
|
1790
4098
|
'follow-ups': AgentFollowUpsSpan;
|
|
4099
|
+
'structured-output': AgentStructuredOutputSpan;
|
|
1791
4100
|
};
|
|
1792
4101
|
}
|
|
1793
4102
|
}
|
|
@@ -1800,6 +4109,9 @@ declare function publishAgentRunFailed(payload: AgentRunFailed): void;
|
|
|
1800
4109
|
declare function publishAgentDelegated(payload: AgentDelegated): void;
|
|
1801
4110
|
declare function publishAgentRetrieved(payload: AgentRetrieved): void;
|
|
1802
4111
|
declare function publishAgentToolRetry(payload: AgentToolRetry): void;
|
|
4112
|
+
declare function publishAgentSkillsResolved(payload: AgentSkillsResolved): void;
|
|
4113
|
+
declare function publishAgentMemoryResolved(payload: AgentMemoryResolved): void;
|
|
4114
|
+
declare function publishAgentMemoryWritten(payload: AgentMemoryWritten): void;
|
|
1803
4115
|
/**
|
|
1804
4116
|
* Events published ONLY as spans — via `trace('agent', ...)` on the five `:start`/`:end`/
|
|
1805
4117
|
* `:asyncStart`/`:asyncEnd`/`:error` sub-channels — never as point events on the base channel.
|
|
@@ -1807,13 +4119,13 @@ declare function publishAgentToolRetry(payload: AgentToolRetry): void;
|
|
|
1807
4119
|
* nothing to subscribe to on their base channels, and claiming their keys would be meaningless
|
|
1808
4120
|
* (the generic bridge only records point traffic).
|
|
1809
4121
|
*/
|
|
1810
|
-
type AgentSpanEvent = 'llm.turn' | 'tool.execution' | 'retrieval' | 'follow-ups';
|
|
4122
|
+
type AgentSpanEvent = 'llm.turn' | 'tool.execution' | 'retrieval' | 'follow-ups' | 'structured-output';
|
|
1811
4123
|
/** All span-only events, in a stable order — for a future span recorder to derive sub-channels from. */
|
|
1812
4124
|
declare const AGENT_SPAN_EVENTS: readonly AgentSpanEvent[];
|
|
1813
4125
|
/** Every POINT event key declared on `ChannelRegistry['agent']` above — derived, not hand-copied. */
|
|
1814
4126
|
type AgentDiagnosticEvent = Exclude<keyof ChannelRegistry['agent'], AgentSpanEvent>;
|
|
1815
4127
|
/**
|
|
1816
|
-
* All
|
|
4128
|
+
* All 12 point events on `ChannelRegistry['agent']`, in a stable order — handy for wiring
|
|
1817
4129
|
* subscribers (mirrors nestjs-media's `MEDIA_DIAGNOSTIC_EVENTS`). Span-only events (see
|
|
1818
4130
|
* {@link AgentSpanEvent}) are excluded. A drift between this list and the registry is a compile
|
|
1819
4131
|
* error in both directions: an extra/misspelled entry fails this array's own
|
|
@@ -1835,4 +4147,4 @@ type AgentDiagnosticKey = `agent:${AgentDiagnosticEvent}`;
|
|
|
1835
4147
|
*/
|
|
1836
4148
|
declare function agentDiagnosticKey(event: AgentDiagnosticEvent): AgentDiagnosticKey;
|
|
1837
4149
|
|
|
1838
|
-
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type ApprovalWhere, type AttachmentStagingStore, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type DetailThreadRef, type EmbeddingProvider, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceRunDetail, type GovernanceThreadDetail, type GovernanceThreadDetailQuery, type GovernanceUsageInput, type InvokeWithTransientRetryOptions, type LlmStepEnvelope, type MessageAttachment, type MessageRole, type MessageUsage, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type PageContext, type Passage, type PendingApprovalRow, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RerankOptions, type Reranker, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, type RunErrorBreakdownRow, type RunMetrics, type RunToolCallRow, type RunTrendPoint, type RunWhere, type SinkWriter, type StageAttachmentInput, type StoredMessage, type StreamError, THREAD_DETAIL_CONTENT_CHARS, type ThreadActivityRow, type ThreadDetail, type ThreadMessageRow, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadUsageRollup, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolDisabledError, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, agentDiagnosticKey, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, canActorUseTool, dayBoundsUtc, encodeStreamEvent, estimateCost, filterToolsByAllowList, filterToolsByCanUse, filterToolsByEnabled, filterToolsByRole, invokeWithTransientRetry, isToolEnabled, isTransientToolError, publishAgentDelegated, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentToolCall, publishAgentToolRetry, resolveToolTransientRetryNumbers, rollupThreadUsage, runAgentLoop, seedModelPrices, traceLlmTurn, traceToolExecution, truncateDetailContent, withToolTimeout };
|
|
4150
|
+
export { AGENT_ACTOR_DIRECTORY, AGENT_ACTOR_RESOLVER, AGENT_APPROVAL_PORT, AGENT_ATTACHMENT_STAGING, AGENT_DEPS_FACTORY, AGENT_DIAGNOSTIC_EVENTS, AGENT_DURABLE_RUNNER, AGENT_EMBEDDING_PROVIDER, AGENT_GOVERNANCE_QUERIES, AGENT_MEMORY, AGENT_MODEL, AGENT_OPTIONS, AGENT_PRICING_STORE, AGENT_PROMPT_CONTRIBUTORS, AGENT_QUOTA_STORE, AGENT_REGISTRY, AGENT_RETRIEVER, AGENT_ROLES_POLICY, AGENT_RUNNER, AGENT_SINK, AGENT_SKILLS, AGENT_SKILL_SOURCES, AGENT_SPAN_EVENTS, AGENT_STORE, AGENT_TOOL_REGISTRY, ASK_TOOL_DESCRIPTION, ASK_TOOL_NAME, type Actor, type ActorDirectory, type ActorResolver, type ActorSpendRow, type AgentApprovalPort, type AgentCatalogEntry, type AgentDefinition, type AgentDelegated, type AgentDelegation, type AgentDiagnosticEvent, type AgentDiagnosticKey, type AgentFollowUpsSpan, type AgentGovernanceQueries, type AgentHistoryWindow, type AgentIntake, type AgentLlmTurnSpan, type AgentLoopDeps, type AgentLoopHooks, type AgentLoopResult, type AgentMemoryResolved, type AgentMemoryWritten, type AgentMessageEvent, type AgentPricingStore, type AgentQuotaExceeded, AgentRegistry, type AgentRetrievalSpan, type AgentRetrieved, type AgentRunFailed, type AgentRunFinished, type AgentRunInput, type AgentRunStarted, type AgentRunner, type AgentSkillsResolved, type AgentSpanEvent, type AgentStore, AgentStreamError, type AgentStreamEvent, type AgentStructuredOutputSpan, type AgentToolCallEvent, type AgentToolExecutionSpan, type AgentToolRetry, type AiToolCtx, type AppendMessageInput, type ApprovalWhere, type AskToolInput, type AttachmentRef, type AttachmentStagingStore, type BufferedModelTurnResult, type BuildMemoryBlockInput, type CostUsage, type CreateThreadInput, type CurrentModelPrice, DEFAULT_HISTORY_SUMMARY_INSTRUCTION, DEFAULT_INCREMENTAL_LOOKBACK_CHARS, DEFAULT_INTAKE_PREAMBLE, DEFAULT_MAX_FACT_CHARS, DEFAULT_MAX_MEMORIES, DEFAULT_MAX_SKILLS, DEFAULT_STRUCTURED_OUTPUT_INSTRUCTION, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS, type Decision, DefaultRolesPolicy, type DetachedDelegationOutcome, type DetachedDelegationReceipt, type DetachedDelivery, type DetailThreadRef, type ElicitationOption, type ElicitationOutcome, type ElicitationQuestion, type ElicitationReply, type ElicitationRequest, type ElicitationResult, type EmbeddingProvider, type ForgetMemoryInput, type FrameBuffer, GLOBAL_SCOPE, type GovernancePage, type GovernancePageQuery, type GovernanceRange, type GovernanceRunDetail, type GovernanceThreadDetail, type GovernanceThreadDetailQuery, type GovernanceUsageInput, type HistoryPolicy, type HistoryPolicyContext, type HistorySelection, type HistorySummary, type HumanReply, type IncrementalGate, type IncrementalGating, type InputProcessor, type InvokeWithTransientRetryOptions, type ListMemoriesInput, type ListSkillsInput, type ListStagedAttachmentsInput, type LlmStepEnvelope, type LoadSkillInput, MAX_ASK_QUESTIONS, type MemoryAuthor, type MemoryConfig, type MemoryDigest, type MemoryDigestEntry, type MemoryFact, type MemoryForgetRequest, type MemoryOrigin, type MemoryProvider, type MemoryRecord, type MemoryVerdict, type MemoryWriteOutcome, type MemoryWriteRequest, type MessageAttachment, type MessageRole, type MessageUsage, type ModelAnswer, type ModelMessage, type ModelPrice, type ModelPriceInput, type ModelProvider, type ModelSpendRow, type ModelTurnArgs, type ModelTurnResult, type OfferMemoriesInput, type OutputGateMode, type OutputGateResult, type OutputProcessor, OutputRejectedError, type OutputVerdict, type OverriddenMemory, type PageContext, type Passage, type PendingApprovalRow, type ProcessedPrompt, type ProcessorContext, ProcessorFailedError, type PromptBuilder, type PromptContext, type PromptContributor, QuotaExceededError, type QuotaState, type QuotaStore, type QuotaView, REMEMBER_TOOL_DESCRIPTION, REMEMBER_TOOL_NAME, type RecentRunRow, type RecordRunEndInput, type RecordRunStartInput, type RecordToolCallInput, type RecordUsageInput, type RememberToolInput, type RerankOptions, type Reranker, type ResolveAttachmentInput, type ResolveMemoryDigestInput, type ResolvedDelegation, type RetrieveOptions, type Retriever, type RolesPolicy, type RunAgentBreakdownRow, RunCancelledError, type RunErrorBreakdownRow, type RunMetrics, type RunToolCallRow, type RunTrendPoint, type RunWhere, SKILL_TOOL_DESCRIPTION, SKILL_TOOL_NAME, type ScopeContext, type ScopeResolver, type SearchMemoriesInput, type SettledTask, type SinkWriter, type Skill, type SkillAuthor, type SkillCatalogEntry, type SkillContext, type SkillLoadOutcome, type SkillOffer, type SkillProvider, type SkillSummary, type SkillToolInput, type SkillWriteRequest, type SkillWriteVerdict, type SkillsConfig, type StageAttachmentInput, type StagedAttachment, type StoreMemoryInput, type StoredMessage, type StreamError, type StructuredOutcome, StructuredOutputError, THREAD_DETAIL_CONTENT_CHARS, type ThreadActivityRow, type ThreadDetail, type ThreadMessageRow, type ThreadMeta, type ThreadSpendRow, type ThreadSummary, type ThreadTurnPage, type ThreadTurnQuery, type ThreadTurnReader, type ThreadUsageRollup, type ThreadWhere, type TokenStreamSink, type ToolCallActivityRow, type ToolCallRequest, type ToolCallStatus, type ToolCallWhere, type ToolDefinition, ToolDisabledError, ToolForbiddenError, type ToolHandler, ToolInputInvalidError, type ToolKind, ToolNotFoundError, ToolRegistry, type ToolResult, type ToolSpec, type ToolStatRow, type ToolStepCtx, type ToolStepEnvelope, type ToolTransientRetryNumbers, type ToolTransientRetryOptions, type ToolTransientRetrySetting, type UpdateThreadInput, type UpdateToolCallInput, type UsagePurpose, type UsageTrendPoint, type WindowHistoryOptions, type WithMemoryToolInput, type WriteMemoryInput, actorScope, agentDiagnosticKey, agentFailureCode, askInputSchema, askToolDefinition, bucketByActor, bucketByModel, bucketByThread, bucketUsageTrend, buildMemoryBlock, buildSkillsBlock, canActorUseTool, compositeSkillProvider, createFrameBuffer, createIncrementalGate, dayBoundsUtc, decodeStreamEvent, defaultScopeResolver, detachedDelivered, detachedStarted, detachedUnsettled, encodeStreamEvent, estimateCost, estimateMessageTokens, extractJson, filterToolsByAllowList, filterToolsByCanUse, filterToolsByEnabled, filterToolsByRole, gateFollowUps, gateTail, invokeWithTransientRetry, isControlFlowSignal, isReplayIntegrityError, isToolEnabled, isTransientToolError, loadSkill, memoryForgetVerdict, memoryWriteVerdict, normalizeDelegation, normalizeElicitationReply, offerMemories, offerSkills, publishAgentDelegated, publishAgentMemoryResolved, publishAgentMemoryWritten, publishAgentMessage, publishAgentQuotaExceeded, publishAgentRetrieved, publishAgentRunFailed, publishAgentRunFinished, publishAgentRunStarted, publishAgentSkillsResolved, publishAgentToolCall, publishAgentToolRetry, releaseGatedFrames, rememberInputSchema, rememberToolDefinition, renderElicitationAnswers, repairInstruction, resolveElicitation, resolveGateLookback, resolveMemoryDigest, resolveOutputGateMode, resolveSkillCatalog, resolveToolTransientRetryNumbers, rollupThreadUsage, runAgentLoop, runInputProcessors, runOutputProcessors, seedModelPrices, settleAll, settleElicitation, settleUnsettledDelegation, skillInputSchema, skillToolDefinition, skillWriteVerdict, staticSkillProvider, summarizeWithModel, tenantScope, traceLlmTurn, traceToolExecution, truncateDetailContent, validateStructured, windowHistory, withAskTool, withMemoryTool, withSkillTool, withToolTimeout, writeMemory };
|