@kindgi/agents 0.1.2 → 0.1.4-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/blocks.d.ts +101 -0
- package/dist/blocks.d.ts.map +1 -0
- package/dist/blocks.js +149 -0
- package/dist/blocks.js.map +1 -0
- package/dist/conversation-binding.d.ts +9 -1
- package/dist/conversation-binding.d.ts.map +1 -1
- package/dist/define.d.ts +17 -2
- package/dist/define.d.ts.map +1 -1
- package/dist/define.js +93 -6
- package/dist/define.js.map +1 -1
- package/dist/guardrails-gate.d.ts +7 -4
- package/dist/guardrails-gate.d.ts.map +1 -1
- package/dist/guardrails-gate.js +5 -2
- package/dist/guardrails-gate.js.map +1 -1
- package/dist/handlers/compose-result.d.ts.map +1 -1
- package/dist/handlers/compose-result.js +3 -0
- package/dist/handlers/compose-result.js.map +1 -1
- package/dist/handlers/context.d.ts +24 -0
- package/dist/handlers/context.d.ts.map +1 -1
- package/dist/handlers/dispatch-tools.d.ts.map +1 -1
- package/dist/handlers/dispatch-tools.js +69 -43
- package/dist/handlers/dispatch-tools.js.map +1 -1
- package/dist/handlers/errors.d.ts +12 -1
- package/dist/handlers/errors.d.ts.map +1 -1
- package/dist/handlers/errors.js.map +1 -1
- package/dist/handlers/evaluate-guardrails.d.ts.map +1 -1
- package/dist/handlers/evaluate-guardrails.js +36 -1
- package/dist/handlers/evaluate-guardrails.js.map +1 -1
- package/dist/handlers/gate-decision.d.ts +62 -0
- package/dist/handlers/gate-decision.d.ts.map +1 -0
- package/dist/handlers/gate-decision.js +55 -0
- package/dist/handlers/gate-decision.js.map +1 -0
- package/dist/handlers/model-call.d.ts +10 -0
- package/dist/handlers/model-call.d.ts.map +1 -1
- package/dist/handlers/model-call.js +120 -34
- package/dist/handlers/model-call.js.map +1 -1
- package/dist/handlers/persist-provenance.d.ts.map +1 -1
- package/dist/handlers/persist-provenance.js +3 -1
- package/dist/handlers/persist-provenance.js.map +1 -1
- package/dist/handlers/persist-user-message.d.ts.map +1 -1
- package/dist/handlers/persist-user-message.js +5 -16
- package/dist/handlers/persist-user-message.js.map +1 -1
- package/dist/handlers/public-types.d.ts +37 -3
- package/dist/handlers/public-types.d.ts.map +1 -1
- package/dist/handlers/rehydrate.d.ts.map +1 -1
- package/dist/handlers/rehydrate.js +91 -1
- package/dist/handlers/rehydrate.js.map +1 -1
- package/dist/handlers/render-prompt.d.ts.map +1 -1
- package/dist/handlers/render-prompt.js +4 -1
- package/dist/handlers/render-prompt.js.map +1 -1
- package/dist/handlers/replay.d.ts +143 -0
- package/dist/handlers/replay.d.ts.map +1 -0
- package/dist/handlers/replay.js +177 -0
- package/dist/handlers/replay.js.map +1 -0
- package/dist/handlers/resolve-blocks.d.ts +32 -0
- package/dist/handlers/resolve-blocks.d.ts.map +1 -0
- package/dist/handlers/resolve-blocks.js +129 -0
- package/dist/handlers/resolve-blocks.js.map +1 -0
- package/dist/handlers/resolve-tools.d.ts +6 -2
- package/dist/handlers/resolve-tools.d.ts.map +1 -1
- package/dist/handlers/resolve-tools.js +59 -27
- package/dist/handlers/resolve-tools.js.map +1 -1
- package/dist/handlers/result-shape.d.ts +7 -0
- package/dist/handlers/result-shape.d.ts.map +1 -1
- package/dist/handlers/result-shape.js.map +1 -1
- package/dist/handlers/run-retrievals.d.ts.map +1 -1
- package/dist/handlers/run-retrievals.js +35 -35
- package/dist/handlers/run-retrievals.js.map +1 -1
- package/dist/handlers/run-snapshot.d.ts.map +1 -1
- package/dist/handlers/run-snapshot.js +1 -0
- package/dist/handlers/run-snapshot.js.map +1 -1
- package/dist/handlers/setup.d.ts +2 -0
- package/dist/handlers/setup.d.ts.map +1 -1
- package/dist/handlers/setup.js +29 -12
- package/dist/handlers/setup.js.map +1 -1
- package/dist/handlers/tool-hitl.d.ts +5 -9
- package/dist/handlers/tool-hitl.d.ts.map +1 -1
- package/dist/handlers/tool-hitl.js +5 -0
- package/dist/handlers/tool-hitl.js.map +1 -1
- package/dist/handlers/turn-environment.d.ts +13 -2
- package/dist/handlers/turn-environment.d.ts.map +1 -1
- package/dist/handlers/turn-environment.js +12 -3
- package/dist/handlers/turn-environment.js.map +1 -1
- package/dist/handlers/turn-provenance.d.ts +67 -0
- package/dist/handlers/turn-provenance.d.ts.map +1 -0
- package/dist/handlers/turn-provenance.js +160 -0
- package/dist/handlers/turn-provenance.js.map +1 -0
- package/dist/index.d.ts +10 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -1
- package/dist/invoke.d.ts +14 -1
- package/dist/invoke.d.ts.map +1 -1
- package/dist/invoke.js +37 -19
- package/dist/invoke.js.map +1 -1
- package/dist/pins.d.ts +54 -0
- package/dist/pins.d.ts.map +1 -0
- package/dist/pins.js +51 -0
- package/dist/pins.js.map +1 -0
- package/dist/prompt.d.ts +12 -2
- package/dist/prompt.d.ts.map +1 -1
- package/dist/prompt.js +41 -4
- package/dist/prompt.js.map +1 -1
- package/dist/provenance-emit.d.ts +4 -2
- package/dist/provenance-emit.d.ts.map +1 -1
- package/dist/provenance-emit.js +4 -2
- package/dist/provenance-emit.js.map +1 -1
- package/dist/run-snapshot-binding.d.ts +5 -0
- package/dist/run-snapshot-binding.d.ts.map +1 -1
- package/dist/schema.d.ts +34 -0
- package/dist/schema.d.ts.map +1 -1
- package/dist/schema.js +16 -0
- package/dist/schema.js.map +1 -1
- package/dist/streaming.d.ts +5 -0
- package/dist/streaming.d.ts.map +1 -1
- package/dist/streaming.js.map +1 -1
- package/dist/types.d.ts +65 -2
- package/dist/types.d.ts.map +1 -1
- package/migrations/0003_hesitant_captain_cross.sql +2 -0
- package/migrations/0004_stormy_moondragon.sql +1 -0
- package/migrations/meta/0003_snapshot.json +321 -0
- package/migrations/meta/0004_snapshot.json +327 -0
- package/migrations/meta/_journal.json +14 -0
- package/package.json +15 -15
- package/src/blocks.ts +231 -0
- package/src/conversation-binding.ts +9 -1
- package/src/define.ts +104 -7
- package/src/guardrails-gate.ts +17 -3
- package/src/handlers/compose-result.ts +3 -0
- package/src/handlers/context.ts +24 -0
- package/src/handlers/dispatch-tools.ts +75 -44
- package/src/handlers/errors.ts +13 -0
- package/src/handlers/evaluate-guardrails.ts +41 -1
- package/src/handlers/gate-decision.ts +101 -0
- package/src/handlers/model-call.ts +163 -33
- package/src/handlers/persist-provenance.ts +3 -1
- package/src/handlers/persist-user-message.ts +3 -16
- package/src/handlers/public-types.ts +38 -3
- package/src/handlers/rehydrate.ts +141 -9
- package/src/handlers/render-prompt.ts +13 -7
- package/src/handlers/replay.ts +314 -0
- package/src/handlers/resolve-blocks.ts +188 -0
- package/src/handlers/resolve-tools.ts +76 -28
- package/src/handlers/result-shape.ts +8 -0
- package/src/handlers/run-retrievals.ts +45 -41
- package/src/handlers/run-snapshot.ts +1 -0
- package/src/handlers/setup.ts +30 -22
- package/src/handlers/tool-hitl.ts +6 -10
- package/src/handlers/turn-environment.ts +28 -3
- package/src/handlers/turn-provenance.ts +230 -0
- package/src/index.ts +44 -0
- package/src/invoke.ts +49 -20
- package/src/pins.ts +98 -0
- package/src/prompt.ts +52 -4
- package/src/provenance-emit.ts +4 -1
- package/src/run-snapshot-binding.ts +6 -0
- package/src/schema.ts +16 -0
- package/src/streaming.ts +5 -0
- package/src/types.ts +68 -2
|
@@ -10,7 +10,9 @@
|
|
|
10
10
|
*
|
|
11
11
|
* - the environment — conversation, approval rules, guardrails, tools,
|
|
12
12
|
* policies — is resolved again, routed to the provider and model
|
|
13
|
-
* `setup` journaled
|
|
13
|
+
* `setup` journaled, with the tool versions it journaled (a range
|
|
14
|
+
* isn't resolved again: a version published meanwhile doesn't run
|
|
15
|
+
* mid-turn);
|
|
14
16
|
* - the messages the turn stored come back from the conversation, from
|
|
15
17
|
* the turn's user message (`persist-user-message` journals its
|
|
16
18
|
* sequence);
|
|
@@ -27,26 +29,47 @@
|
|
|
27
29
|
* A turn parked inside `setup` (the session gate) re-runs `setup`, and
|
|
28
30
|
* nothing here applies.
|
|
29
31
|
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
+
* - the provenance nodes of the steps that ran before the park are
|
|
33
|
+
* added again from what they left (`turn-provenance.ts`), so a
|
|
34
|
+
* resumed turn's DAG is whole: its user message, its retrievals, each
|
|
35
|
+
* model call, and the tool calls of each completed step. The step the
|
|
36
|
+
* turn parked in adds its own when it runs again;
|
|
37
|
+
* - the tool-call approvals decided so far come from the journal (the
|
|
38
|
+
* waitpoint each gate recorded, when the call parked, the decision
|
|
39
|
+
* and when it came), so each call's provenance shows the approval it
|
|
40
|
+
* waited on and who decided it.
|
|
32
41
|
*/
|
|
33
42
|
|
|
34
43
|
import type { LoopContext } from '@kindgi/handler';
|
|
35
|
-
import { type JournalEntry, bodyStepKey } from '@kindgi/runtime';
|
|
44
|
+
import { type JournalEntry, type ValueRecordedPayload, bodyStepKey } from '@kindgi/runtime';
|
|
45
|
+
import type { Timestamp } from '@kindgi/types';
|
|
36
46
|
|
|
37
47
|
import type { ConversationMessage, RetrievedFact } from '../types.js';
|
|
38
48
|
import type { TurnContext } from './context.js';
|
|
39
49
|
import { throwAgentTurnFailure } from './errors.js';
|
|
50
|
+
import { readGateDecision } from './gate-decision.js';
|
|
51
|
+
import { rehydrateReplay } from './replay.js';
|
|
52
|
+
import type { PinnedBlockVersions } from './resolve-blocks.js';
|
|
53
|
+
import { TOOL_GATE_RECORD_PREFIX } from './tool-hitl.js';
|
|
40
54
|
import {
|
|
41
55
|
loadTurnConversation,
|
|
42
56
|
resolveTurnEnvironment,
|
|
43
57
|
resolveTurnHitlPolicy,
|
|
44
58
|
} from './turn-environment.js';
|
|
59
|
+
import {
|
|
60
|
+
addInputNode,
|
|
61
|
+
addModelCallNode,
|
|
62
|
+
addRetrievalNodes,
|
|
63
|
+
addStepToolNodes,
|
|
64
|
+
} from './turn-provenance.js';
|
|
65
|
+
import type { ToolApproval } from './turn-provenance.js';
|
|
45
66
|
|
|
46
67
|
interface StepRecord {
|
|
47
68
|
readonly nodeId: string;
|
|
48
69
|
readonly output: unknown;
|
|
49
70
|
readonly inLoop: boolean;
|
|
71
|
+
/** When the step completed. */
|
|
72
|
+
readonly at: Timestamp;
|
|
50
73
|
}
|
|
51
74
|
|
|
52
75
|
/** Each step's last completion — a step at one loop iteration counts once. */
|
|
@@ -67,6 +90,7 @@ function completedSteps(journal: readonly JournalEntry[]): readonly StepRecord[]
|
|
|
67
90
|
nodeId: e.nodeId as unknown as string,
|
|
68
91
|
output: payload.output,
|
|
69
92
|
inLoop: payload.loopContext !== undefined,
|
|
93
|
+
at: e.timestamp,
|
|
70
94
|
});
|
|
71
95
|
}
|
|
72
96
|
return [...byStep.values()];
|
|
@@ -87,10 +111,12 @@ export async function rehydrateTurnContext(
|
|
|
87
111
|
journal: readonly JournalEntry[],
|
|
88
112
|
): Promise<boolean> {
|
|
89
113
|
const steps = completedSteps(journal);
|
|
90
|
-
const setup = outputOf<{
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
114
|
+
const setup = outputOf<{
|
|
115
|
+
readonly providerId: string;
|
|
116
|
+
readonly providerModel: string;
|
|
117
|
+
readonly toolVersions?: Readonly<Record<string, string>>;
|
|
118
|
+
readonly blockVersions?: PinnedBlockVersions;
|
|
119
|
+
}>(steps, 'setup');
|
|
94
120
|
if (setup === undefined) return false;
|
|
95
121
|
|
|
96
122
|
if (ctx.bindings.provenance?.newBuilder !== undefined) {
|
|
@@ -99,7 +125,15 @@ export async function rehydrateTurnContext(
|
|
|
99
125
|
}
|
|
100
126
|
await loadTurnConversation(ctx);
|
|
101
127
|
ctx.hitlPolicy = await resolveTurnHitlPolicy(ctx);
|
|
102
|
-
|
|
128
|
+
// The route and the tool versions `setup` resolved: a resumed turn
|
|
129
|
+
// runs those, never a range resolved again (a journal from before
|
|
130
|
+
// `toolVersions` was recorded resolves the ranges, as it did).
|
|
131
|
+
await resolveTurnEnvironment(
|
|
132
|
+
ctx,
|
|
133
|
+
{ providerId: setup.providerId, model: setup.providerModel },
|
|
134
|
+
setup.toolVersions,
|
|
135
|
+
setup.blockVersions,
|
|
136
|
+
);
|
|
103
137
|
|
|
104
138
|
const userMessage = outputOf<{ readonly sequence: number }>(steps, 'persist-user-message');
|
|
105
139
|
if (userMessage !== undefined) {
|
|
@@ -129,9 +163,107 @@ export async function rehydrateTurnContext(
|
|
|
129
163
|
ctx.usage.totalCostUsd += out.iterationUsage?.costUsd ?? 0;
|
|
130
164
|
if (out.provider !== undefined) ctx.lastProvider = out.provider;
|
|
131
165
|
}
|
|
166
|
+
ctx.toolApprovals = toolApprovalsOf(journal);
|
|
167
|
+
rebuildProvenance(ctx, steps, retrievals?.retrieved);
|
|
168
|
+
rehydrateReplay(ctx, journal);
|
|
132
169
|
return true;
|
|
133
170
|
}
|
|
134
171
|
|
|
172
|
+
/**
|
|
173
|
+
* The tool-call approvals the journal shows decided, by invocation id: the
|
|
174
|
+
* waitpoint each call's gate recorded (`dispatch-tools`), when the call
|
|
175
|
+
* parked on it (`wait.suspended`, the first), and the decision that
|
|
176
|
+
* resolved it (`wait.resumed`).
|
|
177
|
+
*/
|
|
178
|
+
function toolApprovalsOf(journal: readonly JournalEntry[]): ReadonlyMap<string, ToolApproval> {
|
|
179
|
+
const callOf = new Map<string, string>();
|
|
180
|
+
const parkedAt = new Map<string, Timestamp>();
|
|
181
|
+
const resumed = new Map<string, { readonly at: Timestamp; readonly value: unknown }>();
|
|
182
|
+
for (const e of journal) {
|
|
183
|
+
const gate = toolGateOf(e);
|
|
184
|
+
if (gate !== undefined) callOf.set(gate.waitTokenId, gate.invocationId);
|
|
185
|
+
const p = (e.payload ?? {}) as { readonly tokenId?: unknown; readonly value?: unknown };
|
|
186
|
+
if (typeof p.tokenId !== 'string') continue;
|
|
187
|
+
if (e.kind === 'wait.suspended' && !parkedAt.has(p.tokenId)) {
|
|
188
|
+
parkedAt.set(p.tokenId, e.timestamp);
|
|
189
|
+
}
|
|
190
|
+
if (e.kind === 'wait.resumed') resumed.set(p.tokenId, { at: e.timestamp, value: p.value });
|
|
191
|
+
}
|
|
192
|
+
const approvals = new Map<string, ToolApproval>();
|
|
193
|
+
for (const [waitTokenId, decided] of resumed) {
|
|
194
|
+
const invocationId = callOf.get(waitTokenId);
|
|
195
|
+
if (invocationId === undefined) continue;
|
|
196
|
+
approvals.set(invocationId, {
|
|
197
|
+
waitTokenId,
|
|
198
|
+
parkedAt: parkedAt.get(waitTokenId) ?? decided.at,
|
|
199
|
+
decidedAt: decided.at,
|
|
200
|
+
decision: readGateDecision(decided.value),
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
return approvals;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/** The waitpoint a tool call's gate recorded (`dispatch-tools`), if `e` is that record. */
|
|
207
|
+
function toolGateOf(
|
|
208
|
+
e: JournalEntry,
|
|
209
|
+
): { readonly waitTokenId: string; readonly invocationId: string } | undefined {
|
|
210
|
+
if (e.kind !== 'value.recorded') return undefined;
|
|
211
|
+
const p = (e.payload ?? {}) as Partial<ValueRecordedPayload>;
|
|
212
|
+
if (typeof p.key !== 'string' || !p.key.startsWith(TOOL_GATE_RECORD_PREFIX)) return undefined;
|
|
213
|
+
const waitTokenId = (p.value as { readonly waitTokenId?: unknown } | undefined)?.waitTokenId;
|
|
214
|
+
return typeof waitTokenId === 'string'
|
|
215
|
+
? { waitTokenId, invocationId: p.key.slice(TOOL_GATE_RECORD_PREFIX.length) }
|
|
216
|
+
: undefined;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/** A completed model-call step's output, as far as provenance reads it. */
|
|
220
|
+
interface ModelCallStepOutput {
|
|
221
|
+
readonly step?: number;
|
|
222
|
+
readonly callId?: string;
|
|
223
|
+
readonly finishReason?: string;
|
|
224
|
+
readonly provider?: { readonly id: string; readonly model: string };
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* The provenance of the steps that ran before the park, in the order
|
|
229
|
+
* they ran: the user message, the retrievals, then each model call with
|
|
230
|
+
* the tool calls its completed step stored.
|
|
231
|
+
*/
|
|
232
|
+
function rebuildProvenance(
|
|
233
|
+
ctx: TurnContext,
|
|
234
|
+
steps: readonly StepRecord[],
|
|
235
|
+
retrieved: readonly RetrievedFact[] | undefined,
|
|
236
|
+
): void {
|
|
237
|
+
const input = ctx.userMessage;
|
|
238
|
+
if (ctx.provenance === undefined || input === undefined) return;
|
|
239
|
+
addInputNode(ctx.provenance, input);
|
|
240
|
+
if (retrieved !== undefined) addRetrievalNodes(ctx.provenance, retrieved, input);
|
|
241
|
+
|
|
242
|
+
const storedByStep = new Map<number, readonly ConversationMessage[]>();
|
|
243
|
+
for (const s of steps.filter((s) => s.nodeId === 'dispatch-tools' && s.inLoop)) {
|
|
244
|
+
const out = s.output as {
|
|
245
|
+
readonly step?: number;
|
|
246
|
+
readonly iterationAppended?: readonly ConversationMessage[];
|
|
247
|
+
};
|
|
248
|
+
if (out.step !== undefined) storedByStep.set(out.step, out.iterationAppended ?? []);
|
|
249
|
+
}
|
|
250
|
+
const calls = steps
|
|
251
|
+
.filter((s) => s.nodeId === 'model-call' && s.inLoop)
|
|
252
|
+
.map((s) => ({ at: s.at, out: s.output as ModelCallStepOutput }))
|
|
253
|
+
.sort((a, b) => (a.out.step ?? 0) - (b.out.step ?? 0));
|
|
254
|
+
for (const { at, out } of calls) {
|
|
255
|
+
const { step, callId, provider, finishReason } = out;
|
|
256
|
+
if (step === undefined || callId === undefined || provider === undefined) continue;
|
|
257
|
+
addModelCallNode(
|
|
258
|
+
ctx.provenance,
|
|
259
|
+
{ step, callId, provider, finishReason: finishReason ?? 'stop', at },
|
|
260
|
+
input.sequence,
|
|
261
|
+
[...(ctx.toolResultIds ?? [])],
|
|
262
|
+
);
|
|
263
|
+
addStepToolNodes(ctx, step, storedByStep.get(step) ?? []);
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
135
267
|
/**
|
|
136
268
|
* The messages no completed step accounts for: those the step the turn
|
|
137
269
|
* parked in stored before it parked.
|
|
@@ -25,14 +25,20 @@ export function buildRenderPromptHandler(ctx: TurnContext): NodeHandler {
|
|
|
25
25
|
cause: null,
|
|
26
26
|
});
|
|
27
27
|
}
|
|
28
|
-
const rendered = renderInstructions(
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
28
|
+
const rendered = renderInstructions(
|
|
29
|
+
ctx.input.agent,
|
|
30
|
+
{
|
|
31
|
+
parameters: ctx.input.parameters ?? {},
|
|
32
|
+
...(ctx.input.input !== undefined && { input: ctx.input.input }),
|
|
33
|
+
conversation: {
|
|
34
|
+
id: ctx.input.conversationId,
|
|
35
|
+
turn: ctx.conversation.turnCount + 1,
|
|
36
|
+
},
|
|
37
|
+
...(ctx.blocks !== undefined && { settings: ctx.blocks.settings }),
|
|
34
38
|
},
|
|
35
|
-
|
|
39
|
+
// The pinned prompt block, when the instructions come from one.
|
|
40
|
+
ctx.blocks?.prompt?.content,
|
|
41
|
+
);
|
|
36
42
|
if (!rendered.ok) {
|
|
37
43
|
throwAgentTurnFailure({
|
|
38
44
|
code: 'model-invocation-failed',
|
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Replay turns: an eval run re-running a past run (`InvokeAgentInput.replay`)
|
|
6
|
+
* on an agent version, without doing anything the past run didn't already
|
|
7
|
+
* do. Each tool call is decided by the deployment's `ReplayBinding`:
|
|
8
|
+
*
|
|
9
|
+
* - `live`: the tool runs. Only a tool declared read-only (see
|
|
10
|
+
* `isReadOnlyTool`) with no approval to wait for can; a `live`
|
|
11
|
+
* decision for any other is refused here, whatever the binding says;
|
|
12
|
+
* - `recorded`: the past run's result for the same call is used;
|
|
13
|
+
* - `refused`: the tool doesn't run, and the model gets the given result.
|
|
14
|
+
*
|
|
15
|
+
* A turn marked as a replay with no binding refuses every call. Each
|
|
16
|
+
* decision is journaled (`NodeContext.record`), so a resumed turn keeps
|
|
17
|
+
* the decisions it made, and the turn's result lists them
|
|
18
|
+
* (`AgentTurnResult.replay`).
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import type { NodeContext } from '@kindgi/handler';
|
|
22
|
+
import type { RunReplayRef } from '@kindgi/runtime';
|
|
23
|
+
import type { JournalEntry, ValueRecordedPayload } from '@kindgi/runtime';
|
|
24
|
+
import type { Tool } from '@kindgi/tools';
|
|
25
|
+
import type { RunId, TenantId } from '@kindgi/types';
|
|
26
|
+
|
|
27
|
+
import type { RetrievedFact } from '../types.js';
|
|
28
|
+
|
|
29
|
+
import type { TurnContext } from './context.js';
|
|
30
|
+
import { throwAgentTurnFailure } from './errors.js';
|
|
31
|
+
|
|
32
|
+
/** The replay turn a `ReplayBinding` is asked about. */
|
|
33
|
+
export interface ReplayTurnRef {
|
|
34
|
+
readonly tenantId: TenantId;
|
|
35
|
+
/** The replay turn's own run. */
|
|
36
|
+
readonly runId: RunId;
|
|
37
|
+
readonly replay: RunReplayRef;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export interface ReplayToolInput extends ReplayTurnRef {
|
|
41
|
+
readonly tool: {
|
|
42
|
+
readonly id: string;
|
|
43
|
+
readonly version: string;
|
|
44
|
+
/** As the tool declares it; absent means it changes things. */
|
|
45
|
+
readonly mutating?: boolean;
|
|
46
|
+
readonly effects?: readonly string[];
|
|
47
|
+
};
|
|
48
|
+
readonly arguments: unknown;
|
|
49
|
+
/** The model's id for the call. */
|
|
50
|
+
readonly callId: string;
|
|
51
|
+
/** `true` when the tool's approval rules would ask for a review before it runs. */
|
|
52
|
+
readonly gated: boolean;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export type ReplayToolDecision =
|
|
56
|
+
| { readonly kind: 'live' }
|
|
57
|
+
| { readonly kind: 'recorded'; readonly result: unknown }
|
|
58
|
+
| { readonly kind: 'refused'; readonly result: unknown; readonly reason: string };
|
|
59
|
+
|
|
60
|
+
/** An approval decision the past run recorded. */
|
|
61
|
+
export interface ReplayApproval {
|
|
62
|
+
readonly approved: boolean;
|
|
63
|
+
readonly rationale?: string;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* How a deployment replays turns. Consulted only for a turn marked as a
|
|
68
|
+
* replay (`InvokeAgentInput.replay`).
|
|
69
|
+
*/
|
|
70
|
+
export interface ReplayBinding {
|
|
71
|
+
/** Decide one tool call. */
|
|
72
|
+
decideTool(input: ReplayToolInput): Promise<ReplayToolDecision>;
|
|
73
|
+
/**
|
|
74
|
+
* What the turn's retrievals return: the past run's, or `undefined` to
|
|
75
|
+
* retrieve live. Absent: retrieve live.
|
|
76
|
+
*/
|
|
77
|
+
retrievals?(input: ReplayTurnRef): Promise<readonly RetrievedFact[] | undefined>;
|
|
78
|
+
/**
|
|
79
|
+
* The past run's decision at the session approval gate, when the replay
|
|
80
|
+
* reaches that gate: the replay follows it. `undefined` (or absent): the
|
|
81
|
+
* gate is skipped, and the result says so.
|
|
82
|
+
*/
|
|
83
|
+
sessionApproval?(input: ReplayTurnRef): Promise<ReplayApproval | undefined>;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** One tool call of a replay turn, and what happened to it. */
|
|
87
|
+
export interface ReplayToolTrace {
|
|
88
|
+
readonly step: number;
|
|
89
|
+
readonly callId: string;
|
|
90
|
+
readonly toolId: string;
|
|
91
|
+
readonly toolVersion: string;
|
|
92
|
+
readonly arguments: unknown;
|
|
93
|
+
readonly source: 'live' | 'recorded' | 'refused';
|
|
94
|
+
/** Why it was refused. A refused call is what the turn would have done. */
|
|
95
|
+
readonly reason?: string;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** What a replay turn did differently from a live one (`AgentTurnResult.replay`). */
|
|
99
|
+
export interface ReplayTurnReport extends RunReplayRef {
|
|
100
|
+
readonly tools: readonly ReplayToolTrace[];
|
|
101
|
+
/**
|
|
102
|
+
* The session approval gate, when the replay reached it: `followed` the
|
|
103
|
+
* past run's recorded decision, or `skipped` (none was recorded).
|
|
104
|
+
*/
|
|
105
|
+
readonly approval?: 'followed' | 'skipped';
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** The `record` key of a replay's tool decision, per call. */
|
|
109
|
+
export const REPLAY_TOOL_RECORD_PREFIX = 'replay-tool:';
|
|
110
|
+
/** The `record` key of a replay's session approval. */
|
|
111
|
+
export const REPLAY_APPROVAL_RECORD = 'replay-session-approval';
|
|
112
|
+
|
|
113
|
+
/** Effects a read-only tool can't declare. */
|
|
114
|
+
const CHANGING_EFFECTS: ReadonlySet<string> = new Set([
|
|
115
|
+
'writes',
|
|
116
|
+
'deletes',
|
|
117
|
+
'spawns-run',
|
|
118
|
+
'emits-event',
|
|
119
|
+
'external-side-effect',
|
|
120
|
+
]);
|
|
121
|
+
|
|
122
|
+
/** A tool that declares it changes nothing: `mutating: false`, and no changing effect. */
|
|
123
|
+
export function isReadOnlyTool(tool: {
|
|
124
|
+
readonly mutating?: boolean;
|
|
125
|
+
readonly effects?: readonly { readonly kind: string }[] | readonly string[];
|
|
126
|
+
}): boolean {
|
|
127
|
+
if (tool.mutating !== false) return false;
|
|
128
|
+
return !(tool.effects ?? []).some((e) =>
|
|
129
|
+
CHANGING_EFFECTS.has(typeof e === 'string' ? e : e.kind),
|
|
130
|
+
);
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** What the model gets for a call a replay refuses because nothing decided it. */
|
|
134
|
+
const NO_BINDING_REASON = 'replay: no replay binding is wired, so no tool runs';
|
|
135
|
+
const CHANGES_REASON = 'replay: this call changes things and has no recorded result';
|
|
136
|
+
const GATED_REASON = 'replay: this call needs an approval and has no recorded result';
|
|
137
|
+
|
|
138
|
+
function refusedResult(reason: string): Readonly<Record<string, unknown>> {
|
|
139
|
+
return { status: 'not-executed', reason };
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** The journaled decision of one call: its trace entry, and the result the model got. */
|
|
143
|
+
interface RecordedDecision extends ReplayToolTrace {
|
|
144
|
+
readonly result?: unknown;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Decide one tool call of a replay turn, once: the binding's decision,
|
|
149
|
+
* held to `isReadOnlyTool`, journaled, and added to the turn's trace.
|
|
150
|
+
*/
|
|
151
|
+
export async function decideReplayTool(
|
|
152
|
+
ctx: TurnContext,
|
|
153
|
+
kctx: NodeContext,
|
|
154
|
+
call: {
|
|
155
|
+
readonly step: number;
|
|
156
|
+
readonly callId: string;
|
|
157
|
+
readonly tool: Tool;
|
|
158
|
+
readonly version: string;
|
|
159
|
+
readonly arguments: unknown;
|
|
160
|
+
readonly gated: boolean;
|
|
161
|
+
},
|
|
162
|
+
): Promise<ReplayToolDecision> {
|
|
163
|
+
const replay = ctx.input.replay;
|
|
164
|
+
if (replay === undefined) return { kind: 'live' };
|
|
165
|
+
const base = {
|
|
166
|
+
step: call.step,
|
|
167
|
+
callId: call.callId,
|
|
168
|
+
toolId: call.tool.id as unknown as string,
|
|
169
|
+
toolVersion: call.version,
|
|
170
|
+
arguments: call.arguments,
|
|
171
|
+
};
|
|
172
|
+
const refuse = (reason: string): RecordedDecision => ({
|
|
173
|
+
...base,
|
|
174
|
+
source: 'refused',
|
|
175
|
+
reason,
|
|
176
|
+
result: refusedResult(reason),
|
|
177
|
+
});
|
|
178
|
+
const decided = await kctx.record(
|
|
179
|
+
`${REPLAY_TOOL_RECORD_PREFIX}${call.callId}`,
|
|
180
|
+
async (): Promise<RecordedDecision> => {
|
|
181
|
+
const binding = ctx.bindings.replay;
|
|
182
|
+
if (binding === undefined) return refuse(NO_BINDING_REASON);
|
|
183
|
+
const decision = await binding.decideTool({
|
|
184
|
+
tenantId: ctx.input.tenantId,
|
|
185
|
+
runId: kctx.runId,
|
|
186
|
+
replay,
|
|
187
|
+
tool: {
|
|
188
|
+
id: base.toolId,
|
|
189
|
+
version: call.version,
|
|
190
|
+
...(call.tool.mutating !== undefined && { mutating: call.tool.mutating }),
|
|
191
|
+
...(call.tool.effects !== undefined && {
|
|
192
|
+
effects: call.tool.effects.map((e) => e.kind),
|
|
193
|
+
}),
|
|
194
|
+
},
|
|
195
|
+
arguments: call.arguments,
|
|
196
|
+
callId: call.callId,
|
|
197
|
+
gated: call.gated,
|
|
198
|
+
});
|
|
199
|
+
if (decision.kind === 'recorded') {
|
|
200
|
+
return { ...base, source: 'recorded', result: decision.result };
|
|
201
|
+
}
|
|
202
|
+
if (decision.kind === 'refused') {
|
|
203
|
+
return { ...base, source: 'refused', reason: decision.reason, result: decision.result };
|
|
204
|
+
}
|
|
205
|
+
// `live` runs only a read-only tool, and never waits for an approval.
|
|
206
|
+
if (!isReadOnlyTool(call.tool)) return refuse(CHANGES_REASON);
|
|
207
|
+
if (call.gated) return refuse(GATED_REASON);
|
|
208
|
+
return { ...base, source: 'live' };
|
|
209
|
+
},
|
|
210
|
+
);
|
|
211
|
+
addReplayTrace(ctx, traceOf(decided));
|
|
212
|
+
if (decided.source === 'live') return { kind: 'live' };
|
|
213
|
+
if (decided.source === 'recorded') return { kind: 'recorded', result: decided.result };
|
|
214
|
+
return { kind: 'refused', result: decided.result, reason: decided.reason ?? CHANGES_REASON };
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/** A journaled decision's trace entry (without the result). */
|
|
218
|
+
function traceOf(decided: RecordedDecision): ReplayToolTrace {
|
|
219
|
+
const { result: _result, ...trace } = decided;
|
|
220
|
+
return trace;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
function addReplayTrace(ctx: TurnContext, entry: ReplayToolTrace): void {
|
|
224
|
+
const trace = ctx.replayTrace ?? [];
|
|
225
|
+
ctx.replayTrace = trace;
|
|
226
|
+
// A step run again after a resume decides its calls again (from the journal).
|
|
227
|
+
if (trace.some((t) => t.callId === entry.callId)) return;
|
|
228
|
+
trace.push(entry);
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/** The session approval a replay follows, decided once: the past run's, or none. */
|
|
232
|
+
export async function replaySessionApproval(
|
|
233
|
+
ctx: TurnContext,
|
|
234
|
+
kctx: NodeContext,
|
|
235
|
+
): Promise<ReplayApproval | undefined> {
|
|
236
|
+
const replay = ctx.input.replay;
|
|
237
|
+
if (replay === undefined) return undefined;
|
|
238
|
+
const recorded = await kctx.record(
|
|
239
|
+
REPLAY_APPROVAL_RECORD,
|
|
240
|
+
async (): Promise<{ readonly approval: ReplayApproval | null }> => {
|
|
241
|
+
const approval = await ctx.bindings.replay?.sessionApproval?.({
|
|
242
|
+
tenantId: ctx.input.tenantId,
|
|
243
|
+
runId: kctx.runId,
|
|
244
|
+
replay,
|
|
245
|
+
});
|
|
246
|
+
return { approval: approval ?? null };
|
|
247
|
+
},
|
|
248
|
+
);
|
|
249
|
+
ctx.replayApproval = recorded.approval === null ? 'skipped' : 'followed';
|
|
250
|
+
return recorded.approval ?? undefined;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* A replay at the session approval gate: it goes on when the past run's
|
|
255
|
+
* reviewer approved (or none was recorded), and fails as the past run did
|
|
256
|
+
* (`hitl-rejected`) when they rejected.
|
|
257
|
+
*/
|
|
258
|
+
export async function followReplaySessionGate(ctx: TurnContext, kctx: NodeContext): Promise<void> {
|
|
259
|
+
const approval = await replaySessionApproval(ctx, kctx);
|
|
260
|
+
if (approval === undefined || approval.approved) return;
|
|
261
|
+
throwAgentTurnFailure({
|
|
262
|
+
code: 'hitl-rejected',
|
|
263
|
+
message: `The replayed run's reviewer rejected the session-HITL gate${
|
|
264
|
+
approval.rationale !== undefined ? `: ${approval.rationale}` : ''
|
|
265
|
+
}`,
|
|
266
|
+
...(approval.rationale !== undefined && { rationale: approval.rationale }),
|
|
267
|
+
} as never);
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
/**
|
|
271
|
+
* A resumed replay turn's trace and approval, from the journal: every call
|
|
272
|
+
* decided before the park, and the session approval if `setup` reached it.
|
|
273
|
+
* The step the turn parked in decides its calls again, from the journal.
|
|
274
|
+
*/
|
|
275
|
+
export function rehydrateReplay(ctx: TurnContext, journal: readonly JournalEntry[]): void {
|
|
276
|
+
if (ctx.input.replay === undefined) return;
|
|
277
|
+
for (const e of journal) {
|
|
278
|
+
if (e.kind !== 'value.recorded') continue;
|
|
279
|
+
const p = (e.payload ?? {}) as Partial<ValueRecordedPayload>;
|
|
280
|
+
if (p.key === REPLAY_APPROVAL_RECORD) {
|
|
281
|
+
const approval = (p.value as { readonly approval?: unknown } | undefined)?.approval;
|
|
282
|
+
ctx.replayApproval = approval == null ? 'skipped' : 'followed';
|
|
283
|
+
} else if (p.key?.startsWith(REPLAY_TOOL_RECORD_PREFIX) === true) {
|
|
284
|
+
restoreDecision(ctx, p.value);
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
function restoreDecision(ctx: TurnContext, value: unknown): void {
|
|
290
|
+
const decided = value as RecordedDecision | undefined;
|
|
291
|
+
if (decided?.source !== undefined && decided.callId !== undefined) {
|
|
292
|
+
addReplayTrace(ctx, traceOf(decided));
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/** The turn's replay report, for its result. */
|
|
297
|
+
export function replayReport(ctx: TurnContext): ReplayTurnReport | undefined {
|
|
298
|
+
const replay = ctx.input.replay;
|
|
299
|
+
if (replay === undefined) return undefined;
|
|
300
|
+
return {
|
|
301
|
+
of: replay.of,
|
|
302
|
+
evalRunId: replay.evalRunId,
|
|
303
|
+
tools: [...(ctx.replayTrace ?? [])].sort((a, b) => a.step - b.step),
|
|
304
|
+
...(ctx.replayApproval !== undefined && { approval: ctx.replayApproval }),
|
|
305
|
+
};
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/** A replay's tag on its usage records (`ModelUsageRecord.replay`). */
|
|
309
|
+
export function replayTag(replay: RunReplayRef): {
|
|
310
|
+
readonly of: string;
|
|
311
|
+
readonly evalRunId: string;
|
|
312
|
+
} {
|
|
313
|
+
return { of: replay.of as unknown as string, evalRunId: replay.evalRunId };
|
|
314
|
+
}
|