@pikku/core 0.12.80 → 0.12.83
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +345 -0
- package/dist/errors/index.d.ts +1 -1
- package/dist/errors/index.js +1 -1
- package/dist/function/function-runner.js +2 -5
- package/dist/function/index.d.ts +1 -1
- package/dist/index.d.ts +11 -11
- package/dist/index.js +3 -3
- package/dist/pikku-state.js +4 -0
- package/dist/services/ai-agent-runner-service.d.ts +7 -0
- package/dist/services/ai-run-state-service.d.ts +10 -0
- package/dist/services/in-memory-ai-run-state-service.d.ts +5 -1
- package/dist/services/in-memory-ai-run-state-service.js +9 -0
- package/dist/services/index.d.ts +15 -16
- package/dist/services/index.js +5 -5
- package/dist/services/meta-service.d.ts +2 -1
- package/dist/services/scoped-credential-service.d.ts +21 -0
- package/dist/services/scoped-credential-service.js +53 -0
- package/dist/testing/service-tests/ai-storage-service-tests.js +76 -0
- package/dist/types/core.types.d.ts +2 -3
- package/dist/types/state.types.d.ts +19 -1
- package/dist/wirings/actor-flow/index.d.ts +1 -1
- package/dist/wirings/ai-agent/ai-agent-finalize.d.ts +58 -0
- package/dist/wirings/ai-agent/ai-agent-finalize.js +138 -0
- package/dist/wirings/ai-agent/ai-agent-interrupt.js +1 -0
- package/dist/wirings/ai-agent/ai-agent-memory.d.ts +2 -8
- package/dist/wirings/ai-agent/ai-agent-memory.js +34 -17
- package/dist/wirings/ai-agent/ai-agent-model-config.d.ts +7 -0
- package/dist/wirings/ai-agent/ai-agent-model-config.js +44 -1
- package/dist/wirings/ai-agent/ai-agent-prepare.js +4 -0
- package/dist/wirings/ai-agent/ai-agent-runner.js +61 -40
- package/dist/wirings/ai-agent/ai-agent-stream.js +89 -36
- package/dist/wirings/ai-agent/ai-agent-turn.d.ts +1 -0
- package/dist/wirings/ai-agent/ai-agent-turn.js +1 -0
- package/dist/wirings/ai-agent/ai-agent.types.d.ts +46 -1
- package/dist/wirings/ai-agent/index.d.ts +8 -7
- package/dist/wirings/ai-agent/index.js +5 -4
- package/dist/wirings/ai-scorer/ai-scorer-grade.d.ts +26 -0
- package/dist/wirings/ai-scorer/ai-scorer-grade.js +33 -0
- package/dist/wirings/ai-scorer/ai-scorer-judge.d.ts +17 -0
- package/dist/wirings/ai-scorer/ai-scorer-judge.js +92 -0
- package/dist/wirings/ai-scorer/ai-scorer-live.d.ts +15 -0
- package/dist/wirings/ai-scorer/ai-scorer-live.js +38 -0
- package/dist/wirings/ai-scorer/ai-scorer-registry.d.ts +18 -0
- package/dist/wirings/ai-scorer/ai-scorer-registry.js +46 -0
- package/dist/wirings/ai-scorer/ai-scorer-sampling.d.ts +8 -0
- package/dist/wirings/ai-scorer/ai-scorer-sampling.js +31 -0
- package/dist/wirings/ai-scorer/ai-scorer-snapshots.d.ts +10 -0
- package/dist/wirings/ai-scorer/ai-scorer-snapshots.js +40 -0
- package/dist/wirings/ai-scorer/ai-scorer-worker.d.ts +15 -0
- package/dist/wirings/ai-scorer/ai-scorer-worker.js +58 -0
- package/dist/wirings/ai-scorer/ai-scorer.d.ts +39 -0
- package/dist/wirings/ai-scorer/ai-scorer.js +40 -0
- package/dist/wirings/ai-scorer/ai-scorer.types.d.ts +90 -0
- package/dist/wirings/ai-scorer/ai-scorer.types.js +4 -0
- package/dist/wirings/ai-scorer/index.d.ts +6 -0
- package/dist/wirings/ai-scorer/index.js +5 -0
- package/dist/wirings/channel/index.d.ts +5 -6
- package/dist/wirings/channel/index.js +3 -4
- package/dist/wirings/channel/local/local-channel-runner.js +8 -1
- package/dist/wirings/cli/channel/cli-raw-channel-runner.js +9 -1
- package/dist/wirings/cli/channel/index.d.ts +1 -2
- package/dist/wirings/cli/channel/index.js +0 -1
- package/dist/wirings/cli/cli-runner.js +13 -1
- package/dist/wirings/credential/index.d.ts +1 -1
- package/dist/wirings/gateway/index.d.ts +1 -1
- package/dist/wirings/http/http-runner.js +8 -2
- package/dist/wirings/http/index.d.ts +1 -2
- package/dist/wirings/mcp/index.d.ts +1 -1
- package/dist/wirings/mcp/mcp-runner.d.ts +15 -0
- package/dist/wirings/mcp/mcp-runner.js +18 -5
- package/dist/wirings/persona/index.d.ts +3 -4
- package/dist/wirings/persona/index.js +2 -3
- package/dist/wirings/queue/index.d.ts +1 -3
- package/dist/wirings/queue/index.js +1 -3
- package/dist/wirings/rpc/addon-runner.d.ts +8 -0
- package/dist/wirings/rpc/addon-runner.js +31 -3
- package/dist/wirings/rpc/rpc-runner.js +4 -0
- package/dist/wirings/rpc/rpc-types.d.ts +8 -0
- package/dist/wirings/rpc/wire-addon.d.ts +25 -0
- package/dist/wirings/rpc/wire-addon.js +8 -0
- package/dist/wirings/scheduler/index.d.ts +1 -1
- package/dist/wirings/trigger/index.d.ts +1 -1
- package/dist/wirings/virtual-user/index.d.ts +5 -6
- package/dist/wirings/virtual-user/index.js +2 -4
- package/dist/wirings/workflow/dsl/workflow-dsl.types.d.ts +85 -15
- package/dist/wirings/workflow/feature.d.ts +2 -1
- package/dist/wirings/workflow/index.d.ts +5 -16
- package/dist/wirings/workflow/index.js +1 -9
- package/dist/wirings/workflow/pikku-scenario-service.d.ts +17 -7
- package/dist/wirings/workflow/pikku-scenario-service.js +48 -13
- package/dist/wirings/workflow/pikku-workflow-service.js +17 -3
- package/dist/wirings/workflow/scenario-step.types.d.ts +8 -0
- package/dist/wirings/workflow/scenario.types.d.ts +37 -0
- package/dist/wirings/workflow/workflow-approval-audit.d.ts +16 -0
- package/dist/wirings/workflow/workflow-approval-audit.js +40 -0
- package/dist/wirings/workflow/workflow-approval-policy.d.ts +20 -0
- package/dist/wirings/workflow/workflow-approval-policy.js +48 -0
- package/dist/wirings/workflow/workflow-approval.d.ts +29 -1
- package/dist/wirings/workflow/workflow-approval.js +65 -2
- package/dist/wirings/workflow/workflow-run-ownership.d.ts +2 -1
- package/dist/wirings/workflow/workflow-run-ownership.js +2 -1
- package/dist/wirings/workflow/workflow.types.d.ts +2 -37
- package/knowledge/decisions/internals/addon-pikku-meta-ships-at-the-package-root-or-under-dist.md +32 -0
- package/knowledge/decisions/internals/an-addon-scope-root-loses-to-a-root-the-host-already-declares.md +39 -0
- package/knowledge/decisions/internals/index.md +30 -3
- package/knowledge/decisions/internals/validate-runs-checks-by-precondition.md +115 -0
- package/knowledge/decisions/security/a-function-never-receives-the-secret-service.md +37 -0
- package/knowledge/decisions/security/a-workflow-run-is-read-and-approved-by-its-owner.md +30 -14
- package/knowledge/decisions/security/an-approval-answer-outlives-the-run-it-answered.md +59 -0
- package/knowledge/decisions/security/index.md +3 -1
- package/knowledge/questions/index.md +1 -1
- package/package.json +3 -2
- package/scripts/generate-api-report.mts +143 -18
- package/src/api-report.test.ts +2 -2
- package/src/errors/index.ts +1 -1
- package/src/function/function-runner.test.ts +52 -0
- package/src/function/function-runner.ts +5 -9
- package/src/function/index.ts +0 -2
- package/src/index.ts +0 -35
- package/src/pikku-state.ts +5 -0
- package/src/public-surface.json +81 -118
- package/src/services/ai-agent-runner-service.ts +12 -1
- package/src/services/ai-run-state-service.ts +11 -0
- package/src/services/in-memory-ai-run-state-service.ts +13 -0
- package/src/services/index.ts +7 -58
- package/src/services/meta-service.ts +2 -4
- package/src/services/scoped-credential-service.test.ts +86 -0
- package/src/services/scoped-credential-service.ts +63 -0
- package/src/testing/service-tests/ai-storage-service-tests.ts +93 -0
- package/src/types/core.types.ts +4 -7
- package/src/types/state.types.ts +21 -1
- package/src/wirings/actor-flow/index.ts +0 -3
- package/src/wirings/ai-agent/ai-agent-finalize.test.ts +186 -0
- package/src/wirings/ai-agent/ai-agent-finalize.ts +197 -0
- package/src/wirings/ai-agent/ai-agent-interrupt.ts +1 -0
- package/src/wirings/ai-agent/ai-agent-memory.ts +54 -38
- package/src/wirings/ai-agent/ai-agent-model-config.test.ts +72 -3
- package/src/wirings/ai-agent/ai-agent-model-config.ts +49 -1
- package/src/wirings/ai-agent/ai-agent-prepare.ts +4 -0
- package/src/wirings/ai-agent/ai-agent-runner.ts +71 -40
- package/src/wirings/ai-agent/ai-agent-stream-output-hooks.test.ts +353 -0
- package/src/wirings/ai-agent/ai-agent-stream.ts +116 -54
- package/src/wirings/ai-agent/ai-agent-turn.test.ts +67 -0
- package/src/wirings/ai-agent/ai-agent-turn.ts +1 -0
- package/src/wirings/ai-agent/ai-agent.types.ts +64 -4
- package/src/wirings/ai-agent/index.ts +2 -16
- package/src/wirings/ai-scorer/ai-scorer-grade.test.ts +106 -0
- package/src/wirings/ai-scorer/ai-scorer-grade.ts +55 -0
- package/src/wirings/ai-scorer/ai-scorer-judge.test.ts +143 -0
- package/src/wirings/ai-scorer/ai-scorer-judge.ts +120 -0
- package/src/wirings/ai-scorer/ai-scorer-live.test.ts +174 -0
- package/src/wirings/ai-scorer/ai-scorer-live.ts +56 -0
- package/src/wirings/ai-scorer/ai-scorer-registry.ts +63 -0
- package/src/wirings/ai-scorer/ai-scorer-sampling.test.ts +34 -0
- package/src/wirings/ai-scorer/ai-scorer-sampling.ts +36 -0
- package/src/wirings/ai-scorer/ai-scorer-snapshots.test.ts +49 -0
- package/src/wirings/ai-scorer/ai-scorer-snapshots.ts +46 -0
- package/src/wirings/ai-scorer/ai-scorer-worker.test.ts +122 -0
- package/src/wirings/ai-scorer/ai-scorer-worker.ts +69 -0
- package/src/wirings/ai-scorer/ai-scorer.ts +76 -0
- package/src/wirings/ai-scorer/ai-scorer.types.ts +107 -0
- package/src/wirings/ai-scorer/index.ts +24 -0
- package/src/wirings/channel/index.ts +1 -20
- package/src/wirings/channel/local/local-channel-runner.test.ts +68 -0
- package/src/wirings/channel/local/local-channel-runner.ts +8 -1
- package/src/wirings/cli/channel/cli-raw-channel-runner.test.ts +23 -0
- package/src/wirings/cli/channel/cli-raw-channel-runner.ts +12 -1
- package/src/wirings/cli/channel/index.ts +0 -7
- package/src/wirings/cli/cli-runner.test.ts +68 -0
- package/src/wirings/cli/cli-runner.ts +18 -1
- package/src/wirings/credential/index.ts +0 -1
- package/src/wirings/gateway/index.ts +0 -3
- package/src/wirings/http/http-runner.test.ts +66 -0
- package/src/wirings/http/http-runner.ts +10 -2
- package/src/wirings/http/index.ts +1 -1
- package/src/wirings/mcp/index.ts +0 -1
- package/src/wirings/mcp/mcp-runner.test.ts +181 -0
- package/src/wirings/mcp/mcp-runner.ts +35 -5
- package/src/wirings/persona/index.ts +0 -8
- package/src/wirings/queue/index.ts +0 -14
- package/src/wirings/rpc/addon-runner.ts +62 -3
- package/src/wirings/rpc/addon-secrets.test.ts +391 -0
- package/src/wirings/rpc/rpc-runner.test.ts +2 -0
- package/src/wirings/rpc/rpc-runner.ts +4 -0
- package/src/wirings/rpc/rpc-types.ts +8 -0
- package/src/wirings/rpc/wire-addon.ts +33 -0
- package/src/wirings/scheduler/index.ts +0 -1
- package/src/wirings/trigger/index.ts +0 -1
- package/src/wirings/virtual-user/index.ts +0 -16
- package/src/wirings/workflow/dsl/workflow-dsl.types.ts +96 -16
- package/src/wirings/workflow/feature.ts +2 -5
- package/src/wirings/workflow/graph/graph-runner.test.ts +72 -0
- package/src/wirings/workflow/index.ts +2 -68
- package/src/wirings/workflow/pikku-scenario-service.ts +81 -16
- package/src/wirings/workflow/pikku-workflow-service.test.ts +13 -12
- package/src/wirings/workflow/pikku-workflow-service.ts +28 -4
- package/src/wirings/workflow/scenario-expectations.test.ts +75 -0
- package/src/wirings/workflow/scenario-hooks.test.ts +3 -2
- package/src/wirings/workflow/scenario-step.types.ts +8 -0
- package/src/wirings/workflow/scenario.types.ts +63 -0
- package/src/wirings/workflow/workflow-approval-audit.ts +47 -0
- package/src/wirings/workflow/workflow-approval-policy.test.ts +524 -0
- package/src/wirings/workflow/workflow-approval-policy.ts +68 -0
- package/src/wirings/workflow/workflow-approval.ts +113 -9
- package/src/wirings/workflow/workflow-run-authority.test.ts +12 -15
- package/src/wirings/workflow/workflow-run-ownership.ts +2 -1
- package/src/wirings/workflow/workflow.types.ts +1 -63
- package/src/wirings-stay-decoupled.test.ts +6 -2
- package/tsconfig.tsbuildinfo +1 -1
- package/dist/internal.d.ts +0 -3
- package/dist/internal.js +0 -2
- package/dist/middleware/timeout.d.ts +0 -9
- package/dist/middleware/timeout.js +0 -15
- package/dist/pikku-response.d.ts +0 -6
- package/dist/pikku-response.js +0 -6
- package/dist/services/gopass-secrets.d.ts +0 -15
- package/dist/services/gopass-secrets.js +0 -76
- package/dist/services/http-scenario-actors.d.ts +0 -75
- package/dist/services/http-scenario-actors.js +0 -195
- package/dist/services/http-user-flow-actors.d.ts +0 -67
- package/dist/services/http-user-flow-actors.js +0 -193
- package/dist/services/scenario-actors-service.d.ts +0 -127
- package/dist/services/scenario-actors-service.js +0 -40
- package/dist/services/user-flow-actors-service.d.ts +0 -39
- package/dist/wirings/credential/wire-credential.d.ts +0 -48
- package/dist/wirings/credential/wire-credential.js +0 -47
- package/dist/wirings/oauth2/oauth2-client.d.ts +0 -47
- package/dist/wirings/oauth2/oauth2-client.js +0 -263
- package/dist/wirings/oauth2/oauth2-routes.d.ts +0 -35
- package/dist/wirings/oauth2/oauth2-routes.js +0 -146
- package/dist/wirings/scope/wire-scope.d.ts +0 -33
- package/dist/wirings/scope/wire-scope.js +0 -32
- package/dist/wirings/workflow/dsl/index.d.ts +0 -5
- package/dist/wirings/workflow/dsl/index.js +0 -4
- package/dist/wirings/workflow/graph/index.d.ts +0 -5
- package/dist/wirings/workflow/graph/index.js +0 -4
- /package/dist/{services/user-flow-actors-service.js → wirings/workflow/scenario.types.js} +0 -0
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type {
|
|
2
2
|
AIStreamChannel,
|
|
3
3
|
AIStreamEvent,
|
|
4
|
+
AIAgentStep,
|
|
4
5
|
AIMessage,
|
|
5
6
|
AIToolCall,
|
|
6
7
|
AIToolResult,
|
|
@@ -9,6 +10,10 @@ import type {
|
|
|
9
10
|
CoreAIAgent,
|
|
10
11
|
AIAgentMemoryConfig,
|
|
11
12
|
} from './ai-agent.types.js'
|
|
13
|
+
import {
|
|
14
|
+
finalizeAgentRun,
|
|
15
|
+
lastUserMessageText,
|
|
16
|
+
} from './ai-agent-finalize.js'
|
|
12
17
|
import { pikkuState, getSingletonServices } from '../../pikku-state.js'
|
|
13
18
|
import { applyInputMiddleware } from './ai-agent-turn.js'
|
|
14
19
|
import { AIProviderNotConfiguredError } from '../../errors/errors.js'
|
|
@@ -68,6 +73,8 @@ type PersistingChannel = AIStreamChannel & {
|
|
|
68
73
|
fullText: string
|
|
69
74
|
flush: (opts?: { interrupted?: boolean }) => Promise<void>
|
|
70
75
|
totalUsage: { inputTokens: number; outputTokens: number; model?: string }
|
|
76
|
+
/** Every tool the run called, kept for the whole run rather than per step. */
|
|
77
|
+
runToolCalls: NonNullable<AIAgentStep['toolCalls']>
|
|
71
78
|
}
|
|
72
79
|
|
|
73
80
|
function createPersistingChannel(
|
|
@@ -89,6 +96,9 @@ function createPersistingChannel(
|
|
|
89
96
|
inputTokens: 0,
|
|
90
97
|
outputTokens: 0,
|
|
91
98
|
}
|
|
99
|
+
// Survives the per-step flush below, which clears its own buffers: the run
|
|
100
|
+
// record needs every call the run made, not just the last step's.
|
|
101
|
+
const runToolCalls: NonNullable<AIAgentStep['toolCalls']> = []
|
|
92
102
|
|
|
93
103
|
const flushStep = async (opts?: { interrupted?: boolean }) => {
|
|
94
104
|
if (!storage) return
|
|
@@ -137,6 +147,8 @@ function createPersistingChannel(
|
|
|
137
147
|
})
|
|
138
148
|
}
|
|
139
149
|
|
|
150
|
+
const runToolCallIndex = new Map<string, number>()
|
|
151
|
+
|
|
140
152
|
const channel: PersistingChannel = {
|
|
141
153
|
channelId: parent.channelId,
|
|
142
154
|
openingData: parent.openingData,
|
|
@@ -149,6 +161,9 @@ function createPersistingChannel(
|
|
|
149
161
|
get totalUsage() {
|
|
150
162
|
return totalUsage
|
|
151
163
|
},
|
|
164
|
+
get runToolCalls() {
|
|
165
|
+
return runToolCalls
|
|
166
|
+
},
|
|
152
167
|
flush: flushStep,
|
|
153
168
|
close: () => parent.close(),
|
|
154
169
|
sendBinary: (data) => parent.sendBinary(data),
|
|
@@ -157,6 +172,26 @@ function createPersistingChannel(
|
|
|
157
172
|
// the client was streamed, and an interrupted run has to be able to
|
|
158
173
|
// report the fragment it got through even with persistence turned off.
|
|
159
174
|
if (event.type === 'text-delta') fullText += event.text
|
|
175
|
+
if (event.type === 'tool-call') {
|
|
176
|
+
runToolCallIndex.set(event.toolCallId, runToolCalls.length)
|
|
177
|
+
runToolCalls.push({
|
|
178
|
+
name: event.toolName,
|
|
179
|
+
args: event.args as Record<string, unknown>,
|
|
180
|
+
result: '',
|
|
181
|
+
})
|
|
182
|
+
}
|
|
183
|
+
if (event.type === 'tool-result') {
|
|
184
|
+
const index = runToolCallIndex.get(event.toolCallId)
|
|
185
|
+
const result =
|
|
186
|
+
typeof event.result === 'string'
|
|
187
|
+
? event.result
|
|
188
|
+
: JSON.stringify(event.result)
|
|
189
|
+
const call = index === undefined ? undefined : runToolCalls[index]
|
|
190
|
+
if (call) {
|
|
191
|
+
call.result = result
|
|
192
|
+
if (event.error) call.error = event.error
|
|
193
|
+
}
|
|
194
|
+
}
|
|
160
195
|
if (storage) {
|
|
161
196
|
switch (event.type) {
|
|
162
197
|
case 'text-delta':
|
|
@@ -177,6 +212,7 @@ function createPersistingChannel(
|
|
|
177
212
|
typeof event.result === 'string'
|
|
178
213
|
? event.result
|
|
179
214
|
: JSON.stringify(event.result),
|
|
215
|
+
...(event.error ? { error: event.error } : {}),
|
|
180
216
|
})
|
|
181
217
|
break
|
|
182
218
|
case 'generative-ui':
|
|
@@ -203,44 +239,67 @@ function createPersistingChannel(
|
|
|
203
239
|
return channel
|
|
204
240
|
}
|
|
205
241
|
|
|
242
|
+
/**
|
|
243
|
+
* Agents already warned about, so a per-request hook does not become a
|
|
244
|
+
* per-request log line.
|
|
245
|
+
*/
|
|
246
|
+
const warnedUnstreamedOutputHooks = new Set<string>()
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* `modifyOutput` does not run on a streamed run at all. Nothing here could act
|
|
250
|
+
* on what it returns — the text has already reached the client, and
|
|
251
|
+
* `createPersistingChannel` flushes each step to storage as it goes, so by the
|
|
252
|
+
* time the run ends the transcript is already written.
|
|
253
|
+
*
|
|
254
|
+
* Rewriting on this path belongs to `modifyOutputStream`, which genuinely
|
|
255
|
+
* works: the stream middleware wraps the persisting channel, so what is stored
|
|
256
|
+
* and accumulated is already what the client was sent. A middleware that
|
|
257
|
+
* rewrites in `modifyOutput` only — a redaction hook, typically — is therefore
|
|
258
|
+
* silently ineffective when the agent is streamed, and is told so once.
|
|
259
|
+
*/
|
|
260
|
+
const warnUnstreamedOutputHooks = (
|
|
261
|
+
agentName: string,
|
|
262
|
+
aiMiddlewares: PikkuAIMiddlewareHooks[],
|
|
263
|
+
logger?: { warn: (...args: any[]) => void }
|
|
264
|
+
) => {
|
|
265
|
+
if (warnedUnstreamedOutputHooks.has(agentName)) return
|
|
266
|
+
const unstreamed = aiMiddlewares.some(
|
|
267
|
+
(mw) => mw.modifyOutput && !mw.modifyOutputStream
|
|
268
|
+
)
|
|
269
|
+
if (!unstreamed) return
|
|
270
|
+
warnedUnstreamedOutputHooks.add(agentName)
|
|
271
|
+
logger?.warn(
|
|
272
|
+
`Agent '${agentName}' has AI middleware with modifyOutput but no modifyOutputStream — modifyOutput does not apply to streamed runs. Implement modifyOutputStream to affect a streamed reply.`
|
|
273
|
+
)
|
|
274
|
+
}
|
|
275
|
+
|
|
206
276
|
async function postStreamCleanup(
|
|
207
277
|
persistingChannel: PersistingChannel,
|
|
208
|
-
aiMiddlewares: PikkuAIMiddlewareHooks[],
|
|
209
|
-
singletonServices: any,
|
|
210
|
-
messages: AIMessage[],
|
|
211
278
|
aiRunState: AIRunStateService,
|
|
212
|
-
runId: string
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
const mw = aiMiddlewares[i]
|
|
219
|
-
if (mw.modifyOutput) {
|
|
220
|
-
const result = await mw.modifyOutput(singletonServices, {
|
|
221
|
-
text: outputText,
|
|
222
|
-
messages: outputMessages,
|
|
223
|
-
usage: {
|
|
224
|
-
inputTokens: usage.inputTokens,
|
|
225
|
-
outputTokens: usage.outputTokens,
|
|
226
|
-
},
|
|
227
|
-
})
|
|
228
|
-
outputText = result.text
|
|
229
|
-
outputMessages = result.messages
|
|
230
|
-
}
|
|
279
|
+
runId: string,
|
|
280
|
+
run: {
|
|
281
|
+
agentName: string
|
|
282
|
+
threadId: string
|
|
283
|
+
resourceId?: string
|
|
284
|
+
input: string
|
|
231
285
|
}
|
|
232
|
-
|
|
233
|
-
await aiRunState
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
286
|
+
): Promise<void> {
|
|
287
|
+
await finalizeAgentRun(aiRunState, {
|
|
288
|
+
runId,
|
|
289
|
+
agentName: run.agentName,
|
|
290
|
+
threadId: run.threadId,
|
|
291
|
+
resourceId: run.resourceId,
|
|
292
|
+
input: run.input,
|
|
293
|
+
// Already what the client received: the stream middleware wraps the
|
|
294
|
+
// persisting channel, so both were accumulated post-rewrite.
|
|
295
|
+
text: persistingChannel.fullText,
|
|
296
|
+
steps: [
|
|
297
|
+
{
|
|
298
|
+
usage: persistingChannel.totalUsage,
|
|
299
|
+
toolCalls: persistingChannel.runToolCalls,
|
|
300
|
+
},
|
|
301
|
+
],
|
|
302
|
+
usage: persistingChannel.totalUsage,
|
|
244
303
|
})
|
|
245
304
|
}
|
|
246
305
|
|
|
@@ -723,6 +782,8 @@ export async function streamAIAgent(
|
|
|
723
782
|
await storage.saveMessages(threadId, [persistedUserMessage])
|
|
724
783
|
}
|
|
725
784
|
|
|
785
|
+
warnUnstreamedOutputHooks(agentName, aiMiddlewares, singletonServices.logger)
|
|
786
|
+
|
|
726
787
|
const streamMiddleware = aiMiddlewares
|
|
727
788
|
.filter((mw) => mw.modifyOutputStream)
|
|
728
789
|
.map((mw) => {
|
|
@@ -849,14 +910,12 @@ export async function streamAIAgent(
|
|
|
849
910
|
return persistingChannel.fullText
|
|
850
911
|
}
|
|
851
912
|
|
|
852
|
-
await postStreamCleanup(
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
runnerParams.messages,
|
|
857
|
-
|
|
858
|
-
runId
|
|
859
|
-
)
|
|
913
|
+
await postStreamCleanup(persistingChannel, aiRunState, runId, {
|
|
914
|
+
agentName,
|
|
915
|
+
threadId,
|
|
916
|
+
resourceId: input.resourceId,
|
|
917
|
+
input: lastUserMessageText(runnerParams.messages),
|
|
918
|
+
})
|
|
860
919
|
|
|
861
920
|
// knowledge: decisions/internals/the-agent-done-event-goes-through-the-middleware-and-is-awaited.md
|
|
862
921
|
await outputChannel.send({ type: 'done' })
|
|
@@ -1184,16 +1243,17 @@ export async function resumeAIAgent(
|
|
|
1184
1243
|
typeof pending.args === 'string' ? JSON.parse(pending.args) : pending.args
|
|
1185
1244
|
|
|
1186
1245
|
let toolResult: unknown
|
|
1187
|
-
let
|
|
1246
|
+
let toolError: string | undefined
|
|
1188
1247
|
try {
|
|
1189
1248
|
toolResult = await matchingTool.execute(toolArgs)
|
|
1190
1249
|
} catch (execErr: any) {
|
|
1191
1250
|
if (execErr?.payload?.error === 'missing_credential') {
|
|
1192
1251
|
toolResult = execErr.payload
|
|
1252
|
+
toolError = 'missing_credential'
|
|
1193
1253
|
} else {
|
|
1194
|
-
|
|
1254
|
+
toolError = execErr instanceof Error ? execErr.message : String(execErr)
|
|
1255
|
+
toolResult = `Error: ${toolError}`
|
|
1195
1256
|
}
|
|
1196
|
-
isError = true
|
|
1197
1257
|
}
|
|
1198
1258
|
|
|
1199
1259
|
const resultStr =
|
|
@@ -1220,7 +1280,7 @@ export async function resumeAIAgent(
|
|
|
1220
1280
|
toolCallId: input.toolCallId,
|
|
1221
1281
|
toolName: pending.toolName,
|
|
1222
1282
|
result: toolResult,
|
|
1223
|
-
...(
|
|
1283
|
+
...(toolError ? { error: toolError } : {}),
|
|
1224
1284
|
})
|
|
1225
1285
|
}
|
|
1226
1286
|
|
|
@@ -1313,6 +1373,12 @@ async function continueAfterToolResult(
|
|
|
1313
1373
|
// knowledge: decisions/internals/a-resumed-agent-turn-is-as-interruptible-as-the-first.md
|
|
1314
1374
|
const interruptHandle = registerInterruptibleRun(run.runId)
|
|
1315
1375
|
|
|
1376
|
+
warnUnstreamedOutputHooks(
|
|
1377
|
+
run.agentName,
|
|
1378
|
+
aiMiddlewares,
|
|
1379
|
+
singletonServices.logger
|
|
1380
|
+
)
|
|
1381
|
+
|
|
1316
1382
|
const streamMiddleware = aiMiddlewares
|
|
1317
1383
|
.filter((mw) => mw.modifyOutputStream)
|
|
1318
1384
|
.map((mw) => {
|
|
@@ -1434,14 +1500,10 @@ async function continueAfterToolResult(
|
|
|
1434
1500
|
return
|
|
1435
1501
|
}
|
|
1436
1502
|
|
|
1437
|
-
await postStreamCleanup(
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
runnerParams.messages,
|
|
1442
|
-
aiRunState,
|
|
1443
|
-
run.runId
|
|
1444
|
-
)
|
|
1503
|
+
await postStreamCleanup(persistingChannel, aiRunState, run.runId, {
|
|
1504
|
+
...run,
|
|
1505
|
+
input: lastUserMessageText(runnerParams.messages),
|
|
1506
|
+
})
|
|
1445
1507
|
|
|
1446
1508
|
// knowledge: decisions/internals/the-agent-done-event-goes-through-the-middleware-and-is-awaited.md
|
|
1447
1509
|
await wrappedChannel.send({ type: 'done' })
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { describe, test } from 'node:test'
|
|
2
|
+
import assert from 'node:assert/strict'
|
|
3
|
+
|
|
4
|
+
import { toAccumulatedStep } from './ai-agent-turn.js'
|
|
5
|
+
import type { AIAgentStepResult } from '../../services/ai-agent-runner-service.js'
|
|
6
|
+
|
|
7
|
+
const stepResult = (
|
|
8
|
+
overrides?: Partial<AIAgentStepResult>
|
|
9
|
+
): AIAgentStepResult => ({
|
|
10
|
+
text: '',
|
|
11
|
+
toolCalls: [],
|
|
12
|
+
toolResults: [],
|
|
13
|
+
usage: { inputTokens: 0, outputTokens: 0 },
|
|
14
|
+
finishReason: 'stop',
|
|
15
|
+
...overrides,
|
|
16
|
+
})
|
|
17
|
+
|
|
18
|
+
describe('toAccumulatedStep', () => {
|
|
19
|
+
test('carries a tool failure as its own field, not only as rendered text', () => {
|
|
20
|
+
const step = toAccumulatedStep(
|
|
21
|
+
stepResult({
|
|
22
|
+
toolCalls: [
|
|
23
|
+
{ toolCallId: 'call-1', toolName: 'lookupOrder', args: { id: 7 } },
|
|
24
|
+
],
|
|
25
|
+
toolResults: [
|
|
26
|
+
{
|
|
27
|
+
toolCallId: 'call-1',
|
|
28
|
+
toolName: 'lookupOrder',
|
|
29
|
+
result: 'Error: order service unreachable',
|
|
30
|
+
error: 'order service unreachable',
|
|
31
|
+
},
|
|
32
|
+
],
|
|
33
|
+
})
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
assert.deepEqual(step.toolCalls, [
|
|
37
|
+
{
|
|
38
|
+
name: 'lookupOrder',
|
|
39
|
+
args: { id: 7 },
|
|
40
|
+
result: 'Error: order service unreachable',
|
|
41
|
+
error: 'order service unreachable',
|
|
42
|
+
},
|
|
43
|
+
])
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
test('leaves error unset on a tool that returned normally, even if it says Error', () => {
|
|
47
|
+
// A tool is allowed to return the word "Error" — which is exactly why
|
|
48
|
+
// "did this fail" cannot be answered by matching on the result text.
|
|
49
|
+
const step = toAccumulatedStep(
|
|
50
|
+
stepResult({
|
|
51
|
+
toolCalls: [
|
|
52
|
+
{ toolCallId: 'call-1', toolName: 'searchLogs', args: { q: 'x' } },
|
|
53
|
+
],
|
|
54
|
+
toolResults: [
|
|
55
|
+
{
|
|
56
|
+
toolCallId: 'call-1',
|
|
57
|
+
toolName: 'searchLogs',
|
|
58
|
+
result: 'Error: connection refused (1 match)',
|
|
59
|
+
},
|
|
60
|
+
],
|
|
61
|
+
})
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
assert.equal(step.toolCalls[0].error, undefined)
|
|
65
|
+
assert.ok(!('error' in step.toolCalls[0]))
|
|
66
|
+
})
|
|
67
|
+
})
|
|
@@ -51,6 +51,13 @@ export interface AIToolResult {
|
|
|
51
51
|
id: string
|
|
52
52
|
name: string
|
|
53
53
|
result: string
|
|
54
|
+
/**
|
|
55
|
+
* Set when the tool threw rather than returned. Carried separately from
|
|
56
|
+
* `result`, which is a rendered string by the time it is persisted — a tool
|
|
57
|
+
* may legitimately return text beginning `Error:`, so the prefix cannot be
|
|
58
|
+
* read as a failure signal.
|
|
59
|
+
*/
|
|
60
|
+
error?: string
|
|
54
61
|
}
|
|
55
62
|
|
|
56
63
|
export interface AIMessage {
|
|
@@ -82,7 +89,13 @@ export interface AIMessage {
|
|
|
82
89
|
|
|
83
90
|
export interface AIAgentStep {
|
|
84
91
|
usage: { inputTokens: number; outputTokens: number }
|
|
85
|
-
toolCalls?: {
|
|
92
|
+
toolCalls?: {
|
|
93
|
+
name: string
|
|
94
|
+
args: Record<string, unknown>
|
|
95
|
+
result: string
|
|
96
|
+
/** The failure message, when the tool threw rather than returned. */
|
|
97
|
+
error?: string
|
|
98
|
+
}[]
|
|
86
99
|
}
|
|
87
100
|
|
|
88
101
|
export interface AIAgentInputAttachment {
|
|
@@ -233,16 +246,40 @@ export interface PikkuAIMiddlewareHooks<
|
|
|
233
246
|
| AIStreamEvent[]
|
|
234
247
|
| null
|
|
235
248
|
|
|
249
|
+
/**
|
|
250
|
+
* The last chance to rewrite what the run produced, before it is persisted
|
|
251
|
+
* and returned.
|
|
252
|
+
*
|
|
253
|
+
* It does **not** run on a streamed run: there the text has already reached
|
|
254
|
+
* the client and each step is flushed to storage as it goes, so nothing could
|
|
255
|
+
* act on what this returned. Use {@link modifyOutputStream} to rewrite a
|
|
256
|
+
* streamed reply — a middleware that implements only this one is warned about
|
|
257
|
+
* when an agent it is attached to streams.
|
|
258
|
+
*
|
|
259
|
+
* `toolCalls` is here so a redaction pass covers the whole run record rather
|
|
260
|
+
* than just the visible answer: the tool arguments and results are persisted
|
|
261
|
+
* and handed to anything that grades the run, and scrubbing the reply alone
|
|
262
|
+
* leaves them untouched.
|
|
263
|
+
*/
|
|
236
264
|
modifyOutput?: (
|
|
237
265
|
services: Services,
|
|
238
266
|
ctx: {
|
|
239
267
|
text: string
|
|
240
268
|
messages: AIMessage[]
|
|
241
269
|
usage: { inputTokens: number; outputTokens: number }
|
|
270
|
+
toolCalls: NonNullable<AIAgentStep['toolCalls']>
|
|
242
271
|
}
|
|
243
272
|
) =>
|
|
244
|
-
| Promise<{
|
|
245
|
-
|
|
273
|
+
| Promise<{
|
|
274
|
+
text: string
|
|
275
|
+
messages: AIMessage[]
|
|
276
|
+
toolCalls?: NonNullable<AIAgentStep['toolCalls']>
|
|
277
|
+
}>
|
|
278
|
+
| {
|
|
279
|
+
text: string
|
|
280
|
+
messages: AIMessage[]
|
|
281
|
+
toolCalls?: NonNullable<AIAgentStep['toolCalls']>
|
|
282
|
+
}
|
|
246
283
|
|
|
247
284
|
beforeToolCall?: (
|
|
248
285
|
services: Services,
|
|
@@ -273,7 +310,13 @@ export interface PikkuAIMiddlewareHooks<
|
|
|
273
310
|
stepNumber: number
|
|
274
311
|
text: string
|
|
275
312
|
toolCalls: { toolCallId: string; toolName: string; args: unknown }[]
|
|
276
|
-
toolResults: {
|
|
313
|
+
toolResults: {
|
|
314
|
+
toolCallId: string
|
|
315
|
+
toolName: string
|
|
316
|
+
result: unknown
|
|
317
|
+
/** Set when the tool threw rather than returned. */
|
|
318
|
+
error?: string
|
|
319
|
+
}[]
|
|
277
320
|
usage: { inputTokens: number; outputTokens: number }
|
|
278
321
|
finishReason: string
|
|
279
322
|
}
|
|
@@ -301,6 +344,7 @@ export type CoreAIAgent<
|
|
|
301
344
|
PikkuPermission = CorePikkuPermission<any, any>,
|
|
302
345
|
PikkuMiddleware = CorePikkuMiddleware<any>,
|
|
303
346
|
Scope extends string = string,
|
|
347
|
+
Scorer extends string = string,
|
|
304
348
|
> = {
|
|
305
349
|
name: string
|
|
306
350
|
description: string
|
|
@@ -327,6 +371,16 @@ export type CoreAIAgent<
|
|
|
327
371
|
tools?: unknown[]
|
|
328
372
|
agents?: unknown[]
|
|
329
373
|
workflows?: unknown[]
|
|
374
|
+
/**
|
|
375
|
+
* Grades this agent's finished runs on live traffic, named by the generated
|
|
376
|
+
* `ScorerName` union rather than by `ref()` — a scorer is not a function, so
|
|
377
|
+
* there is nothing in the function map for a ref to resolve against.
|
|
378
|
+
*
|
|
379
|
+
* A reference-based judge listed here is never sampled: live traffic has no
|
|
380
|
+
* answer key. Scenarios name scorers directly and may grade with scorers an
|
|
381
|
+
* agent does not ship with.
|
|
382
|
+
*/
|
|
383
|
+
scorers?: Scorer[]
|
|
330
384
|
agentMode?: 'delegate' | 'supervise'
|
|
331
385
|
memory?: AIAgentMemoryConfig
|
|
332
386
|
maxSteps?: number
|
|
@@ -392,6 +446,12 @@ export type AIStreamEvent =
|
|
|
392
446
|
toolCallId: string
|
|
393
447
|
toolName: string
|
|
394
448
|
result: unknown
|
|
449
|
+
/**
|
|
450
|
+
* The failure message, set when the tool threw rather than returned.
|
|
451
|
+
* Carried explicitly because a tool may legitimately return text that
|
|
452
|
+
* reads like an error, so `result` cannot be matched on to tell.
|
|
453
|
+
*/
|
|
454
|
+
error?: string
|
|
395
455
|
agent?: string
|
|
396
456
|
session?: string
|
|
397
457
|
}
|
|
@@ -5,8 +5,9 @@ export {
|
|
|
5
5
|
agentApprove,
|
|
6
6
|
agentInterrupt,
|
|
7
7
|
} from './ai-agent-helpers.js'
|
|
8
|
-
export { wrapChannelWithAGUI
|
|
8
|
+
export { wrapChannelWithAGUI } from './ai-agent-agui.js'
|
|
9
9
|
export { runAIAgent, resumeAIAgentSync } from './ai-agent-runner.js'
|
|
10
|
+
export { resolveModelAlias } from './ai-agent-model-config.js'
|
|
10
11
|
export {
|
|
11
12
|
streamAIAgent,
|
|
12
13
|
resumeAIAgent,
|
|
@@ -14,7 +15,6 @@ export {
|
|
|
14
15
|
} from './ai-agent-stream.js'
|
|
15
16
|
export {
|
|
16
17
|
voiceInput,
|
|
17
|
-
readsAsNonSpeech,
|
|
18
18
|
NoSpeechDetectedError,
|
|
19
19
|
SPOKEN_TURN,
|
|
20
20
|
SPOKEN_TRANSCRIPT,
|
|
@@ -27,21 +27,12 @@ export {
|
|
|
27
27
|
} from './voice-output.js'
|
|
28
28
|
export {
|
|
29
29
|
AgentInterruptedError,
|
|
30
|
-
awaitPendingInterruptNote,
|
|
31
|
-
getInFlightTools,
|
|
32
|
-
isAbortError,
|
|
33
|
-
isRunInterruptible,
|
|
34
|
-
persistOrphanedToolResults,
|
|
35
|
-
registerInterruptibleRun,
|
|
36
30
|
signalRunInterrupt,
|
|
37
|
-
trackInterruptNote,
|
|
38
|
-
trackToolExecution,
|
|
39
31
|
} from './ai-agent-interrupt.js'
|
|
40
32
|
export type {
|
|
41
33
|
AgentInterruption,
|
|
42
34
|
AgentInterruptResult,
|
|
43
35
|
InterruptibleRunHandle,
|
|
44
|
-
OrphanedToolResult,
|
|
45
36
|
} from './ai-agent-interrupt.js'
|
|
46
37
|
export {
|
|
47
38
|
type RunAIAgentParams,
|
|
@@ -50,18 +41,13 @@ export {
|
|
|
50
41
|
ToolCredentialRequired,
|
|
51
42
|
canAccessThread,
|
|
52
43
|
isOwnedByPrincipal,
|
|
53
|
-
sessionPrincipals,
|
|
54
44
|
threadOwnerConstraint,
|
|
55
45
|
} from './ai-agent-prepare.js'
|
|
56
46
|
export {
|
|
57
47
|
addAIAgent,
|
|
58
|
-
approveAIAgent,
|
|
59
|
-
getAIAgents,
|
|
60
|
-
getAIAgentsMeta,
|
|
61
48
|
} from './ai-agent-registry.js'
|
|
62
49
|
export type {
|
|
63
50
|
AIAgentInput,
|
|
64
|
-
AIAgentInputAttachment,
|
|
65
51
|
AIAgentMeta,
|
|
66
52
|
AIAgentMemoryConfig,
|
|
67
53
|
AIAgentStep,
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { beforeEach, describe, test } from 'node:test'
|
|
2
|
+
import assert from 'node:assert/strict'
|
|
3
|
+
import { pikkuState, resetPikkuState } from '../../pikku-state.js'
|
|
4
|
+
import { gradeRun } from './ai-scorer-grade.js'
|
|
5
|
+
import { pikkuAIJudge, pikkuAIScorer } from './ai-scorer.js'
|
|
6
|
+
import type { ScoreJob } from './ai-scorer.types.js'
|
|
7
|
+
|
|
8
|
+
const job = (overrides: Partial<ScoreJob> = {}): ScoreJob => ({
|
|
9
|
+
scorerName: 'correctness',
|
|
10
|
+
runId: 'run-1',
|
|
11
|
+
agentName: 'assistant',
|
|
12
|
+
input: 'what is the capital of France?',
|
|
13
|
+
output: 'Paris',
|
|
14
|
+
toolCalls: [],
|
|
15
|
+
usage: { inputTokens: 10, outputTokens: 5 },
|
|
16
|
+
...overrides,
|
|
17
|
+
})
|
|
18
|
+
|
|
19
|
+
describe('gradeRun', () => {
|
|
20
|
+
beforeEach(() => resetPikkuState())
|
|
21
|
+
|
|
22
|
+
test('a scenario grade is returned and not written to the live record', async () => {
|
|
23
|
+
pikkuState(null, 'agent', 'scorers').set(
|
|
24
|
+
'correctness',
|
|
25
|
+
pikkuAIScorer({
|
|
26
|
+
name: 'correctness',
|
|
27
|
+
description: 'Matches the answer key',
|
|
28
|
+
requiresReference: true,
|
|
29
|
+
score: (input) => ({
|
|
30
|
+
score: input.output === input.reference ? 1 : 0,
|
|
31
|
+
}),
|
|
32
|
+
})
|
|
33
|
+
)
|
|
34
|
+
let saves = 0
|
|
35
|
+
|
|
36
|
+
const result = await gradeRun(
|
|
37
|
+
job({ reference: 'Paris' }),
|
|
38
|
+
{
|
|
39
|
+
aiRunState: {
|
|
40
|
+
saveScore: async () => {
|
|
41
|
+
saves++
|
|
42
|
+
},
|
|
43
|
+
},
|
|
44
|
+
},
|
|
45
|
+
{ persist: false }
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
assert.deepEqual(result, { score: 1 })
|
|
49
|
+
assert.equal(saves, 0)
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
test('a reference-based scorer sees the answer key it was given', async () => {
|
|
53
|
+
pikkuState(null, 'agent', 'scorers').set(
|
|
54
|
+
'correctness',
|
|
55
|
+
pikkuAIScorer({
|
|
56
|
+
name: 'correctness',
|
|
57
|
+
description: 'Matches the answer key',
|
|
58
|
+
requiresReference: true,
|
|
59
|
+
score: (input) => ({
|
|
60
|
+
score: input.output === input.reference ? 1 : 0,
|
|
61
|
+
}),
|
|
62
|
+
})
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
const result = await gradeRun(
|
|
66
|
+
job({ reference: 'Lyon' }),
|
|
67
|
+
{},
|
|
68
|
+
{
|
|
69
|
+
persist: false,
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
assert.equal(result.score, 0)
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
test('a judge grades without a score function, and reports the model it used', async () => {
|
|
77
|
+
pikkuState(null, 'agent', 'scorers').set(
|
|
78
|
+
'helpfulness',
|
|
79
|
+
pikkuAIJudge({
|
|
80
|
+
name: 'helpfulness',
|
|
81
|
+
description: 'Is the answer useful',
|
|
82
|
+
model: 'claude-opus-5',
|
|
83
|
+
goal: 'Grade helpfulness.',
|
|
84
|
+
})
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
const result = await gradeRun(
|
|
88
|
+
job({ scorerName: 'helpfulness' }),
|
|
89
|
+
{
|
|
90
|
+
aiAgentRunner: {
|
|
91
|
+
run: async () => ({
|
|
92
|
+
object: { score: 0.75, reason: 'Correct but terse.' },
|
|
93
|
+
usage: { inputTokens: 80, outputTokens: 20 },
|
|
94
|
+
}),
|
|
95
|
+
},
|
|
96
|
+
},
|
|
97
|
+
{ persist: false }
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
assert.equal(result.score, 0.75)
|
|
101
|
+
assert.deepEqual(result.metadata, {
|
|
102
|
+
judgeModel: 'claude-opus-5',
|
|
103
|
+
judgeTokens: 100,
|
|
104
|
+
})
|
|
105
|
+
})
|
|
106
|
+
})
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import { runJudge } from './ai-scorer-judge.js'
|
|
2
|
+
import { resolveAIScorer } from './ai-scorer-registry.js'
|
|
3
|
+
import type { ScoreJob, ScorerOutput } from './ai-scorer.types.js'
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Grade one run with one scorer.
|
|
7
|
+
*
|
|
8
|
+
* The single path both callers take — the lane worker on live traffic and the
|
|
9
|
+
* scenario grading RPC — so a scenario's grade is the same computation the
|
|
10
|
+
* production sampler would have made, not an approximation of it.
|
|
11
|
+
*
|
|
12
|
+
* Persisting is optional because the two callers differ on it: a live grade is
|
|
13
|
+
* only useful once recorded, while a scenario asserts on the returned value and
|
|
14
|
+
* runs against servers that may have no run-state adapter at all.
|
|
15
|
+
*/
|
|
16
|
+
export const gradeRun = async (
|
|
17
|
+
job: ScoreJob,
|
|
18
|
+
services: {
|
|
19
|
+
aiAgentRunner?: unknown
|
|
20
|
+
aiRunState?: {
|
|
21
|
+
saveScore: (score: {
|
|
22
|
+
runId: string
|
|
23
|
+
scorerName: string
|
|
24
|
+
score: number
|
|
25
|
+
reason?: string
|
|
26
|
+
metadata?: Record<string, unknown>
|
|
27
|
+
}) => Promise<void>
|
|
28
|
+
}
|
|
29
|
+
},
|
|
30
|
+
options: { persist: boolean }
|
|
31
|
+
): Promise<ScorerOutput> => {
|
|
32
|
+
const { scorerName, ...input } = job
|
|
33
|
+
const scorer = resolveAIScorer(scorerName)
|
|
34
|
+
|
|
35
|
+
const result = scorer.score
|
|
36
|
+
? await scorer.score(input, services)
|
|
37
|
+
: await runJudge(scorer, input, services.aiAgentRunner as never)
|
|
38
|
+
|
|
39
|
+
if (options.persist) {
|
|
40
|
+
if (!services.aiRunState) {
|
|
41
|
+
throw new Error(
|
|
42
|
+
`AI run state service not initialized: cannot record the '${scorerName}' grade of run ${job.runId}`
|
|
43
|
+
)
|
|
44
|
+
}
|
|
45
|
+
await services.aiRunState.saveScore({
|
|
46
|
+
runId: job.runId,
|
|
47
|
+
scorerName,
|
|
48
|
+
score: result.score,
|
|
49
|
+
...(result.reason !== undefined ? { reason: result.reason } : {}),
|
|
50
|
+
...(result.metadata !== undefined ? { metadata: result.metadata } : {}),
|
|
51
|
+
})
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
return result
|
|
55
|
+
}
|