@pikku/core 0.12.80 → 0.12.83

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. package/CHANGELOG.md +345 -0
  2. package/dist/errors/index.d.ts +1 -1
  3. package/dist/errors/index.js +1 -1
  4. package/dist/function/function-runner.js +2 -5
  5. package/dist/function/index.d.ts +1 -1
  6. package/dist/index.d.ts +11 -11
  7. package/dist/index.js +3 -3
  8. package/dist/pikku-state.js +4 -0
  9. package/dist/services/ai-agent-runner-service.d.ts +7 -0
  10. package/dist/services/ai-run-state-service.d.ts +10 -0
  11. package/dist/services/in-memory-ai-run-state-service.d.ts +5 -1
  12. package/dist/services/in-memory-ai-run-state-service.js +9 -0
  13. package/dist/services/index.d.ts +15 -16
  14. package/dist/services/index.js +5 -5
  15. package/dist/services/meta-service.d.ts +2 -1
  16. package/dist/services/scoped-credential-service.d.ts +21 -0
  17. package/dist/services/scoped-credential-service.js +53 -0
  18. package/dist/testing/service-tests/ai-storage-service-tests.js +76 -0
  19. package/dist/types/core.types.d.ts +2 -3
  20. package/dist/types/state.types.d.ts +19 -1
  21. package/dist/wirings/actor-flow/index.d.ts +1 -1
  22. package/dist/wirings/ai-agent/ai-agent-finalize.d.ts +58 -0
  23. package/dist/wirings/ai-agent/ai-agent-finalize.js +138 -0
  24. package/dist/wirings/ai-agent/ai-agent-interrupt.js +1 -0
  25. package/dist/wirings/ai-agent/ai-agent-memory.d.ts +2 -8
  26. package/dist/wirings/ai-agent/ai-agent-memory.js +34 -17
  27. package/dist/wirings/ai-agent/ai-agent-model-config.d.ts +7 -0
  28. package/dist/wirings/ai-agent/ai-agent-model-config.js +44 -1
  29. package/dist/wirings/ai-agent/ai-agent-prepare.js +4 -0
  30. package/dist/wirings/ai-agent/ai-agent-runner.js +61 -40
  31. package/dist/wirings/ai-agent/ai-agent-stream.js +89 -36
  32. package/dist/wirings/ai-agent/ai-agent-turn.d.ts +1 -0
  33. package/dist/wirings/ai-agent/ai-agent-turn.js +1 -0
  34. package/dist/wirings/ai-agent/ai-agent.types.d.ts +46 -1
  35. package/dist/wirings/ai-agent/index.d.ts +8 -7
  36. package/dist/wirings/ai-agent/index.js +5 -4
  37. package/dist/wirings/ai-scorer/ai-scorer-grade.d.ts +26 -0
  38. package/dist/wirings/ai-scorer/ai-scorer-grade.js +33 -0
  39. package/dist/wirings/ai-scorer/ai-scorer-judge.d.ts +17 -0
  40. package/dist/wirings/ai-scorer/ai-scorer-judge.js +92 -0
  41. package/dist/wirings/ai-scorer/ai-scorer-live.d.ts +15 -0
  42. package/dist/wirings/ai-scorer/ai-scorer-live.js +38 -0
  43. package/dist/wirings/ai-scorer/ai-scorer-registry.d.ts +18 -0
  44. package/dist/wirings/ai-scorer/ai-scorer-registry.js +46 -0
  45. package/dist/wirings/ai-scorer/ai-scorer-sampling.d.ts +8 -0
  46. package/dist/wirings/ai-scorer/ai-scorer-sampling.js +31 -0
  47. package/dist/wirings/ai-scorer/ai-scorer-snapshots.d.ts +10 -0
  48. package/dist/wirings/ai-scorer/ai-scorer-snapshots.js +40 -0
  49. package/dist/wirings/ai-scorer/ai-scorer-worker.d.ts +15 -0
  50. package/dist/wirings/ai-scorer/ai-scorer-worker.js +58 -0
  51. package/dist/wirings/ai-scorer/ai-scorer.d.ts +39 -0
  52. package/dist/wirings/ai-scorer/ai-scorer.js +40 -0
  53. package/dist/wirings/ai-scorer/ai-scorer.types.d.ts +90 -0
  54. package/dist/wirings/ai-scorer/ai-scorer.types.js +4 -0
  55. package/dist/wirings/ai-scorer/index.d.ts +6 -0
  56. package/dist/wirings/ai-scorer/index.js +5 -0
  57. package/dist/wirings/channel/index.d.ts +5 -6
  58. package/dist/wirings/channel/index.js +3 -4
  59. package/dist/wirings/channel/local/local-channel-runner.js +8 -1
  60. package/dist/wirings/cli/channel/cli-raw-channel-runner.js +9 -1
  61. package/dist/wirings/cli/channel/index.d.ts +1 -2
  62. package/dist/wirings/cli/channel/index.js +0 -1
  63. package/dist/wirings/cli/cli-runner.js +13 -1
  64. package/dist/wirings/credential/index.d.ts +1 -1
  65. package/dist/wirings/gateway/index.d.ts +1 -1
  66. package/dist/wirings/http/http-runner.js +8 -2
  67. package/dist/wirings/http/index.d.ts +1 -2
  68. package/dist/wirings/mcp/index.d.ts +1 -1
  69. package/dist/wirings/mcp/mcp-runner.d.ts +15 -0
  70. package/dist/wirings/mcp/mcp-runner.js +18 -5
  71. package/dist/wirings/persona/index.d.ts +3 -4
  72. package/dist/wirings/persona/index.js +2 -3
  73. package/dist/wirings/queue/index.d.ts +1 -3
  74. package/dist/wirings/queue/index.js +1 -3
  75. package/dist/wirings/rpc/addon-runner.d.ts +8 -0
  76. package/dist/wirings/rpc/addon-runner.js +31 -3
  77. package/dist/wirings/rpc/rpc-runner.js +4 -0
  78. package/dist/wirings/rpc/rpc-types.d.ts +8 -0
  79. package/dist/wirings/rpc/wire-addon.d.ts +25 -0
  80. package/dist/wirings/rpc/wire-addon.js +8 -0
  81. package/dist/wirings/scheduler/index.d.ts +1 -1
  82. package/dist/wirings/trigger/index.d.ts +1 -1
  83. package/dist/wirings/virtual-user/index.d.ts +5 -6
  84. package/dist/wirings/virtual-user/index.js +2 -4
  85. package/dist/wirings/workflow/dsl/workflow-dsl.types.d.ts +85 -15
  86. package/dist/wirings/workflow/feature.d.ts +2 -1
  87. package/dist/wirings/workflow/index.d.ts +5 -16
  88. package/dist/wirings/workflow/index.js +1 -9
  89. package/dist/wirings/workflow/pikku-scenario-service.d.ts +17 -7
  90. package/dist/wirings/workflow/pikku-scenario-service.js +48 -13
  91. package/dist/wirings/workflow/pikku-workflow-service.js +17 -3
  92. package/dist/wirings/workflow/scenario-step.types.d.ts +8 -0
  93. package/dist/wirings/workflow/scenario.types.d.ts +37 -0
  94. package/dist/wirings/workflow/workflow-approval-audit.d.ts +16 -0
  95. package/dist/wirings/workflow/workflow-approval-audit.js +40 -0
  96. package/dist/wirings/workflow/workflow-approval-policy.d.ts +20 -0
  97. package/dist/wirings/workflow/workflow-approval-policy.js +48 -0
  98. package/dist/wirings/workflow/workflow-approval.d.ts +29 -1
  99. package/dist/wirings/workflow/workflow-approval.js +65 -2
  100. package/dist/wirings/workflow/workflow-run-ownership.d.ts +2 -1
  101. package/dist/wirings/workflow/workflow-run-ownership.js +2 -1
  102. package/dist/wirings/workflow/workflow.types.d.ts +2 -37
  103. package/knowledge/decisions/internals/addon-pikku-meta-ships-at-the-package-root-or-under-dist.md +32 -0
  104. package/knowledge/decisions/internals/an-addon-scope-root-loses-to-a-root-the-host-already-declares.md +39 -0
  105. package/knowledge/decisions/internals/index.md +30 -3
  106. package/knowledge/decisions/internals/validate-runs-checks-by-precondition.md +115 -0
  107. package/knowledge/decisions/security/a-function-never-receives-the-secret-service.md +37 -0
  108. package/knowledge/decisions/security/a-workflow-run-is-read-and-approved-by-its-owner.md +30 -14
  109. package/knowledge/decisions/security/an-approval-answer-outlives-the-run-it-answered.md +59 -0
  110. package/knowledge/decisions/security/index.md +3 -1
  111. package/knowledge/questions/index.md +1 -1
  112. package/package.json +3 -2
  113. package/scripts/generate-api-report.mts +143 -18
  114. package/src/api-report.test.ts +2 -2
  115. package/src/errors/index.ts +1 -1
  116. package/src/function/function-runner.test.ts +52 -0
  117. package/src/function/function-runner.ts +5 -9
  118. package/src/function/index.ts +0 -2
  119. package/src/index.ts +0 -35
  120. package/src/pikku-state.ts +5 -0
  121. package/src/public-surface.json +81 -118
  122. package/src/services/ai-agent-runner-service.ts +12 -1
  123. package/src/services/ai-run-state-service.ts +11 -0
  124. package/src/services/in-memory-ai-run-state-service.ts +13 -0
  125. package/src/services/index.ts +7 -58
  126. package/src/services/meta-service.ts +2 -4
  127. package/src/services/scoped-credential-service.test.ts +86 -0
  128. package/src/services/scoped-credential-service.ts +63 -0
  129. package/src/testing/service-tests/ai-storage-service-tests.ts +93 -0
  130. package/src/types/core.types.ts +4 -7
  131. package/src/types/state.types.ts +21 -1
  132. package/src/wirings/actor-flow/index.ts +0 -3
  133. package/src/wirings/ai-agent/ai-agent-finalize.test.ts +186 -0
  134. package/src/wirings/ai-agent/ai-agent-finalize.ts +197 -0
  135. package/src/wirings/ai-agent/ai-agent-interrupt.ts +1 -0
  136. package/src/wirings/ai-agent/ai-agent-memory.ts +54 -38
  137. package/src/wirings/ai-agent/ai-agent-model-config.test.ts +72 -3
  138. package/src/wirings/ai-agent/ai-agent-model-config.ts +49 -1
  139. package/src/wirings/ai-agent/ai-agent-prepare.ts +4 -0
  140. package/src/wirings/ai-agent/ai-agent-runner.ts +71 -40
  141. package/src/wirings/ai-agent/ai-agent-stream-output-hooks.test.ts +353 -0
  142. package/src/wirings/ai-agent/ai-agent-stream.ts +116 -54
  143. package/src/wirings/ai-agent/ai-agent-turn.test.ts +67 -0
  144. package/src/wirings/ai-agent/ai-agent-turn.ts +1 -0
  145. package/src/wirings/ai-agent/ai-agent.types.ts +64 -4
  146. package/src/wirings/ai-agent/index.ts +2 -16
  147. package/src/wirings/ai-scorer/ai-scorer-grade.test.ts +106 -0
  148. package/src/wirings/ai-scorer/ai-scorer-grade.ts +55 -0
  149. package/src/wirings/ai-scorer/ai-scorer-judge.test.ts +143 -0
  150. package/src/wirings/ai-scorer/ai-scorer-judge.ts +120 -0
  151. package/src/wirings/ai-scorer/ai-scorer-live.test.ts +174 -0
  152. package/src/wirings/ai-scorer/ai-scorer-live.ts +56 -0
  153. package/src/wirings/ai-scorer/ai-scorer-registry.ts +63 -0
  154. package/src/wirings/ai-scorer/ai-scorer-sampling.test.ts +34 -0
  155. package/src/wirings/ai-scorer/ai-scorer-sampling.ts +36 -0
  156. package/src/wirings/ai-scorer/ai-scorer-snapshots.test.ts +49 -0
  157. package/src/wirings/ai-scorer/ai-scorer-snapshots.ts +46 -0
  158. package/src/wirings/ai-scorer/ai-scorer-worker.test.ts +122 -0
  159. package/src/wirings/ai-scorer/ai-scorer-worker.ts +69 -0
  160. package/src/wirings/ai-scorer/ai-scorer.ts +76 -0
  161. package/src/wirings/ai-scorer/ai-scorer.types.ts +107 -0
  162. package/src/wirings/ai-scorer/index.ts +24 -0
  163. package/src/wirings/channel/index.ts +1 -20
  164. package/src/wirings/channel/local/local-channel-runner.test.ts +68 -0
  165. package/src/wirings/channel/local/local-channel-runner.ts +8 -1
  166. package/src/wirings/cli/channel/cli-raw-channel-runner.test.ts +23 -0
  167. package/src/wirings/cli/channel/cli-raw-channel-runner.ts +12 -1
  168. package/src/wirings/cli/channel/index.ts +0 -7
  169. package/src/wirings/cli/cli-runner.test.ts +68 -0
  170. package/src/wirings/cli/cli-runner.ts +18 -1
  171. package/src/wirings/credential/index.ts +0 -1
  172. package/src/wirings/gateway/index.ts +0 -3
  173. package/src/wirings/http/http-runner.test.ts +66 -0
  174. package/src/wirings/http/http-runner.ts +10 -2
  175. package/src/wirings/http/index.ts +1 -1
  176. package/src/wirings/mcp/index.ts +0 -1
  177. package/src/wirings/mcp/mcp-runner.test.ts +181 -0
  178. package/src/wirings/mcp/mcp-runner.ts +35 -5
  179. package/src/wirings/persona/index.ts +0 -8
  180. package/src/wirings/queue/index.ts +0 -14
  181. package/src/wirings/rpc/addon-runner.ts +62 -3
  182. package/src/wirings/rpc/addon-secrets.test.ts +391 -0
  183. package/src/wirings/rpc/rpc-runner.test.ts +2 -0
  184. package/src/wirings/rpc/rpc-runner.ts +4 -0
  185. package/src/wirings/rpc/rpc-types.ts +8 -0
  186. package/src/wirings/rpc/wire-addon.ts +33 -0
  187. package/src/wirings/scheduler/index.ts +0 -1
  188. package/src/wirings/trigger/index.ts +0 -1
  189. package/src/wirings/virtual-user/index.ts +0 -16
  190. package/src/wirings/workflow/dsl/workflow-dsl.types.ts +96 -16
  191. package/src/wirings/workflow/feature.ts +2 -5
  192. package/src/wirings/workflow/graph/graph-runner.test.ts +72 -0
  193. package/src/wirings/workflow/index.ts +2 -68
  194. package/src/wirings/workflow/pikku-scenario-service.ts +81 -16
  195. package/src/wirings/workflow/pikku-workflow-service.test.ts +13 -12
  196. package/src/wirings/workflow/pikku-workflow-service.ts +28 -4
  197. package/src/wirings/workflow/scenario-expectations.test.ts +75 -0
  198. package/src/wirings/workflow/scenario-hooks.test.ts +3 -2
  199. package/src/wirings/workflow/scenario-step.types.ts +8 -0
  200. package/src/wirings/workflow/scenario.types.ts +63 -0
  201. package/src/wirings/workflow/workflow-approval-audit.ts +47 -0
  202. package/src/wirings/workflow/workflow-approval-policy.test.ts +524 -0
  203. package/src/wirings/workflow/workflow-approval-policy.ts +68 -0
  204. package/src/wirings/workflow/workflow-approval.ts +113 -9
  205. package/src/wirings/workflow/workflow-run-authority.test.ts +12 -15
  206. package/src/wirings/workflow/workflow-run-ownership.ts +2 -1
  207. package/src/wirings/workflow/workflow.types.ts +1 -63
  208. package/src/wirings-stay-decoupled.test.ts +6 -2
  209. package/tsconfig.tsbuildinfo +1 -1
  210. package/dist/internal.d.ts +0 -3
  211. package/dist/internal.js +0 -2
  212. package/dist/middleware/timeout.d.ts +0 -9
  213. package/dist/middleware/timeout.js +0 -15
  214. package/dist/pikku-response.d.ts +0 -6
  215. package/dist/pikku-response.js +0 -6
  216. package/dist/services/gopass-secrets.d.ts +0 -15
  217. package/dist/services/gopass-secrets.js +0 -76
  218. package/dist/services/http-scenario-actors.d.ts +0 -75
  219. package/dist/services/http-scenario-actors.js +0 -195
  220. package/dist/services/http-user-flow-actors.d.ts +0 -67
  221. package/dist/services/http-user-flow-actors.js +0 -193
  222. package/dist/services/scenario-actors-service.d.ts +0 -127
  223. package/dist/services/scenario-actors-service.js +0 -40
  224. package/dist/services/user-flow-actors-service.d.ts +0 -39
  225. package/dist/wirings/credential/wire-credential.d.ts +0 -48
  226. package/dist/wirings/credential/wire-credential.js +0 -47
  227. package/dist/wirings/oauth2/oauth2-client.d.ts +0 -47
  228. package/dist/wirings/oauth2/oauth2-client.js +0 -263
  229. package/dist/wirings/oauth2/oauth2-routes.d.ts +0 -35
  230. package/dist/wirings/oauth2/oauth2-routes.js +0 -146
  231. package/dist/wirings/scope/wire-scope.d.ts +0 -33
  232. package/dist/wirings/scope/wire-scope.js +0 -32
  233. package/dist/wirings/workflow/dsl/index.d.ts +0 -5
  234. package/dist/wirings/workflow/dsl/index.js +0 -4
  235. package/dist/wirings/workflow/graph/index.d.ts +0 -5
  236. package/dist/wirings/workflow/graph/index.js +0 -4
  237. /package/dist/{services/user-flow-actors-service.js → wirings/workflow/scenario.types.js} +0 -0
@@ -1,6 +1,7 @@
1
1
  import type {
2
2
  AIStreamChannel,
3
3
  AIStreamEvent,
4
+ AIAgentStep,
4
5
  AIMessage,
5
6
  AIToolCall,
6
7
  AIToolResult,
@@ -9,6 +10,10 @@ import type {
9
10
  CoreAIAgent,
10
11
  AIAgentMemoryConfig,
11
12
  } from './ai-agent.types.js'
13
+ import {
14
+ finalizeAgentRun,
15
+ lastUserMessageText,
16
+ } from './ai-agent-finalize.js'
12
17
  import { pikkuState, getSingletonServices } from '../../pikku-state.js'
13
18
  import { applyInputMiddleware } from './ai-agent-turn.js'
14
19
  import { AIProviderNotConfiguredError } from '../../errors/errors.js'
@@ -68,6 +73,8 @@ type PersistingChannel = AIStreamChannel & {
68
73
  fullText: string
69
74
  flush: (opts?: { interrupted?: boolean }) => Promise<void>
70
75
  totalUsage: { inputTokens: number; outputTokens: number; model?: string }
76
+ /** Every tool the run called, kept for the whole run rather than per step. */
77
+ runToolCalls: NonNullable<AIAgentStep['toolCalls']>
71
78
  }
72
79
 
73
80
  function createPersistingChannel(
@@ -89,6 +96,9 @@ function createPersistingChannel(
89
96
  inputTokens: 0,
90
97
  outputTokens: 0,
91
98
  }
99
+ // Survives the per-step flush below, which clears its own buffers: the run
100
+ // record needs every call the run made, not just the last step's.
101
+ const runToolCalls: NonNullable<AIAgentStep['toolCalls']> = []
92
102
 
93
103
  const flushStep = async (opts?: { interrupted?: boolean }) => {
94
104
  if (!storage) return
@@ -137,6 +147,8 @@ function createPersistingChannel(
137
147
  })
138
148
  }
139
149
 
150
+ const runToolCallIndex = new Map<string, number>()
151
+
140
152
  const channel: PersistingChannel = {
141
153
  channelId: parent.channelId,
142
154
  openingData: parent.openingData,
@@ -149,6 +161,9 @@ function createPersistingChannel(
149
161
  get totalUsage() {
150
162
  return totalUsage
151
163
  },
164
+ get runToolCalls() {
165
+ return runToolCalls
166
+ },
152
167
  flush: flushStep,
153
168
  close: () => parent.close(),
154
169
  sendBinary: (data) => parent.sendBinary(data),
@@ -157,6 +172,26 @@ function createPersistingChannel(
157
172
  // the client was streamed, and an interrupted run has to be able to
158
173
  // report the fragment it got through even with persistence turned off.
159
174
  if (event.type === 'text-delta') fullText += event.text
175
+ if (event.type === 'tool-call') {
176
+ runToolCallIndex.set(event.toolCallId, runToolCalls.length)
177
+ runToolCalls.push({
178
+ name: event.toolName,
179
+ args: event.args as Record<string, unknown>,
180
+ result: '',
181
+ })
182
+ }
183
+ if (event.type === 'tool-result') {
184
+ const index = runToolCallIndex.get(event.toolCallId)
185
+ const result =
186
+ typeof event.result === 'string'
187
+ ? event.result
188
+ : JSON.stringify(event.result)
189
+ const call = index === undefined ? undefined : runToolCalls[index]
190
+ if (call) {
191
+ call.result = result
192
+ if (event.error) call.error = event.error
193
+ }
194
+ }
160
195
  if (storage) {
161
196
  switch (event.type) {
162
197
  case 'text-delta':
@@ -177,6 +212,7 @@ function createPersistingChannel(
177
212
  typeof event.result === 'string'
178
213
  ? event.result
179
214
  : JSON.stringify(event.result),
215
+ ...(event.error ? { error: event.error } : {}),
180
216
  })
181
217
  break
182
218
  case 'generative-ui':
@@ -203,44 +239,67 @@ function createPersistingChannel(
203
239
  return channel
204
240
  }
205
241
 
242
+ /**
243
+ * Agents already warned about, so a per-request hook does not become a
244
+ * per-request log line.
245
+ */
246
+ const warnedUnstreamedOutputHooks = new Set<string>()
247
+
248
+ /**
249
+ * `modifyOutput` does not run on a streamed run at all. Nothing here could act
250
+ * on what it returns — the text has already reached the client, and
251
+ * `createPersistingChannel` flushes each step to storage as it goes, so by the
252
+ * time the run ends the transcript is already written.
253
+ *
254
+ * Rewriting on this path belongs to `modifyOutputStream`, which genuinely
255
+ * works: the stream middleware wraps the persisting channel, so what is stored
256
+ * and accumulated is already what the client was sent. A middleware that
257
+ * rewrites in `modifyOutput` only — a redaction hook, typically — is therefore
258
+ * silently ineffective when the agent is streamed, and is told so once.
259
+ */
260
+ const warnUnstreamedOutputHooks = (
261
+ agentName: string,
262
+ aiMiddlewares: PikkuAIMiddlewareHooks[],
263
+ logger?: { warn: (...args: any[]) => void }
264
+ ) => {
265
+ if (warnedUnstreamedOutputHooks.has(agentName)) return
266
+ const unstreamed = aiMiddlewares.some(
267
+ (mw) => mw.modifyOutput && !mw.modifyOutputStream
268
+ )
269
+ if (!unstreamed) return
270
+ warnedUnstreamedOutputHooks.add(agentName)
271
+ logger?.warn(
272
+ `Agent '${agentName}' has AI middleware with modifyOutput but no modifyOutputStream — modifyOutput does not apply to streamed runs. Implement modifyOutputStream to affect a streamed reply.`
273
+ )
274
+ }
275
+
206
276
  async function postStreamCleanup(
207
277
  persistingChannel: PersistingChannel,
208
- aiMiddlewares: PikkuAIMiddlewareHooks[],
209
- singletonServices: any,
210
- messages: AIMessage[],
211
278
  aiRunState: AIRunStateService,
212
- runId: string
213
- ): Promise<void> {
214
- const usage = persistingChannel.totalUsage
215
- let outputText = persistingChannel.fullText
216
- let outputMessages = messages
217
- for (let i = aiMiddlewares.length - 1; i >= 0; i--) {
218
- const mw = aiMiddlewares[i]
219
- if (mw.modifyOutput) {
220
- const result = await mw.modifyOutput(singletonServices, {
221
- text: outputText,
222
- messages: outputMessages,
223
- usage: {
224
- inputTokens: usage.inputTokens,
225
- outputTokens: usage.outputTokens,
226
- },
227
- })
228
- outputText = result.text
229
- outputMessages = result.messages
230
- }
279
+ runId: string,
280
+ run: {
281
+ agentName: string
282
+ threadId: string
283
+ resourceId?: string
284
+ input: string
231
285
  }
232
-
233
- await aiRunState.updateRun(runId, {
234
- status: 'completed',
235
- ...(usage.model
236
- ? {
237
- usage: {
238
- inputTokens: usage.inputTokens,
239
- outputTokens: usage.outputTokens,
240
- model: usage.model,
241
- },
242
- }
243
- : {}),
286
+ ): Promise<void> {
287
+ await finalizeAgentRun(aiRunState, {
288
+ runId,
289
+ agentName: run.agentName,
290
+ threadId: run.threadId,
291
+ resourceId: run.resourceId,
292
+ input: run.input,
293
+ // Already what the client received: the stream middleware wraps the
294
+ // persisting channel, so both were accumulated post-rewrite.
295
+ text: persistingChannel.fullText,
296
+ steps: [
297
+ {
298
+ usage: persistingChannel.totalUsage,
299
+ toolCalls: persistingChannel.runToolCalls,
300
+ },
301
+ ],
302
+ usage: persistingChannel.totalUsage,
244
303
  })
245
304
  }
246
305
 
@@ -723,6 +782,8 @@ export async function streamAIAgent(
723
782
  await storage.saveMessages(threadId, [persistedUserMessage])
724
783
  }
725
784
 
785
+ warnUnstreamedOutputHooks(agentName, aiMiddlewares, singletonServices.logger)
786
+
726
787
  const streamMiddleware = aiMiddlewares
727
788
  .filter((mw) => mw.modifyOutputStream)
728
789
  .map((mw) => {
@@ -849,14 +910,12 @@ export async function streamAIAgent(
849
910
  return persistingChannel.fullText
850
911
  }
851
912
 
852
- await postStreamCleanup(
853
- persistingChannel,
854
- aiMiddlewares,
855
- singletonServices,
856
- runnerParams.messages,
857
- aiRunState,
858
- runId
859
- )
913
+ await postStreamCleanup(persistingChannel, aiRunState, runId, {
914
+ agentName,
915
+ threadId,
916
+ resourceId: input.resourceId,
917
+ input: lastUserMessageText(runnerParams.messages),
918
+ })
860
919
 
861
920
  // knowledge: decisions/internals/the-agent-done-event-goes-through-the-middleware-and-is-awaited.md
862
921
  await outputChannel.send({ type: 'done' })
@@ -1184,16 +1243,17 @@ export async function resumeAIAgent(
1184
1243
  typeof pending.args === 'string' ? JSON.parse(pending.args) : pending.args
1185
1244
 
1186
1245
  let toolResult: unknown
1187
- let isError = false
1246
+ let toolError: string | undefined
1188
1247
  try {
1189
1248
  toolResult = await matchingTool.execute(toolArgs)
1190
1249
  } catch (execErr: any) {
1191
1250
  if (execErr?.payload?.error === 'missing_credential') {
1192
1251
  toolResult = execErr.payload
1252
+ toolError = 'missing_credential'
1193
1253
  } else {
1194
- toolResult = `Error: ${execErr instanceof Error ? execErr.message : String(execErr)}`
1254
+ toolError = execErr instanceof Error ? execErr.message : String(execErr)
1255
+ toolResult = `Error: ${toolError}`
1195
1256
  }
1196
- isError = true
1197
1257
  }
1198
1258
 
1199
1259
  const resultStr =
@@ -1220,7 +1280,7 @@ export async function resumeAIAgent(
1220
1280
  toolCallId: input.toolCallId,
1221
1281
  toolName: pending.toolName,
1222
1282
  result: toolResult,
1223
- ...(isError ? { isError: true } : {}),
1283
+ ...(toolError ? { error: toolError } : {}),
1224
1284
  })
1225
1285
  }
1226
1286
 
@@ -1313,6 +1373,12 @@ async function continueAfterToolResult(
1313
1373
  // knowledge: decisions/internals/a-resumed-agent-turn-is-as-interruptible-as-the-first.md
1314
1374
  const interruptHandle = registerInterruptibleRun(run.runId)
1315
1375
 
1376
+ warnUnstreamedOutputHooks(
1377
+ run.agentName,
1378
+ aiMiddlewares,
1379
+ singletonServices.logger
1380
+ )
1381
+
1316
1382
  const streamMiddleware = aiMiddlewares
1317
1383
  .filter((mw) => mw.modifyOutputStream)
1318
1384
  .map((mw) => {
@@ -1434,14 +1500,10 @@ async function continueAfterToolResult(
1434
1500
  return
1435
1501
  }
1436
1502
 
1437
- await postStreamCleanup(
1438
- persistingChannel,
1439
- aiMiddlewares,
1440
- singletonServices,
1441
- runnerParams.messages,
1442
- aiRunState,
1443
- run.runId
1444
- )
1503
+ await postStreamCleanup(persistingChannel, aiRunState, run.runId, {
1504
+ ...run,
1505
+ input: lastUserMessageText(runnerParams.messages),
1506
+ })
1445
1507
 
1446
1508
  // knowledge: decisions/internals/the-agent-done-event-goes-through-the-middleware-and-is-awaited.md
1447
1509
  await wrappedChannel.send({ type: 'done' })
@@ -0,0 +1,67 @@
1
+ import { describe, test } from 'node:test'
2
+ import assert from 'node:assert/strict'
3
+
4
+ import { toAccumulatedStep } from './ai-agent-turn.js'
5
+ import type { AIAgentStepResult } from '../../services/ai-agent-runner-service.js'
6
+
7
+ const stepResult = (
8
+ overrides?: Partial<AIAgentStepResult>
9
+ ): AIAgentStepResult => ({
10
+ text: '',
11
+ toolCalls: [],
12
+ toolResults: [],
13
+ usage: { inputTokens: 0, outputTokens: 0 },
14
+ finishReason: 'stop',
15
+ ...overrides,
16
+ })
17
+
18
+ describe('toAccumulatedStep', () => {
19
+ test('carries a tool failure as its own field, not only as rendered text', () => {
20
+ const step = toAccumulatedStep(
21
+ stepResult({
22
+ toolCalls: [
23
+ { toolCallId: 'call-1', toolName: 'lookupOrder', args: { id: 7 } },
24
+ ],
25
+ toolResults: [
26
+ {
27
+ toolCallId: 'call-1',
28
+ toolName: 'lookupOrder',
29
+ result: 'Error: order service unreachable',
30
+ error: 'order service unreachable',
31
+ },
32
+ ],
33
+ })
34
+ )
35
+
36
+ assert.deepEqual(step.toolCalls, [
37
+ {
38
+ name: 'lookupOrder',
39
+ args: { id: 7 },
40
+ result: 'Error: order service unreachable',
41
+ error: 'order service unreachable',
42
+ },
43
+ ])
44
+ })
45
+
46
+ test('leaves error unset on a tool that returned normally, even if it says Error', () => {
47
+ // A tool is allowed to return the word "Error" — which is exactly why
48
+ // "did this fail" cannot be answered by matching on the result text.
49
+ const step = toAccumulatedStep(
50
+ stepResult({
51
+ toolCalls: [
52
+ { toolCallId: 'call-1', toolName: 'searchLogs', args: { q: 'x' } },
53
+ ],
54
+ toolResults: [
55
+ {
56
+ toolCallId: 'call-1',
57
+ toolName: 'searchLogs',
58
+ result: 'Error: connection refused (1 match)',
59
+ },
60
+ ],
61
+ })
62
+ )
63
+
64
+ assert.equal(step.toolCalls[0].error, undefined)
65
+ assert.ok(!('error' in step.toolCalls[0]))
66
+ })
67
+ })
@@ -83,6 +83,7 @@ export const toAccumulatedStep = (stepResult: StepResult) => ({
83
83
  typeof tr?.result === 'string'
84
84
  ? tr.result
85
85
  : JSON.stringify(tr?.result ?? ''),
86
+ ...(tr?.error ? { error: tr.error } : {}),
86
87
  }
87
88
  }),
88
89
  })
@@ -51,6 +51,13 @@ export interface AIToolResult {
51
51
  id: string
52
52
  name: string
53
53
  result: string
54
+ /**
55
+ * Set when the tool threw rather than returned. Carried separately from
56
+ * `result`, which is a rendered string by the time it is persisted — a tool
57
+ * may legitimately return text beginning `Error:`, so the prefix cannot be
58
+ * read as a failure signal.
59
+ */
60
+ error?: string
54
61
  }
55
62
 
56
63
  export interface AIMessage {
@@ -82,7 +89,13 @@ export interface AIMessage {
82
89
 
83
90
  export interface AIAgentStep {
84
91
  usage: { inputTokens: number; outputTokens: number }
85
- toolCalls?: { name: string; args: Record<string, unknown>; result: string }[]
92
+ toolCalls?: {
93
+ name: string
94
+ args: Record<string, unknown>
95
+ result: string
96
+ /** The failure message, when the tool threw rather than returned. */
97
+ error?: string
98
+ }[]
86
99
  }
87
100
 
88
101
  export interface AIAgentInputAttachment {
@@ -233,16 +246,40 @@ export interface PikkuAIMiddlewareHooks<
233
246
  | AIStreamEvent[]
234
247
  | null
235
248
 
249
+ /**
250
+ * The last chance to rewrite what the run produced, before it is persisted
251
+ * and returned.
252
+ *
253
+ * It does **not** run on a streamed run: there the text has already reached
254
+ * the client and each step is flushed to storage as it goes, so nothing could
255
+ * act on what this returned. Use {@link modifyOutputStream} to rewrite a
256
+ * streamed reply — a middleware that implements only this one is warned about
257
+ * when an agent it is attached to streams.
258
+ *
259
+ * `toolCalls` is here so a redaction pass covers the whole run record rather
260
+ * than just the visible answer: the tool arguments and results are persisted
261
+ * and handed to anything that grades the run, and scrubbing the reply alone
262
+ * leaves them untouched.
263
+ */
236
264
  modifyOutput?: (
237
265
  services: Services,
238
266
  ctx: {
239
267
  text: string
240
268
  messages: AIMessage[]
241
269
  usage: { inputTokens: number; outputTokens: number }
270
+ toolCalls: NonNullable<AIAgentStep['toolCalls']>
242
271
  }
243
272
  ) =>
244
- | Promise<{ text: string; messages: AIMessage[] }>
245
- | { text: string; messages: AIMessage[] }
273
+ | Promise<{
274
+ text: string
275
+ messages: AIMessage[]
276
+ toolCalls?: NonNullable<AIAgentStep['toolCalls']>
277
+ }>
278
+ | {
279
+ text: string
280
+ messages: AIMessage[]
281
+ toolCalls?: NonNullable<AIAgentStep['toolCalls']>
282
+ }
246
283
 
247
284
  beforeToolCall?: (
248
285
  services: Services,
@@ -273,7 +310,13 @@ export interface PikkuAIMiddlewareHooks<
273
310
  stepNumber: number
274
311
  text: string
275
312
  toolCalls: { toolCallId: string; toolName: string; args: unknown }[]
276
- toolResults: { toolCallId: string; toolName: string; result: unknown }[]
313
+ toolResults: {
314
+ toolCallId: string
315
+ toolName: string
316
+ result: unknown
317
+ /** Set when the tool threw rather than returned. */
318
+ error?: string
319
+ }[]
277
320
  usage: { inputTokens: number; outputTokens: number }
278
321
  finishReason: string
279
322
  }
@@ -301,6 +344,7 @@ export type CoreAIAgent<
301
344
  PikkuPermission = CorePikkuPermission<any, any>,
302
345
  PikkuMiddleware = CorePikkuMiddleware<any>,
303
346
  Scope extends string = string,
347
+ Scorer extends string = string,
304
348
  > = {
305
349
  name: string
306
350
  description: string
@@ -327,6 +371,16 @@ export type CoreAIAgent<
327
371
  tools?: unknown[]
328
372
  agents?: unknown[]
329
373
  workflows?: unknown[]
374
+ /**
375
+ * Grades this agent's finished runs on live traffic, named by the generated
376
+ * `ScorerName` union rather than by `ref()` — a scorer is not a function, so
377
+ * there is nothing in the function map for a ref to resolve against.
378
+ *
379
+ * A reference-based judge listed here is never sampled: live traffic has no
380
+ * answer key. Scenarios name scorers directly and may grade with scorers an
381
+ * agent does not ship with.
382
+ */
383
+ scorers?: Scorer[]
330
384
  agentMode?: 'delegate' | 'supervise'
331
385
  memory?: AIAgentMemoryConfig
332
386
  maxSteps?: number
@@ -392,6 +446,12 @@ export type AIStreamEvent =
392
446
  toolCallId: string
393
447
  toolName: string
394
448
  result: unknown
449
+ /**
450
+ * The failure message, set when the tool threw rather than returned.
451
+ * Carried explicitly because a tool may legitimately return text that
452
+ * reads like an error, so `result` cannot be matched on to tell.
453
+ */
454
+ error?: string
395
455
  agent?: string
396
456
  session?: string
397
457
  }
@@ -5,8 +5,9 @@ export {
5
5
  agentApprove,
6
6
  agentInterrupt,
7
7
  } from './ai-agent-helpers.js'
8
- export { wrapChannelWithAGUI, type AGUIEvent } from './ai-agent-agui.js'
8
+ export { wrapChannelWithAGUI } from './ai-agent-agui.js'
9
9
  export { runAIAgent, resumeAIAgentSync } from './ai-agent-runner.js'
10
+ export { resolveModelAlias } from './ai-agent-model-config.js'
10
11
  export {
11
12
  streamAIAgent,
12
13
  resumeAIAgent,
@@ -14,7 +15,6 @@ export {
14
15
  } from './ai-agent-stream.js'
15
16
  export {
16
17
  voiceInput,
17
- readsAsNonSpeech,
18
18
  NoSpeechDetectedError,
19
19
  SPOKEN_TURN,
20
20
  SPOKEN_TRANSCRIPT,
@@ -27,21 +27,12 @@ export {
27
27
  } from './voice-output.js'
28
28
  export {
29
29
  AgentInterruptedError,
30
- awaitPendingInterruptNote,
31
- getInFlightTools,
32
- isAbortError,
33
- isRunInterruptible,
34
- persistOrphanedToolResults,
35
- registerInterruptibleRun,
36
30
  signalRunInterrupt,
37
- trackInterruptNote,
38
- trackToolExecution,
39
31
  } from './ai-agent-interrupt.js'
40
32
  export type {
41
33
  AgentInterruption,
42
34
  AgentInterruptResult,
43
35
  InterruptibleRunHandle,
44
- OrphanedToolResult,
45
36
  } from './ai-agent-interrupt.js'
46
37
  export {
47
38
  type RunAIAgentParams,
@@ -50,18 +41,13 @@ export {
50
41
  ToolCredentialRequired,
51
42
  canAccessThread,
52
43
  isOwnedByPrincipal,
53
- sessionPrincipals,
54
44
  threadOwnerConstraint,
55
45
  } from './ai-agent-prepare.js'
56
46
  export {
57
47
  addAIAgent,
58
- approveAIAgent,
59
- getAIAgents,
60
- getAIAgentsMeta,
61
48
  } from './ai-agent-registry.js'
62
49
  export type {
63
50
  AIAgentInput,
64
- AIAgentInputAttachment,
65
51
  AIAgentMeta,
66
52
  AIAgentMemoryConfig,
67
53
  AIAgentStep,
@@ -0,0 +1,106 @@
1
+ import { beforeEach, describe, test } from 'node:test'
2
+ import assert from 'node:assert/strict'
3
+ import { pikkuState, resetPikkuState } from '../../pikku-state.js'
4
+ import { gradeRun } from './ai-scorer-grade.js'
5
+ import { pikkuAIJudge, pikkuAIScorer } from './ai-scorer.js'
6
+ import type { ScoreJob } from './ai-scorer.types.js'
7
+
8
+ const job = (overrides: Partial<ScoreJob> = {}): ScoreJob => ({
9
+ scorerName: 'correctness',
10
+ runId: 'run-1',
11
+ agentName: 'assistant',
12
+ input: 'what is the capital of France?',
13
+ output: 'Paris',
14
+ toolCalls: [],
15
+ usage: { inputTokens: 10, outputTokens: 5 },
16
+ ...overrides,
17
+ })
18
+
19
+ describe('gradeRun', () => {
20
+ beforeEach(() => resetPikkuState())
21
+
22
+ test('a scenario grade is returned and not written to the live record', async () => {
23
+ pikkuState(null, 'agent', 'scorers').set(
24
+ 'correctness',
25
+ pikkuAIScorer({
26
+ name: 'correctness',
27
+ description: 'Matches the answer key',
28
+ requiresReference: true,
29
+ score: (input) => ({
30
+ score: input.output === input.reference ? 1 : 0,
31
+ }),
32
+ })
33
+ )
34
+ let saves = 0
35
+
36
+ const result = await gradeRun(
37
+ job({ reference: 'Paris' }),
38
+ {
39
+ aiRunState: {
40
+ saveScore: async () => {
41
+ saves++
42
+ },
43
+ },
44
+ },
45
+ { persist: false }
46
+ )
47
+
48
+ assert.deepEqual(result, { score: 1 })
49
+ assert.equal(saves, 0)
50
+ })
51
+
52
+ test('a reference-based scorer sees the answer key it was given', async () => {
53
+ pikkuState(null, 'agent', 'scorers').set(
54
+ 'correctness',
55
+ pikkuAIScorer({
56
+ name: 'correctness',
57
+ description: 'Matches the answer key',
58
+ requiresReference: true,
59
+ score: (input) => ({
60
+ score: input.output === input.reference ? 1 : 0,
61
+ }),
62
+ })
63
+ )
64
+
65
+ const result = await gradeRun(
66
+ job({ reference: 'Lyon' }),
67
+ {},
68
+ {
69
+ persist: false,
70
+ }
71
+ )
72
+
73
+ assert.equal(result.score, 0)
74
+ })
75
+
76
+ test('a judge grades without a score function, and reports the model it used', async () => {
77
+ pikkuState(null, 'agent', 'scorers').set(
78
+ 'helpfulness',
79
+ pikkuAIJudge({
80
+ name: 'helpfulness',
81
+ description: 'Is the answer useful',
82
+ model: 'claude-opus-5',
83
+ goal: 'Grade helpfulness.',
84
+ })
85
+ )
86
+
87
+ const result = await gradeRun(
88
+ job({ scorerName: 'helpfulness' }),
89
+ {
90
+ aiAgentRunner: {
91
+ run: async () => ({
92
+ object: { score: 0.75, reason: 'Correct but terse.' },
93
+ usage: { inputTokens: 80, outputTokens: 20 },
94
+ }),
95
+ },
96
+ },
97
+ { persist: false }
98
+ )
99
+
100
+ assert.equal(result.score, 0.75)
101
+ assert.deepEqual(result.metadata, {
102
+ judgeModel: 'claude-opus-5',
103
+ judgeTokens: 100,
104
+ })
105
+ })
106
+ })
@@ -0,0 +1,55 @@
1
+ import { runJudge } from './ai-scorer-judge.js'
2
+ import { resolveAIScorer } from './ai-scorer-registry.js'
3
+ import type { ScoreJob, ScorerOutput } from './ai-scorer.types.js'
4
+
5
+ /**
6
+ * Grade one run with one scorer.
7
+ *
8
+ * The single path both callers take — the lane worker on live traffic and the
9
+ * scenario grading RPC — so a scenario's grade is the same computation the
10
+ * production sampler would have made, not an approximation of it.
11
+ *
12
+ * Persisting is optional because the two callers differ on it: a live grade is
13
+ * only useful once recorded, while a scenario asserts on the returned value and
14
+ * runs against servers that may have no run-state adapter at all.
15
+ */
16
+ export const gradeRun = async (
17
+ job: ScoreJob,
18
+ services: {
19
+ aiAgentRunner?: unknown
20
+ aiRunState?: {
21
+ saveScore: (score: {
22
+ runId: string
23
+ scorerName: string
24
+ score: number
25
+ reason?: string
26
+ metadata?: Record<string, unknown>
27
+ }) => Promise<void>
28
+ }
29
+ },
30
+ options: { persist: boolean }
31
+ ): Promise<ScorerOutput> => {
32
+ const { scorerName, ...input } = job
33
+ const scorer = resolveAIScorer(scorerName)
34
+
35
+ const result = scorer.score
36
+ ? await scorer.score(input, services)
37
+ : await runJudge(scorer, input, services.aiAgentRunner as never)
38
+
39
+ if (options.persist) {
40
+ if (!services.aiRunState) {
41
+ throw new Error(
42
+ `AI run state service not initialized: cannot record the '${scorerName}' grade of run ${job.runId}`
43
+ )
44
+ }
45
+ await services.aiRunState.saveScore({
46
+ runId: job.runId,
47
+ scorerName,
48
+ score: result.score,
49
+ ...(result.reason !== undefined ? { reason: result.reason } : {}),
50
+ ...(result.metadata !== undefined ? { metadata: result.metadata } : {}),
51
+ })
52
+ }
53
+
54
+ return result
55
+ }