@kindgi/api 0.1.3 → 0.1.4-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (222) hide show
  1. package/dist/agent-binding.d.ts +26 -4
  2. package/dist/agent-binding.d.ts.map +1 -1
  3. package/dist/agent-pins.d.ts +48 -0
  4. package/dist/agent-pins.d.ts.map +1 -0
  5. package/dist/agent-pins.js +102 -0
  6. package/dist/agent-pins.js.map +1 -0
  7. package/dist/app.d.ts +22 -0
  8. package/dist/app.d.ts.map +1 -1
  9. package/dist/app.js +27 -3
  10. package/dist/app.js.map +1 -1
  11. package/dist/block-binding.d.ts +132 -0
  12. package/dist/block-binding.d.ts.map +1 -0
  13. package/dist/block-binding.js +4 -0
  14. package/dist/block-binding.js.map +1 -0
  15. package/dist/block-pins.d.ts +22 -0
  16. package/dist/block-pins.d.ts.map +1 -0
  17. package/dist/block-pins.js +112 -0
  18. package/dist/block-pins.js.map +1 -0
  19. package/dist/cost-binding.d.ts +20 -1
  20. package/dist/cost-binding.d.ts.map +1 -1
  21. package/dist/cost-binding.js +4 -0
  22. package/dist/cost-binding.js.map +1 -1
  23. package/dist/deploy-versions.d.ts +60 -0
  24. package/dist/deploy-versions.d.ts.map +1 -0
  25. package/dist/deploy-versions.js +91 -0
  26. package/dist/deploy-versions.js.map +1 -0
  27. package/dist/deployment-binding.d.ts +24 -3
  28. package/dist/deployment-binding.d.ts.map +1 -1
  29. package/dist/derive-agent-version.d.ts +77 -0
  30. package/dist/derive-agent-version.d.ts.map +1 -0
  31. package/dist/derive-agent-version.js +149 -0
  32. package/dist/derive-agent-version.js.map +1 -0
  33. package/dist/errors.d.ts.map +1 -1
  34. package/dist/errors.js +18 -0
  35. package/dist/errors.js.map +1 -1
  36. package/dist/eval-case-binding.d.ts +65 -0
  37. package/dist/eval-case-binding.d.ts.map +1 -0
  38. package/dist/eval-case-binding.js +4 -0
  39. package/dist/eval-case-binding.js.map +1 -0
  40. package/dist/eval-run-binding.d.ts +37 -0
  41. package/dist/eval-run-binding.d.ts.map +1 -1
  42. package/dist/eval-run-binding.js.map +1 -1
  43. package/dist/eval-run-dispatcher.d.ts +41 -3
  44. package/dist/eval-run-dispatcher.d.ts.map +1 -1
  45. package/dist/eval-run-dispatcher.js +21 -15
  46. package/dist/eval-run-dispatcher.js.map +1 -1
  47. package/dist/eval-suite-binding.d.ts +1 -1
  48. package/dist/eval-suite-binding.d.ts.map +1 -1
  49. package/dist/eval-suite-binding.js +2 -0
  50. package/dist/eval-suite-binding.js.map +1 -1
  51. package/dist/flow-binding.d.ts +18 -4
  52. package/dist/flow-binding.d.ts.map +1 -1
  53. package/dist/flow-pins.d.ts +36 -0
  54. package/dist/flow-pins.d.ts.map +1 -0
  55. package/dist/flow-pins.js +81 -0
  56. package/dist/flow-pins.js.map +1 -0
  57. package/dist/guardrail-binding.d.ts +8 -0
  58. package/dist/guardrail-binding.d.ts.map +1 -1
  59. package/dist/handler-binding.d.ts +3 -0
  60. package/dist/handler-binding.d.ts.map +1 -1
  61. package/dist/index.d.ts +18 -6
  62. package/dist/index.d.ts.map +1 -1
  63. package/dist/index.js +8 -2
  64. package/dist/index.js.map +1 -1
  65. package/dist/judged-dispatcher.d.ts +138 -0
  66. package/dist/judged-dispatcher.d.ts.map +1 -0
  67. package/dist/judged-dispatcher.js +308 -0
  68. package/dist/judged-dispatcher.js.map +1 -0
  69. package/dist/judged-items.d.ts +86 -0
  70. package/dist/judged-items.d.ts.map +1 -0
  71. package/dist/judged-items.js +184 -0
  72. package/dist/judged-items.js.map +1 -0
  73. package/dist/judgment-binding.d.ts +316 -0
  74. package/dist/judgment-binding.d.ts.map +1 -0
  75. package/dist/judgment-binding.js +19 -0
  76. package/dist/judgment-binding.js.map +1 -0
  77. package/dist/openapi/generate.d.ts.map +1 -1
  78. package/dist/openapi/generate.js +4 -1
  79. package/dist/openapi/generate.js.map +1 -1
  80. package/dist/openapi/operations.d.ts.map +1 -1
  81. package/dist/openapi/operations.js +482 -13
  82. package/dist/openapi/operations.js.map +1 -1
  83. package/dist/openapi/schemas.d.ts +46 -0
  84. package/dist/openapi/schemas.d.ts.map +1 -1
  85. package/dist/openapi/schemas.js +1350 -175
  86. package/dist/openapi/schemas.js.map +1 -1
  87. package/dist/provider-binding.d.ts +12 -7
  88. package/dist/provider-binding.d.ts.map +1 -1
  89. package/dist/registry-read-only.d.ts +32 -0
  90. package/dist/registry-read-only.d.ts.map +1 -0
  91. package/dist/registry-read-only.js +22 -0
  92. package/dist/registry-read-only.js.map +1 -0
  93. package/dist/routes/agents.d.ts +9 -1
  94. package/dist/routes/agents.d.ts.map +1 -1
  95. package/dist/routes/agents.js +181 -11
  96. package/dist/routes/agents.js.map +1 -1
  97. package/dist/routes/blocks.d.ts +19 -0
  98. package/dist/routes/blocks.d.ts.map +1 -0
  99. package/dist/routes/blocks.js +306 -0
  100. package/dist/routes/blocks.js.map +1 -0
  101. package/dist/routes/cost.d.ts.map +1 -1
  102. package/dist/routes/cost.js +47 -2
  103. package/dist/routes/cost.js.map +1 -1
  104. package/dist/routes/deployments.d.ts +3 -0
  105. package/dist/routes/deployments.d.ts.map +1 -1
  106. package/dist/routes/deployments.js +224 -55
  107. package/dist/routes/deployments.js.map +1 -1
  108. package/dist/routes/eval-comparison.d.ts +15 -0
  109. package/dist/routes/eval-comparison.d.ts.map +1 -0
  110. package/dist/routes/eval-comparison.js +123 -0
  111. package/dist/routes/eval-comparison.js.map +1 -0
  112. package/dist/routes/eval-runs.d.ts +7 -1
  113. package/dist/routes/eval-runs.d.ts.map +1 -1
  114. package/dist/routes/eval-runs.js +42 -3
  115. package/dist/routes/eval-runs.js.map +1 -1
  116. package/dist/routes/eval-versions.d.ts +25 -0
  117. package/dist/routes/eval-versions.d.ts.map +1 -0
  118. package/dist/routes/eval-versions.js +66 -0
  119. package/dist/routes/eval-versions.js.map +1 -0
  120. package/dist/routes/flows.d.ts +13 -1
  121. package/dist/routes/flows.d.ts.map +1 -1
  122. package/dist/routes/flows.js +48 -3
  123. package/dist/routes/flows.js.map +1 -1
  124. package/dist/routes/guardrails.d.ts.map +1 -1
  125. package/dist/routes/guardrails.js +4 -0
  126. package/dist/routes/guardrails.js.map +1 -1
  127. package/dist/routes/hierarchy-errors.d.ts +45 -0
  128. package/dist/routes/hierarchy-errors.d.ts.map +1 -0
  129. package/dist/routes/hierarchy-errors.js +47 -0
  130. package/dist/routes/hierarchy-errors.js.map +1 -0
  131. package/dist/routes/judged-suites.d.ts +20 -0
  132. package/dist/routes/judged-suites.d.ts.map +1 -0
  133. package/dist/routes/judged-suites.js +272 -0
  134. package/dist/routes/judged-suites.js.map +1 -0
  135. package/dist/routes/judgment-context.d.ts +22 -0
  136. package/dist/routes/judgment-context.d.ts.map +1 -0
  137. package/dist/routes/judgment-context.js +88 -0
  138. package/dist/routes/judgment-context.js.map +1 -0
  139. package/dist/routes/judgment-flow-context.d.ts +32 -0
  140. package/dist/routes/judgment-flow-context.d.ts.map +1 -0
  141. package/dist/routes/judgment-flow-context.js +195 -0
  142. package/dist/routes/judgment-flow-context.js.map +1 -0
  143. package/dist/routes/judgments.d.ts +41 -0
  144. package/dist/routes/judgments.d.ts.map +1 -0
  145. package/dist/routes/judgments.js +566 -0
  146. package/dist/routes/judgments.js.map +1 -0
  147. package/dist/routes/orgs.d.ts +5 -2
  148. package/dist/routes/orgs.d.ts.map +1 -1
  149. package/dist/routes/orgs.js +38 -22
  150. package/dist/routes/orgs.js.map +1 -1
  151. package/dist/routes/policies.d.ts.map +1 -1
  152. package/dist/routes/policies.js +12 -1
  153. package/dist/routes/policies.js.map +1 -1
  154. package/dist/routes/projects.d.ts +10 -2
  155. package/dist/routes/projects.d.ts.map +1 -1
  156. package/dist/routes/projects.js +94 -80
  157. package/dist/routes/projects.js.map +1 -1
  158. package/dist/routes/providers.d.ts.map +1 -1
  159. package/dist/routes/providers.js +6 -1
  160. package/dist/routes/providers.js.map +1 -1
  161. package/dist/routes/runs.js +25 -3
  162. package/dist/routes/runs.js.map +1 -1
  163. package/dist/routes/teams.d.ts +6 -2
  164. package/dist/routes/teams.d.ts.map +1 -1
  165. package/dist/routes/teams.js +83 -73
  166. package/dist/routes/teams.js.map +1 -1
  167. package/dist/routes/tools.d.ts.map +1 -1
  168. package/dist/routes/tools.js +4 -0
  169. package/dist/routes/tools.js.map +1 -1
  170. package/dist/tool-binding.d.ts +8 -0
  171. package/dist/tool-binding.d.ts.map +1 -1
  172. package/openapi.json +14253 -10099
  173. package/package.json +21 -21
  174. package/src/agent-binding.ts +28 -4
  175. package/src/agent-pins.ts +147 -0
  176. package/src/app.ts +81 -3
  177. package/src/block-binding.ts +137 -0
  178. package/src/block-pins.ts +148 -0
  179. package/src/cost-binding.ts +21 -1
  180. package/src/deploy-versions.ts +157 -0
  181. package/src/deployment-binding.ts +27 -3
  182. package/src/derive-agent-version.ts +217 -0
  183. package/src/errors.ts +18 -0
  184. package/src/eval-case-binding.ts +71 -0
  185. package/src/eval-run-binding.ts +40 -0
  186. package/src/eval-run-dispatcher.ts +60 -16
  187. package/src/eval-suite-binding.ts +2 -0
  188. package/src/flow-binding.ts +20 -4
  189. package/src/flow-pins.ts +113 -0
  190. package/src/guardrail-binding.ts +9 -0
  191. package/src/handler-binding.ts +3 -0
  192. package/src/index.ts +86 -2
  193. package/src/judged-dispatcher.ts +530 -0
  194. package/src/judged-items.ts +263 -0
  195. package/src/judgment-binding.ts +349 -0
  196. package/src/openapi/generate.ts +7 -1
  197. package/src/openapi/operations.ts +579 -13
  198. package/src/openapi/schemas.ts +1408 -128
  199. package/src/provider-binding.ts +12 -7
  200. package/src/registry-read-only.ts +43 -0
  201. package/src/routes/agents.ts +252 -19
  202. package/src/routes/blocks.ts +387 -0
  203. package/src/routes/cost.ts +55 -1
  204. package/src/routes/deployments.ts +291 -56
  205. package/src/routes/eval-comparison.ts +135 -0
  206. package/src/routes/eval-runs.ts +57 -3
  207. package/src/routes/eval-versions.ts +110 -0
  208. package/src/routes/flows.ts +70 -5
  209. package/src/routes/guardrails.ts +7 -0
  210. package/src/routes/hierarchy-errors.ts +60 -0
  211. package/src/routes/judged-suites.ts +363 -0
  212. package/src/routes/judgment-context.ts +128 -0
  213. package/src/routes/judgment-flow-context.ts +245 -0
  214. package/src/routes/judgments.ts +743 -0
  215. package/src/routes/orgs.ts +44 -27
  216. package/src/routes/policies.ts +19 -0
  217. package/src/routes/projects.ts +118 -96
  218. package/src/routes/providers.ts +5 -0
  219. package/src/routes/runs.ts +29 -3
  220. package/src/routes/teams.ts +104 -90
  221. package/src/routes/tools.ts +7 -0
  222. package/src/tool-binding.ts +9 -0
@@ -0,0 +1,530 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ /**
5
+ * The comparison eval run of a `judged` suite (a test set): each case is a
6
+ * past run people judged; the candidate agent version re-runs it as a
7
+ * replay (`EvalRunSubjectInvokeInput.replay`, so it does nothing the past
8
+ * run didn't), `repetitions` times, and its output's items are scored
9
+ * against the judgments, beside the baseline's.
10
+ *
11
+ * Per case: the replay runs, the items kept, dropped and new (new ones
12
+ * for experts to judge), the tool calls and what happened to each, and
13
+ * whether the replay diverged (a read with no recording ran live under
14
+ * `reads: 'recorded'`). The summary (`JudgedComparisonSummary`) is what a
15
+ * promotion gate reads.
16
+ */
17
+
18
+ import type { ReplayTurnReport } from '@kindgi/agents';
19
+ import type { FlowVersionOverrides } from '@kindgi/flow';
20
+ import type { RunId } from '@kindgi/types';
21
+
22
+ import type { EvalCaseStoreBinding, JudgedEvalCase } from './eval-case-binding.js';
23
+ import type { AgentRef, EvalComparison, FlowRef } from './eval-run-binding.js';
24
+ import type {
25
+ DispatchContext,
26
+ DispatchResult,
27
+ EvalRunDispatcher,
28
+ EvalRunSubjectInvokeOutcome,
29
+ } from './eval-run-dispatcher.js';
30
+ import {
31
+ type ItemChanges,
32
+ type OutputScore,
33
+ itemChanges,
34
+ matchJudged,
35
+ outputItems,
36
+ scoreItems,
37
+ } from './judged-items.js';
38
+
39
+ export const DEFAULT_COMPARISON: EvalComparison = {
40
+ baseline: 'recorded',
41
+ reads: 'recorded',
42
+ repetitions: 1,
43
+ k: 10,
44
+ };
45
+
46
+ /** One metric, baseline beside candidate. `null` where a side had no judged evidence. */
47
+ export interface ComparisonMetric {
48
+ readonly baseline: number | null;
49
+ readonly candidate: number | null;
50
+ readonly delta: number | null;
51
+ /** The candidate's evidence: cases with judged items, and the judgment weight behind it. */
52
+ readonly n: number;
53
+ readonly weight: number;
54
+ readonly baselineN: number;
55
+ readonly baselineWeight: number;
56
+ readonly direction: 'higher';
57
+ /** `weightedPrecisionAtK`: the ranked items it looked at. */
58
+ readonly k?: number;
59
+ /** With more than one repetition: the candidate's max − min across them. */
60
+ readonly spread?: number;
61
+ }
62
+
63
+ export type ComparisonBaselineSummary =
64
+ | {
65
+ readonly kind: 'recorded';
66
+ /** The versions that served the recorded runs, with how many cases each. */
67
+ readonly versions: readonly RecordedVersion[];
68
+ }
69
+ | {
70
+ readonly kind: 'version';
71
+ readonly agentId: string;
72
+ readonly version: string;
73
+ readonly via: 'explicit' | 'live';
74
+ readonly liveScope?: Readonly<Record<string, unknown>>;
75
+ };
76
+
77
+ /** A version behind recorded runs: an agent's or a flow's. */
78
+ export type RecordedVersion =
79
+ | { readonly agentId: string; readonly version: string; readonly cases: number }
80
+ | { readonly flowId: string; readonly version: string; readonly cases: number };
81
+
82
+ /** What ran on the cases: an agent version, or a flow version. */
83
+ export type ComparisonCandidate =
84
+ | { readonly kind: 'agent'; readonly agentId: string; readonly version: string }
85
+ | {
86
+ readonly kind: 'flow';
87
+ readonly flowId: string;
88
+ readonly version: string;
89
+ /** Agents and tools its replays ran at other versions than the flow version's pins. */
90
+ readonly versions?: FlowVersionOverrides;
91
+ };
92
+
93
+ /** What a comparison eval run concluded: what a promotion gate reads. */
94
+ export interface JudgedComparisonSummary {
95
+ readonly evalRunId: string;
96
+ readonly status: 'completed' | 'partial' | 'failed';
97
+ readonly completedAt: string;
98
+ readonly suite: { readonly id: string; readonly version: string };
99
+ readonly candidate: ComparisonCandidate;
100
+ readonly baseline: ComparisonBaselineSummary;
101
+ /** Where the test set's judgments came from. */
102
+ readonly scope: { readonly projectId?: string };
103
+ readonly cases: number;
104
+ /** Cases where a read with no recording ran live under `reads: 'recorded'`. */
105
+ readonly diverged: number;
106
+ /** Tool calls refused across the cases (what the candidate would have done). */
107
+ readonly refusedWrites: number;
108
+ /** Cases that errored: none of their repetitions ran. */
109
+ readonly errors: number;
110
+ /**
111
+ * Cases that stopped at a refused write (a flow's tool node the replay
112
+ * refused): no output to score, so they're left out of the metrics.
113
+ */
114
+ readonly stopped: number;
115
+ readonly reads: EvalComparison['reads'];
116
+ /** The models that answered the candidate's replays, and how many replays each. */
117
+ readonly sampling: {
118
+ readonly models: readonly {
119
+ readonly providerId: string;
120
+ readonly model: string;
121
+ readonly runs: number;
122
+ }[];
123
+ };
124
+ readonly repetitions: number;
125
+ readonly metrics: {
126
+ readonly weightedYesShare: ComparisonMetric;
127
+ readonly judgedCoverage: ComparisonMetric;
128
+ readonly weightedPrecisionAtK: ComparisonMetric;
129
+ };
130
+ }
131
+
132
+ /** One case's result. */
133
+ export interface JudgedCaseResult {
134
+ readonly caseId: string;
135
+ /** The candidate's replay runs, one per repetition (absent for one that couldn't start). */
136
+ readonly runIds: readonly string[];
137
+ readonly baseline: OutputScore;
138
+ /** One per repetition that ran. */
139
+ readonly candidate: readonly OutputScore[];
140
+ /** The first repetition's items against the judged ones. */
141
+ readonly changes?: ItemChanges;
142
+ /** The first repetition's tool calls. */
143
+ readonly tools?: ReplayTurnReport['tools'];
144
+ readonly diverged: boolean;
145
+ readonly refusedWrites: number;
146
+ readonly noContext: boolean;
147
+ readonly approvalSkipped: boolean;
148
+ readonly error?: string;
149
+ /** Set when the replay stopped at a refused write: what it would have done. */
150
+ readonly stopped?: NonNullable<EvalRunSubjectInvokeOutcome['stopped']>;
151
+ }
152
+
153
+ export interface JudgedDispatcherOptions {
154
+ readonly cases: EvalCaseStoreBinding;
155
+ }
156
+
157
+ /** Cases read per page. */
158
+ const CASE_PAGE = 100;
159
+
160
+ export const VERSIONS_NEED_A_FLOW =
161
+ '`versions` runs a flow with some of its agents or tools at other versions: it needs `flowRef`.';
162
+
163
+ function validateComparison(
164
+ suite: { readonly spec: Readonly<Record<string, unknown>> },
165
+ target: AgentRef | FlowRef,
166
+ comparison: EvalComparison | undefined,
167
+ ): { kind: 'ok' } | { kind: 'err'; message: string } {
168
+ if (target.version === undefined) {
169
+ return {
170
+ kind: 'err',
171
+ message:
172
+ 'agentId' in target
173
+ ? '`agentRef.version` is required: the candidate is one version of the agent.'
174
+ : '`flowRef.version` is required: the candidate is one version of the flow.',
175
+ };
176
+ }
177
+ if (suite.spec.caseCount === 0) {
178
+ return { kind: 'err', message: 'The test set has no cases.' };
179
+ }
180
+ const c = comparison ?? DEFAULT_COMPARISON;
181
+ if (c.versions !== undefined && 'agentId' in target) {
182
+ return { kind: 'err', message: VERSIONS_NEED_A_FLOW };
183
+ }
184
+ if (c.baseline !== 'recorded') {
185
+ return { kind: 'err', message: "Only `baseline: 'recorded'` runs today." };
186
+ }
187
+ return { kind: 'ok' };
188
+ }
189
+
190
+ export function createJudgedDispatcher(options: JudgedDispatcherOptions): EvalRunDispatcher {
191
+ return {
192
+ kind: 'judged',
193
+ validate: validateComparison,
194
+ async dispatch(ctx): Promise<DispatchResult> {
195
+ const comparison = ctx.comparison ?? DEFAULT_COMPARISON;
196
+ const all = await allCases(options.cases, ctx);
197
+ if (ctx.dryRun) {
198
+ return { result: { dryRun: true, cases: all.length, comparison } };
199
+ }
200
+ const results: JudgedCaseResult[] = [];
201
+ const models = new Map<string, { providerId: string; model: string; runs: number }>();
202
+ for (const judgedCase of all) {
203
+ if (ctx.abortSignal.aborted) break;
204
+ const result = await runCase(ctx, comparison, judgedCase, models);
205
+ results.push(result);
206
+ ctx.onProgress(result as unknown as Readonly<Record<string, unknown>>);
207
+ }
208
+ const summary = summarize(ctx, comparison, all, results, [...models.values()]);
209
+ return {
210
+ result: { summary, perCase: results },
211
+ ...(ctx.abortSignal.aborted && { error: 'cancelled' }),
212
+ };
213
+ },
214
+ };
215
+ }
216
+
217
+ async function allCases(
218
+ store: EvalCaseStoreBinding,
219
+ ctx: DispatchContext,
220
+ ): Promise<readonly JudgedEvalCase[]> {
221
+ const out: JudgedEvalCase[] = [];
222
+ let cursor: Parameters<EvalCaseStoreBinding['listCases']>[0]['cursor'];
223
+ for (;;) {
224
+ const page = await store.listCases({
225
+ tenantId: ctx.tenantId,
226
+ suiteId: ctx.suite.id,
227
+ version: ctx.suite.version,
228
+ limit: CASE_PAGE,
229
+ ...(cursor !== undefined && { cursor }),
230
+ });
231
+ out.push(...page.data);
232
+ if (!page.hasMore || page.nextCursor === undefined) return out;
233
+ cursor = page.nextCursor;
234
+ }
235
+ }
236
+
237
+ /** One case's repetitions as they come back. */
238
+ class CaseTally {
239
+ readonly runIds: string[] = [];
240
+ readonly candidate: OutputScore[] = [];
241
+ first: { changes: ItemChanges; replay?: ReplayTurnReport } | undefined;
242
+ error: string | undefined;
243
+ stopped: JudgedCaseResult['stopped'];
244
+ diverged = false;
245
+ refusedWrites = 0;
246
+ approvalSkipped = false;
247
+
248
+ constructor(
249
+ private readonly judgedCase: JudgedEvalCase,
250
+ private readonly comparison: EvalComparison,
251
+ private readonly agentTurn: boolean,
252
+ ) {}
253
+
254
+ add(outcome: EvalRunSubjectInvokeOutcome): 'ran' | 'stopped' | 'error' {
255
+ if (outcome.runId !== undefined) this.runIds.push(outcome.runId as unknown as string);
256
+ if (outcome.stopped !== undefined) {
257
+ this.stopped ??= outcome.stopped;
258
+ this.refusedWrites = Math.max(this.refusedWrites, refusedCount(outcome));
259
+ this.first ??= { changes: { kept: [], dropped: [], new: [] }, ...replayOf(outcome) };
260
+ return 'stopped';
261
+ }
262
+ if (outcome.error !== undefined) {
263
+ this.error ??= outcome.error;
264
+ return 'error';
265
+ }
266
+ this.ran(outcome);
267
+ return 'ran';
268
+ }
269
+
270
+ private ran(outcome: EvalRunSubjectInvokeOutcome): void {
271
+ const { judgedCase, comparison } = this;
272
+ const matched = matchJudged(
273
+ outputItems(outcome.output, this.agentTurn),
274
+ judgedCase.items,
275
+ judgedCase.output,
276
+ );
277
+ this.candidate.push(scoreItems(matched, comparison.k));
278
+ const tools = outcome.replay?.tools ?? [];
279
+ this.refusedWrites = Math.max(this.refusedWrites, refusedCount(outcome));
280
+ if (comparison.reads === 'recorded' && tools.some((t) => t.source === 'live')) {
281
+ this.diverged = true;
282
+ }
283
+ if (outcome.replay?.approval === 'skipped') this.approvalSkipped = true;
284
+ this.first ??= { changes: itemChanges(matched, judgedCase.items), ...replayOf(outcome) };
285
+ }
286
+
287
+ result(baseline: OutputScore): JudgedCaseResult {
288
+ const none = this.candidate.length === 0;
289
+ return {
290
+ caseId: this.judgedCase.caseId,
291
+ runIds: this.runIds,
292
+ baseline,
293
+ candidate: this.candidate,
294
+ ...(this.first !== undefined && { changes: this.first.changes }),
295
+ ...(this.first?.replay !== undefined && { tools: this.first.replay.tools }),
296
+ diverged: this.diverged,
297
+ refusedWrites: this.refusedWrites,
298
+ noContext: this.judgedCase.context === undefined,
299
+ approvalSkipped: this.approvalSkipped,
300
+ // A case none of whose repetitions ran stopped (at a refused write) or errored.
301
+ ...(none && this.stopped !== undefined && { stopped: this.stopped }),
302
+ ...(none && this.stopped === undefined && this.error !== undefined && { error: this.error }),
303
+ };
304
+ }
305
+ }
306
+
307
+ async function runCase(
308
+ ctx: DispatchContext,
309
+ comparison: EvalComparison,
310
+ judgedCase: JudgedEvalCase,
311
+ models: Map<string, { providerId: string; model: string; runs: number }>,
312
+ ): Promise<JudgedCaseResult> {
313
+ // An agent turn's items are its answer and typed result; a flow run's, its whole output.
314
+ const agentTurn = judgedCase.subject.kind === 'agent';
315
+ const baseline = scoreItems(
316
+ matchJudged(outputItems(judgedCase.output, agentTurn), judgedCase.items, judgedCase.output),
317
+ comparison.k,
318
+ );
319
+ const tally = new CaseTally(judgedCase, comparison, agentTurn);
320
+ for (let rep = 0; rep < comparison.repetitions; rep++) {
321
+ const outcome = await invokeCase(ctx, judgedCase);
322
+ if (tally.add(outcome) === 'ran') countModel(models, outcome);
323
+ }
324
+ return tally.result(baseline);
325
+ }
326
+
327
+ function refusedCount(outcome: EvalRunSubjectInvokeOutcome): number {
328
+ return (outcome.replay?.tools ?? []).filter((t) => t.source === 'refused').length;
329
+ }
330
+
331
+ function replayOf(outcome: EvalRunSubjectInvokeOutcome): { replay?: ReplayTurnReport } {
332
+ return outcome.replay !== undefined ? { replay: outcome.replay } : {};
333
+ }
334
+
335
+ async function invokeCase(
336
+ ctx: DispatchContext,
337
+ judgedCase: JudgedEvalCase,
338
+ ): Promise<EvalRunSubjectInvokeOutcome> {
339
+ try {
340
+ return await ctx.subject.invoke({
341
+ tenantId: ctx.tenantId,
342
+ target: ctx.target,
343
+ // An agent turn's input is its user message.
344
+ input: judgedCase.input,
345
+ dryRun: false,
346
+ abortSignal: ctx.abortSignal,
347
+ ...(ctx.projectId !== undefined && { projectId: ctx.projectId }),
348
+ replay: {
349
+ of: judgedCase.caseId as unknown as RunId,
350
+ evalRunId: ctx.runId as unknown as string,
351
+ },
352
+ ...(judgedCase.subject.kind === 'agent' && { history: judgedCase.context?.history ?? [] }),
353
+ ...(!('agentId' in ctx.target) &&
354
+ ctx.comparison?.versions !== undefined && { versions: ctx.comparison.versions }),
355
+ });
356
+ } catch (cause) {
357
+ return { error: cause instanceof Error ? cause.message : String(cause) };
358
+ }
359
+ }
360
+
361
+ function countModel(
362
+ models: Map<string, { providerId: string; model: string; runs: number }>,
363
+ outcome: EvalRunSubjectInvokeOutcome,
364
+ ): void {
365
+ if (outcome.provider === undefined) return;
366
+ const key = `${outcome.provider.id}\u0000${outcome.provider.model}`;
367
+ const kept = models.get(key) ?? {
368
+ providerId: outcome.provider.id,
369
+ model: outcome.provider.model,
370
+ runs: 0,
371
+ };
372
+ kept.runs += 1;
373
+ models.set(key, kept);
374
+ }
375
+
376
+ type Side = (s: OutputScore) => { readonly yes: number; readonly total: number };
377
+
378
+ const yesShare: Side = (s) => ({ yes: s.yesWeight, total: s.totalWeight });
379
+ const precisionAtK: Side = (s) => ({ yes: s.topK.yesWeight, total: s.topK.totalWeight });
380
+ /** Coverage's ratio is judged items over items; its evidence weight is the judged weight. */
381
+ const coverage: Side = (s) => ({ yes: s.judgedItems, total: s.items });
382
+
383
+ /** A metric over cases: Σ numerator / Σ denominator, with its evidence. */
384
+ function pooled(
385
+ scores: readonly OutputScore[],
386
+ side: Side,
387
+ weightOf: (s: OutputScore) => number,
388
+ ): { readonly value: number | null; readonly n: number; readonly weight: number } {
389
+ let yes = 0;
390
+ let total = 0;
391
+ let n = 0;
392
+ let weight = 0;
393
+ for (const s of scores) {
394
+ const { yes: y, total: t } = side(s);
395
+ if (t <= 0) continue;
396
+ yes += y;
397
+ total += t;
398
+ n += 1;
399
+ weight += weightOf(s);
400
+ }
401
+ return { value: total > 0 ? yes / total : null, n, weight };
402
+ }
403
+
404
+ function metric(
405
+ results: readonly JudgedCaseResult[],
406
+ repetitions: number,
407
+ side: Side,
408
+ weightOf: (s: OutputScore) => number,
409
+ extra: { readonly k?: number } = {},
410
+ ): ComparisonMetric {
411
+ // Only cases the candidate ran: errored and stopped cases leave both sides.
412
+ const scored = results.filter((r) => r.candidate.length > 0);
413
+ const baseline = pooled(
414
+ scored.map((r) => r.baseline),
415
+ side,
416
+ weightOf,
417
+ );
418
+ // Each repetition is a full pass over the cases: its value, then the spread across them.
419
+ const perRep = Array.from({ length: repetitions }, (_, rep) =>
420
+ pooled(
421
+ scored.flatMap((r) => (r.candidate[rep] !== undefined ? [r.candidate[rep]] : [])),
422
+ side,
423
+ weightOf,
424
+ ),
425
+ );
426
+ const values = perRep.map((p) => p.value).filter((v): v is number => v !== null);
427
+ const candidate = values.length > 0 ? values.reduce((a, b) => a + b, 0) / values.length : null;
428
+ const cases = new Set<number>();
429
+ scored.forEach((r, i) => {
430
+ if (r.candidate.some((s) => side(s).total > 0)) cases.add(i);
431
+ });
432
+ const weight = perRep.reduce((a, p) => a + p.weight, 0) / Math.max(1, perRep.length);
433
+ return {
434
+ baseline: baseline.value,
435
+ candidate,
436
+ delta: baseline.value !== null && candidate !== null ? candidate - baseline.value : null,
437
+ n: cases.size,
438
+ weight,
439
+ baselineN: baseline.n,
440
+ baselineWeight: baseline.weight,
441
+ direction: 'higher',
442
+ ...(extra.k !== undefined && { k: extra.k }),
443
+ ...(repetitions > 1 &&
444
+ values.length > 1 && { spread: Math.max(...values) - Math.min(...values) }),
445
+ };
446
+ }
447
+
448
+ function summarize(
449
+ ctx: DispatchContext,
450
+ comparison: EvalComparison,
451
+ cases: readonly JudgedEvalCase[],
452
+ results: readonly JudgedCaseResult[],
453
+ models: readonly { providerId: string; model: string; runs: number }[],
454
+ ): JudgedComparisonSummary {
455
+ const errors = results.filter((r) => r.error !== undefined).length;
456
+ const stopped = results.filter((r) => r.stopped !== undefined).length;
457
+ const versions = new Map<string, { id: string; flow: boolean; version: string; cases: number }>();
458
+ for (const c of cases) {
459
+ const key = `${c.subject.kind}:${c.subject.id}@${c.subject.version}`;
460
+ const kept = versions.get(key) ?? {
461
+ id: c.subject.id,
462
+ flow: c.subject.kind === 'flow',
463
+ version: c.subject.version,
464
+ cases: 0,
465
+ };
466
+ kept.cases += 1;
467
+ versions.set(key, kept);
468
+ }
469
+ const specProject = ctx.suite.spec.projectId;
470
+ const projectId =
471
+ typeof specProject === 'string'
472
+ ? specProject
473
+ : (ctx.projectId as unknown as string | undefined);
474
+ const judgedWeight = (s: OutputScore) => s.totalWeight;
475
+ return {
476
+ evalRunId: ctx.runId as unknown as string,
477
+ status:
478
+ results.length > 0 && errors === 0
479
+ ? 'completed'
480
+ : errors < results.length
481
+ ? 'partial'
482
+ : 'failed',
483
+ completedAt: new Date().toISOString(),
484
+ suite: { id: ctx.suite.id, version: ctx.suite.version },
485
+ candidate: candidateOf(ctx.target, ctx.comparison?.versions),
486
+ baseline: {
487
+ kind: 'recorded',
488
+ versions: [...versions.values()].map(
489
+ (v): RecordedVersion =>
490
+ v.flow
491
+ ? { flowId: v.id, version: v.version, cases: v.cases }
492
+ : { agentId: v.id, version: v.version, cases: v.cases },
493
+ ),
494
+ },
495
+ scope: { ...(projectId !== undefined && { projectId }) },
496
+ cases: results.length,
497
+ diverged: results.filter((r) => r.diverged).length,
498
+ refusedWrites: results.reduce((a, r) => a + r.refusedWrites, 0),
499
+ errors,
500
+ stopped,
501
+ reads: comparison.reads,
502
+ sampling: { models },
503
+ repetitions: comparison.repetitions,
504
+ metrics: {
505
+ weightedYesShare: metric(results, comparison.repetitions, yesShare, judgedWeight),
506
+ judgedCoverage: metric(results, comparison.repetitions, coverage, judgedWeight),
507
+ weightedPrecisionAtK: metric(
508
+ results,
509
+ comparison.repetitions,
510
+ precisionAtK,
511
+ (s) => s.topK.totalWeight,
512
+ { k: comparison.k },
513
+ ),
514
+ },
515
+ };
516
+ }
517
+
518
+ function candidateOf(
519
+ target: AgentRef | FlowRef,
520
+ versions: FlowVersionOverrides | undefined,
521
+ ): ComparisonCandidate {
522
+ return 'agentId' in target
523
+ ? { kind: 'agent', agentId: target.agentId as unknown as string, version: target.version ?? '' }
524
+ : {
525
+ kind: 'flow',
526
+ flowId: target.flowId as unknown as string,
527
+ version: target.version ?? '',
528
+ ...(versions !== undefined && { versions }),
529
+ };
530
+ }