@agentproto/eval 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jeremy (agentik.net)
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,236 @@
1
+ # @agentproto/eval
2
+
3
+ Deterministic reference scorers for agentproto — shipped as AIP-14 TOOL
4
+ contracts plus one builtin AIP-30 PROVIDER.
5
+
6
+ ## A scorer is a tool
7
+
8
+ There is **no separate "scorer port"**. A scorer is an ordinary AIP-14 tool
9
+ authored with `defineTool`, whose `outputSchema` is the shared `Score` shape,
10
+ and implemented with `implementTool` bundled in a `defineDriver` — exactly like
11
+ every other builtin. That reuse buys the registry, retries, and the
12
+ `toMastraTool` / `toAiSdkTool` projections for free.
13
+
14
+ ```ts
15
+ export const scoreSchema = z.object({
16
+ value: z.number().min(0).max(1), // normalized score
17
+ passed: z.boolean(), // clears the scorer's threshold
18
+ label: z.string(), // scorer id, e.g. "exact-match"
19
+ rationale: z.string().optional(),// human-readable why
20
+ })
21
+ ```
22
+
23
+ ## The four deterministic scorers
24
+
25
+ All are pure functions of their input — no LLM / model calls. For a
26
+ model-backed scorer, see [LLM-as-judge scorer](#llm-as-judge-scorer) below.
27
+
28
+ | Tool id | Input | Behavior |
29
+ | ------------------------ | -------------------------------------------- | -------- |
30
+ | `eval.exact-match` | `{ actual, expected, trim? }` | `value` 1 when equal (after optional trim), else 0. |
31
+ | `eval.regex-match` | `{ actual, pattern, flags? }` | 1 when the compiled RegExp matches, else 0. An **invalid pattern** returns a typed failure `Score` (`value 0`, `rationale: "invalid pattern: …"`) — it never throws. |
32
+ | `eval.json-schema-valid` | `{ actual, schema }` | Minimal structural check of top-level `required` + `properties.type` only. 1 when valid, else 0 with the first violation in `rationale`. |
33
+ | `eval.latency-budget` | `{ durationMs, budgetMs }` | `passed = durationMs <= budgetMs`; `value` is 1 within budget, else `max(0, 1 - (durationMs - budgetMs) / budgetMs)`. |
34
+
35
+ ### `eval.json-schema-valid` limitation
36
+
37
+ To keep this package light (no `ajv`), the schema check is intentionally
38
+ minimal: it verifies the top-level `type`, that every name in `required` is
39
+ present, and that any `properties.<name>.type` matches the value's primitive
40
+ type. It does **not** recurse into nested schemas, and does not implement the
41
+ rest of JSON Schema (`enum`, `format`, `oneOf`, `items`, …). A fuller
42
+ validator is a later step.
43
+
44
+ ## Running a scorer
45
+
46
+ Invoke any scorer through the driver's `runTool`, passing the provider as the
47
+ sole candidate:
48
+
49
+ ```ts
50
+ import { runTool } from "@agentproto/driver"
51
+ import { exactMatchTool, evalScorersProvider } from "@agentproto/eval"
52
+
53
+ const score = await runTool({
54
+ tool: exactMatchTool,
55
+ candidates: [evalScorersProvider],
56
+ input: { actual: "hello", expected: "hello" },
57
+ })
58
+ // → { value: 1, passed: true, label: "exact-match", rationale: "actual equals expected" }
59
+ ```
60
+
61
+ ## Running a suite
62
+
63
+ An **eval is a workflow**: for each case it runs a `target`, then scores the
64
+ target's output with one scorer `tool` step per binding, aggregates, and reports
65
+ through the `@agentproto/telemetry` port. `runEval` composes a
66
+ `@agentproto/workflow-runtime` `RuntimeWorkflow` per case — the target's
67
+ structured output is the input selector for every scorer step — and rolls the
68
+ per-case results up into an `EvalReport`.
69
+
70
+ Each scorer is authored as a `ScorerBinding` (its `tool`, its `driver`, and a
71
+ `mapInput` that shapes the target output into the scorer's input) and boxed with
72
+ `bindScorer` so scorers with different input types coexist in one suite.
73
+
74
+ ```ts
75
+ import { arrayTelemetry } from "@agentproto/telemetry"
76
+ import {
77
+ runEval,
78
+ bindScorer,
79
+ exactMatchTool,
80
+ latencyBudgetTool,
81
+ evalScorersProvider,
82
+ type EvalEvent,
83
+ type TypedEvalSuite,
84
+ } from "@agentproto/eval"
85
+
86
+ interface Answer { text: string; latencyMs: number }
87
+
88
+ const suite: TypedEvalSuite<{ prompt: string }, Answer> = {
89
+ id: "greeting-suite",
90
+ cases: [
91
+ { id: "hello", input: { prompt: "hi" }, expected: "hello" },
92
+ { id: "world", input: { prompt: "wo" }, expected: "world" },
93
+ ],
94
+ scorers: [
95
+ bindScorer<Answer, { actual: string; expected: string }>({
96
+ id: "exact",
97
+ tool: exactMatchTool,
98
+ driver: evalScorersProvider,
99
+ mapInput: ({ output, expected }) => ({
100
+ actual: output.text,
101
+ expected: typeof expected === "string" ? expected : "",
102
+ }),
103
+ }),
104
+ bindScorer<Answer, { durationMs: number; budgetMs: number }>({
105
+ id: "latency",
106
+ tool: latencyBudgetTool,
107
+ driver: evalScorersProvider,
108
+ mapInput: ({ output }) => ({ durationMs: output.latencyMs, budgetMs: 100 }),
109
+ }),
110
+ ],
111
+ }
112
+
113
+ const telemetry = arrayTelemetry<EvalEvent>()
114
+ const report = await runEval(suite, {
115
+ target: async ({ prompt }) => ({ text: prompt === "hi" ? "hello" : "world", latencyMs: 10 }),
116
+ telemetry,
117
+ })
118
+ // report.passedCount / report.meanValue aggregate across cases;
119
+ // telemetry.events runs eval.started … eval.finished in order.
120
+ ```
121
+
122
+ The `target` is a plain async function (not required to be a tool); a
123
+ tool/agent target is a documented follow-up. Events ride the shared telemetry
124
+ port — `Telemetry<EvalEvent>` — under the `agentproto/eval/v1` schema
125
+ (`eval.started`, `eval.case.started`, `eval.case.scored`, `eval.case.finished`,
126
+ `eval.finished`). Sinks MUST tolerate unknown kinds (forward-compat).
127
+
128
+ ## LLM-as-judge scorer
129
+
130
+ `eval.llm-judge` is a **model-backed** scorer — but it stays a TOOL, exactly
131
+ like the four deterministic scorers above. What differs is where the model
132
+ lives: the injected judge is closed over by the **driver**, not threaded
133
+ through tool input/context, via `makeLlmJudgeDriver(judge)`. That means it
134
+ composes with the existing `bindScorer` / `runEval` with zero changes to
135
+ either.
136
+
137
+ The judge itself is an injected function — `JudgeFn` — so this package stays
138
+ vendor-neutral: no LLM SDK, no network dependency here.
139
+
140
+ ```ts
141
+ export type JudgeFn = (args: {
142
+ output: JsonValue
143
+ criteria: string
144
+ expected?: JsonValue
145
+ }) => Promise<JudgeVerdict>
146
+ // JudgeVerdict = { value: number (0..1); passed?: boolean; rationale?: string }
147
+ ```
148
+
149
+ Build a driver around a judge, then either call the tool directly or use the
150
+ `llmJudge(...)` convenience to get a ready-to-use `ScorerBinding`:
151
+
152
+ ```ts
153
+ import { runTool } from "@agentproto/driver"
154
+ import {
155
+ llmJudgeTool,
156
+ makeLlmJudgeDriver,
157
+ llmJudge,
158
+ bindScorer,
159
+ type JudgeFn,
160
+ } from "@agentproto/eval"
161
+
162
+ // A real judge is injected by the caller — an LLM call, an agent session,
163
+ // or the supervisor's judge-gate. Shown here as a stand-in.
164
+ const judge: JudgeFn = async ({ output, criteria }) => {
165
+ // ... call out to a model, or whatever satisfies JudgeFn ...
166
+ return { value: 0.9, rationale: "meets the criteria" }
167
+ }
168
+
169
+ // Direct tool invocation:
170
+ const score = await runTool({
171
+ tool: llmJudgeTool,
172
+ candidates: [makeLlmJudgeDriver(judge, { threshold: 0.7 })],
173
+ input: { output: "the produced answer", criteria: "Is the answer helpful?" },
174
+ })
175
+
176
+ // Or as a suite scorer:
177
+ const binding = bindScorer(
178
+ llmJudge<Answer>({
179
+ id: "helpfulness",
180
+ judge,
181
+ criteria: "Is the answer helpful and correct?",
182
+ threshold: 0.7,
183
+ mapOutput: ({ output }) => output.text,
184
+ }),
185
+ )
186
+ ```
187
+
188
+ `passed` resolution: when the judge's verdict includes an explicit `passed`,
189
+ it **wins** over the threshold comparison — a judge that says `passed: false`
190
+ is never overridden by a high `value`, and vice versa. Otherwise `passed =
191
+ value >= threshold` (`threshold` defaults to `0.5`).
192
+
193
+ A real adapter wiring `JudgeFn` up to an agent session or the supervisor's
194
+ judge-gate is a documented follow-up — not built in this package.
195
+
196
+ ## As a CI gate
197
+
198
+ `toVitest` turns a suite into vitest test registrations — one `it(caseId)` per
199
+ case that runs the case and asserts every scorer passed. Vitest's own
200
+ `describe` / `it` / `expect` are **injected**, so this package carries no vitest
201
+ runtime dependency:
202
+
203
+ ```ts
204
+ import { describe, it, expect } from "vitest"
205
+ import { toVitest } from "@agentproto/eval"
206
+
207
+ toVitest(
208
+ suite,
209
+ { target: async ({ prompt }) => ({ text: prompt === "hi" ? "hello" : "world", latencyMs: 10 }) },
210
+ { describe, it, expect },
211
+ )
212
+ ```
213
+
214
+ ## Exports
215
+
216
+ - `runEval`, `bindScorer` — the eval harness (`EvalCase`, `ScorerBinding`,
217
+ `BoundScorer`, `EvalSuite`, `TypedEvalSuite`, `EvalReport`, `CaseReport`,
218
+ `CaseScore`, `RunEvalOptions`)
219
+ - `EVAL_EVENT_SCHEMA`, `EvalEvent` — the telemetry-port event union
220
+ - `toVitest` — the CI-gate bridge (`VitestHooks`, `ExpectApi`, `ToVitestOptions`)
221
+ - `JsonValue`, `JsonObject`, `jsonValueSchema` — the canonical JSON model
222
+ - `scoreSchema`, `Score`
223
+ - `exactMatchTool` / `exactMatchImpl`
224
+ - `regexMatchTool` / `regexMatchImpl`
225
+ - `jsonSchemaValidTool` / `jsonSchemaValidImpl` (`MinimalJsonSchema`)
226
+ - `latencyBudgetTool` / `latencyBudgetImpl`
227
+ - `evalScorersProvider` — the builtin PROVIDER bundling all four
228
+ - `llmJudgeTool` — the `eval.llm-judge` TOOL contract
229
+ - `makeLlmJudgeDriver` — build a DRIVER that closes over an injected `JudgeFn`
230
+ - `llmJudge` — convenience: build a ready-to-use `ScorerBinding` around a judge
231
+ - `JudgeFn`, `JudgeVerdict`, `judgeVerdictSchema`, `LlmJudgeInput`,
232
+ `MakeLlmJudgeDriverOptions`, `LlmJudgeBinding`
233
+
234
+ ## License
235
+
236
+ MIT
@@ -0,0 +1,449 @@
1
+ import * as _agentproto_driver from '@agentproto/driver';
2
+ import { DriverHandle } from '@agentproto/driver';
3
+ import { z } from 'zod';
4
+ import * as _agentproto_tool from '@agentproto/tool';
5
+ import { ToolHandle } from '@agentproto/tool';
6
+ import { RuntimeWorkflow } from '@agentproto/workflow-runtime';
7
+ import { Telemetry } from '@agentproto/telemetry';
8
+
9
+ /**
10
+ * The shared output shape every scorer produces.
11
+ *
12
+ * A scorer is an AIP-14 TOOL whose `outputSchema` is exactly this — there is
13
+ * no separate "scorer port". Reusing `Score` as the tool output means the
14
+ * registry, retries, and `toMastraTool` / `toAiSdkTool` projections all apply
15
+ * to scorers for free, identical to any other builtin tool.
16
+ */
17
+ declare const scoreSchema: z.ZodObject<{
18
+ value: z.ZodNumber;
19
+ passed: z.ZodBoolean;
20
+ label: z.ZodString;
21
+ rationale: z.ZodOptional<z.ZodString>;
22
+ }, z.core.$strip>;
23
+ type Score = z.infer<typeof scoreSchema>;
24
+
25
+ /**
26
+ * Recursive JSON value union. Modelling the JSON payload explicitly (rather
27
+ * than reaching for `any` / `unknown`) lets consumers inspect values with
28
+ * proper narrowing and no casts. This is the ONE canonical `JsonValue` for the
29
+ * package — scorers, the eval harness, and the vitest bridge all import it from
30
+ * here so there is no divergent redeclaration.
31
+ */
32
+ type JsonValue = string | number | boolean | null | JsonValue[] | {
33
+ [key: string]: JsonValue;
34
+ };
35
+ /** A plain JSON object (not `null`, not an array). */
36
+ type JsonObject = {
37
+ [key: string]: JsonValue;
38
+ };
39
+ /** Zod schema mirroring {@link JsonValue}. */
40
+ declare const jsonValueSchema: z.ZodType<JsonValue>;
41
+
42
+ declare const exactMatchTool: _agentproto_tool.ToolHandle<{
43
+ actual: string;
44
+ expected: string;
45
+ trim?: boolean | undefined;
46
+ }, {
47
+ value: number;
48
+ passed: boolean;
49
+ label: string;
50
+ rationale?: string | undefined;
51
+ }, _agentproto_tool.ToolContext>;
52
+ declare const exactMatchImpl: _agentproto_driver.ToolImplementation<{
53
+ actual: string;
54
+ expected: string;
55
+ trim?: boolean | undefined;
56
+ }, {
57
+ value: number;
58
+ passed: boolean;
59
+ label: string;
60
+ rationale?: string | undefined;
61
+ }, _agentproto_tool.ToolContext>;
62
+ declare const regexMatchTool: _agentproto_tool.ToolHandle<{
63
+ actual: string;
64
+ pattern: string;
65
+ flags?: string | undefined;
66
+ }, {
67
+ value: number;
68
+ passed: boolean;
69
+ label: string;
70
+ rationale?: string | undefined;
71
+ }, _agentproto_tool.ToolContext>;
72
+ declare const regexMatchImpl: _agentproto_driver.ToolImplementation<{
73
+ actual: string;
74
+ pattern: string;
75
+ flags?: string | undefined;
76
+ }, {
77
+ value: number;
78
+ passed: boolean;
79
+ label: string;
80
+ rationale?: string | undefined;
81
+ }, _agentproto_tool.ToolContext>;
82
+ /**
83
+ * A deliberately minimal, dependency-free JSON-schema shape.
84
+ *
85
+ * LIMITATION: this scorer checks ONLY the two most common structural
86
+ * constraints on a top-level object — that every name in `required` is
87
+ * present, and that any name in `properties` declaring a `type` has a value
88
+ * of that primitive type. It does NOT recurse into nested schemas, and does
89
+ * not implement the rest of JSON Schema (enum, format, oneOf, items, …). A
90
+ * full validator (e.g. ajv) is a later step; keeping this package light is
91
+ * the point.
92
+ */
93
+ interface MinimalJsonSchema {
94
+ readonly type?: string;
95
+ readonly required?: readonly string[];
96
+ readonly properties?: {
97
+ readonly [key: string]: {
98
+ readonly type?: string;
99
+ };
100
+ };
101
+ }
102
+ declare const jsonSchemaValidTool: _agentproto_tool.ToolHandle<{
103
+ actual: JsonValue;
104
+ schema: MinimalJsonSchema;
105
+ }, {
106
+ value: number;
107
+ passed: boolean;
108
+ label: string;
109
+ rationale?: string | undefined;
110
+ }, _agentproto_tool.ToolContext>;
111
+ declare const jsonSchemaValidImpl: _agentproto_driver.ToolImplementation<{
112
+ actual: JsonValue;
113
+ schema: MinimalJsonSchema;
114
+ }, {
115
+ value: number;
116
+ passed: boolean;
117
+ label: string;
118
+ rationale?: string | undefined;
119
+ }, _agentproto_tool.ToolContext>;
120
+ declare const latencyBudgetTool: _agentproto_tool.ToolHandle<{
121
+ durationMs: number;
122
+ budgetMs: number;
123
+ }, {
124
+ value: number;
125
+ passed: boolean;
126
+ label: string;
127
+ rationale?: string | undefined;
128
+ }, _agentproto_tool.ToolContext>;
129
+ declare const latencyBudgetImpl: _agentproto_driver.ToolImplementation<{
130
+ durationMs: number;
131
+ budgetMs: number;
132
+ }, {
133
+ value: number;
134
+ passed: boolean;
135
+ label: string;
136
+ rationale?: string | undefined;
137
+ }, _agentproto_tool.ToolContext>;
138
+
139
+ /**
140
+ * Structured events emitted while an eval suite runs.
141
+ *
142
+ * The union mirrors the shape of `@agentproto/telemetry`'s `TelemetryEvent`:
143
+ * a discriminated union keyed on `kind`, every member carrying a correlation
144
+ * id (`runId`) and an ISO `at` timestamp so a sink can rebuild a per-run span
145
+ * tree. It rides the same {@link Telemetry} port — a `runEval` call takes a
146
+ * `Telemetry<EvalEvent>` sink.
147
+ *
148
+ * Sinks MUST tolerate unknown kinds — future versions may add event types
149
+ * under the same `agentproto/eval/v1` schema (identical forward-compat
150
+ * contract as the telemetry port).
151
+ */
152
+ /** Schema identifier for this event family. */
153
+ declare const EVAL_EVENT_SCHEMA: "agentproto/eval/v1";
154
+ type EvalEvent = {
155
+ readonly kind: "eval.started";
156
+ readonly runId: string;
157
+ readonly at: string;
158
+ readonly suiteId: string;
159
+ readonly caseCount: number;
160
+ readonly scorerCount: number;
161
+ } | {
162
+ readonly kind: "eval.case.started";
163
+ readonly runId: string;
164
+ readonly at: string;
165
+ readonly caseId: string;
166
+ } | {
167
+ readonly kind: "eval.case.scored";
168
+ readonly runId: string;
169
+ readonly at: string;
170
+ readonly caseId: string;
171
+ readonly scorerId: string;
172
+ readonly value: number;
173
+ readonly passed: boolean;
174
+ } | {
175
+ readonly kind: "eval.case.finished";
176
+ readonly runId: string;
177
+ readonly at: string;
178
+ readonly caseId: string;
179
+ /** True when EVERY scorer bound to the case passed. */
180
+ readonly passed: boolean;
181
+ } | {
182
+ readonly kind: "eval.finished";
183
+ readonly runId: string;
184
+ readonly at: string;
185
+ readonly suiteId: string;
186
+ readonly total: number;
187
+ readonly passedCount: number;
188
+ readonly meanValue: number;
189
+ readonly durationMs: number;
190
+ };
191
+
192
+ /**
193
+ * `runEval` — the eval harness.
194
+ *
195
+ * The thesis this file proves: **an eval is a workflow that runs a target then
196
+ * scores its output with scorer-tools, reporting through the telemetry port.**
197
+ *
198
+ * For each case we build a {@link RuntimeWorkflow} that composes
199
+ * target → one scorer `tool` step per binding → a per-case aggregate transform
200
+ * and execute it with `runWorkflow` (AIP-15). The target's structured output is
201
+ * the input selector for every scorer step — that is where composable tool I/O
202
+ * is demonstrated. `runEval` then aggregates across cases and emits
203
+ * {@link EvalEvent}s through an injected `Telemetry<EvalEvent>` sink.
204
+ *
205
+ * The `target` is a plain async function here (not required to be a tool); a
206
+ * tool/agent target is a documented follow-up.
207
+ */
208
+
209
+ /** One case: an input for the target and an optional expected reference. */
210
+ interface EvalCase<I> {
211
+ readonly id: string;
212
+ readonly input: I;
213
+ /** Optional reference value, modelled as JSON (never `unknown`/`any`). */
214
+ readonly expected?: JsonValue;
215
+ }
216
+ /** Context handed to a binding's `mapInput` when scoring one case's output. */
217
+ interface ScorerInputContext<A> {
218
+ readonly output: A;
219
+ readonly expected?: JsonValue;
220
+ }
221
+ /**
222
+ * One scorer applied to the target output. Generic over the target output type
223
+ * `A` and the scorer's own input type `S` (captured when the binding is
224
+ * authored). {@link bindScorer} erases `S` so heterogeneous scorers coexist in
225
+ * one `EvalSuite.scorers` array without an `any`.
226
+ */
227
+ interface ScorerBinding<A, S = JsonValue> {
228
+ /** Label for this binding, e.g. "exact". */
229
+ readonly id: string;
230
+ /** The scorer TOOL contract — its `outputSchema` is the shared {@link Score}. */
231
+ readonly tool: ToolHandle<S, Score>;
232
+ /** The DRIVER that implements the scorer tool. */
233
+ readonly driver: DriverHandle;
234
+ /** Derive the scorer's input from the target's output (+ optional expected). */
235
+ readonly mapInput: (ctx: ScorerInputContext<A>) => S;
236
+ }
237
+ /**
238
+ * The uniform, `S`-erased view of a binding stored in a suite. It exposes the
239
+ * authored `id` plus a closure that runs the scorer for a given case output —
240
+ * the existential box that lets a suite hold scorers with different input
241
+ * types with no `any` at the collection boundary.
242
+ */
243
+ interface BoundScorer<A> {
244
+ readonly id: string;
245
+ /** Build the AIP-15 tool step that scores one case's output. */
246
+ toStep(stepId: string, output: A, expected?: JsonValue): RuntimeWorkflow["steps"][number];
247
+ }
248
+ /**
249
+ * Box a typed {@link ScorerBinding} into a {@link BoundScorer}, capturing the
250
+ * scorer input type `S` inside the closure so the outer type is `S`-free.
251
+ */
252
+ declare function bindScorer<A, S>(binding: ScorerBinding<A, S>): BoundScorer<A>;
253
+ interface EvalSuite<A> {
254
+ readonly id: string;
255
+ readonly cases: readonly EvalCase<unknown>[];
256
+ readonly scorers: readonly BoundScorer<A>[];
257
+ }
258
+ /** One scorer's outcome on one case. */
259
+ interface CaseScore {
260
+ readonly scorerId: string;
261
+ readonly score: Score;
262
+ }
263
+ /** The per-case rollup carried in the final report. */
264
+ interface CaseReport {
265
+ readonly caseId: string;
266
+ readonly passed: boolean;
267
+ readonly scores: readonly CaseScore[];
268
+ }
269
+ interface EvalReport {
270
+ readonly runId: string;
271
+ readonly suiteId: string;
272
+ readonly total: number;
273
+ readonly passedCount: number;
274
+ /** Mean of every scorer value across every case (0 when there are none). */
275
+ readonly meanValue: number;
276
+ readonly cases: readonly CaseReport[];
277
+ }
278
+ interface RunEvalOptions<I, A> {
279
+ /** Produce the target output for a case input. */
280
+ readonly target: (input: I) => Promise<A>;
281
+ /** Sink for {@link EvalEvent}s. Defaults to a no-op. */
282
+ readonly telemetry?: Telemetry<EvalEvent>;
283
+ /** Correlation id for this run. Defaults to a deterministic suite-based id. */
284
+ readonly runId?: string;
285
+ }
286
+ /** A suite whose cases share the target input type `I`. */
287
+ interface TypedEvalSuite<I, A> extends EvalSuite<A> {
288
+ readonly cases: readonly EvalCase<I>[];
289
+ }
290
+ /**
291
+ * Run every case in the suite, scoring each with all bound scorers, and return
292
+ * the aggregated {@link EvalReport}. Emits `eval.started` … `eval.finished`.
293
+ */
294
+ declare function runEval<I, A>(suite: TypedEvalSuite<I, A>, opts: RunEvalOptions<I, A>): Promise<EvalReport>;
295
+
296
+ /**
297
+ * `llmJudge` — a model-backed scorer.
298
+ *
299
+ * Design (locked): a model-backed scorer stays a TOOL, exactly like the
300
+ * deterministic scorers in `scorers.ts`. What differs is where the model
301
+ * lives: the injected {@link JudgeFn} is closed over by the DRIVER (built by
302
+ * {@link makeLlmJudgeDriver}), never threaded through tool input/context. That
303
+ * keeps `eval.llm-judge`'s contract identical in shape to every other scorer
304
+ * tool, so it composes with the existing `bindScorer` / `runEval` in
305
+ * run-eval.ts with ZERO changes there.
306
+ *
307
+ * The judge itself is vendor-neutral: {@link JudgeFn} is just an injected
308
+ * async function. This file has no LLM SDK and no network dependency — a real
309
+ * adapter (an agent session, a supervisor judge-gate, …) is a documented
310
+ * follow-up, not built here.
311
+ */
312
+ /** Zod schema for the raw verdict a {@link JudgeFn} produces. */
313
+ declare const judgeVerdictSchema: z.ZodObject<{
314
+ value: z.ZodNumber;
315
+ passed: z.ZodOptional<z.ZodBoolean>;
316
+ rationale: z.ZodOptional<z.ZodString>;
317
+ }, z.core.$strip>;
318
+ type JudgeVerdict = z.infer<typeof judgeVerdictSchema>;
319
+ /**
320
+ * The seam a real judge satisfies: given the produced `output`, free-form
321
+ * grading `criteria`, and an optional `expected` reference, return a
322
+ * {@link JudgeVerdict}. Callers supply this — a real LLM call, an agent
323
+ * session, or the supervisor's judge-gate all satisfy the same shape.
324
+ * Deliberately no LLM SDK / network types here: keeping this package
325
+ * vendor-neutral is the point.
326
+ */
327
+ type JudgeFn = (args: {
328
+ readonly output: JsonValue;
329
+ readonly criteria: string;
330
+ readonly expected?: JsonValue;
331
+ }) => Promise<JudgeVerdict>;
332
+ /** Tool input for `eval.llm-judge`. */
333
+ interface LlmJudgeInput {
334
+ readonly output: JsonValue;
335
+ readonly criteria: string;
336
+ readonly expected?: JsonValue;
337
+ }
338
+ declare const llmJudgeTool: _agentproto_tool.ToolHandle<{
339
+ output: JsonValue;
340
+ criteria: string;
341
+ expected?: JsonValue | undefined;
342
+ }, {
343
+ value: number;
344
+ passed: boolean;
345
+ label: string;
346
+ rationale?: string | undefined;
347
+ }, _agentproto_tool.ToolContext>;
348
+ interface MakeLlmJudgeDriverOptions {
349
+ /** Minimum `value` to count as passed when the judge omits `passed`. Default 0.5. */
350
+ readonly threshold?: number;
351
+ }
352
+ /**
353
+ * Build a DRIVER that implements `eval.llm-judge` by delegating to `judge`.
354
+ * This is the one seam where the injected model-backed capability enters the
355
+ * system — everything downstream (`bindScorer`, `runEval`) stays unaware that
356
+ * this scorer is model-backed at all.
357
+ */
358
+ declare function makeLlmJudgeDriver(judge: JudgeFn, opts?: MakeLlmJudgeDriverOptions): DriverHandle;
359
+ interface LlmJudgeBinding<A> {
360
+ /** Label for this binding, e.g. "helpfulness". */
361
+ readonly id: string;
362
+ /** The injected judge capability. */
363
+ readonly judge: JudgeFn;
364
+ /** Free-form grading criteria/rubric handed to the judge on every call. */
365
+ readonly criteria: string;
366
+ /** Minimum value to count as passed when the judge omits `passed`. Default 0.5. */
367
+ readonly threshold?: number;
368
+ /** Derive the judge's `output` input from the target's output. */
369
+ readonly mapOutput: (ctx: {
370
+ output: A;
371
+ expected?: JsonValue;
372
+ }) => JsonValue;
373
+ }
374
+ /**
375
+ * Convenience: build a {@link ScorerBinding} ready to drop into a suite's
376
+ * `scorers` via `bindScorer` — wires up `eval.llm-judge`, a driver built with
377
+ * {@link makeLlmJudgeDriver}, and the input mapping in one call.
378
+ */
379
+ declare function llmJudge<A>(binding: LlmJudgeBinding<A>): ScorerBinding<A, LlmJudgeInput>;
380
+
381
+ /**
382
+ * `toVitest` — turn an {@link TypedEvalSuite} into vitest test registrations.
383
+ *
384
+ * The vitest primitives (`describe` / `it` / `expect`) are INJECTED by the
385
+ * host's own `import ... from "vitest"`, so THIS source carries no vitest
386
+ * runtime dependency (nothing to bundle, nothing to version-pin). It registers
387
+ * one `it(caseId)` per case that runs the case through {@link runEval} and
388
+ * asserts every scorer passed — the same workflow the harness runs, wired as a
389
+ * CI gate.
390
+ */
391
+
392
+ /** The subset of vitest's `expect` this bridge needs. */
393
+ interface ExpectApi {
394
+ (actual: unknown): {
395
+ toBe(expected: unknown): void;
396
+ };
397
+ }
398
+ /** The subset of vitest's `describe`/`it` this bridge needs. */
399
+ interface VitestHooks {
400
+ describe(name: string, body: () => void): void;
401
+ it(name: string, body: () => Promise<void> | void): void;
402
+ expect: ExpectApi;
403
+ }
404
+ type ToVitestOptions<I, A> = RunEvalOptions<I, A>;
405
+ /**
406
+ * Register `describe(suite.id)` with one `it` per case. Each `it` runs the
407
+ * single-case suite through {@link runEval} and asserts the case passed (all
408
+ * scorers green). Failing scorers surface as a failing vitest assertion.
409
+ */
410
+ declare function toVitest<I, A>(suite: TypedEvalSuite<I, A>, opts: ToVitestOptions<I, A>, hooks: VitestHooks): void;
411
+
412
+ /**
413
+ * @agentproto/eval — deterministic reference scorers.
414
+ *
415
+ * A scorer IS an AIP-14 TOOL whose `outputSchema` is the shared {@link Score}
416
+ * shape; there is no separate "scorer port". Each scorer is authored with
417
+ * `defineTool` + `implementTool` and bundled in a single builtin AIP-30
418
+ * PROVIDER, exactly like every other builtin. Invoke one through the driver:
419
+ *
420
+ * ```ts
421
+ * import { runTool } from "@agentproto/driver"
422
+ * import { exactMatchTool, evalScorersProvider } from "@agentproto/eval"
423
+ *
424
+ * const score = await runTool({
425
+ * tool: exactMatchTool,
426
+ * candidates: [evalScorersProvider],
427
+ * input: { actual: "hello", expected: "hello" },
428
+ * })
429
+ * // → { value: 1, passed: true, label: "exact-match", rationale: "…" }
430
+ * ```
431
+ *
432
+ * Alongside the deterministic scorers, `eval.llm-judge` (see judge.ts) is a
433
+ * model-backed scorer whose judge is an INJECTED function — the driver
434
+ * closes over it via `makeLlmJudgeDriver(judge)`. This package still carries
435
+ * no LLM SDK or network dependency of its own; a real adapter (agent
436
+ * session, supervisor judge-gate, …) is a documented follow-up.
437
+ *
438
+ * Spec: https://agentproto.sh/docs/aip-14 (TOOL), /docs/aip-30 (PROVIDER)
439
+ */
440
+ declare const SPEC_NAME: "agenteval/v1";
441
+ declare const SPEC_VERSION: "1.0.0-alpha";
442
+
443
+ /**
444
+ * Builtin AIP-30 PROVIDER bundling the four deterministic ref scorers.
445
+ * `kind: "builtin"` — pure in-process functions, no subprocess / network hop.
446
+ */
447
+ declare const evalScorersProvider: _agentproto_driver.DriverHandle;
448
+
449
+ export { type BoundScorer, type CaseReport, type CaseScore, EVAL_EVENT_SCHEMA, type EvalCase, type EvalEvent, type EvalReport, type EvalSuite, type ExpectApi, type JsonObject, type JsonValue, type JudgeFn, type JudgeVerdict, type LlmJudgeBinding, type LlmJudgeInput, type MakeLlmJudgeDriverOptions, type MinimalJsonSchema, type RunEvalOptions, SPEC_NAME, SPEC_VERSION, type Score, type ScorerBinding, type ScorerInputContext, type ToVitestOptions, type TypedEvalSuite, type VitestHooks, bindScorer, evalScorersProvider, exactMatchImpl, exactMatchTool, jsonSchemaValidImpl, jsonSchemaValidTool, jsonValueSchema, judgeVerdictSchema, latencyBudgetImpl, latencyBudgetTool, llmJudge, llmJudgeTool, makeLlmJudgeDriver, regexMatchImpl, regexMatchTool, runEval, scoreSchema, toVitest };
package/dist/index.mjs ADDED
@@ -0,0 +1,444 @@
1
+ import { implementTool, defineDriver } from '@agentproto/driver';
2
+ import { z } from 'zod';
3
+ import { defineTool } from '@agentproto/tool';
4
+ import { runWorkflow } from '@agentproto/workflow-runtime';
5
+
6
+ /**
7
+ * @agentproto/eval v0.1.0
8
+ * Deterministic reference scorers as AIP-14 TOOL contracts + builtin PROVIDER.
9
+ */
10
+
11
+ var scoreSchema = z.object({
12
+ /** Normalized score in [0, 1]. */
13
+ value: z.number().min(0).max(1),
14
+ /** Whether the score clears the scorer's own threshold. */
15
+ passed: z.boolean(),
16
+ /** Scorer id, e.g. "exact-match". */
17
+ label: z.string(),
18
+ /** Human-readable explanation of the outcome. */
19
+ rationale: z.string().optional()
20
+ });
21
+ var jsonValueSchema = z.lazy(
22
+ () => z.union([
23
+ z.string(),
24
+ z.number(),
25
+ z.boolean(),
26
+ z.null(),
27
+ z.array(jsonValueSchema),
28
+ z.record(z.string(), jsonValueSchema)
29
+ ])
30
+ );
31
+
32
+ // src/scorers.ts
33
+ var exactMatchTool = defineTool({
34
+ id: "eval.exact-match",
35
+ description: "Deterministic scorer: value 1 when `actual` equals `expected` (optionally after trimming whitespace on both sides), else 0.",
36
+ version: "0.1.0",
37
+ inputSchema: z.object({
38
+ actual: z.string().describe("The produced string to score."),
39
+ expected: z.string().describe("The reference string to compare against."),
40
+ trim: z.boolean().optional().describe("Trim leading/trailing whitespace on both sides before comparing.")
41
+ }),
42
+ outputSchema: scoreSchema,
43
+ mutates: [],
44
+ approval: "auto",
45
+ riskLevel: 0
46
+ });
47
+ var exactMatchImpl = implementTool(exactMatchTool, ({ input }) => {
48
+ const a = input.trim ? input.actual.trim() : input.actual;
49
+ const b = input.trim ? input.expected.trim() : input.expected;
50
+ const equal = a === b;
51
+ return {
52
+ value: equal ? 1 : 0,
53
+ passed: equal,
54
+ label: "exact-match",
55
+ rationale: equal ? "actual equals expected" : "actual does not equal expected"
56
+ };
57
+ });
58
+ var regexMatchTool = defineTool({
59
+ id: "eval.regex-match",
60
+ description: "Deterministic scorer: value 1 when `actual` matches the RegExp built from `pattern`/`flags`, else 0. An invalid pattern yields a typed failure Score rather than a thrown error.",
61
+ version: "0.1.0",
62
+ inputSchema: z.object({
63
+ actual: z.string().describe("The produced string to test."),
64
+ pattern: z.string().describe("RegExp source, compiled inside the body."),
65
+ flags: z.string().optional().describe("RegExp flags, e.g. 'i', 'm', 'gimsuy'.")
66
+ }),
67
+ outputSchema: scoreSchema,
68
+ mutates: [],
69
+ approval: "auto",
70
+ riskLevel: 0
71
+ });
72
+ var regexMatchImpl = implementTool(regexMatchTool, ({ input }) => {
73
+ let regex;
74
+ try {
75
+ regex = new RegExp(input.pattern, input.flags);
76
+ } catch (err) {
77
+ const reason = err instanceof Error ? err.message : String(err);
78
+ return {
79
+ value: 0,
80
+ passed: false,
81
+ label: "regex-match",
82
+ rationale: `invalid pattern: ${reason}`
83
+ };
84
+ }
85
+ const matched = regex.test(input.actual);
86
+ return {
87
+ value: matched ? 1 : 0,
88
+ passed: matched,
89
+ label: "regex-match",
90
+ rationale: matched ? `actual matches /${input.pattern}/${input.flags ?? ""}` : `actual does not match /${input.pattern}/${input.flags ?? ""}`
91
+ };
92
+ });
93
+ var minimalJsonSchemaSchema = z.object({
94
+ type: z.string().optional(),
95
+ required: z.array(z.string()).optional(),
96
+ properties: z.record(z.string(), z.object({ type: z.string().optional() })).optional()
97
+ });
98
+ function isJsonObject(value) {
99
+ return typeof value === "object" && value !== null && !Array.isArray(value);
100
+ }
101
+ function jsonTypeOf(value) {
102
+ if (value === null) return "null";
103
+ if (Array.isArray(value)) return "array";
104
+ if (typeof value === "number") return "number";
105
+ if (typeof value === "boolean") return "boolean";
106
+ if (typeof value === "string") return "string";
107
+ return "object";
108
+ }
109
+ var jsonSchemaValidTool = defineTool({
110
+ id: "eval.json-schema-valid",
111
+ description: "Deterministic scorer: minimal structural check of a JSON value against a schema's top-level `required` and `properties` type constraints only. Not a full JSON Schema validator (no recursion, enum, format, \u2026).",
112
+ version: "0.1.0",
113
+ inputSchema: z.object({
114
+ actual: jsonValueSchema.describe("The JSON value to validate."),
115
+ schema: minimalJsonSchemaSchema.describe(
116
+ "Minimal JSON schema: top-level `required` and `properties.type` only."
117
+ )
118
+ }),
119
+ outputSchema: scoreSchema,
120
+ mutates: [],
121
+ approval: "auto",
122
+ riskLevel: 0
123
+ });
124
+ var jsonSchemaValidImpl = implementTool(
125
+ jsonSchemaValidTool,
126
+ ({ input }) => {
127
+ const { actual, schema } = input;
128
+ const violation = firstSchemaViolation(actual, schema);
129
+ const valid = violation === void 0;
130
+ return {
131
+ value: valid ? 1 : 0,
132
+ passed: valid,
133
+ label: "json-schema-valid",
134
+ rationale: valid ? "value satisfies the schema" : violation
135
+ };
136
+ }
137
+ );
138
+ function firstSchemaViolation(actual, schema) {
139
+ const declaredType = schema.type ?? "object";
140
+ const actualType = jsonTypeOf(actual);
141
+ if (declaredType !== actualType) {
142
+ return `expected top-level type '${declaredType}' but got '${actualType}'`;
143
+ }
144
+ if (!isJsonObject(actual)) {
145
+ return void 0;
146
+ }
147
+ for (const key of schema.required ?? []) {
148
+ if (!(key in actual)) {
149
+ return `missing required property '${key}'`;
150
+ }
151
+ }
152
+ for (const [key, propSchema] of Object.entries(schema.properties ?? {})) {
153
+ if (propSchema.type === void 0) continue;
154
+ const propValue = actual[key];
155
+ if (propValue === void 0) continue;
156
+ const propType = jsonTypeOf(propValue);
157
+ if (propType !== propSchema.type) {
158
+ return `property '${key}' expected type '${propSchema.type}' but got '${propType}'`;
159
+ }
160
+ }
161
+ return void 0;
162
+ }
163
+ var latencyBudgetTool = defineTool({
164
+ id: "eval.latency-budget",
165
+ description: "Deterministic scorer: passes when `durationMs` is within `budgetMs`. value is 1 when within budget, else linearly decays toward 0 as the overrun approaches one full budget.",
166
+ version: "0.1.0",
167
+ inputSchema: z.object({
168
+ durationMs: z.number().min(0).describe("Observed duration in milliseconds."),
169
+ budgetMs: z.number().positive().describe("Allowed budget in milliseconds (must be > 0).")
170
+ }),
171
+ outputSchema: scoreSchema,
172
+ mutates: [],
173
+ approval: "auto",
174
+ riskLevel: 0
175
+ });
176
+ var latencyBudgetImpl = implementTool(latencyBudgetTool, ({ input }) => {
177
+ const { durationMs, budgetMs } = input;
178
+ const passed = durationMs <= budgetMs;
179
+ const value = passed ? 1 : Math.max(0, 1 - (durationMs - budgetMs) / budgetMs);
180
+ return {
181
+ value,
182
+ passed,
183
+ label: "latency-budget",
184
+ rationale: `durationMs=${durationMs}, budgetMs=${budgetMs}`
185
+ };
186
+ });
187
+ var judgeVerdictSchema = z.object({
188
+ /** Normalized judge score in [0, 1]. */
189
+ value: z.number().min(0).max(1),
190
+ /**
191
+ * Explicit pass/fail from the judge. When present it WINS over the
192
+ * threshold comparison — see {@link makeLlmJudgeDriver}.
193
+ */
194
+ passed: z.boolean().optional(),
195
+ /** Human-readable explanation of the verdict. */
196
+ rationale: z.string().optional()
197
+ });
198
+ var llmJudgeTool = defineTool({
199
+ id: "eval.llm-judge",
200
+ description: "Model-backed scorer: hands `output` (plus free-form `criteria` and an optional `expected` reference) to an injected judge and normalizes the judge's verdict into the shared Score shape. The judge itself lives in the driver (see makeLlmJudgeDriver) \u2014 this tool contract carries no model call of its own.",
201
+ version: "0.1.0",
202
+ inputSchema: z.object({
203
+ output: jsonValueSchema.describe("The produced value to judge."),
204
+ criteria: z.string().describe("Free-form grading criteria/rubric for the judge."),
205
+ expected: jsonValueSchema.optional().describe("Optional reference value the judge may compare against.")
206
+ }),
207
+ outputSchema: scoreSchema,
208
+ mutates: [],
209
+ approval: "auto",
210
+ riskLevel: 0
211
+ });
212
+ function clamp01(x) {
213
+ return Math.min(1, Math.max(0, x));
214
+ }
215
+ function makeLlmJudgeDriver(judge, opts) {
216
+ const threshold = opts?.threshold ?? 0.5;
217
+ return defineDriver({
218
+ id: "eval-llm-judge",
219
+ name: "Eval LLM Judge (model-backed)",
220
+ description: "Model-backed scorer driver: implements eval.llm-judge by awaiting an injected JudgeFn and mapping its verdict to the shared Score shape. The judge is supplied by the caller \u2014 no LLM SDK or network call here.",
221
+ version: "0.1.0",
222
+ kind: "builtin",
223
+ implements: [{ tool: "eval.llm-judge", version: "0.1.0" }],
224
+ implementations: [
225
+ implementTool(llmJudgeTool, async ({ input }) => {
226
+ const verdict = await judge({
227
+ output: input.output,
228
+ criteria: input.criteria,
229
+ expected: input.expected
230
+ });
231
+ const value = clamp01(verdict.value);
232
+ const passed = verdict.passed ?? value >= threshold;
233
+ return {
234
+ value,
235
+ passed,
236
+ label: "llm-judge",
237
+ ...verdict.rationale ? { rationale: verdict.rationale } : {}
238
+ };
239
+ })
240
+ ]
241
+ });
242
+ }
243
+ function llmJudge(binding) {
244
+ const driver = makeLlmJudgeDriver(binding.judge, { threshold: binding.threshold });
245
+ return {
246
+ id: binding.id,
247
+ tool: llmJudgeTool,
248
+ driver,
249
+ mapInput: (ctx) => ({
250
+ output: binding.mapOutput({ output: ctx.output, expected: ctx.expected }),
251
+ criteria: binding.criteria,
252
+ expected: ctx.expected
253
+ })
254
+ };
255
+ }
256
+
257
+ // src/events.ts
258
+ var EVAL_EVENT_SCHEMA = "agentproto/eval/v1";
259
+ function bindScorer(binding) {
260
+ return {
261
+ id: binding.id,
262
+ toStep(stepId, output, expected) {
263
+ return {
264
+ kind: "tool",
265
+ id: stepId,
266
+ tool: binding.tool,
267
+ candidates: [binding.driver],
268
+ input: () => binding.mapInput({ output, expected })
269
+ };
270
+ }
271
+ };
272
+ }
273
+ var runCounter = 0;
274
+ function defaultRunId(suiteId) {
275
+ runCounter += 1;
276
+ return `${suiteId}#${runCounter}`;
277
+ }
278
+ function asScore(value) {
279
+ if (typeof value === "object" && value !== null && "value" in value && "passed" in value && "label" in value) {
280
+ const v = value.value;
281
+ const passed = value.passed;
282
+ const label = value.label;
283
+ const rationale = "rationale" in value ? value.rationale : void 0;
284
+ if (typeof v === "number" && typeof passed === "boolean" && typeof label === "string" && (rationale === void 0 || typeof rationale === "string")) {
285
+ return { value: v, passed, label, rationale };
286
+ }
287
+ }
288
+ throw new Error("scorer step did not produce a valid Score");
289
+ }
290
+ async function runEval(suite, opts) {
291
+ const telemetry = opts.telemetry;
292
+ const runId = opts.runId ?? defaultRunId(suite.id);
293
+ const start = Date.now();
294
+ telemetry?.emit({
295
+ kind: "eval.started",
296
+ runId,
297
+ at: (/* @__PURE__ */ new Date()).toISOString(),
298
+ suiteId: suite.id,
299
+ caseCount: suite.cases.length,
300
+ scorerCount: suite.scorers.length
301
+ });
302
+ const caseReports = [];
303
+ let valueSum = 0;
304
+ let valueCount = 0;
305
+ for (const evalCase of suite.cases) {
306
+ telemetry?.emit({
307
+ kind: "eval.case.started",
308
+ runId,
309
+ at: (/* @__PURE__ */ new Date()).toISOString(),
310
+ caseId: evalCase.id
311
+ });
312
+ const aggregate = await runCase(suite, evalCase, opts.target);
313
+ for (const cs of aggregate.scores) {
314
+ valueSum += cs.score.value;
315
+ valueCount += 1;
316
+ telemetry?.emit({
317
+ kind: "eval.case.scored",
318
+ runId,
319
+ at: (/* @__PURE__ */ new Date()).toISOString(),
320
+ caseId: evalCase.id,
321
+ scorerId: cs.scorerId,
322
+ value: cs.score.value,
323
+ passed: cs.score.passed
324
+ });
325
+ }
326
+ telemetry?.emit({
327
+ kind: "eval.case.finished",
328
+ runId,
329
+ at: (/* @__PURE__ */ new Date()).toISOString(),
330
+ caseId: evalCase.id,
331
+ passed: aggregate.passed
332
+ });
333
+ caseReports.push({
334
+ caseId: evalCase.id,
335
+ passed: aggregate.passed,
336
+ scores: aggregate.scores
337
+ });
338
+ }
339
+ const passedCount = caseReports.filter((c) => c.passed).length;
340
+ const meanValue = valueCount === 0 ? 0 : valueSum / valueCount;
341
+ const report = {
342
+ runId,
343
+ suiteId: suite.id,
344
+ total: suite.cases.length,
345
+ passedCount,
346
+ meanValue,
347
+ cases: caseReports
348
+ };
349
+ telemetry?.emit({
350
+ kind: "eval.finished",
351
+ runId,
352
+ at: (/* @__PURE__ */ new Date()).toISOString(),
353
+ suiteId: suite.id,
354
+ total: report.total,
355
+ passedCount,
356
+ meanValue,
357
+ durationMs: Date.now() - start
358
+ });
359
+ return report;
360
+ }
361
+ async function runCase(suite, evalCase, target) {
362
+ const output = await target(evalCase.input);
363
+ const scorerStepIds = suite.scorers.map((_, i) => `score_${i}`);
364
+ const workflow = {
365
+ id: `${suite.id}:${evalCase.id}`,
366
+ steps: [
367
+ ...suite.scorers.map(
368
+ (scorer, i) => scorer.toStep(scorerStepIds[i], output, evalCase.expected)
369
+ ),
370
+ {
371
+ kind: "transform",
372
+ id: "aggregate",
373
+ compute: (b) => {
374
+ const scores = suite.scorers.map((scorer, i) => ({
375
+ scorerId: scorer.id,
376
+ score: asScore(b.steps[scorerStepIds[i]])
377
+ }));
378
+ return { scores, passed: scores.every((s) => s.score.passed) };
379
+ }
380
+ }
381
+ ],
382
+ output: (b) => b.steps.aggregate
383
+ };
384
+ const { output: runOutput } = await runWorkflow({ workflow });
385
+ return toCaseAggregate(runOutput);
386
+ }
387
+ function toCaseAggregate(value) {
388
+ if (typeof value === "object" && value !== null && "scores" in value && "passed" in value && Array.isArray(value.scores) && typeof value.passed === "boolean") {
389
+ const scores = value.scores.map((entry) => {
390
+ if (typeof entry === "object" && entry !== null && "scorerId" in entry && "score" in entry && typeof entry.scorerId === "string") {
391
+ return { scorerId: entry.scorerId, score: asScore(entry.score) };
392
+ }
393
+ throw new Error("invalid CaseScore in aggregate");
394
+ });
395
+ return { scores, passed: value.passed };
396
+ }
397
+ throw new Error("workflow did not produce a CaseAggregate");
398
+ }
399
+
400
+ // src/to-vitest.ts
401
+ function toVitest(suite, opts, hooks) {
402
+ const { describe, it, expect } = hooks;
403
+ describe(suite.id, () => {
404
+ for (const evalCase of suite.cases) {
405
+ it(evalCase.id, async () => {
406
+ const singleCaseSuite = {
407
+ id: `${suite.id}:${evalCase.id}`,
408
+ cases: [evalCase],
409
+ scorers: suite.scorers
410
+ };
411
+ const report = await runEval(singleCaseSuite, opts);
412
+ const caseReport = report.cases[0];
413
+ expect(caseReport?.passed).toBe(true);
414
+ });
415
+ }
416
+ });
417
+ }
418
+
419
+ // src/index.ts
420
+ var SPEC_NAME = "agenteval/v1";
421
+ var SPEC_VERSION = "1.0.0-alpha";
422
+ var evalScorersProvider = defineDriver({
423
+ id: "eval-scorers",
424
+ name: "Eval Reference Scorers (built-in)",
425
+ description: "In-process deterministic scorers: exact-match, regex-match, json-schema-valid (minimal structural check), and latency-budget. Each is an AIP-14 TOOL whose output is the shared Score shape.",
426
+ version: "0.1.0",
427
+ kind: "builtin",
428
+ implements: [
429
+ { tool: "eval.exact-match", version: "0.1.0" },
430
+ { tool: "eval.regex-match", version: "0.1.0" },
431
+ { tool: "eval.json-schema-valid", version: "0.1.0" },
432
+ { tool: "eval.latency-budget", version: "0.1.0" }
433
+ ],
434
+ implementations: [
435
+ exactMatchImpl,
436
+ regexMatchImpl,
437
+ jsonSchemaValidImpl,
438
+ latencyBudgetImpl
439
+ ]
440
+ });
441
+
442
+ export { EVAL_EVENT_SCHEMA, SPEC_NAME, SPEC_VERSION, bindScorer, evalScorersProvider, exactMatchImpl, exactMatchTool, jsonSchemaValidImpl, jsonSchemaValidTool, jsonValueSchema, judgeVerdictSchema, latencyBudgetImpl, latencyBudgetTool, llmJudge, llmJudgeTool, makeLlmJudgeDriver, regexMatchImpl, regexMatchTool, runEval, scoreSchema, toVitest };
443
+ //# sourceMappingURL=index.mjs.map
444
+ //# sourceMappingURL=index.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/score.ts","../src/json.ts","../src/scorers.ts","../src/judge.ts","../src/events.ts","../src/run-eval.ts","../src/to-vitest.ts","../src/index.ts"],"names":["z","defineTool","implementTool","defineDriver"],"mappings":";;;;;;;;;;AAUO,IAAM,WAAA,GAAc,EAAE,MAAA,CAAO;AAAA;AAAA,EAElC,KAAA,EAAO,EAAE,MAAA,EAAO,CAAE,IAAI,CAAC,CAAA,CAAE,IAAI,CAAC,CAAA;AAAA;AAAA,EAE9B,MAAA,EAAQ,EAAE,OAAA,EAAQ;AAAA;AAAA,EAElB,KAAA,EAAO,EAAE,MAAA,EAAO;AAAA;AAAA,EAEhB,SAAA,EAAW,CAAA,CAAE,MAAA,EAAO,CAAE,QAAA;AACxB,CAAC;ACEM,IAAM,kBAAwCA,CAAAA,CAAE,IAAA;AAAA,EAAK,MAC1DA,EAAE,KAAA,CAAM;AAAA,IACNA,EAAE,MAAA,EAAO;AAAA,IACTA,EAAE,MAAA,EAAO;AAAA,IACTA,EAAE,OAAA,EAAQ;AAAA,IACVA,EAAE,IAAA,EAAK;AAAA,IACPA,CAAAA,CAAE,MAAM,eAAe,CAAA;AAAA,IACvBA,CAAAA,CAAE,MAAA,CAAOA,CAAAA,CAAE,MAAA,IAAU,eAAe;AAAA,GACrC;AACH;;;AClBO,IAAM,iBAAiB,UAAA,CAAW;AAAA,EACvC,EAAA,EAAI,kBAAA;AAAA,EACJ,WAAA,EACE,6HAAA;AAAA,EAEF,OAAA,EAAS,OAAA;AAAA,EACT,WAAA,EAAaA,EAAE,MAAA,CAAO;AAAA,IACpB,MAAA,EAAQA,CAAAA,CAAE,MAAA,EAAO,CAAE,SAAS,+BAA+B,CAAA;AAAA,IAC3D,QAAA,EAAUA,CAAAA,CAAE,MAAA,EAAO,CAAE,SAAS,0CAA0C,CAAA;AAAA,IACxE,MAAMA,CAAAA,CACH,OAAA,GACA,QAAA,EAAS,CACT,SAAS,kEAAkE;AAAA,GAC/E,CAAA;AAAA,EACD,YAAA,EAAc,WAAA;AAAA,EACd,SAAS,EAAC;AAAA,EACV,QAAA,EAAU,MAAA;AAAA,EACV,SAAA,EAAW;AACb,CAAC;AAEM,IAAM,iBAAiB,aAAA,CAAc,cAAA,EAAgB,CAAC,EAAE,OAAM,KAAM;AACzE,EAAA,MAAM,IAAI,KAAA,CAAM,IAAA,GAAO,MAAM,MAAA,CAAO,IAAA,KAAS,KAAA,CAAM,MAAA;AACnD,EAAA,MAAM,IAAI,KAAA,CAAM,IAAA,GAAO,MAAM,QAAA,CAAS,IAAA,KAAS,KAAA,CAAM,QAAA;AACrD,EAAA,MAAM,QAAQ,CAAA,KAAM,CAAA;AACpB,EAAA,OAAO;AAAA,IACL,KAAA,EAAO,QAAQ,CAAA,GAAI,CAAA;AAAA,IACnB,MAAA,EAAQ,KAAA;AAAA,IACR,KAAA,EAAO,aAAA;AAAA,IACP,SAAA,EAAW,QACP,wBAAA,GACA;AAAA,GACN;AACF,CAAC;AAMM,IAAM,iBAAiB,UAAA,CAAW;AAAA,EACvC,EAAA,EAAI,kBAAA;AAAA,EACJ,WAAA,EACE,kLAAA;AAAA,EAGF,OAAA,EAAS,OAAA;AAAA,EACT,WAAA,EAAaA,EAAE,MAAA,CAAO;AAAA,IACpB,MAAA,EAAQA,CAAAA,CAAE,MAAA,EAAO,CAAE,SAAS,8BAA8B,CAAA;AAAA,IAC1D,OAAA,EAASA,CAAAA,CAAE,MAAA,EAAO,CAAE,SAAS,0CAA0C,CAAA;AAAA,IACvE,OAAOA,CAAAA,CACJ,MAAA,GACA,QAAA,EAAS,CACT,SAAS,wCAAwC;AAAA,GACrD,CAAA;AAAA,EACD,YAAA,EAAc,WAAA;AAAA,EACd,SAAS,EAAC;AAAA,EACV,QAAA,EAAU,MAAA;AAAA,EACV,SAAA,EAAW;AACb,CAAC;AAEM,IAAM,iBAAiB,aAAA,CAAc,cAAA,EAAgB,CAAC,EAAE,OAAM,KAAM;AACzE,EAAA,IAAI,KAAA;AACJ,EAAA,IAAI;AACF,IAAA,KAAA,GAAQ,IAAI,MAAA,CAAO,KAAA,CAAM,OAAA,EAAS,MAAM,KAAK,CAAA;AAAA,EAC/C,SAAS,GAAA,EAAK;AACZ,IAAA,MAAM,SAAS,GAAA,YAAe,KAAA,GAAQ,GAAA,CAAI,OAAA,GAAU,OAAO,GAAG,CAAA;AAC9D,IAAA,OAAO;AAAA,MACL,KAAA,EAAO,CAAA;AAAA,MACP,MAAA,EAAQ,KAAA;AAAA,MACR,KAAA,EAAO,aAAA;AAAA,MACP,SAAA,EAAW,oBAAoB,MAAM,CAAA;AAAA,KACvC;AAAA,EACF;AACA,EAAA,MAAM,OAAA,GAAU,KAAA,CAAM,IAAA,CAAK,KAAA,CAAM,MAAM,CAAA;AACvC,EAAA,OAAO;AAAA,IACL,KAAA,EAAO,UAAU,CAAA,GAAI,CAAA;AAAA,IACrB,MAAA,EAAQ,OAAA;AAAA,IACR,KAAA,EAAO,aAAA;AAAA,IACP,WAAW,OAAA,GACP,CAAA,gBAAA,EAAmB,KAAA,CAAM,OAAO,IAAI,KAAA,CAAM,KAAA,IAAS,EAAE,CAAA,CAAA,GACrD,0BAA0B,KAAA,CAAM,OAAO,CAAA,CAAA,EAAI,KAAA,CAAM,SAAS,EAAE,CAAA;AAAA,GAClE;AACF,CAAC;AAyBD,IAAM,uBAAA,GAAwDA,EAAE,MAAA,CAAO;AAAA,EACrE,IAAA,EAAMA,CAAAA,CAAE,MAAA,EAAO,CAAE,QAAA,EAAS;AAAA,EAC1B,UAAUA,CAAAA,CAAE,KAAA,CAAMA,EAAE,MAAA,EAAQ,EAAE,QAAA,EAAS;AAAA,EACvC,YAAYA,CAAAA,CACT,MAAA,CAAOA,EAAE,MAAA,EAAO,EAAGA,EAAE,MAAA,CAAO,EAAE,IAAA,EAAMA,CAAAA,CAAE,QAAO,CAAE,QAAA,IAAY,CAAC,EAC5D,QAAA;AACL,CAAC,CAAA;AAGD,SAAS,aAAa,KAAA,EAAuC;AAC3D,EAAA,OAAO,OAAO,UAAU,QAAA,IAAY,KAAA,KAAU,QAAQ,CAAC,KAAA,CAAM,QAAQ,KAAK,CAAA;AAC5E;AAGA,SAAS,WAAW,KAAA,EAA0B;AAC5C,EAAA,IAAI,KAAA,KAAU,MAAM,OAAO,MAAA;AAC3B,EAAA,IAAI,KAAA,CAAM,OAAA,CAAQ,KAAK,CAAA,EAAG,OAAO,OAAA;AACjC,EAAA,IAAI,OAAO,KAAA,KAAU,QAAA,EAAU,OAAO,QAAA;AACtC,EAAA,IAAI,OAAO,KAAA,KAAU,SAAA,EAAW,OAAO,SAAA;AACvC,EAAA,IAAI,OAAO,KAAA,KAAU,QAAA,EAAU,OAAO,QAAA;AACtC,EAAA,OAAO,QAAA;AACT;AAEO,IAAM,sBAAsB,UAAA,CAAW;AAAA,EAC5C,EAAA,EAAI,wBAAA;AAAA,EACJ,WAAA,EACE,uNAAA;AAAA,EAGF,OAAA,EAAS,OAAA;AAAA,EACT,WAAA,EAAaA,EAAE,MAAA,CAAO;AAAA,IACpB,MAAA,EAAQ,eAAA,CAAgB,QAAA,CAAS,6BAA6B,CAAA;AAAA,IAC9D,QAAQ,uBAAA,CAAwB,QAAA;AAAA,MAC9B;AAAA;AACF,GACD,CAAA;AAAA,EACD,YAAA,EAAc,WAAA;AAAA,EACd,SAAS,EAAC;AAAA,EACV,QAAA,EAAU,MAAA;AAAA,EACV,SAAA,EAAW;AACb,CAAC;AAEM,IAAM,mBAAA,GAAsB,aAAA;AAAA,EACjC,mBAAA;AAAA,EACA,CAAC,EAAE,KAAA,EAAM,KAAM;AACb,IAAA,MAAM,EAAE,MAAA,EAAQ,MAAA,EAAO,GAAI,KAAA;AAC3B,IAAA,MAAM,SAAA,GAAY,oBAAA,CAAqB,MAAA,EAAQ,MAAM,CAAA;AACrD,IAAA,MAAM,QAAQ,SAAA,KAAc,MAAA;AAC5B,IAAA,OAAO;AAAA,MACL,KAAA,EAAO,QAAQ,CAAA,GAAI,CAAA;AAAA,MACnB,MAAA,EAAQ,KAAA;AAAA,MACR,KAAA,EAAO,mBAAA;AAAA,MACP,SAAA,EAAW,QAAQ,4BAAA,GAA+B;AAAA,KACpD;AAAA,EACF;AACF;AAGA,SAAS,oBAAA,CACP,QACA,MAAA,EACoB;AACpB,EAAA,MAAM,YAAA,GAAe,OAAO,IAAA,IAAQ,QAAA;AACpC,EAAA,MAAM,UAAA,GAAa,WAAW,MAAM,CAAA;AACpC,EAAA,IAAI,iBAAiB,UAAA,EAAY;AAC/B,IAAA,OAAO,CAAA,yBAAA,EAA4B,YAAY,CAAA,WAAA,EAAc,UAAU,CAAA,CAAA,CAAA;AAAA,EACzE;AAGA,EAAA,IAAI,CAAC,YAAA,CAAa,MAAM,CAAA,EAAG;AACzB,IAAA,OAAO,MAAA;AAAA,EACT;AAEA,EAAA,KAAA,MAAW,GAAA,IAAO,MAAA,CAAO,QAAA,IAAY,EAAC,EAAG;AACvC,IAAA,IAAI,EAAE,OAAO,MAAA,CAAA,EAAS;AACpB,MAAA,OAAO,8BAA8B,GAAG,CAAA,CAAA,CAAA;AAAA,IAC1C;AAAA,EACF;AAEA,EAAA,KAAA,MAAW,CAAC,GAAA,EAAK,UAAU,CAAA,IAAK,MAAA,CAAO,QAAQ,MAAA,CAAO,UAAA,IAAc,EAAE,CAAA,EAAG;AACvE,IAAA,IAAI,UAAA,CAAW,SAAS,MAAA,EAAW;AACnC,IAAA,MAAM,SAAA,GAAY,OAAO,GAAG,CAAA;AAC5B,IAAA,IAAI,cAAc,MAAA,EAAW;AAC7B,IAAA,MAAM,QAAA,GAAW,WAAW,SAAS,CAAA;AACrC,IAAA,IAAI,QAAA,KAAa,WAAW,IAAA,EAAM;AAChC,MAAA,OAAO,aAAa,GAAG,CAAA,iBAAA,EAAoB,UAAA,CAAW,IAAI,cAAc,QAAQ,CAAA,CAAA,CAAA;AAAA,IAClF;AAAA,EACF;AAEA,EAAA,OAAO,MAAA;AACT;AAMO,IAAM,oBAAoB,UAAA,CAAW;AAAA,EAC1C,EAAA,EAAI,qBAAA;AAAA,EACJ,WAAA,EACE,8KAAA;AAAA,EAGF,OAAA,EAAS,OAAA;AAAA,EACT,WAAA,EAAaA,EAAE,MAAA,CAAO;AAAA,IACpB,UAAA,EAAYA,EACT,MAAA,EAAO,CACP,IAAI,CAAC,CAAA,CACL,SAAS,oCAAoC,CAAA;AAAA,IAChD,UAAUA,CAAAA,CACP,MAAA,GACA,QAAA,EAAS,CACT,SAAS,+CAA+C;AAAA,GAC5D,CAAA;AAAA,EACD,YAAA,EAAc,WAAA;AAAA,EACd,SAAS,EAAC;AAAA,EACV,QAAA,EAAU,MAAA;AAAA,EACV,SAAA,EAAW;AACb,CAAC;AAEM,IAAM,oBAAoB,aAAA,CAAc,iBAAA,EAAmB,CAAC,EAAE,OAAM,KAAM;AAC/E,EAAA,MAAM,EAAE,UAAA,EAAY,QAAA,EAAS,GAAI,KAAA;AACjC,EAAA,MAAM,SAAS,UAAA,IAAc,QAAA;AAC7B,EAAA,MAAM,KAAA,GAAQ,SACV,CAAA,GACA,IAAA,CAAK,IAAI,CAAA,EAAG,CAAA,GAAA,CAAK,UAAA,GAAa,QAAA,IAAY,QAAQ,CAAA;AACtD,EAAA,OAAO;AAAA,IACL,KAAA;AAAA,IACA,MAAA;AAAA,IACA,KAAA,EAAO,gBAAA;AAAA,IACP,SAAA,EAAW,CAAA,WAAA,EAAc,UAAU,CAAA,WAAA,EAAc,QAAQ,CAAA;AAAA,GAC3D;AACF,CAAC;AC5NM,IAAM,kBAAA,GAAqBA,EAAE,MAAA,CAAO;AAAA;AAAA,EAEzC,KAAA,EAAOA,EAAE,MAAA,EAAO,CAAE,IAAI,CAAC,CAAA,CAAE,IAAI,CAAC,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAK9B,MAAA,EAAQA,CAAAA,CAAE,OAAA,EAAQ,CAAE,QAAA,EAAS;AAAA;AAAA,EAE7B,SAAA,EAAWA,CAAAA,CAAE,MAAA,EAAO,CAAE,QAAA;AACxB,CAAC;AAiCM,IAAM,eAAeC,UAAAA,CAAW;AAAA,EACrC,EAAA,EAAI,gBAAA;AAAA,EACJ,WAAA,EACE,qTAAA;AAAA,EAKF,OAAA,EAAS,OAAA;AAAA,EACT,WAAA,EAAaD,EAAE,MAAA,CAAO;AAAA,IACpB,MAAA,EAAQ,eAAA,CAAgB,QAAA,CAAS,8BAA8B,CAAA;AAAA,IAC/D,QAAA,EAAUA,CAAAA,CAAE,MAAA,EAAO,CAAE,SAAS,kDAAkD,CAAA;AAAA,IAChF,QAAA,EAAU,eAAA,CACP,QAAA,EAAS,CACT,SAAS,yDAAyD;AAAA,GACtE,CAAA;AAAA,EACD,YAAA,EAAc,WAAA;AAAA,EACd,SAAS,EAAC;AAAA,EACV,QAAA,EAAU,MAAA;AAAA,EACV,SAAA,EAAW;AACb,CAAC;AAOD,SAAS,QAAQ,CAAA,EAAmB;AAClC,EAAA,OAAO,KAAK,GAAA,CAAI,CAAA,EAAG,KAAK,GAAA,CAAI,CAAA,EAAG,CAAC,CAAC,CAAA;AACnC;AAaO,SAAS,kBAAA,CACd,OACA,IAAA,EACc;AACd,EAAA,MAAM,SAAA,GAAY,MAAM,SAAA,IAAa,GAAA;AACrC,EAAA,OAAO,YAAA,CAAa;AAAA,IAClB,EAAA,EAAI,gBAAA;AAAA,IACJ,IAAA,EAAM,+BAAA;AAAA,IACN,WAAA,EACE,sNAAA;AAAA,IAGF,OAAA,EAAS,OAAA;AAAA,IACT,IAAA,EAAM,SAAA;AAAA,IACN,YAAY,CAAC,EAAE,MAAM,gBAAA,EAAkB,OAAA,EAAS,SAAS,CAAA;AAAA,IACzD,eAAA,EAAiB;AAAA,MACfE,aAAAA,CAAc,YAAA,EAAc,OAAO,EAAE,OAAM,KAAM;AAC/C,QAAA,MAAM,OAAA,GAAU,MAAM,KAAA,CAAM;AAAA,UAC1B,QAAQ,KAAA,CAAM,MAAA;AAAA,UACd,UAAU,KAAA,CAAM,QAAA;AAAA,UAChB,UAAU,KAAA,CAAM;AAAA,SACjB,CAAA;AACD,QAAA,MAAM,KAAA,GAAQ,OAAA,CAAQ,OAAA,CAAQ,KAAK,CAAA;AACnC,QAAA,MAAM,MAAA,GAAS,OAAA,CAAQ,MAAA,IAAU,KAAA,IAAS,SAAA;AAC1C,QAAA,OAAO;AAAA,UACL,KAAA;AAAA,UACA,MAAA;AAAA,UACA,KAAA,EAAO,WAAA;AAAA,UACP,GAAI,QAAQ,SAAA,GAAY,EAAE,WAAW,OAAA,CAAQ,SAAA,KAAc;AAAC,SAC9D;AAAA,MACF,CAAC;AAAA;AACH,GACD,CAAA;AACH;AAwBO,SAAS,SAAY,OAAA,EAA8D;AACxF,EAAA,MAAM,MAAA,GAAS,mBAAmB,OAAA,CAAQ,KAAA,EAAO,EAAE,SAAA,EAAW,OAAA,CAAQ,WAAW,CAAA;AACjF,EAAA,OAAO;AAAA,IACL,IAAI,OAAA,CAAQ,EAAA;AAAA,IACZ,IAAA,EAAM,YAAA;AAAA,IACN,MAAA;AAAA,IACA,QAAA,EAAU,CAAC,GAAA,MAAS;AAAA,MAClB,MAAA,EAAQ,OAAA,CAAQ,SAAA,CAAU,EAAE,MAAA,EAAQ,IAAI,MAAA,EAAQ,QAAA,EAAU,GAAA,CAAI,QAAA,EAAU,CAAA;AAAA,MACxE,UAAU,OAAA,CAAQ,QAAA;AAAA,MAClB,UAAU,GAAA,CAAI;AAAA,KAChB;AAAA,GACF;AACF;;;ACxKO,IAAM,iBAAA,GAAoB;ACyD1B,SAAS,WAAiB,OAAA,EAA8C;AAC7E,EAAA,OAAO;AAAA,IACL,IAAI,OAAA,CAAQ,EAAA;AAAA,IACZ,MAAA,CAAO,MAAA,EAAQ,MAAA,EAAQ,QAAA,EAAU;AAC/B,MAAA,OAAO;AAAA,QACL,IAAA,EAAM,MAAA;AAAA,QACN,EAAA,EAAI,MAAA;AAAA,QACJ,MAAM,OAAA,CAAQ,IAAA;AAAA,QACd,UAAA,EAAY,CAAC,OAAA,CAAQ,MAAM,CAAA;AAAA,QAC3B,OAAO,MAAM,OAAA,CAAQ,SAAS,EAAE,MAAA,EAAQ,UAAU;AAAA,OACpD;AAAA,IACF;AAAA,GACF;AACF;AA8CA,IAAI,UAAA,GAAa,CAAA;AAEjB,SAAS,aAAa,OAAA,EAAyB;AAC7C,EAAA,UAAA,IAAc,CAAA;AACd,EAAA,OAAO,CAAA,EAAG,OAAO,CAAA,CAAA,EAAI,UAAU,CAAA,CAAA;AACjC;AASA,SAAS,QAAQ,KAAA,EAAuB;AACtC,EAAA,IACE,OAAO,KAAA,KAAU,QAAA,IACjB,KAAA,KAAU,IAAA,IACV,WAAW,KAAA,IACX,QAAA,IAAY,KAAA,IACZ,OAAA,IAAW,KAAA,EACX;AACA,IAAA,MAAM,IAAa,KAAA,CAAM,KAAA;AACzB,IAAA,MAAM,SAAkB,KAAA,CAAM,MAAA;AAC9B,IAAA,MAAM,QAAiB,KAAA,CAAM,KAAA;AAC7B,IAAA,MAAM,SAAA,GAAqB,WAAA,IAAe,KAAA,GAAQ,KAAA,CAAM,SAAA,GAAY,MAAA;AACpE,IAAA,IACE,OAAO,CAAA,KAAM,QAAA,IACb,OAAO,MAAA,KAAW,SAAA,IAClB,OAAO,KAAA,KAAU,QAAA,KAChB,SAAA,KAAc,MAAA,IAAa,OAAO,cAAc,QAAA,CAAA,EACjD;AACA,MAAA,OAAO,EAAE,KAAA,EAAO,CAAA,EAAG,MAAA,EAAQ,OAAO,SAAA,EAAU;AAAA,IAC9C;AAAA,EACF;AACA,EAAA,MAAM,IAAI,MAAM,2CAA2C,CAAA;AAC7D;AAMA,eAAsB,OAAA,CACpB,OACA,IAAA,EACqB;AACrB,EAAA,MAAM,YAAY,IAAA,CAAK,SAAA;AACvB,EAAA,MAAM,KAAA,GAAQ,IAAA,CAAK,KAAA,IAAS,YAAA,CAAa,MAAM,EAAE,CAAA;AACjD,EAAA,MAAM,KAAA,GAAQ,KAAK,GAAA,EAAI;AAEvB,EAAA,SAAA,EAAW,IAAA,CAAK;AAAA,IACd,IAAA,EAAM,cAAA;AAAA,IACN,KAAA;AAAA,IACA,EAAA,EAAA,iBAAI,IAAI,IAAA,EAAK,EAAE,WAAA,EAAY;AAAA,IAC3B,SAAS,KAAA,CAAM,EAAA;AAAA,IACf,SAAA,EAAW,MAAM,KAAA,CAAM,MAAA;AAAA,IACvB,WAAA,EAAa,MAAM,OAAA,CAAQ;AAAA,GAC5B,CAAA;AAED,EAAA,MAAM,cAA4B,EAAC;AACnC,EAAA,IAAI,QAAA,GAAW,CAAA;AACf,EAAA,IAAI,UAAA,GAAa,CAAA;AAEjB,EAAA,KAAA,MAAW,QAAA,IAAY,MAAM,KAAA,EAAO;AAClC,IAAA,SAAA,EAAW,IAAA,CAAK;AAAA,MACd,IAAA,EAAM,mBAAA;AAAA,MACN,KAAA;AAAA,MACA,EAAA,EAAA,iBAAI,IAAI,IAAA,EAAK,EAAE,WAAA,EAAY;AAAA,MAC3B,QAAQ,QAAA,CAAS;AAAA,KAClB,CAAA;AAED,IAAA,MAAM,YAAY,MAAM,OAAA,CAAQ,KAAA,EAAO,QAAA,EAAU,KAAK,MAAM,CAAA;AAE5D,IAAA,KAAA,MAAW,EAAA,IAAM,UAAU,MAAA,EAAQ;AACjC,MAAA,QAAA,IAAY,GAAG,KAAA,CAAM,KAAA;AACrB,MAAA,UAAA,IAAc,CAAA;AACd,MAAA,SAAA,EAAW,IAAA,CAAK;AAAA,QACd,IAAA,EAAM,kBAAA;AAAA,QACN,KAAA;AAAA,QACA,EAAA,EAAA,iBAAI,IAAI,IAAA,EAAK,EAAE,WAAA,EAAY;AAAA,QAC3B,QAAQ,QAAA,CAAS,EAAA;AAAA,QACjB,UAAU,EAAA,CAAG,QAAA;AAAA,QACb,KAAA,EAAO,GAAG,KAAA,CAAM,KAAA;AAAA,QAChB,MAAA,EAAQ,GAAG,KAAA,CAAM;AAAA,OAClB,CAAA;AAAA,IACH;AAEA,IAAA,SAAA,EAAW,IAAA,CAAK;AAAA,MACd,IAAA,EAAM,oBAAA;AAAA,MACN,KAAA;AAAA,MACA,EAAA,EAAA,iBAAI,IAAI,IAAA,EAAK,EAAE,WAAA,EAAY;AAAA,MAC3B,QAAQ,QAAA,CAAS,EAAA;AAAA,MACjB,QAAQ,SAAA,CAAU;AAAA,KACnB,CAAA;AAED,IAAA,WAAA,CAAY,IAAA,CAAK;AAAA,MACf,QAAQ,QAAA,CAAS,EAAA;AAAA,MACjB,QAAQ,SAAA,CAAU,MAAA;AAAA,MAClB,QAAQ,SAAA,CAAU;AAAA,KACnB,CAAA;AAAA,EACH;AAEA,EAAA,MAAM,cAAc,WAAA,CAAY,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,MAAM,CAAA,CAAE,MAAA;AACxD,EAAA,MAAM,SAAA,GAAY,UAAA,KAAe,CAAA,GAAI,CAAA,GAAI,QAAA,GAAW,UAAA;AAEpD,EAAA,MAAM,MAAA,GAAqB;AAAA,IACzB,KAAA;AAAA,IACA,SAAS,KAAA,CAAM,EAAA;AAAA,IACf,KAAA,EAAO,MAAM,KAAA,CAAM,MAAA;AAAA,IACnB,WAAA;AAAA,IACA,SAAA;AAAA,IACA,KAAA,EAAO;AAAA,GACT;AAEA,EAAA,SAAA,EAAW,IAAA,CAAK;AAAA,IACd,IAAA,EAAM,eAAA;AAAA,IACN,KAAA;AAAA,IACA,EAAA,EAAA,iBAAI,IAAI,IAAA,EAAK,EAAE,WAAA,EAAY;AAAA,IAC3B,SAAS,KAAA,CAAM,EAAA;AAAA,IACf,OAAO,MAAA,CAAO,KAAA;AAAA,IACd,WAAA;AAAA,IACA,SAAA;AAAA,IACA,UAAA,EAAY,IAAA,CAAK,GAAA,EAAI,GAAI;AAAA,GAC1B,CAAA;AAED,EAAA,OAAO,MAAA;AACT;AAMA,eAAe,OAAA,CACb,KAAA,EACA,QAAA,EACA,MAAA,EACwB;AACxB,EAAA,MAAM,MAAA,GAAS,MAAM,MAAA,CAAO,QAAA,CAAS,KAAK,CAAA;AAC1C,EAAA,MAAM,aAAA,GAAgB,MAAM,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,MAAA,EAAS,CAAC,CAAA,CAAE,CAAA;AAE9D,EAAA,MAAM,QAAA,GAA4B;AAAA,IAChC,IAAI,CAAA,EAAG,KAAA,CAAM,EAAE,CAAA,CAAA,EAAI,SAAS,EAAE,CAAA,CAAA;AAAA,IAC9B,KAAA,EAAO;AAAA,MACL,GAAG,MAAM,OAAA,CAAQ,GAAA;AAAA,QAAI,CAAC,MAAA,EAAQ,CAAA,KAC5B,MAAA,CAAO,MAAA,CAAO,cAAc,CAAC,CAAA,EAAI,MAAA,EAAQ,QAAA,CAAS,QAAQ;AAAA,OAC5D;AAAA,MACA;AAAA,QACE,IAAA,EAAM,WAAA;AAAA,QACN,EAAA,EAAI,WAAA;AAAA,QACJ,OAAA,EAAS,CAAC,CAAA,KAAqB;AAC7B,UAAA,MAAM,SAAsB,KAAA,CAAM,OAAA,CAAQ,GAAA,CAAI,CAAC,QAAQ,CAAA,MAAO;AAAA,YAC5D,UAAU,MAAA,CAAO,EAAA;AAAA,YACjB,OAAO,OAAA,CAAQ,CAAA,CAAE,MAAM,aAAA,CAAc,CAAC,CAAE,CAAC;AAAA,WAC3C,CAAE,CAAA;AACF,UAAA,OAAO,EAAE,MAAA,EAAQ,MAAA,EAAQ,MAAA,CAAO,KAAA,CAAM,CAAC,CAAA,KAAM,CAAA,CAAE,KAAA,CAAM,MAAM,CAAA,EAAE;AAAA,QAC/D;AAAA;AACF,KACF;AAAA,IACA,MAAA,EAAQ,CAAC,CAAA,KAAM,CAAA,CAAE,KAAA,CAAM;AAAA,GACzB;AAEA,EAAA,MAAM,EAAE,QAAQ,SAAA,EAAU,GAAI,MAAM,WAAA,CAAY,EAAE,UAAU,CAAA;AAC5D,EAAA,OAAO,gBAAgB,SAAS,CAAA;AAClC;AAGA,SAAS,gBAAgB,KAAA,EAA+B;AACtD,EAAA,IACE,OAAO,KAAA,KAAU,QAAA,IACjB,KAAA,KAAU,IAAA,IACV,YAAY,KAAA,IACZ,QAAA,IAAY,KAAA,IACZ,KAAA,CAAM,QAAQ,KAAA,CAAM,MAAM,KAC1B,OAAO,KAAA,CAAM,WAAW,SAAA,EACxB;AACA,IAAA,MAAM,MAAA,GAAS,KAAA,CAAM,MAAA,CAAO,GAAA,CAAI,CAAC,KAAA,KAAmB;AAClD,MAAA,IACE,OAAO,KAAA,KAAU,QAAA,IACjB,KAAA,KAAU,IAAA,IACV,UAAA,IAAc,KAAA,IACd,OAAA,IAAW,KAAA,IACX,OAAO,KAAA,CAAM,QAAA,KAAa,QAAA,EAC1B;AACA,QAAA,OAAO,EAAE,UAAU,KAAA,CAAM,QAAA,EAAU,OAAO,OAAA,CAAQ,KAAA,CAAM,KAAK,CAAA,EAAE;AAAA,MACjE;AACA,MAAA,MAAM,IAAI,MAAM,gCAAgC,CAAA;AAAA,IAClD,CAAC,CAAA;AACD,IAAA,OAAO,EAAE,MAAA,EAAQ,MAAA,EAAQ,KAAA,CAAM,MAAA,EAAO;AAAA,EACxC;AACA,EAAA,MAAM,IAAI,MAAM,0CAA0C,CAAA;AAC5D;;;AC/RO,SAAS,QAAA,CACd,KAAA,EACA,IAAA,EACA,KAAA,EACM;AACN,EAAA,MAAM,EAAE,QAAA,EAAU,EAAA,EAAI,MAAA,EAAO,GAAI,KAAA;AACjC,EAAA,QAAA,CAAS,KAAA,CAAM,IAAI,MAAM;AACvB,IAAA,KAAA,MAAW,QAAA,IAAY,MAAM,KAAA,EAAO;AAClC,MAAA,EAAA,CAAG,QAAA,CAAS,IAAI,YAAY;AAC1B,QAAA,MAAM,eAAA,GAAwC;AAAA,UAC5C,IAAI,CAAA,EAAG,KAAA,CAAM,EAAE,CAAA,CAAA,EAAI,SAAS,EAAE,CAAA,CAAA;AAAA,UAC9B,KAAA,EAAO,CAAC,QAAQ,CAAA;AAAA,UAChB,SAAS,KAAA,CAAM;AAAA,SACjB;AACA,QAAA,MAAM,MAAA,GAAS,MAAM,OAAA,CAAQ,eAAA,EAAiB,IAAI,CAAA;AAClD,QAAA,MAAM,UAAA,GAAa,MAAA,CAAO,KAAA,CAAM,CAAC,CAAA;AACjC,QAAA,MAAA,CAAO,UAAA,EAAY,MAAM,CAAA,CAAE,IAAA,CAAK,IAAI,CAAA;AAAA,MACtC,CAAC,CAAA;AAAA,IACH;AAAA,EACF,CAAC,CAAA;AACH;;;ACbO,IAAM,SAAA,GAAY;AAClB,IAAM,YAAA,GAAe;AA8DrB,IAAM,sBAAsBC,YAAAA,CAAa;AAAA,EAC9C,EAAA,EAAI,cAAA;AAAA,EACJ,IAAA,EAAM,mCAAA;AAAA,EACN,WAAA,EACE,8LAAA;AAAA,EAGF,OAAA,EAAS,OAAA;AAAA,EACT,IAAA,EAAM,SAAA;AAAA,EACN,UAAA,EAAY;AAAA,IACV,EAAE,IAAA,EAAM,kBAAA,EAAoB,OAAA,EAAS,OAAA,EAAQ;AAAA,IAC7C,EAAE,IAAA,EAAM,kBAAA,EAAoB,OAAA,EAAS,OAAA,EAAQ;AAAA,IAC7C,EAAE,IAAA,EAAM,wBAAA,EAA0B,OAAA,EAAS,OAAA,EAAQ;AAAA,IACnD,EAAE,IAAA,EAAM,qBAAA,EAAuB,OAAA,EAAS,OAAA;AAAQ,GAClD;AAAA,EACA,eAAA,EAAiB;AAAA,IACf,cAAA;AAAA,IACA,cAAA;AAAA,IACA,mBAAA;AAAA,IACA;AAAA;AAEJ,CAAC","file":"index.mjs","sourcesContent":["import { z } from \"zod\"\n\n/**\n * The shared output shape every scorer produces.\n *\n * A scorer is an AIP-14 TOOL whose `outputSchema` is exactly this — there is\n * no separate \"scorer port\". Reusing `Score` as the tool output means the\n * registry, retries, and `toMastraTool` / `toAiSdkTool` projections all apply\n * to scorers for free, identical to any other builtin tool.\n */\nexport const scoreSchema = z.object({\n /** Normalized score in [0, 1]. */\n value: z.number().min(0).max(1),\n /** Whether the score clears the scorer's own threshold. */\n passed: z.boolean(),\n /** Scorer id, e.g. \"exact-match\". */\n label: z.string(),\n /** Human-readable explanation of the outcome. */\n rationale: z.string().optional(),\n})\n\nexport type Score = z.infer<typeof scoreSchema>\n","import { z } from \"zod\"\n\n/**\n * Recursive JSON value union. Modelling the JSON payload explicitly (rather\n * than reaching for `any` / `unknown`) lets consumers inspect values with\n * proper narrowing and no casts. This is the ONE canonical `JsonValue` for the\n * package — scorers, the eval harness, and the vitest bridge all import it from\n * here so there is no divergent redeclaration.\n */\nexport type JsonValue =\n | string\n | number\n | boolean\n | null\n | JsonValue[]\n | { [key: string]: JsonValue }\n\n/** A plain JSON object (not `null`, not an array). */\nexport type JsonObject = { [key: string]: JsonValue }\n\n/** Zod schema mirroring {@link JsonValue}. */\nexport const jsonValueSchema: z.ZodType<JsonValue> = z.lazy(() =>\n z.union([\n z.string(),\n z.number(),\n z.boolean(),\n z.null(),\n z.array(jsonValueSchema),\n z.record(z.string(), jsonValueSchema),\n ]),\n)\n","import { z } from \"zod\"\nimport { defineTool } from \"@agentproto/tool\"\nimport { implementTool } from \"@agentproto/driver\"\nimport { scoreSchema } from \"./score.js\"\nimport { jsonValueSchema, type JsonObject, type JsonValue } from \"./json.js\"\n\nexport type { JsonValue } from \"./json.js\"\n\n// ---------------------------------------------------------------------------\n// eval.exact-match\n// ---------------------------------------------------------------------------\n\nexport const exactMatchTool = defineTool({\n id: \"eval.exact-match\",\n description:\n \"Deterministic scorer: value 1 when `actual` equals `expected` \" +\n \"(optionally after trimming whitespace on both sides), else 0.\",\n version: \"0.1.0\",\n inputSchema: z.object({\n actual: z.string().describe(\"The produced string to score.\"),\n expected: z.string().describe(\"The reference string to compare against.\"),\n trim: z\n .boolean()\n .optional()\n .describe(\"Trim leading/trailing whitespace on both sides before comparing.\"),\n }),\n outputSchema: scoreSchema,\n mutates: [],\n approval: \"auto\",\n riskLevel: 0,\n})\n\nexport const exactMatchImpl = implementTool(exactMatchTool, ({ input }) => {\n const a = input.trim ? input.actual.trim() : input.actual\n const b = input.trim ? input.expected.trim() : input.expected\n const equal = a === b\n return {\n value: equal ? 1 : 0,\n passed: equal,\n label: \"exact-match\",\n rationale: equal\n ? \"actual equals expected\"\n : \"actual does not equal expected\",\n }\n})\n\n// ---------------------------------------------------------------------------\n// eval.regex-match\n// ---------------------------------------------------------------------------\n\nexport const regexMatchTool = defineTool({\n id: \"eval.regex-match\",\n description:\n \"Deterministic scorer: value 1 when `actual` matches the RegExp built \" +\n \"from `pattern`/`flags`, else 0. An invalid pattern yields a typed \" +\n \"failure Score rather than a thrown error.\",\n version: \"0.1.0\",\n inputSchema: z.object({\n actual: z.string().describe(\"The produced string to test.\"),\n pattern: z.string().describe(\"RegExp source, compiled inside the body.\"),\n flags: z\n .string()\n .optional()\n .describe(\"RegExp flags, e.g. 'i', 'm', 'gimsuy'.\"),\n }),\n outputSchema: scoreSchema,\n mutates: [],\n approval: \"auto\",\n riskLevel: 0,\n})\n\nexport const regexMatchImpl = implementTool(regexMatchTool, ({ input }) => {\n let regex: RegExp\n try {\n regex = new RegExp(input.pattern, input.flags)\n } catch (err) {\n const reason = err instanceof Error ? err.message : String(err)\n return {\n value: 0,\n passed: false,\n label: \"regex-match\",\n rationale: `invalid pattern: ${reason}`,\n }\n }\n const matched = regex.test(input.actual)\n return {\n value: matched ? 1 : 0,\n passed: matched,\n label: \"regex-match\",\n rationale: matched\n ? `actual matches /${input.pattern}/${input.flags ?? \"\"}`\n : `actual does not match /${input.pattern}/${input.flags ?? \"\"}`,\n }\n})\n\n// ---------------------------------------------------------------------------\n// eval.json-schema-valid\n// ---------------------------------------------------------------------------\n\n/**\n * A deliberately minimal, dependency-free JSON-schema shape.\n *\n * LIMITATION: this scorer checks ONLY the two most common structural\n * constraints on a top-level object — that every name in `required` is\n * present, and that any name in `properties` declaring a `type` has a value\n * of that primitive type. It does NOT recurse into nested schemas, and does\n * not implement the rest of JSON Schema (enum, format, oneOf, items, …). A\n * full validator (e.g. ajv) is a later step; keeping this package light is\n * the point.\n */\nexport interface MinimalJsonSchema {\n readonly type?: string\n readonly required?: readonly string[]\n readonly properties?: {\n readonly [key: string]: { readonly type?: string }\n }\n}\n\nconst minimalJsonSchemaSchema: z.ZodType<MinimalJsonSchema> = z.object({\n type: z.string().optional(),\n required: z.array(z.string()).optional(),\n properties: z\n .record(z.string(), z.object({ type: z.string().optional() }))\n .optional(),\n})\n\n/** Narrows a JSON value to a plain JSON object (not null, not an array). */\nfunction isJsonObject(value: JsonValue): value is JsonObject {\n return typeof value === \"object\" && value !== null && !Array.isArray(value)\n}\n\n/** The JSON Schema primitive-type name for a runtime JSON value. */\nfunction jsonTypeOf(value: JsonValue): string {\n if (value === null) return \"null\"\n if (Array.isArray(value)) return \"array\"\n if (typeof value === \"number\") return \"number\"\n if (typeof value === \"boolean\") return \"boolean\"\n if (typeof value === \"string\") return \"string\"\n return \"object\"\n}\n\nexport const jsonSchemaValidTool = defineTool({\n id: \"eval.json-schema-valid\",\n description:\n \"Deterministic scorer: minimal structural check of a JSON value against \" +\n \"a schema's top-level `required` and `properties` type constraints only. \" +\n \"Not a full JSON Schema validator (no recursion, enum, format, …).\",\n version: \"0.1.0\",\n inputSchema: z.object({\n actual: jsonValueSchema.describe(\"The JSON value to validate.\"),\n schema: minimalJsonSchemaSchema.describe(\n \"Minimal JSON schema: top-level `required` and `properties.type` only.\",\n ),\n }),\n outputSchema: scoreSchema,\n mutates: [],\n approval: \"auto\",\n riskLevel: 0,\n})\n\nexport const jsonSchemaValidImpl = implementTool(\n jsonSchemaValidTool,\n ({ input }) => {\n const { actual, schema } = input\n const violation = firstSchemaViolation(actual, schema)\n const valid = violation === undefined\n return {\n value: valid ? 1 : 0,\n passed: valid,\n label: \"json-schema-valid\",\n rationale: valid ? \"value satisfies the schema\" : violation,\n }\n },\n)\n\n/** Returns the first violation message, or undefined when the value is valid. */\nfunction firstSchemaViolation(\n actual: JsonValue,\n schema: MinimalJsonSchema,\n): string | undefined {\n const declaredType = schema.type ?? \"object\"\n const actualType = jsonTypeOf(actual)\n if (declaredType !== actualType) {\n return `expected top-level type '${declaredType}' but got '${actualType}'`\n }\n\n // Only object payloads carry required/properties semantics here.\n if (!isJsonObject(actual)) {\n return undefined\n }\n\n for (const key of schema.required ?? []) {\n if (!(key in actual)) {\n return `missing required property '${key}'`\n }\n }\n\n for (const [key, propSchema] of Object.entries(schema.properties ?? {})) {\n if (propSchema.type === undefined) continue\n const propValue = actual[key]\n if (propValue === undefined) continue\n const propType = jsonTypeOf(propValue)\n if (propType !== propSchema.type) {\n return `property '${key}' expected type '${propSchema.type}' but got '${propType}'`\n }\n }\n\n return undefined\n}\n\n// ---------------------------------------------------------------------------\n// eval.latency-budget\n// ---------------------------------------------------------------------------\n\nexport const latencyBudgetTool = defineTool({\n id: \"eval.latency-budget\",\n description:\n \"Deterministic scorer: passes when `durationMs` is within `budgetMs`. \" +\n \"value is 1 when within budget, else linearly decays toward 0 as the \" +\n \"overrun approaches one full budget.\",\n version: \"0.1.0\",\n inputSchema: z.object({\n durationMs: z\n .number()\n .min(0)\n .describe(\"Observed duration in milliseconds.\"),\n budgetMs: z\n .number()\n .positive()\n .describe(\"Allowed budget in milliseconds (must be > 0).\"),\n }),\n outputSchema: scoreSchema,\n mutates: [],\n approval: \"auto\",\n riskLevel: 0,\n})\n\nexport const latencyBudgetImpl = implementTool(latencyBudgetTool, ({ input }) => {\n const { durationMs, budgetMs } = input\n const passed = durationMs <= budgetMs\n const value = passed\n ? 1\n : Math.max(0, 1 - (durationMs - budgetMs) / budgetMs)\n return {\n value,\n passed,\n label: \"latency-budget\",\n rationale: `durationMs=${durationMs}, budgetMs=${budgetMs}`,\n }\n})\n","import { z } from \"zod\"\nimport { defineTool } from \"@agentproto/tool\"\nimport { defineDriver, implementTool, type DriverHandle } from \"@agentproto/driver\"\nimport { scoreSchema } from \"./score.js\"\nimport { jsonValueSchema, type JsonValue } from \"./json.js\"\nimport type { ScorerBinding } from \"./run-eval.js\"\n\n/**\n * `llmJudge` — a model-backed scorer.\n *\n * Design (locked): a model-backed scorer stays a TOOL, exactly like the\n * deterministic scorers in `scorers.ts`. What differs is where the model\n * lives: the injected {@link JudgeFn} is closed over by the DRIVER (built by\n * {@link makeLlmJudgeDriver}), never threaded through tool input/context. That\n * keeps `eval.llm-judge`'s contract identical in shape to every other scorer\n * tool, so it composes with the existing `bindScorer` / `runEval` in\n * run-eval.ts with ZERO changes there.\n *\n * The judge itself is vendor-neutral: {@link JudgeFn} is just an injected\n * async function. This file has no LLM SDK and no network dependency — a real\n * adapter (an agent session, a supervisor judge-gate, …) is a documented\n * follow-up, not built here.\n */\n\n// ---------------------------------------------------------------------------\n// JudgeVerdict — what an injected judge returns\n// ---------------------------------------------------------------------------\n\n/** Zod schema for the raw verdict a {@link JudgeFn} produces. */\nexport const judgeVerdictSchema = z.object({\n /** Normalized judge score in [0, 1]. */\n value: z.number().min(0).max(1),\n /**\n * Explicit pass/fail from the judge. When present it WINS over the\n * threshold comparison — see {@link makeLlmJudgeDriver}.\n */\n passed: z.boolean().optional(),\n /** Human-readable explanation of the verdict. */\n rationale: z.string().optional(),\n})\n\nexport type JudgeVerdict = z.infer<typeof judgeVerdictSchema>\n\n// ---------------------------------------------------------------------------\n// JudgeFn — the injected capability\n// ---------------------------------------------------------------------------\n\n/**\n * The seam a real judge satisfies: given the produced `output`, free-form\n * grading `criteria`, and an optional `expected` reference, return a\n * {@link JudgeVerdict}. Callers supply this — a real LLM call, an agent\n * session, or the supervisor's judge-gate all satisfy the same shape.\n * Deliberately no LLM SDK / network types here: keeping this package\n * vendor-neutral is the point.\n */\nexport type JudgeFn = (args: {\n readonly output: JsonValue\n readonly criteria: string\n readonly expected?: JsonValue\n}) => Promise<JudgeVerdict>\n\n// ---------------------------------------------------------------------------\n// eval.llm-judge — the TOOL contract\n// ---------------------------------------------------------------------------\n\n/** Tool input for `eval.llm-judge`. */\nexport interface LlmJudgeInput {\n readonly output: JsonValue\n readonly criteria: string\n readonly expected?: JsonValue\n}\n\nexport const llmJudgeTool = defineTool({\n id: \"eval.llm-judge\",\n description:\n \"Model-backed scorer: hands `output` (plus free-form `criteria` and an \" +\n \"optional `expected` reference) to an injected judge and normalizes the \" +\n \"judge's verdict into the shared Score shape. The judge itself lives in \" +\n \"the driver (see makeLlmJudgeDriver) — this tool contract carries no \" +\n \"model call of its own.\",\n version: \"0.1.0\",\n inputSchema: z.object({\n output: jsonValueSchema.describe(\"The produced value to judge.\"),\n criteria: z.string().describe(\"Free-form grading criteria/rubric for the judge.\"),\n expected: jsonValueSchema\n .optional()\n .describe(\"Optional reference value the judge may compare against.\"),\n }),\n outputSchema: scoreSchema,\n mutates: [],\n approval: \"auto\",\n riskLevel: 0,\n})\n\n// ---------------------------------------------------------------------------\n// makeLlmJudgeDriver — closes over the injected JudgeFn\n// ---------------------------------------------------------------------------\n\n/** Clamp a number into [0, 1]. */\nfunction clamp01(x: number): number {\n return Math.min(1, Math.max(0, x))\n}\n\nexport interface MakeLlmJudgeDriverOptions {\n /** Minimum `value` to count as passed when the judge omits `passed`. Default 0.5. */\n readonly threshold?: number\n}\n\n/**\n * Build a DRIVER that implements `eval.llm-judge` by delegating to `judge`.\n * This is the one seam where the injected model-backed capability enters the\n * system — everything downstream (`bindScorer`, `runEval`) stays unaware that\n * this scorer is model-backed at all.\n */\nexport function makeLlmJudgeDriver(\n judge: JudgeFn,\n opts?: MakeLlmJudgeDriverOptions,\n): DriverHandle {\n const threshold = opts?.threshold ?? 0.5\n return defineDriver({\n id: \"eval-llm-judge\",\n name: \"Eval LLM Judge (model-backed)\",\n description:\n \"Model-backed scorer driver: implements eval.llm-judge by awaiting an \" +\n \"injected JudgeFn and mapping its verdict to the shared Score shape. \" +\n \"The judge is supplied by the caller — no LLM SDK or network call here.\",\n version: \"0.1.0\",\n kind: \"builtin\",\n implements: [{ tool: \"eval.llm-judge\", version: \"0.1.0\" }],\n implementations: [\n implementTool(llmJudgeTool, async ({ input }) => {\n const verdict = await judge({\n output: input.output,\n criteria: input.criteria,\n expected: input.expected,\n })\n const value = clamp01(verdict.value)\n const passed = verdict.passed ?? value >= threshold\n return {\n value,\n passed,\n label: \"llm-judge\",\n ...(verdict.rationale ? { rationale: verdict.rationale } : {}),\n }\n }),\n ],\n })\n}\n\n// ---------------------------------------------------------------------------\n// llmJudge — convenience ScorerBinding factory\n// ---------------------------------------------------------------------------\n\nexport interface LlmJudgeBinding<A> {\n /** Label for this binding, e.g. \"helpfulness\". */\n readonly id: string\n /** The injected judge capability. */\n readonly judge: JudgeFn\n /** Free-form grading criteria/rubric handed to the judge on every call. */\n readonly criteria: string\n /** Minimum value to count as passed when the judge omits `passed`. Default 0.5. */\n readonly threshold?: number\n /** Derive the judge's `output` input from the target's output. */\n readonly mapOutput: (ctx: { output: A; expected?: JsonValue }) => JsonValue\n}\n\n/**\n * Convenience: build a {@link ScorerBinding} ready to drop into a suite's\n * `scorers` via `bindScorer` — wires up `eval.llm-judge`, a driver built with\n * {@link makeLlmJudgeDriver}, and the input mapping in one call.\n */\nexport function llmJudge<A>(binding: LlmJudgeBinding<A>): ScorerBinding<A, LlmJudgeInput> {\n const driver = makeLlmJudgeDriver(binding.judge, { threshold: binding.threshold })\n return {\n id: binding.id,\n tool: llmJudgeTool,\n driver,\n mapInput: (ctx) => ({\n output: binding.mapOutput({ output: ctx.output, expected: ctx.expected }),\n criteria: binding.criteria,\n expected: ctx.expected,\n }),\n }\n}\n","/**\n * Structured events emitted while an eval suite runs.\n *\n * The union mirrors the shape of `@agentproto/telemetry`'s `TelemetryEvent`:\n * a discriminated union keyed on `kind`, every member carrying a correlation\n * id (`runId`) and an ISO `at` timestamp so a sink can rebuild a per-run span\n * tree. It rides the same {@link Telemetry} port — a `runEval` call takes a\n * `Telemetry<EvalEvent>` sink.\n *\n * Sinks MUST tolerate unknown kinds — future versions may add event types\n * under the same `agentproto/eval/v1` schema (identical forward-compat\n * contract as the telemetry port).\n */\n\n/** Schema identifier for this event family. */\nexport const EVAL_EVENT_SCHEMA = \"agentproto/eval/v1\" as const\n\nexport type EvalEvent =\n | {\n readonly kind: \"eval.started\"\n readonly runId: string\n readonly at: string\n readonly suiteId: string\n readonly caseCount: number\n readonly scorerCount: number\n }\n | {\n readonly kind: \"eval.case.started\"\n readonly runId: string\n readonly at: string\n readonly caseId: string\n }\n | {\n readonly kind: \"eval.case.scored\"\n readonly runId: string\n readonly at: string\n readonly caseId: string\n readonly scorerId: string\n readonly value: number\n readonly passed: boolean\n }\n | {\n readonly kind: \"eval.case.finished\"\n readonly runId: string\n readonly at: string\n readonly caseId: string\n /** True when EVERY scorer bound to the case passed. */\n readonly passed: boolean\n }\n | {\n readonly kind: \"eval.finished\"\n readonly runId: string\n readonly at: string\n readonly suiteId: string\n readonly total: number\n readonly passedCount: number\n readonly meanValue: number\n readonly durationMs: number\n }\n","/**\n * `runEval` — the eval harness.\n *\n * The thesis this file proves: **an eval is a workflow that runs a target then\n * scores its output with scorer-tools, reporting through the telemetry port.**\n *\n * For each case we build a {@link RuntimeWorkflow} that composes\n * target → one scorer `tool` step per binding → a per-case aggregate transform\n * and execute it with `runWorkflow` (AIP-15). The target's structured output is\n * the input selector for every scorer step — that is where composable tool I/O\n * is demonstrated. `runEval` then aggregates across cases and emits\n * {@link EvalEvent}s through an injected `Telemetry<EvalEvent>` sink.\n *\n * The `target` is a plain async function here (not required to be a tool); a\n * tool/agent target is a documented follow-up.\n */\n\nimport { runWorkflow, type RuntimeWorkflow } from \"@agentproto/workflow-runtime\"\nimport type { Telemetry } from \"@agentproto/telemetry\"\nimport type { DriverHandle } from \"@agentproto/driver\"\nimport type { ToolHandle } from \"@agentproto/tool\"\nimport type { Score } from \"./score.js\"\nimport type { JsonValue } from \"./json.js\"\nimport type { EvalEvent } from \"./events.js\"\n\n/** One case: an input for the target and an optional expected reference. */\nexport interface EvalCase<I> {\n readonly id: string\n readonly input: I\n /** Optional reference value, modelled as JSON (never `unknown`/`any`). */\n readonly expected?: JsonValue\n}\n\n/** Context handed to a binding's `mapInput` when scoring one case's output. */\nexport interface ScorerInputContext<A> {\n readonly output: A\n readonly expected?: JsonValue\n}\n\n/**\n * One scorer applied to the target output. Generic over the target output type\n * `A` and the scorer's own input type `S` (captured when the binding is\n * authored). {@link bindScorer} erases `S` so heterogeneous scorers coexist in\n * one `EvalSuite.scorers` array without an `any`.\n */\nexport interface ScorerBinding<A, S = JsonValue> {\n /** Label for this binding, e.g. \"exact\". */\n readonly id: string\n /** The scorer TOOL contract — its `outputSchema` is the shared {@link Score}. */\n readonly tool: ToolHandle<S, Score>\n /** The DRIVER that implements the scorer tool. */\n readonly driver: DriverHandle\n /** Derive the scorer's input from the target's output (+ optional expected). */\n readonly mapInput: (ctx: ScorerInputContext<A>) => S\n}\n\n/**\n * The uniform, `S`-erased view of a binding stored in a suite. It exposes the\n * authored `id` plus a closure that runs the scorer for a given case output —\n * the existential box that lets a suite hold scorers with different input\n * types with no `any` at the collection boundary.\n */\nexport interface BoundScorer<A> {\n readonly id: string\n /** Build the AIP-15 tool step that scores one case's output. */\n toStep(stepId: string, output: A, expected?: JsonValue): RuntimeWorkflow[\"steps\"][number]\n}\n\n/**\n * Box a typed {@link ScorerBinding} into a {@link BoundScorer}, capturing the\n * scorer input type `S` inside the closure so the outer type is `S`-free.\n */\nexport function bindScorer<A, S>(binding: ScorerBinding<A, S>): BoundScorer<A> {\n return {\n id: binding.id,\n toStep(stepId, output, expected) {\n return {\n kind: \"tool\",\n id: stepId,\n tool: binding.tool,\n candidates: [binding.driver],\n input: () => binding.mapInput({ output, expected }),\n }\n },\n }\n}\n\nexport interface EvalSuite<A> {\n readonly id: string\n readonly cases: readonly EvalCase<unknown>[]\n readonly scorers: readonly BoundScorer<A>[]\n}\n\n/** One scorer's outcome on one case. */\nexport interface CaseScore {\n readonly scorerId: string\n readonly score: Score\n}\n\n/** The per-case rollup carried in the final report. */\nexport interface CaseReport {\n readonly caseId: string\n readonly passed: boolean\n readonly scores: readonly CaseScore[]\n}\n\nexport interface EvalReport {\n readonly runId: string\n readonly suiteId: string\n readonly total: number\n readonly passedCount: number\n /** Mean of every scorer value across every case (0 when there are none). */\n readonly meanValue: number\n readonly cases: readonly CaseReport[]\n}\n\nexport interface RunEvalOptions<I, A> {\n /** Produce the target output for a case input. */\n readonly target: (input: I) => Promise<A>\n /** Sink for {@link EvalEvent}s. Defaults to a no-op. */\n readonly telemetry?: Telemetry<EvalEvent>\n /** Correlation id for this run. Defaults to a deterministic suite-based id. */\n readonly runId?: string\n}\n\n/** A suite whose cases share the target input type `I`. */\nexport interface TypedEvalSuite<I, A> extends EvalSuite<A> {\n readonly cases: readonly EvalCase<I>[]\n}\n\n/** Per-process counter so a default runId is deterministic within a process. */\nlet runCounter = 0\n\nfunction defaultRunId(suiteId: string): string {\n runCounter += 1\n return `${suiteId}#${runCounter}`\n}\n\n/** The per-case aggregate: which scorers ran and whether all passed. */\ninterface CaseAggregate {\n readonly scores: readonly CaseScore[]\n readonly passed: boolean\n}\n\n/** Narrow a step-binding value to a {@link Score} without a cast. */\nfunction asScore(value: unknown): Score {\n if (\n typeof value === \"object\" &&\n value !== null &&\n \"value\" in value &&\n \"passed\" in value &&\n \"label\" in value\n ) {\n const v: unknown = value.value\n const passed: unknown = value.passed\n const label: unknown = value.label\n const rationale: unknown = \"rationale\" in value ? value.rationale : undefined\n if (\n typeof v === \"number\" &&\n typeof passed === \"boolean\" &&\n typeof label === \"string\" &&\n (rationale === undefined || typeof rationale === \"string\")\n ) {\n return { value: v, passed, label, rationale }\n }\n }\n throw new Error(\"scorer step did not produce a valid Score\")\n}\n\n/**\n * Run every case in the suite, scoring each with all bound scorers, and return\n * the aggregated {@link EvalReport}. Emits `eval.started` … `eval.finished`.\n */\nexport async function runEval<I, A>(\n suite: TypedEvalSuite<I, A>,\n opts: RunEvalOptions<I, A>,\n): Promise<EvalReport> {\n const telemetry = opts.telemetry\n const runId = opts.runId ?? defaultRunId(suite.id)\n const start = Date.now()\n\n telemetry?.emit({\n kind: \"eval.started\",\n runId,\n at: new Date().toISOString(),\n suiteId: suite.id,\n caseCount: suite.cases.length,\n scorerCount: suite.scorers.length,\n })\n\n const caseReports: CaseReport[] = []\n let valueSum = 0\n let valueCount = 0\n\n for (const evalCase of suite.cases) {\n telemetry?.emit({\n kind: \"eval.case.started\",\n runId,\n at: new Date().toISOString(),\n caseId: evalCase.id,\n })\n\n const aggregate = await runCase(suite, evalCase, opts.target)\n\n for (const cs of aggregate.scores) {\n valueSum += cs.score.value\n valueCount += 1\n telemetry?.emit({\n kind: \"eval.case.scored\",\n runId,\n at: new Date().toISOString(),\n caseId: evalCase.id,\n scorerId: cs.scorerId,\n value: cs.score.value,\n passed: cs.score.passed,\n })\n }\n\n telemetry?.emit({\n kind: \"eval.case.finished\",\n runId,\n at: new Date().toISOString(),\n caseId: evalCase.id,\n passed: aggregate.passed,\n })\n\n caseReports.push({\n caseId: evalCase.id,\n passed: aggregate.passed,\n scores: aggregate.scores,\n })\n }\n\n const passedCount = caseReports.filter((c) => c.passed).length\n const meanValue = valueCount === 0 ? 0 : valueSum / valueCount\n\n const report: EvalReport = {\n runId,\n suiteId: suite.id,\n total: suite.cases.length,\n passedCount,\n meanValue,\n cases: caseReports,\n }\n\n telemetry?.emit({\n kind: \"eval.finished\",\n runId,\n at: new Date().toISOString(),\n suiteId: suite.id,\n total: report.total,\n passedCount,\n meanValue,\n durationMs: Date.now() - start,\n })\n\n return report\n}\n\n/**\n * Build and run the per-case workflow: `target` → one scorer tool-step per\n * binding → an aggregate transform, then read the aggregate off the run output.\n */\nasync function runCase<I, A>(\n suite: EvalSuite<A>,\n evalCase: EvalCase<I>,\n target: (input: I) => Promise<A>,\n): Promise<CaseAggregate> {\n const output = await target(evalCase.input)\n const scorerStepIds = suite.scorers.map((_, i) => `score_${i}`)\n\n const workflow: RuntimeWorkflow = {\n id: `${suite.id}:${evalCase.id}`,\n steps: [\n ...suite.scorers.map((scorer, i) =>\n scorer.toStep(scorerStepIds[i]!, output, evalCase.expected),\n ),\n {\n kind: \"transform\",\n id: \"aggregate\",\n compute: (b): CaseAggregate => {\n const scores: CaseScore[] = suite.scorers.map((scorer, i) => ({\n scorerId: scorer.id,\n score: asScore(b.steps[scorerStepIds[i]!]),\n }))\n return { scores, passed: scores.every((s) => s.score.passed) }\n },\n },\n ],\n output: (b) => b.steps.aggregate,\n }\n\n const { output: runOutput } = await runWorkflow({ workflow })\n return toCaseAggregate(runOutput)\n}\n\n/** Narrow the workflow output to a {@link CaseAggregate} without a cast. */\nfunction toCaseAggregate(value: unknown): CaseAggregate {\n if (\n typeof value === \"object\" &&\n value !== null &&\n \"scores\" in value &&\n \"passed\" in value &&\n Array.isArray(value.scores) &&\n typeof value.passed === \"boolean\"\n ) {\n const scores = value.scores.map((entry: unknown) => {\n if (\n typeof entry === \"object\" &&\n entry !== null &&\n \"scorerId\" in entry &&\n \"score\" in entry &&\n typeof entry.scorerId === \"string\"\n ) {\n return { scorerId: entry.scorerId, score: asScore(entry.score) }\n }\n throw new Error(\"invalid CaseScore in aggregate\")\n })\n return { scores, passed: value.passed }\n }\n throw new Error(\"workflow did not produce a CaseAggregate\")\n}\n","/**\n * `toVitest` — turn an {@link TypedEvalSuite} into vitest test registrations.\n *\n * The vitest primitives (`describe` / `it` / `expect`) are INJECTED by the\n * host's own `import ... from \"vitest\"`, so THIS source carries no vitest\n * runtime dependency (nothing to bundle, nothing to version-pin). It registers\n * one `it(caseId)` per case that runs the case through {@link runEval} and\n * asserts every scorer passed — the same workflow the harness runs, wired as a\n * CI gate.\n */\n\nimport { runEval, type RunEvalOptions, type TypedEvalSuite } from \"./run-eval.js\"\n\n/** The subset of vitest's `expect` this bridge needs. */\nexport interface ExpectApi {\n (actual: unknown): {\n toBe(expected: unknown): void\n }\n}\n\n/** The subset of vitest's `describe`/`it` this bridge needs. */\nexport interface VitestHooks {\n describe(name: string, body: () => void): void\n it(name: string, body: () => Promise<void> | void): void\n expect: ExpectApi\n}\n\nexport type ToVitestOptions<I, A> = RunEvalOptions<I, A>\n\n/**\n * Register `describe(suite.id)` with one `it` per case. Each `it` runs the\n * single-case suite through {@link runEval} and asserts the case passed (all\n * scorers green). Failing scorers surface as a failing vitest assertion.\n */\nexport function toVitest<I, A>(\n suite: TypedEvalSuite<I, A>,\n opts: ToVitestOptions<I, A>,\n hooks: VitestHooks,\n): void {\n const { describe, it, expect } = hooks\n describe(suite.id, () => {\n for (const evalCase of suite.cases) {\n it(evalCase.id, async () => {\n const singleCaseSuite: TypedEvalSuite<I, A> = {\n id: `${suite.id}:${evalCase.id}`,\n cases: [evalCase],\n scorers: suite.scorers,\n }\n const report = await runEval(singleCaseSuite, opts)\n const caseReport = report.cases[0]\n expect(caseReport?.passed).toBe(true)\n })\n }\n })\n}\n","/**\n * @agentproto/eval — deterministic reference scorers.\n *\n * A scorer IS an AIP-14 TOOL whose `outputSchema` is the shared {@link Score}\n * shape; there is no separate \"scorer port\". Each scorer is authored with\n * `defineTool` + `implementTool` and bundled in a single builtin AIP-30\n * PROVIDER, exactly like every other builtin. Invoke one through the driver:\n *\n * ```ts\n * import { runTool } from \"@agentproto/driver\"\n * import { exactMatchTool, evalScorersProvider } from \"@agentproto/eval\"\n *\n * const score = await runTool({\n * tool: exactMatchTool,\n * candidates: [evalScorersProvider],\n * input: { actual: \"hello\", expected: \"hello\" },\n * })\n * // → { value: 1, passed: true, label: \"exact-match\", rationale: \"…\" }\n * ```\n *\n * Alongside the deterministic scorers, `eval.llm-judge` (see judge.ts) is a\n * model-backed scorer whose judge is an INJECTED function — the driver\n * closes over it via `makeLlmJudgeDriver(judge)`. This package still carries\n * no LLM SDK or network dependency of its own; a real adapter (agent\n * session, supervisor judge-gate, …) is a documented follow-up.\n *\n * Spec: https://agentproto.sh/docs/aip-14 (TOOL), /docs/aip-30 (PROVIDER)\n */\n\nimport { defineDriver } from \"@agentproto/driver\"\nimport {\n exactMatchTool,\n exactMatchImpl,\n regexMatchTool,\n regexMatchImpl,\n jsonSchemaValidTool,\n jsonSchemaValidImpl,\n latencyBudgetTool,\n latencyBudgetImpl,\n} from \"./scorers.js\"\n\nexport const SPEC_NAME = \"agenteval/v1\" as const\nexport const SPEC_VERSION = \"1.0.0-alpha\" as const\n\nexport { scoreSchema, type Score } from \"./score.js\"\n\nexport {\n type JsonValue,\n type JsonObject,\n jsonValueSchema,\n} from \"./json.js\"\n\nexport {\n exactMatchTool,\n exactMatchImpl,\n regexMatchTool,\n regexMatchImpl,\n jsonSchemaValidTool,\n jsonSchemaValidImpl,\n latencyBudgetTool,\n latencyBudgetImpl,\n type MinimalJsonSchema,\n} from \"./scorers.js\"\n\nexport {\n judgeVerdictSchema,\n llmJudgeTool,\n makeLlmJudgeDriver,\n llmJudge,\n type JudgeVerdict,\n type JudgeFn,\n type LlmJudgeInput,\n type MakeLlmJudgeDriverOptions,\n type LlmJudgeBinding,\n} from \"./judge.js\"\n\nexport { EVAL_EVENT_SCHEMA, type EvalEvent } from \"./events.js\"\n\nexport {\n runEval,\n bindScorer,\n type EvalCase,\n type ScorerBinding,\n type ScorerInputContext,\n type BoundScorer,\n type EvalSuite,\n type TypedEvalSuite,\n type CaseScore,\n type CaseReport,\n type EvalReport,\n type RunEvalOptions,\n} from \"./run-eval.js\"\n\nexport {\n toVitest,\n type VitestHooks,\n type ExpectApi,\n type ToVitestOptions,\n} from \"./to-vitest.js\"\n\n/**\n * Builtin AIP-30 PROVIDER bundling the four deterministic ref scorers.\n * `kind: \"builtin\"` — pure in-process functions, no subprocess / network hop.\n */\nexport const evalScorersProvider = defineDriver({\n id: \"eval-scorers\",\n name: \"Eval Reference Scorers (built-in)\",\n description:\n \"In-process deterministic scorers: exact-match, regex-match, \" +\n \"json-schema-valid (minimal structural check), and latency-budget. \" +\n \"Each is an AIP-14 TOOL whose output is the shared Score shape.\",\n version: \"0.1.0\",\n kind: \"builtin\",\n implements: [\n { tool: \"eval.exact-match\", version: \"0.1.0\" },\n { tool: \"eval.regex-match\", version: \"0.1.0\" },\n { tool: \"eval.json-schema-valid\", version: \"0.1.0\" },\n { tool: \"eval.latency-budget\", version: \"0.1.0\" },\n ],\n implementations: [\n exactMatchImpl,\n regexMatchImpl,\n jsonSchemaValidImpl,\n latencyBudgetImpl,\n ],\n})\n"]}
package/package.json ADDED
@@ -0,0 +1,68 @@
1
+ {
2
+ "name": "@agentproto/eval",
3
+ "version": "0.2.0",
4
+ "description": "Deterministic reference scorers as AIP-14 TOOL contracts + one builtin AIP-30 PROVIDER. A scorer IS a tool whose outputSchema is the shared `Score` shape — no new scorer port. Ships exact-match, regex-match, json-schema-valid, and latency-budget. No LLM/model calls.",
5
+ "keywords": [
6
+ "agentproto",
7
+ "aip-14",
8
+ "aip-30",
9
+ "eval",
10
+ "scorer",
11
+ "score",
12
+ "deterministic",
13
+ "open-standard",
14
+ "agentic"
15
+ ],
16
+ "homepage": "https://agentproto.sh",
17
+ "repository": {
18
+ "type": "git",
19
+ "url": "https://github.com/agentproto/ts",
20
+ "directory": "packages/eval"
21
+ },
22
+ "bugs": {
23
+ "url": "https://github.com/agentproto/ts/issues"
24
+ },
25
+ "license": "MIT",
26
+ "type": "module",
27
+ "main": "dist/index.mjs",
28
+ "module": "dist/index.mjs",
29
+ "types": "dist/index.d.ts",
30
+ "exports": {
31
+ ".": {
32
+ "types": "./dist/index.d.ts",
33
+ "import": "./dist/index.mjs",
34
+ "default": "./dist/index.mjs"
35
+ },
36
+ "./package.json": "./package.json"
37
+ },
38
+ "files": [
39
+ "dist",
40
+ "README.md",
41
+ "LICENSE"
42
+ ],
43
+ "publishConfig": {
44
+ "access": "public"
45
+ },
46
+ "dependencies": {
47
+ "zod": "^4.4.3",
48
+ "@agentproto/driver": "0.1.2",
49
+ "@agentproto/tool": "0.2.0",
50
+ "@agentproto/workflow-runtime": "0.3.0",
51
+ "@agentproto/telemetry": "0.2.0"
52
+ },
53
+ "devDependencies": {
54
+ "@types/node": "^25.6.2",
55
+ "tsup": "^8.5.1",
56
+ "typescript": "^5.9.3",
57
+ "vitest": "^3.2.4",
58
+ "@agentproto/tooling": "0.1.0-alpha.0"
59
+ },
60
+ "scripts": {
61
+ "dev": "tsup --watch",
62
+ "build": "tsup",
63
+ "clean": "rm -rf dist",
64
+ "check-types": "tsc --noEmit",
65
+ "test": "vitest run",
66
+ "test:watch": "vitest"
67
+ }
68
+ }