@velum-labs/routekit-eval-service 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ import { Layer } from "effect";
2
+ import { HttpClient } from "effect/unstable/http";
3
+ import type { RouteKitEvalServiceOptions } from "./layer-options.js";
4
+ import { EvalService, type EvalServiceConfiguration } from "./service.js";
5
+ export type { RouteKitEvalServiceOptions };
6
+ declare const EvalServiceCredentialError_base: new <A extends Record<string, any> = {}>(args: import("effect/Types").VoidIfEmpty<{ readonly [P in keyof A as P extends "_tag" ? never : P]: A[P]; }>) => import("effect/Cause").YieldableError & {
7
+ readonly _tag: "EvalServiceCredentialError";
8
+ } & Readonly<A>;
9
+ export declare class EvalServiceCredentialError extends EvalServiceCredentialError_base<{
10
+ readonly detail: string;
11
+ }> {
12
+ get message(): string;
13
+ }
14
+ /**
15
+ * Complete production composition for RouteKit Eval.
16
+ *
17
+ * EvalService consumes the vendored engine's native Effect service. The
18
+ * execution port is the only adapter; no parallel comparison-runner service
19
+ * mirrors or wraps the engine API.
20
+ */
21
+ export declare const makeRouteKitEvalServiceLayer: (configuration: EvalServiceConfiguration, options: RouteKitEvalServiceOptions) => Layer.Layer<EvalService, never, HttpClient.HttpClient>;
@@ -0,0 +1,39 @@
1
+ import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
2
+ import { EvalEngineExecutionError, EvalExecutionPort, makeEvalEngineLayer, makeRouteKitEvalExecutionPortService } from "@velum-labs/routekit-eval-engine";
3
+ import { Data, Effect, Layer, Stream } from "effect";
4
+ import { HttpClient } from "effect/unstable/http";
5
+ import { makeEvalServiceLayer } from "./service.js";
6
+ export class EvalServiceCredentialError extends Data.TaggedError("EvalServiceCredentialError") {
7
+ get message() {
8
+ return this.detail;
9
+ }
10
+ }
11
+ const makeProductionExecutionPort = (options) => Effect.map(HttpClient.HttpClient, (httpClient) => {
12
+ const bearerCredential = options.bearerCredential?.trim();
13
+ if (bearerCredential === undefined || bearerCredential.length === 0) {
14
+ return {
15
+ execute: () => Stream.fail(new EvalEngineExecutionError({
16
+ cause: new EvalServiceCredentialError({
17
+ detail: "RouteKit Eval comparison execution requires an injected bearer credential."
18
+ }),
19
+ detail: "RouteKit Eval comparison execution requires an injected bearer credential."
20
+ }))
21
+ };
22
+ }
23
+ return makeRouteKitEvalExecutionPortService({
24
+ bearerCredential,
25
+ ...(options.childEnvironment === undefined
26
+ ? {}
27
+ : { childEnvironment: options.childEnvironment }),
28
+ ...(options.execPath === undefined ? {} : { execPath: options.execPath })
29
+ }, httpClient);
30
+ });
31
+ const makeProductionEvalEngineLayer = (options) => makeEvalEngineLayer().pipe(Layer.provide(Layer.effect(EvalExecutionPort, makeProductionExecutionPort(options))));
32
+ /**
33
+ * Complete production composition for RouteKit Eval.
34
+ *
35
+ * EvalService consumes the vendored engine's native Effect service. The
36
+ * execution port is the only adapter; no parallel comparison-runner service
37
+ * mirrors or wraps the engine API.
38
+ */
39
+ export const makeRouteKitEvalServiceLayer = (configuration, options) => makeEvalServiceLayer(configuration).pipe(Layer.provide(makeProductionEvalEngineLayer(options)), Layer.provide(NodeServicesLayer));
@@ -0,0 +1,56 @@
1
+ import type { EvalComparisonRequest, EvalComparisonResult, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
2
+ import { EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
3
+ import { EvalEngine } from "@velum-labs/routekit-eval-engine";
4
+ import { Context, Effect, FileSystem, Layer, Path } from "effect";
5
+ import { EvalServiceComparisonError, EvalServiceConfigurationError, EvalServiceEstimateError, EvalServicePolicyError, EvalServicePublicationError, EvalServiceSpendLimitError, EvalServiceValidationError } from "./errors.js";
6
+ export type EvalComparisonMode = "pilot" | "full";
7
+ export type EvalSuiteInspection = {
8
+ readonly suiteDigest: string;
9
+ readonly manifest: EvalRunManifest;
10
+ };
11
+ export type EvalComparisonEstimate = {
12
+ readonly callCount: number;
13
+ readonly maximumCostUsd?: number;
14
+ readonly pricingKnown: boolean;
15
+ };
16
+ export type EvalRunConfiguration = {
17
+ readonly concurrency?: number;
18
+ readonly timeoutMs?: number;
19
+ };
20
+ export type EvalServiceConfiguration = {
21
+ readonly gatewayUrl?: string;
22
+ readonly snapshotRoot?: string;
23
+ readonly full?: EvalRunConfiguration;
24
+ };
25
+ export type DimensionMatrixSuite = {
26
+ readonly dimensionId: string;
27
+ readonly suitePath: string;
28
+ };
29
+ export type DimensionMatrixQualificationInput = {
30
+ readonly basis: RoutingBasis;
31
+ readonly candidateModels: ReadonlyArray<string>;
32
+ readonly classifierModel: string;
33
+ readonly judgeModel: string;
34
+ readonly objective: RoutingObjectivePolicy;
35
+ readonly maximumUnknownWeight: number;
36
+ readonly constraints?: RoutingActivationConstraints;
37
+ readonly suites: ReadonlyArray<DimensionMatrixSuite>;
38
+ };
39
+ export type DimensionMatrixQualificationResult = {
40
+ readonly comparisons: ReadonlyArray<EvalComparisonResult>;
41
+ readonly snapshot: PublishedRoutingActivation;
42
+ };
43
+ export type EvalServiceError = EvalServiceComparisonError | EvalServiceConfigurationError | EvalServiceEstimateError | EvalServicePolicyError | EvalServicePublicationError | EvalServiceSpendLimitError | EvalServiceValidationError;
44
+ export type EvalServiceShape = {
45
+ readonly validate: (suitePath: string) => Effect.Effect<void, EvalServiceValidationError>;
46
+ readonly inspect: (request: EvalComparisonRequest) => Effect.Effect<EvalSuiteInspection, EvalServiceValidationError>;
47
+ readonly estimate: (request: EvalComparisonRequest, mode: EvalComparisonMode) => Effect.Effect<EvalComparisonEstimate, EvalServiceEstimateError | EvalServiceValidationError>;
48
+ readonly runComparison: (request: EvalComparisonRequest, mode: EvalComparisonMode) => Effect.Effect<EvalComparisonResult, EvalServiceComparisonError | EvalServiceSpendLimitError | EvalServiceValidationError>;
49
+ readonly qualifyDimensionMatrix: (input: DimensionMatrixQualificationInput) => Effect.Effect<DimensionMatrixQualificationResult, EvalServiceError>;
50
+ };
51
+ declare const EvalService_base: Context.ServiceClass<EvalService, "@velum-labs/routekit-eval-service/EvalService", EvalServiceShape>;
52
+ export declare class EvalService extends EvalService_base {
53
+ }
54
+ export declare const makeEvalService: (configuration?: EvalServiceConfiguration) => Effect.Effect<EvalService["Service"], never, EvalEngine | FileSystem.FileSystem | Path.Path>;
55
+ export declare const makeEvalServiceLayer: (configuration?: EvalServiceConfiguration) => Layer.Layer<EvalService, never, EvalEngine | FileSystem.FileSystem | Path.Path>;
56
+ export {};
@@ -0,0 +1,345 @@
1
+ import { assertExplicitEvalModel, assertRoutingBasis, EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
2
+ import { EvalEngine } from "@velum-labs/routekit-eval-engine";
3
+ import { makeRoutingActivationStore } from "@velum-labs/routekit-eval-store";
4
+ import { DEFAULT_MODEL_PRICING, PRICING_ALIASES } from "@velum-labs/routekit-registry";
5
+ import { Context, Effect, FileSystem, Layer, Path, Schema } from "effect";
6
+ import { compileDimensionEvidenceMatrix } from "./dimension-evidence.js";
7
+ import { EvalServiceComparisonError, EvalServiceConfigurationError, EvalServiceEstimateError, EvalServicePolicyError, EvalServicePublicationError, EvalServiceSpendLimitError, EvalServiceValidationError } from "./errors.js";
8
+ export class EvalService extends Context.Service()("@velum-labs/routekit-eval-service/EvalService") {
9
+ }
10
+ const detailOf = (cause) => cause instanceof Error ? cause.message : String(cause);
11
+ const validatePositiveOption = (label, value, allowZero = false) => {
12
+ if (value === undefined)
13
+ return;
14
+ if (!Number.isFinite(value) || (allowZero ? value < 0 : value < 1)) {
15
+ throw new Error(`${label} must be ${allowZero ? "non-negative" : "at least 1"}`);
16
+ }
17
+ };
18
+ const validateGatewayUrl = (gatewayUrl) => {
19
+ const gateway = new URL(gatewayUrl);
20
+ if (gateway.protocol !== "http:" && gateway.protocol !== "https:") {
21
+ throw new Error("gatewayUrl must use http or https");
22
+ }
23
+ if (gateway.username.length > 0 || gateway.password.length > 0) {
24
+ throw new Error("gatewayUrl must not contain credentials");
25
+ }
26
+ };
27
+ const validateConfiguration = (configuration) => {
28
+ if (configuration.gatewayUrl === undefined) {
29
+ throw new Error("gatewayUrl is required for dimension matrix qualification");
30
+ }
31
+ validateGatewayUrl(configuration.gatewayUrl);
32
+ if (configuration.snapshotRoot === undefined || configuration.snapshotRoot.trim().length === 0) {
33
+ throw new Error("snapshotRoot is required for dimension matrix qualification");
34
+ }
35
+ validatePositiveOption("full.concurrency", configuration.full?.concurrency);
36
+ validatePositiveOption("full.timeoutMs", configuration.full?.timeoutMs);
37
+ return {
38
+ gatewayUrl: configuration.gatewayUrl,
39
+ snapshotRoot: configuration.snapshotRoot
40
+ };
41
+ };
42
+ const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
43
+ const validateDimensionMatrixInput = (configuration, input) => {
44
+ const validatedConfiguration = validateConfiguration(configuration);
45
+ assertRoutingBasis(input.basis);
46
+ if (input.candidateModels.length < 2) {
47
+ throw new Error("dimension matrix requires at least two candidate models");
48
+ }
49
+ if (new Set(input.candidateModels).size !== input.candidateModels.length) {
50
+ throw new Error("dimension matrix candidate models must be unique");
51
+ }
52
+ for (const model of input.candidateModels)
53
+ assertExplicitEvalModel(model, "candidate");
54
+ assertExplicitEvalModel(input.classifierModel, "classifier");
55
+ assertExplicitEvalModel(input.judgeModel, "judge");
56
+ const expectedDimensions = new Set(input.basis.dimensions.map((dimension) => dimension.id));
57
+ const suites = new Map();
58
+ for (const entry of input.suites) {
59
+ if (!expectedDimensions.has(entry.dimensionId)) {
60
+ throw new Error(`dimension matrix contains unknown dimension ${JSON.stringify(entry.dimensionId)}`);
61
+ }
62
+ if (suites.has(entry.dimensionId)) {
63
+ throw new Error(`dimension matrix contains duplicate dimension ${JSON.stringify(entry.dimensionId)}`);
64
+ }
65
+ if (entry.suitePath.trim().length === 0) {
66
+ throw new Error(`dimension matrix suite path is empty for ${JSON.stringify(entry.dimensionId)}`);
67
+ }
68
+ suites.set(entry.dimensionId, entry);
69
+ }
70
+ for (const dimensionId of expectedDimensions) {
71
+ if (!suites.has(dimensionId)) {
72
+ throw new Error(`dimension matrix is missing dimension ${JSON.stringify(dimensionId)}`);
73
+ }
74
+ }
75
+ return { ...validatedConfiguration, suites };
76
+ };
77
+ const comparisonRequest = (configuration, gatewayUrl, input, suite) => ({
78
+ version: 1,
79
+ profileId: suite.dimensionId,
80
+ suitePath: suite.suitePath,
81
+ candidateModels: [...input.candidateModels],
82
+ judgeModel: input.judgeModel,
83
+ gatewayUrl,
84
+ ...(configuration.full?.concurrency === undefined
85
+ ? {}
86
+ : { concurrency: configuration.full.concurrency }),
87
+ ...(configuration.full?.timeoutMs === undefined
88
+ ? {}
89
+ : { timeoutMs: configuration.full.timeoutMs })
90
+ });
91
+ const MAXIMUM_INPUT_TOKENS_PER_CALL = 256 * 1024;
92
+ const modelPricing = (model) => {
93
+ const names = [model, model.includes("/") ? model.slice(model.indexOf("/") + 1) : model];
94
+ for (const name of names) {
95
+ const direct = Object.entries(DEFAULT_MODEL_PRICING).find(([candidate]) => candidate.toLowerCase() === name.toLowerCase())?.[1];
96
+ if (direct !== undefined)
97
+ return direct;
98
+ const alias = Object.entries(PRICING_ALIASES).find(([candidate]) => candidate.toLowerCase() === name.toLowerCase())?.[1];
99
+ if (alias === undefined)
100
+ continue;
101
+ const resolved = Object.entries(DEFAULT_MODEL_PRICING).find(([candidate]) => candidate.toLowerCase() === alias.toLowerCase())?.[1];
102
+ if (resolved !== undefined)
103
+ return resolved;
104
+ }
105
+ return undefined;
106
+ };
107
+ const maximumCallCost = (pricing, maximumOutputTokens) => (MAXIMUM_INPUT_TOKENS_PER_CALL * pricing.inputPer1mTokens +
108
+ maximumOutputTokens * pricing.outputPer1mTokens) /
109
+ 1_000_000;
110
+ const estimateManifest = (manifest) => {
111
+ const candidatePrices = manifest.candidateModels.map(modelPricing);
112
+ const judgePrice = modelPricing(manifest.judgeModel);
113
+ if (candidatePrices.some((pricing) => pricing === undefined) || judgePrice === undefined) {
114
+ return {
115
+ callCount: manifest.expectedCallCount,
116
+ pricingKnown: false
117
+ };
118
+ }
119
+ const candidateCost = candidatePrices.reduce((total, pricing) => total + manifest.caseCount * maximumCallCost(pricing, manifest.maxOutputTokens), 0);
120
+ const judgeCost = manifest.caseCount *
121
+ manifest.candidateModels.length *
122
+ maximumCallCost(judgePrice, manifest.maxOutputTokens);
123
+ return {
124
+ callCount: manifest.expectedCallCount,
125
+ maximumCostUsd: candidateCost + judgeCost,
126
+ pricingKnown: true
127
+ };
128
+ };
129
+ const enforceSpendLimit = (request, manifest) => {
130
+ if (request.spendLimitUsd === undefined)
131
+ return Effect.void;
132
+ const estimate = estimateManifest(manifest);
133
+ if (!estimate.pricingKnown || estimate.maximumCostUsd === undefined) {
134
+ return Effect.fail(new EvalServiceSpendLimitError({
135
+ operation: "enforce the comparison spend limit",
136
+ detail: "RouteKit Eval cannot enforce spendLimitUsd because pricing is unknown for one or more manifest models."
137
+ }));
138
+ }
139
+ if (estimate.maximumCostUsd > request.spendLimitUsd) {
140
+ return Effect.fail(new EvalServiceSpendLimitError({
141
+ operation: "enforce the comparison spend limit",
142
+ detail: `RouteKit Eval maximum estimated cost $${estimate.maximumCostUsd.toFixed(6)} ` +
143
+ `exceeds spendLimitUsd $${request.spendLimitUsd.toFixed(6)}.`
144
+ }));
145
+ }
146
+ return Effect.void;
147
+ };
148
+ const loadExecutionManifest = Effect.fn("EvalService.loadExecutionManifest")(function* (validation, request) {
149
+ const fs = yield* FileSystem.FileSystem;
150
+ const paths = yield* Path.Path;
151
+ const manifests = yield* fs
152
+ .glob("**/routekit.eval-manifest.json", {
153
+ root: validation.workingDirectory
154
+ })
155
+ .pipe(Effect.mapError((cause) => new EvalServiceValidationError({
156
+ operation: "discover the comparison manifest",
157
+ detail: detailOf(cause),
158
+ cause
159
+ })));
160
+ if (manifests.length !== 1 || manifests[0] === undefined) {
161
+ return yield* new EvalServiceValidationError({
162
+ operation: "discover the comparison manifest",
163
+ detail: "comparison requires exactly one routekit.eval-manifest.json"
164
+ });
165
+ }
166
+ const manifestPath = paths.isAbsolute(manifests[0])
167
+ ? manifests[0]
168
+ : paths.join(validation.workingDirectory, manifests[0]);
169
+ const raw = yield* fs.readFileString(manifestPath).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
170
+ operation: "read the comparison manifest",
171
+ detail: detailOf(cause),
172
+ cause
173
+ })));
174
+ const json = yield* Effect.try({
175
+ try: () => JSON.parse(raw),
176
+ catch: (cause) => new EvalServiceValidationError({
177
+ operation: "parse the comparison manifest",
178
+ detail: "comparison manifest is not JSON",
179
+ cause
180
+ })
181
+ });
182
+ const manifest = yield* Schema.decodeUnknownEffect(EvalRunManifest)(json).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
183
+ operation: "decode the comparison manifest",
184
+ detail: "comparison manifest is invalid",
185
+ cause
186
+ })));
187
+ if (manifest.profileId !== request.profileId ||
188
+ !sameStrings(manifest.candidateModels, request.candidateModels) ||
189
+ manifest.judgeModel !== request.judgeModel) {
190
+ return yield* new EvalServiceValidationError({
191
+ operation: "bind the comparison manifest",
192
+ detail: "comparison request profile or models do not match the authoritative manifest"
193
+ });
194
+ }
195
+ if (manifest.profileId.trim().length === 0 ||
196
+ manifest.caseCount < 1 ||
197
+ manifest.caseIds.length !== manifest.caseCount ||
198
+ new Set(manifest.caseIds).size !== manifest.caseIds.length ||
199
+ manifest.caseIds.some((caseId) => caseId.trim().length === 0)) {
200
+ return yield* new EvalServiceValidationError({
201
+ operation: "validate the comparison manifest cases",
202
+ detail: "comparison manifest case identities are incomplete or duplicated"
203
+ });
204
+ }
205
+ const expectedCallCount = manifest.caseCount * manifest.candidateModels.length * 2;
206
+ if (manifest.expectedCallCount !== expectedCallCount || manifest.maxOutputTokens < 1) {
207
+ return yield* new EvalServiceValidationError({
208
+ operation: "validate the comparison manifest limits",
209
+ detail: "comparison manifest call or output-token limits are inconsistent"
210
+ });
211
+ }
212
+ return {
213
+ suiteDigest: validation.suiteDigest,
214
+ manifest
215
+ };
216
+ });
217
+ export const makeEvalService = (configuration = {}) => Effect.gen(function* () {
218
+ const engine = yield* EvalEngine;
219
+ const fs = yield* FileSystem.FileSystem;
220
+ const paths = yield* Path.Path;
221
+ const inspectRaw = Effect.fn("EvalService.inspectRaw")(function* (request) {
222
+ const validation = yield* engine.validate(request.suitePath);
223
+ return yield* loadExecutionManifest(validation, request).pipe(Effect.provideService(FileSystem.FileSystem, fs), Effect.provideService(Path.Path, paths));
224
+ });
225
+ const validate = Effect.fn("EvalService.validate")(function* (suitePath) {
226
+ yield* engine.validate(suitePath).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
227
+ operation: "validate the comparison suite",
228
+ detail: detailOf(cause),
229
+ cause
230
+ })));
231
+ });
232
+ const inspect = (request) => inspectRaw(request).pipe(Effect.mapError((cause) => cause instanceof EvalServiceValidationError
233
+ ? cause
234
+ : new EvalServiceValidationError({
235
+ operation: "inspect the comparison manifest",
236
+ detail: detailOf(cause),
237
+ cause
238
+ })));
239
+ const estimate = (request) => inspectRaw(request).pipe(Effect.map((inspection) => estimateManifest(inspection.manifest)), Effect.mapError((cause) => new EvalServiceEstimateError({
240
+ operation: "estimate the comparison",
241
+ detail: detailOf(cause),
242
+ cause
243
+ })));
244
+ const runInspectedComparison = (request, inspection) => enforceSpendLimit(request, inspection.manifest).pipe(Effect.flatMap(() => engine
245
+ .runComparison({
246
+ ...request,
247
+ expectedCaseIds: [...inspection.manifest.caseIds],
248
+ expectedCallCount: inspection.manifest.expectedCallCount,
249
+ maxOutputTokens: inspection.manifest.maxOutputTokens,
250
+ suiteDigest: inspection.suiteDigest
251
+ })
252
+ .pipe(Effect.mapError((cause) => new EvalServiceComparisonError({
253
+ operation: "run the comparison",
254
+ detail: detailOf(cause),
255
+ cause
256
+ })))));
257
+ const runComparison = (request) => inspect(request).pipe(Effect.flatMap((inspection) => runInspectedComparison(request, inspection)));
258
+ const qualifyDimensionMatrix = (input) => Effect.gen(function* () {
259
+ const validated = yield* Effect.try({
260
+ try: () => validateDimensionMatrixInput(configuration, input),
261
+ catch: (cause) => new EvalServiceConfigurationError({
262
+ operation: "validate the dimension matrix qualification",
263
+ detail: detailOf(cause),
264
+ cause
265
+ })
266
+ });
267
+ const inspections = new Map();
268
+ for (const dimension of input.basis.dimensions) {
269
+ const suite = validated.suites.get(dimension.id);
270
+ const request = comparisonRequest(configuration, validated.gatewayUrl, input, suite);
271
+ const inspection = yield* inspect(request);
272
+ if (inspection.manifest.profileId !== dimension.id ||
273
+ inspection.manifest.judgeModel !== input.judgeModel ||
274
+ !sameStrings(inspection.manifest.candidateModels, input.candidateModels)) {
275
+ return yield* new EvalServiceValidationError({
276
+ operation: "inspect the full comparison manifest",
277
+ detail: `authoritative manifest does not match dimension ${JSON.stringify(dimension.id)}`
278
+ });
279
+ }
280
+ inspections.set(dimension.id, inspection);
281
+ }
282
+ const comparisons = [];
283
+ const evidenceInputs = [];
284
+ for (const dimension of input.basis.dimensions) {
285
+ const suite = validated.suites.get(dimension.id);
286
+ const inspection = inspections.get(dimension.id);
287
+ const request = comparisonRequest(configuration, validated.gatewayUrl, input, suite);
288
+ const comparison = yield* runInspectedComparison(request, inspection);
289
+ if (comparison.profileId !== dimension.id ||
290
+ comparison.judgeModel !== input.judgeModel ||
291
+ comparison.suiteDigest !== inspection.suiteDigest) {
292
+ return yield* new EvalServiceComparisonError({
293
+ operation: "validate the full comparison",
294
+ detail: `comparison identity does not match dimension ${JSON.stringify(dimension.id)}`
295
+ });
296
+ }
297
+ comparisons.push(comparison);
298
+ evidenceInputs.push({
299
+ dimensionId: dimension.id,
300
+ suiteDigest: inspection.suiteDigest,
301
+ judgeModel: inspection.manifest.judgeModel,
302
+ expectedCaseIds: inspection.manifest.caseIds,
303
+ comparison
304
+ });
305
+ }
306
+ const compiled = yield* Effect.try({
307
+ try: () => compileDimensionEvidenceMatrix({
308
+ basis: input.basis,
309
+ candidateModels: input.candidateModels,
310
+ comparisons: evidenceInputs
311
+ }),
312
+ catch: (cause) => new EvalServicePolicyError({
313
+ operation: "compile the dimension evidence matrix",
314
+ detail: detailOf(cause),
315
+ cause
316
+ })
317
+ });
318
+ const snapshot = yield* makeRoutingActivationStore(validated.snapshotRoot)
319
+ .publish({
320
+ basisDigest: input.basis.basisDigest,
321
+ evidenceDigest: compiled.evidenceDigest,
322
+ classifierModel: input.classifierModel,
323
+ objective: input.objective,
324
+ maximumUnknownWeight: input.maximumUnknownWeight,
325
+ ...(input.constraints === undefined ? {} : { constraints: input.constraints }),
326
+ dimensions: [...input.basis.dimensions],
327
+ candidateModels: [...input.candidateModels],
328
+ evidence: [...compiled.evidence]
329
+ })
330
+ .pipe(Effect.provideService(FileSystem.FileSystem, fs), Effect.provideService(Path.Path, paths), Effect.mapError((cause) => new EvalServicePublicationError({
331
+ operation: "publish the dimension evidence snapshot",
332
+ detail: detailOf(cause),
333
+ cause
334
+ })));
335
+ return { comparisons, snapshot };
336
+ });
337
+ return EvalService.of({
338
+ validate,
339
+ inspect,
340
+ estimate,
341
+ runComparison,
342
+ qualifyDimensionMatrix
343
+ });
344
+ });
345
+ export const makeEvalServiceLayer = (configuration = {}) => Layer.effect(EvalService, makeEvalService(configuration));
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,155 @@
1
+ import assert from "node:assert/strict";
2
+ import { test } from "node:test";
3
+ import { compileDimensionEvidenceMatrix, wilsonLowerBound95 } from "../dimension-evidence.js";
4
+ const dimensionIds = [
5
+ "gateway-protocol",
6
+ "eval-routing",
7
+ "account-pooling",
8
+ "typescript-maintenance",
9
+ "release-operations"
10
+ ];
11
+ const models = ["openai/model-a", "openai/model-b"];
12
+ const caseIds = ["case-1", "case-2", "case-3", "case-4", "case-5"];
13
+ const judgeModel = "openai/judge";
14
+ function mutable(value) {
15
+ return structuredClone(value);
16
+ }
17
+ const basis = {
18
+ version: 2,
19
+ basisDigest: "definition-set-v2",
20
+ dimensions: dimensionIds.map((id) => ({
21
+ id,
22
+ description: `Requests about ${id}`,
23
+ includes: [`Tasks specifically involving ${id}`],
24
+ excludes: [`Tasks unrelated to ${id}`]
25
+ }))
26
+ };
27
+ function cases(options = {}) {
28
+ return caseIds.map((caseId, index) => {
29
+ const passed = !options.failed?.has(caseId);
30
+ return {
31
+ caseId,
32
+ outcome: passed ? "passed" : "failed",
33
+ measurement: {
34
+ judgeScore: passed ? 0.9 : 0.2,
35
+ ...(!options.unpriced?.has(caseId) ? { costUsd: 0.01 + index / 1_000 } : {}),
36
+ ...(!options.missingDuration?.has(caseId) ? { durationMs: 100 + index * 10 } : {}),
37
+ inputTokens: 100 + index,
38
+ outputTokens: 20 + index
39
+ }
40
+ };
41
+ });
42
+ }
43
+ function comparison(dimensionId) {
44
+ return {
45
+ version: 1,
46
+ comparisonId: `comparison-${dimensionId}`,
47
+ profileId: dimensionId,
48
+ suiteDigest: `suite-${dimensionId}`,
49
+ judgeModel,
50
+ startedAt: "2026-08-17T00:00:00.000Z",
51
+ finishedAt: "2026-08-17T00:01:00.000Z",
52
+ models: models.map((model) => ({ model, cases: cases() }))
53
+ };
54
+ }
55
+ function input() {
56
+ return mutable({
57
+ basis,
58
+ candidateModels: models,
59
+ comparisons: dimensionIds.map((dimensionId) => ({
60
+ dimensionId,
61
+ suiteDigest: `suite-${dimensionId}`,
62
+ judgeModel,
63
+ expectedCaseIds: caseIds,
64
+ comparison: comparison(dimensionId)
65
+ }))
66
+ });
67
+ }
68
+ test("complete comparison results compile into a deterministic model-dimension matrix", () => {
69
+ const first = compileDimensionEvidenceMatrix(input());
70
+ const secondInput = input();
71
+ secondInput.comparisons.reverse();
72
+ secondInput.comparisons.forEach((entry) => {
73
+ entry.comparison.models.reverse();
74
+ entry.comparison.models.forEach((model) => model.cases.reverse());
75
+ });
76
+ const second = compileDimensionEvidenceMatrix(secondInput);
77
+ assert.equal(first.evidence.length, dimensionIds.length * models.length);
78
+ assert.match(first.evidenceDigest, /^[a-f0-9]{64}$/u);
79
+ assert.equal(first.evidenceDigest, second.evidenceDigest);
80
+ assert.deepEqual(first.evidence, second.evidence);
81
+ const cell = first.evidence.find((entry) => entry.dimensionId === "gateway-protocol" && entry.model === "openai/model-a");
82
+ assert.equal(cell?.quality.sampleCount, 5);
83
+ assert.equal(cell?.quality.passRate, 1);
84
+ assert.ok((cell?.quality.lowerConfidenceBound ?? 0) > 0.56);
85
+ assert.ok((cell?.quality.lowerConfidenceBound ?? 1) < 0.57);
86
+ assert.equal(cell?.failureRate, 0);
87
+ assert.equal(cell?.averageJudgeScore, 0.9);
88
+ assert.equal(cell?.p95DurationMs, 140);
89
+ assert.equal(cell?.unpricedCalls, 0);
90
+ assert.equal(cell?.averageCostUsd, 0.012);
91
+ });
92
+ test("partial pricing is explicit and never converted into a known average", () => {
93
+ const value = input();
94
+ value.comparisons[0].comparison.models[0].cases = cases({
95
+ unpriced: new Set(["case-4", "case-5"])
96
+ });
97
+ const cell = compileDimensionEvidenceMatrix(value).evidence.find((entry) => entry.dimensionId === dimensionIds[0] && entry.model === models[0]);
98
+ assert.equal(cell?.unpricedCalls, 2);
99
+ assert.equal(cell?.averageCostUsd, undefined);
100
+ });
101
+ test("partial duration measurements do not produce a misleading percentile", () => {
102
+ const value = input();
103
+ value.comparisons[0].comparison.models[0].cases = cases({
104
+ missingDuration: new Set(["case-5"])
105
+ });
106
+ const cell = compileDimensionEvidenceMatrix(value).evidence.find((entry) => entry.dimensionId === dimensionIds[0] && entry.model === models[0]);
107
+ assert.equal(cell?.p95DurationMs, undefined);
108
+ });
109
+ test("matrix compilation rejects missing and duplicate dimensions or candidates", () => {
110
+ const missingDimension = input();
111
+ missingDimension.comparisons.pop();
112
+ assert.throws(() => compileDimensionEvidenceMatrix(missingDimension), /missing dimension/);
113
+ const duplicateDimension = input();
114
+ duplicateDimension.comparisons[1] = duplicateDimension.comparisons[0];
115
+ assert.throws(() => compileDimensionEvidenceMatrix(duplicateDimension), /duplicate dimension/);
116
+ const missingCandidate = input();
117
+ missingCandidate.comparisons[0].comparison.models.pop();
118
+ assert.throws(() => compileDimensionEvidenceMatrix(missingCandidate), /missing candidate/);
119
+ const unexpectedCandidate = input();
120
+ unexpectedCandidate.comparisons[0].comparison.models[0].model = "openai/unexpected";
121
+ assert.throws(() => compileDimensionEvidenceMatrix(unexpectedCandidate), /unexpected candidate/);
122
+ });
123
+ test("matrix compilation rejects incomplete or ambiguous case evidence", () => {
124
+ const missingCase = input();
125
+ missingCase.comparisons[0].comparison.models[0].cases.pop();
126
+ assert.throws(() => compileDimensionEvidenceMatrix(missingCase), /has 4 cases; expected 5/);
127
+ const duplicateCase = input();
128
+ duplicateCase.comparisons[0].comparison.models[0].cases[4] =
129
+ duplicateCase.comparisons[0].comparison.models[0].cases[0];
130
+ assert.throws(() => compileDimensionEvidenceMatrix(duplicateCase), /duplicate case/);
131
+ const missingJudgeScore = input();
132
+ delete missingJudgeScore.comparisons[0].comparison.models[0].cases[0].measurement.judgeScore;
133
+ assert.throws(() => compileDimensionEvidenceMatrix(missingJudgeScore), /missing a judge score/);
134
+ const cutoff = input();
135
+ cutoff.comparisons[0].comparison.models[0].cases[0].outcome = "cutoff";
136
+ assert.throws(() => compileDimensionEvidenceMatrix(cutoff), /non-terminal case/);
137
+ });
138
+ test("matrix compilation binds profile, suite digest, and judge", () => {
139
+ const wrongProfile = input();
140
+ wrongProfile.comparisons[0].comparison.profileId = "wrong";
141
+ assert.throws(() => compileDimensionEvidenceMatrix(wrongProfile), /does not match dimension/);
142
+ const wrongDigest = input();
143
+ wrongDigest.comparisons[0].comparison.suiteDigest = "wrong";
144
+ assert.throws(() => compileDimensionEvidenceMatrix(wrongDigest), /suite digest does not match/);
145
+ const wrongJudge = input();
146
+ wrongJudge.comparisons[0].comparison.judgeModel = "openai/wrong";
147
+ assert.throws(() => compileDimensionEvidenceMatrix(wrongJudge), /comparison judge does not match/);
148
+ });
149
+ test("Wilson lower confidence bounds are conservative and validate counts", () => {
150
+ assert.equal(wilsonLowerBound95(0, 5), 0);
151
+ assert.ok(wilsonLowerBound95(5, 5) < 1);
152
+ assert.ok(wilsonLowerBound95(19, 20) < 0.95);
153
+ assert.throws(() => wilsonLowerBound95(6, 5), /within the sample/);
154
+ assert.throws(() => wilsonLowerBound95(0, 0), /within the sample/);
155
+ });
@@ -0,0 +1 @@
1
+ export {};