@velum-labs/routekit-eval-service 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +18 -0
- package/dist/dimension-evidence.d.ts +27 -0
- package/dist/dimension-evidence.js +263 -0
- package/dist/effect-api.d.ts +7 -0
- package/dist/effect-api.js +4 -0
- package/dist/errors.d.ts +48 -0
- package/dist/errors.js +36 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.js +4 -0
- package/dist/layer-options.d.ts +10 -0
- package/dist/layer-options.js +1 -0
- package/dist/production-runner.d.ts +21 -0
- package/dist/production-runner.js +39 -0
- package/dist/service.d.ts +56 -0
- package/dist/service.js +345 -0
- package/dist/test/dimension-evidence.test.d.ts +1 -0
- package/dist/test/dimension-evidence.test.js +155 -0
- package/dist/test/production-runner.test.d.ts +1 -0
- package/dist/test/production-runner.test.js +218 -0
- package/dist/test/service.test.d.ts +1 -0
- package/dist/test/service.test.js +162 -0
- package/package.json +54 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { Layer } from "effect";
|
|
2
|
+
import { HttpClient } from "effect/unstable/http";
|
|
3
|
+
import type { RouteKitEvalServiceOptions } from "./layer-options.js";
|
|
4
|
+
import { EvalService, type EvalServiceConfiguration } from "./service.js";
|
|
5
|
+
export type { RouteKitEvalServiceOptions };
|
|
6
|
+
declare const EvalServiceCredentialError_base: new <A extends Record<string, any> = {}>(args: import("effect/Types").VoidIfEmpty<{ readonly [P in keyof A as P extends "_tag" ? never : P]: A[P]; }>) => import("effect/Cause").YieldableError & {
|
|
7
|
+
readonly _tag: "EvalServiceCredentialError";
|
|
8
|
+
} & Readonly<A>;
|
|
9
|
+
export declare class EvalServiceCredentialError extends EvalServiceCredentialError_base<{
|
|
10
|
+
readonly detail: string;
|
|
11
|
+
}> {
|
|
12
|
+
get message(): string;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Complete production composition for RouteKit Eval.
|
|
16
|
+
*
|
|
17
|
+
* EvalService consumes the vendored engine's native Effect service. The
|
|
18
|
+
* execution port is the only adapter; no parallel comparison-runner service
|
|
19
|
+
* mirrors or wraps the engine API.
|
|
20
|
+
*/
|
|
21
|
+
export declare const makeRouteKitEvalServiceLayer: (configuration: EvalServiceConfiguration, options: RouteKitEvalServiceOptions) => Layer.Layer<EvalService, never, HttpClient.HttpClient>;
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
|
|
2
|
+
import { EvalEngineExecutionError, EvalExecutionPort, makeEvalEngineLayer, makeRouteKitEvalExecutionPortService } from "@velum-labs/routekit-eval-engine";
|
|
3
|
+
import { Data, Effect, Layer, Stream } from "effect";
|
|
4
|
+
import { HttpClient } from "effect/unstable/http";
|
|
5
|
+
import { makeEvalServiceLayer } from "./service.js";
|
|
6
|
+
export class EvalServiceCredentialError extends Data.TaggedError("EvalServiceCredentialError") {
|
|
7
|
+
get message() {
|
|
8
|
+
return this.detail;
|
|
9
|
+
}
|
|
10
|
+
}
|
|
11
|
+
const makeProductionExecutionPort = (options) => Effect.map(HttpClient.HttpClient, (httpClient) => {
|
|
12
|
+
const bearerCredential = options.bearerCredential?.trim();
|
|
13
|
+
if (bearerCredential === undefined || bearerCredential.length === 0) {
|
|
14
|
+
return {
|
|
15
|
+
execute: () => Stream.fail(new EvalEngineExecutionError({
|
|
16
|
+
cause: new EvalServiceCredentialError({
|
|
17
|
+
detail: "RouteKit Eval comparison execution requires an injected bearer credential."
|
|
18
|
+
}),
|
|
19
|
+
detail: "RouteKit Eval comparison execution requires an injected bearer credential."
|
|
20
|
+
}))
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
return makeRouteKitEvalExecutionPortService({
|
|
24
|
+
bearerCredential,
|
|
25
|
+
...(options.childEnvironment === undefined
|
|
26
|
+
? {}
|
|
27
|
+
: { childEnvironment: options.childEnvironment }),
|
|
28
|
+
...(options.execPath === undefined ? {} : { execPath: options.execPath })
|
|
29
|
+
}, httpClient);
|
|
30
|
+
});
|
|
31
|
+
const makeProductionEvalEngineLayer = (options) => makeEvalEngineLayer().pipe(Layer.provide(Layer.effect(EvalExecutionPort, makeProductionExecutionPort(options))));
|
|
32
|
+
/**
|
|
33
|
+
* Complete production composition for RouteKit Eval.
|
|
34
|
+
*
|
|
35
|
+
* EvalService consumes the vendored engine's native Effect service. The
|
|
36
|
+
* execution port is the only adapter; no parallel comparison-runner service
|
|
37
|
+
* mirrors or wraps the engine API.
|
|
38
|
+
*/
|
|
39
|
+
export const makeRouteKitEvalServiceLayer = (configuration, options) => makeEvalServiceLayer(configuration).pipe(Layer.provide(makeProductionEvalEngineLayer(options)), Layer.provide(NodeServicesLayer));
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import type { EvalComparisonRequest, EvalComparisonResult, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
+
import { EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
|
|
3
|
+
import { EvalEngine } from "@velum-labs/routekit-eval-engine";
|
|
4
|
+
import { Context, Effect, FileSystem, Layer, Path } from "effect";
|
|
5
|
+
import { EvalServiceComparisonError, EvalServiceConfigurationError, EvalServiceEstimateError, EvalServicePolicyError, EvalServicePublicationError, EvalServiceSpendLimitError, EvalServiceValidationError } from "./errors.js";
|
|
6
|
+
export type EvalComparisonMode = "pilot" | "full";
|
|
7
|
+
export type EvalSuiteInspection = {
|
|
8
|
+
readonly suiteDigest: string;
|
|
9
|
+
readonly manifest: EvalRunManifest;
|
|
10
|
+
};
|
|
11
|
+
export type EvalComparisonEstimate = {
|
|
12
|
+
readonly callCount: number;
|
|
13
|
+
readonly maximumCostUsd?: number;
|
|
14
|
+
readonly pricingKnown: boolean;
|
|
15
|
+
};
|
|
16
|
+
export type EvalRunConfiguration = {
|
|
17
|
+
readonly concurrency?: number;
|
|
18
|
+
readonly timeoutMs?: number;
|
|
19
|
+
};
|
|
20
|
+
export type EvalServiceConfiguration = {
|
|
21
|
+
readonly gatewayUrl?: string;
|
|
22
|
+
readonly snapshotRoot?: string;
|
|
23
|
+
readonly full?: EvalRunConfiguration;
|
|
24
|
+
};
|
|
25
|
+
export type DimensionMatrixSuite = {
|
|
26
|
+
readonly dimensionId: string;
|
|
27
|
+
readonly suitePath: string;
|
|
28
|
+
};
|
|
29
|
+
export type DimensionMatrixQualificationInput = {
|
|
30
|
+
readonly basis: RoutingBasis;
|
|
31
|
+
readonly candidateModels: ReadonlyArray<string>;
|
|
32
|
+
readonly classifierModel: string;
|
|
33
|
+
readonly judgeModel: string;
|
|
34
|
+
readonly objective: RoutingObjectivePolicy;
|
|
35
|
+
readonly maximumUnknownWeight: number;
|
|
36
|
+
readonly constraints?: RoutingActivationConstraints;
|
|
37
|
+
readonly suites: ReadonlyArray<DimensionMatrixSuite>;
|
|
38
|
+
};
|
|
39
|
+
export type DimensionMatrixQualificationResult = {
|
|
40
|
+
readonly comparisons: ReadonlyArray<EvalComparisonResult>;
|
|
41
|
+
readonly snapshot: PublishedRoutingActivation;
|
|
42
|
+
};
|
|
43
|
+
export type EvalServiceError = EvalServiceComparisonError | EvalServiceConfigurationError | EvalServiceEstimateError | EvalServicePolicyError | EvalServicePublicationError | EvalServiceSpendLimitError | EvalServiceValidationError;
|
|
44
|
+
export type EvalServiceShape = {
|
|
45
|
+
readonly validate: (suitePath: string) => Effect.Effect<void, EvalServiceValidationError>;
|
|
46
|
+
readonly inspect: (request: EvalComparisonRequest) => Effect.Effect<EvalSuiteInspection, EvalServiceValidationError>;
|
|
47
|
+
readonly estimate: (request: EvalComparisonRequest, mode: EvalComparisonMode) => Effect.Effect<EvalComparisonEstimate, EvalServiceEstimateError | EvalServiceValidationError>;
|
|
48
|
+
readonly runComparison: (request: EvalComparisonRequest, mode: EvalComparisonMode) => Effect.Effect<EvalComparisonResult, EvalServiceComparisonError | EvalServiceSpendLimitError | EvalServiceValidationError>;
|
|
49
|
+
readonly qualifyDimensionMatrix: (input: DimensionMatrixQualificationInput) => Effect.Effect<DimensionMatrixQualificationResult, EvalServiceError>;
|
|
50
|
+
};
|
|
51
|
+
declare const EvalService_base: Context.ServiceClass<EvalService, "@velum-labs/routekit-eval-service/EvalService", EvalServiceShape>;
|
|
52
|
+
export declare class EvalService extends EvalService_base {
|
|
53
|
+
}
|
|
54
|
+
export declare const makeEvalService: (configuration?: EvalServiceConfiguration) => Effect.Effect<EvalService["Service"], never, EvalEngine | FileSystem.FileSystem | Path.Path>;
|
|
55
|
+
export declare const makeEvalServiceLayer: (configuration?: EvalServiceConfiguration) => Layer.Layer<EvalService, never, EvalEngine | FileSystem.FileSystem | Path.Path>;
|
|
56
|
+
export {};
|
package/dist/service.js
ADDED
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
import { assertExplicitEvalModel, assertRoutingBasis, EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
+
import { EvalEngine } from "@velum-labs/routekit-eval-engine";
|
|
3
|
+
import { makeRoutingActivationStore } from "@velum-labs/routekit-eval-store";
|
|
4
|
+
import { DEFAULT_MODEL_PRICING, PRICING_ALIASES } from "@velum-labs/routekit-registry";
|
|
5
|
+
import { Context, Effect, FileSystem, Layer, Path, Schema } from "effect";
|
|
6
|
+
import { compileDimensionEvidenceMatrix } from "./dimension-evidence.js";
|
|
7
|
+
import { EvalServiceComparisonError, EvalServiceConfigurationError, EvalServiceEstimateError, EvalServicePolicyError, EvalServicePublicationError, EvalServiceSpendLimitError, EvalServiceValidationError } from "./errors.js";
|
|
8
|
+
export class EvalService extends Context.Service()("@velum-labs/routekit-eval-service/EvalService") {
|
|
9
|
+
}
|
|
10
|
+
const detailOf = (cause) => cause instanceof Error ? cause.message : String(cause);
|
|
11
|
+
const validatePositiveOption = (label, value, allowZero = false) => {
|
|
12
|
+
if (value === undefined)
|
|
13
|
+
return;
|
|
14
|
+
if (!Number.isFinite(value) || (allowZero ? value < 0 : value < 1)) {
|
|
15
|
+
throw new Error(`${label} must be ${allowZero ? "non-negative" : "at least 1"}`);
|
|
16
|
+
}
|
|
17
|
+
};
|
|
18
|
+
const validateGatewayUrl = (gatewayUrl) => {
|
|
19
|
+
const gateway = new URL(gatewayUrl);
|
|
20
|
+
if (gateway.protocol !== "http:" && gateway.protocol !== "https:") {
|
|
21
|
+
throw new Error("gatewayUrl must use http or https");
|
|
22
|
+
}
|
|
23
|
+
if (gateway.username.length > 0 || gateway.password.length > 0) {
|
|
24
|
+
throw new Error("gatewayUrl must not contain credentials");
|
|
25
|
+
}
|
|
26
|
+
};
|
|
27
|
+
const validateConfiguration = (configuration) => {
|
|
28
|
+
if (configuration.gatewayUrl === undefined) {
|
|
29
|
+
throw new Error("gatewayUrl is required for dimension matrix qualification");
|
|
30
|
+
}
|
|
31
|
+
validateGatewayUrl(configuration.gatewayUrl);
|
|
32
|
+
if (configuration.snapshotRoot === undefined || configuration.snapshotRoot.trim().length === 0) {
|
|
33
|
+
throw new Error("snapshotRoot is required for dimension matrix qualification");
|
|
34
|
+
}
|
|
35
|
+
validatePositiveOption("full.concurrency", configuration.full?.concurrency);
|
|
36
|
+
validatePositiveOption("full.timeoutMs", configuration.full?.timeoutMs);
|
|
37
|
+
return {
|
|
38
|
+
gatewayUrl: configuration.gatewayUrl,
|
|
39
|
+
snapshotRoot: configuration.snapshotRoot
|
|
40
|
+
};
|
|
41
|
+
};
|
|
42
|
+
const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
|
|
43
|
+
const validateDimensionMatrixInput = (configuration, input) => {
|
|
44
|
+
const validatedConfiguration = validateConfiguration(configuration);
|
|
45
|
+
assertRoutingBasis(input.basis);
|
|
46
|
+
if (input.candidateModels.length < 2) {
|
|
47
|
+
throw new Error("dimension matrix requires at least two candidate models");
|
|
48
|
+
}
|
|
49
|
+
if (new Set(input.candidateModels).size !== input.candidateModels.length) {
|
|
50
|
+
throw new Error("dimension matrix candidate models must be unique");
|
|
51
|
+
}
|
|
52
|
+
for (const model of input.candidateModels)
|
|
53
|
+
assertExplicitEvalModel(model, "candidate");
|
|
54
|
+
assertExplicitEvalModel(input.classifierModel, "classifier");
|
|
55
|
+
assertExplicitEvalModel(input.judgeModel, "judge");
|
|
56
|
+
const expectedDimensions = new Set(input.basis.dimensions.map((dimension) => dimension.id));
|
|
57
|
+
const suites = new Map();
|
|
58
|
+
for (const entry of input.suites) {
|
|
59
|
+
if (!expectedDimensions.has(entry.dimensionId)) {
|
|
60
|
+
throw new Error(`dimension matrix contains unknown dimension ${JSON.stringify(entry.dimensionId)}`);
|
|
61
|
+
}
|
|
62
|
+
if (suites.has(entry.dimensionId)) {
|
|
63
|
+
throw new Error(`dimension matrix contains duplicate dimension ${JSON.stringify(entry.dimensionId)}`);
|
|
64
|
+
}
|
|
65
|
+
if (entry.suitePath.trim().length === 0) {
|
|
66
|
+
throw new Error(`dimension matrix suite path is empty for ${JSON.stringify(entry.dimensionId)}`);
|
|
67
|
+
}
|
|
68
|
+
suites.set(entry.dimensionId, entry);
|
|
69
|
+
}
|
|
70
|
+
for (const dimensionId of expectedDimensions) {
|
|
71
|
+
if (!suites.has(dimensionId)) {
|
|
72
|
+
throw new Error(`dimension matrix is missing dimension ${JSON.stringify(dimensionId)}`);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return { ...validatedConfiguration, suites };
|
|
76
|
+
};
|
|
77
|
+
const comparisonRequest = (configuration, gatewayUrl, input, suite) => ({
|
|
78
|
+
version: 1,
|
|
79
|
+
profileId: suite.dimensionId,
|
|
80
|
+
suitePath: suite.suitePath,
|
|
81
|
+
candidateModels: [...input.candidateModels],
|
|
82
|
+
judgeModel: input.judgeModel,
|
|
83
|
+
gatewayUrl,
|
|
84
|
+
...(configuration.full?.concurrency === undefined
|
|
85
|
+
? {}
|
|
86
|
+
: { concurrency: configuration.full.concurrency }),
|
|
87
|
+
...(configuration.full?.timeoutMs === undefined
|
|
88
|
+
? {}
|
|
89
|
+
: { timeoutMs: configuration.full.timeoutMs })
|
|
90
|
+
});
|
|
91
|
+
const MAXIMUM_INPUT_TOKENS_PER_CALL = 256 * 1024;
|
|
92
|
+
const modelPricing = (model) => {
|
|
93
|
+
const names = [model, model.includes("/") ? model.slice(model.indexOf("/") + 1) : model];
|
|
94
|
+
for (const name of names) {
|
|
95
|
+
const direct = Object.entries(DEFAULT_MODEL_PRICING).find(([candidate]) => candidate.toLowerCase() === name.toLowerCase())?.[1];
|
|
96
|
+
if (direct !== undefined)
|
|
97
|
+
return direct;
|
|
98
|
+
const alias = Object.entries(PRICING_ALIASES).find(([candidate]) => candidate.toLowerCase() === name.toLowerCase())?.[1];
|
|
99
|
+
if (alias === undefined)
|
|
100
|
+
continue;
|
|
101
|
+
const resolved = Object.entries(DEFAULT_MODEL_PRICING).find(([candidate]) => candidate.toLowerCase() === alias.toLowerCase())?.[1];
|
|
102
|
+
if (resolved !== undefined)
|
|
103
|
+
return resolved;
|
|
104
|
+
}
|
|
105
|
+
return undefined;
|
|
106
|
+
};
|
|
107
|
+
const maximumCallCost = (pricing, maximumOutputTokens) => (MAXIMUM_INPUT_TOKENS_PER_CALL * pricing.inputPer1mTokens +
|
|
108
|
+
maximumOutputTokens * pricing.outputPer1mTokens) /
|
|
109
|
+
1_000_000;
|
|
110
|
+
const estimateManifest = (manifest) => {
|
|
111
|
+
const candidatePrices = manifest.candidateModels.map(modelPricing);
|
|
112
|
+
const judgePrice = modelPricing(manifest.judgeModel);
|
|
113
|
+
if (candidatePrices.some((pricing) => pricing === undefined) || judgePrice === undefined) {
|
|
114
|
+
return {
|
|
115
|
+
callCount: manifest.expectedCallCount,
|
|
116
|
+
pricingKnown: false
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
const candidateCost = candidatePrices.reduce((total, pricing) => total + manifest.caseCount * maximumCallCost(pricing, manifest.maxOutputTokens), 0);
|
|
120
|
+
const judgeCost = manifest.caseCount *
|
|
121
|
+
manifest.candidateModels.length *
|
|
122
|
+
maximumCallCost(judgePrice, manifest.maxOutputTokens);
|
|
123
|
+
return {
|
|
124
|
+
callCount: manifest.expectedCallCount,
|
|
125
|
+
maximumCostUsd: candidateCost + judgeCost,
|
|
126
|
+
pricingKnown: true
|
|
127
|
+
};
|
|
128
|
+
};
|
|
129
|
+
const enforceSpendLimit = (request, manifest) => {
|
|
130
|
+
if (request.spendLimitUsd === undefined)
|
|
131
|
+
return Effect.void;
|
|
132
|
+
const estimate = estimateManifest(manifest);
|
|
133
|
+
if (!estimate.pricingKnown || estimate.maximumCostUsd === undefined) {
|
|
134
|
+
return Effect.fail(new EvalServiceSpendLimitError({
|
|
135
|
+
operation: "enforce the comparison spend limit",
|
|
136
|
+
detail: "RouteKit Eval cannot enforce spendLimitUsd because pricing is unknown for one or more manifest models."
|
|
137
|
+
}));
|
|
138
|
+
}
|
|
139
|
+
if (estimate.maximumCostUsd > request.spendLimitUsd) {
|
|
140
|
+
return Effect.fail(new EvalServiceSpendLimitError({
|
|
141
|
+
operation: "enforce the comparison spend limit",
|
|
142
|
+
detail: `RouteKit Eval maximum estimated cost $${estimate.maximumCostUsd.toFixed(6)} ` +
|
|
143
|
+
`exceeds spendLimitUsd $${request.spendLimitUsd.toFixed(6)}.`
|
|
144
|
+
}));
|
|
145
|
+
}
|
|
146
|
+
return Effect.void;
|
|
147
|
+
};
|
|
148
|
+
const loadExecutionManifest = Effect.fn("EvalService.loadExecutionManifest")(function* (validation, request) {
|
|
149
|
+
const fs = yield* FileSystem.FileSystem;
|
|
150
|
+
const paths = yield* Path.Path;
|
|
151
|
+
const manifests = yield* fs
|
|
152
|
+
.glob("**/routekit.eval-manifest.json", {
|
|
153
|
+
root: validation.workingDirectory
|
|
154
|
+
})
|
|
155
|
+
.pipe(Effect.mapError((cause) => new EvalServiceValidationError({
|
|
156
|
+
operation: "discover the comparison manifest",
|
|
157
|
+
detail: detailOf(cause),
|
|
158
|
+
cause
|
|
159
|
+
})));
|
|
160
|
+
if (manifests.length !== 1 || manifests[0] === undefined) {
|
|
161
|
+
return yield* new EvalServiceValidationError({
|
|
162
|
+
operation: "discover the comparison manifest",
|
|
163
|
+
detail: "comparison requires exactly one routekit.eval-manifest.json"
|
|
164
|
+
});
|
|
165
|
+
}
|
|
166
|
+
const manifestPath = paths.isAbsolute(manifests[0])
|
|
167
|
+
? manifests[0]
|
|
168
|
+
: paths.join(validation.workingDirectory, manifests[0]);
|
|
169
|
+
const raw = yield* fs.readFileString(manifestPath).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
|
|
170
|
+
operation: "read the comparison manifest",
|
|
171
|
+
detail: detailOf(cause),
|
|
172
|
+
cause
|
|
173
|
+
})));
|
|
174
|
+
const json = yield* Effect.try({
|
|
175
|
+
try: () => JSON.parse(raw),
|
|
176
|
+
catch: (cause) => new EvalServiceValidationError({
|
|
177
|
+
operation: "parse the comparison manifest",
|
|
178
|
+
detail: "comparison manifest is not JSON",
|
|
179
|
+
cause
|
|
180
|
+
})
|
|
181
|
+
});
|
|
182
|
+
const manifest = yield* Schema.decodeUnknownEffect(EvalRunManifest)(json).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
|
|
183
|
+
operation: "decode the comparison manifest",
|
|
184
|
+
detail: "comparison manifest is invalid",
|
|
185
|
+
cause
|
|
186
|
+
})));
|
|
187
|
+
if (manifest.profileId !== request.profileId ||
|
|
188
|
+
!sameStrings(manifest.candidateModels, request.candidateModels) ||
|
|
189
|
+
manifest.judgeModel !== request.judgeModel) {
|
|
190
|
+
return yield* new EvalServiceValidationError({
|
|
191
|
+
operation: "bind the comparison manifest",
|
|
192
|
+
detail: "comparison request profile or models do not match the authoritative manifest"
|
|
193
|
+
});
|
|
194
|
+
}
|
|
195
|
+
if (manifest.profileId.trim().length === 0 ||
|
|
196
|
+
manifest.caseCount < 1 ||
|
|
197
|
+
manifest.caseIds.length !== manifest.caseCount ||
|
|
198
|
+
new Set(manifest.caseIds).size !== manifest.caseIds.length ||
|
|
199
|
+
manifest.caseIds.some((caseId) => caseId.trim().length === 0)) {
|
|
200
|
+
return yield* new EvalServiceValidationError({
|
|
201
|
+
operation: "validate the comparison manifest cases",
|
|
202
|
+
detail: "comparison manifest case identities are incomplete or duplicated"
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
const expectedCallCount = manifest.caseCount * manifest.candidateModels.length * 2;
|
|
206
|
+
if (manifest.expectedCallCount !== expectedCallCount || manifest.maxOutputTokens < 1) {
|
|
207
|
+
return yield* new EvalServiceValidationError({
|
|
208
|
+
operation: "validate the comparison manifest limits",
|
|
209
|
+
detail: "comparison manifest call or output-token limits are inconsistent"
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
return {
|
|
213
|
+
suiteDigest: validation.suiteDigest,
|
|
214
|
+
manifest
|
|
215
|
+
};
|
|
216
|
+
});
|
|
217
|
+
export const makeEvalService = (configuration = {}) => Effect.gen(function* () {
|
|
218
|
+
const engine = yield* EvalEngine;
|
|
219
|
+
const fs = yield* FileSystem.FileSystem;
|
|
220
|
+
const paths = yield* Path.Path;
|
|
221
|
+
const inspectRaw = Effect.fn("EvalService.inspectRaw")(function* (request) {
|
|
222
|
+
const validation = yield* engine.validate(request.suitePath);
|
|
223
|
+
return yield* loadExecutionManifest(validation, request).pipe(Effect.provideService(FileSystem.FileSystem, fs), Effect.provideService(Path.Path, paths));
|
|
224
|
+
});
|
|
225
|
+
const validate = Effect.fn("EvalService.validate")(function* (suitePath) {
|
|
226
|
+
yield* engine.validate(suitePath).pipe(Effect.mapError((cause) => new EvalServiceValidationError({
|
|
227
|
+
operation: "validate the comparison suite",
|
|
228
|
+
detail: detailOf(cause),
|
|
229
|
+
cause
|
|
230
|
+
})));
|
|
231
|
+
});
|
|
232
|
+
const inspect = (request) => inspectRaw(request).pipe(Effect.mapError((cause) => cause instanceof EvalServiceValidationError
|
|
233
|
+
? cause
|
|
234
|
+
: new EvalServiceValidationError({
|
|
235
|
+
operation: "inspect the comparison manifest",
|
|
236
|
+
detail: detailOf(cause),
|
|
237
|
+
cause
|
|
238
|
+
})));
|
|
239
|
+
const estimate = (request) => inspectRaw(request).pipe(Effect.map((inspection) => estimateManifest(inspection.manifest)), Effect.mapError((cause) => new EvalServiceEstimateError({
|
|
240
|
+
operation: "estimate the comparison",
|
|
241
|
+
detail: detailOf(cause),
|
|
242
|
+
cause
|
|
243
|
+
})));
|
|
244
|
+
const runInspectedComparison = (request, inspection) => enforceSpendLimit(request, inspection.manifest).pipe(Effect.flatMap(() => engine
|
|
245
|
+
.runComparison({
|
|
246
|
+
...request,
|
|
247
|
+
expectedCaseIds: [...inspection.manifest.caseIds],
|
|
248
|
+
expectedCallCount: inspection.manifest.expectedCallCount,
|
|
249
|
+
maxOutputTokens: inspection.manifest.maxOutputTokens,
|
|
250
|
+
suiteDigest: inspection.suiteDigest
|
|
251
|
+
})
|
|
252
|
+
.pipe(Effect.mapError((cause) => new EvalServiceComparisonError({
|
|
253
|
+
operation: "run the comparison",
|
|
254
|
+
detail: detailOf(cause),
|
|
255
|
+
cause
|
|
256
|
+
})))));
|
|
257
|
+
const runComparison = (request) => inspect(request).pipe(Effect.flatMap((inspection) => runInspectedComparison(request, inspection)));
|
|
258
|
+
const qualifyDimensionMatrix = (input) => Effect.gen(function* () {
|
|
259
|
+
const validated = yield* Effect.try({
|
|
260
|
+
try: () => validateDimensionMatrixInput(configuration, input),
|
|
261
|
+
catch: (cause) => new EvalServiceConfigurationError({
|
|
262
|
+
operation: "validate the dimension matrix qualification",
|
|
263
|
+
detail: detailOf(cause),
|
|
264
|
+
cause
|
|
265
|
+
})
|
|
266
|
+
});
|
|
267
|
+
const inspections = new Map();
|
|
268
|
+
for (const dimension of input.basis.dimensions) {
|
|
269
|
+
const suite = validated.suites.get(dimension.id);
|
|
270
|
+
const request = comparisonRequest(configuration, validated.gatewayUrl, input, suite);
|
|
271
|
+
const inspection = yield* inspect(request);
|
|
272
|
+
if (inspection.manifest.profileId !== dimension.id ||
|
|
273
|
+
inspection.manifest.judgeModel !== input.judgeModel ||
|
|
274
|
+
!sameStrings(inspection.manifest.candidateModels, input.candidateModels)) {
|
|
275
|
+
return yield* new EvalServiceValidationError({
|
|
276
|
+
operation: "inspect the full comparison manifest",
|
|
277
|
+
detail: `authoritative manifest does not match dimension ${JSON.stringify(dimension.id)}`
|
|
278
|
+
});
|
|
279
|
+
}
|
|
280
|
+
inspections.set(dimension.id, inspection);
|
|
281
|
+
}
|
|
282
|
+
const comparisons = [];
|
|
283
|
+
const evidenceInputs = [];
|
|
284
|
+
for (const dimension of input.basis.dimensions) {
|
|
285
|
+
const suite = validated.suites.get(dimension.id);
|
|
286
|
+
const inspection = inspections.get(dimension.id);
|
|
287
|
+
const request = comparisonRequest(configuration, validated.gatewayUrl, input, suite);
|
|
288
|
+
const comparison = yield* runInspectedComparison(request, inspection);
|
|
289
|
+
if (comparison.profileId !== dimension.id ||
|
|
290
|
+
comparison.judgeModel !== input.judgeModel ||
|
|
291
|
+
comparison.suiteDigest !== inspection.suiteDigest) {
|
|
292
|
+
return yield* new EvalServiceComparisonError({
|
|
293
|
+
operation: "validate the full comparison",
|
|
294
|
+
detail: `comparison identity does not match dimension ${JSON.stringify(dimension.id)}`
|
|
295
|
+
});
|
|
296
|
+
}
|
|
297
|
+
comparisons.push(comparison);
|
|
298
|
+
evidenceInputs.push({
|
|
299
|
+
dimensionId: dimension.id,
|
|
300
|
+
suiteDigest: inspection.suiteDigest,
|
|
301
|
+
judgeModel: inspection.manifest.judgeModel,
|
|
302
|
+
expectedCaseIds: inspection.manifest.caseIds,
|
|
303
|
+
comparison
|
|
304
|
+
});
|
|
305
|
+
}
|
|
306
|
+
const compiled = yield* Effect.try({
|
|
307
|
+
try: () => compileDimensionEvidenceMatrix({
|
|
308
|
+
basis: input.basis,
|
|
309
|
+
candidateModels: input.candidateModels,
|
|
310
|
+
comparisons: evidenceInputs
|
|
311
|
+
}),
|
|
312
|
+
catch: (cause) => new EvalServicePolicyError({
|
|
313
|
+
operation: "compile the dimension evidence matrix",
|
|
314
|
+
detail: detailOf(cause),
|
|
315
|
+
cause
|
|
316
|
+
})
|
|
317
|
+
});
|
|
318
|
+
const snapshot = yield* makeRoutingActivationStore(validated.snapshotRoot)
|
|
319
|
+
.publish({
|
|
320
|
+
basisDigest: input.basis.basisDigest,
|
|
321
|
+
evidenceDigest: compiled.evidenceDigest,
|
|
322
|
+
classifierModel: input.classifierModel,
|
|
323
|
+
objective: input.objective,
|
|
324
|
+
maximumUnknownWeight: input.maximumUnknownWeight,
|
|
325
|
+
...(input.constraints === undefined ? {} : { constraints: input.constraints }),
|
|
326
|
+
dimensions: [...input.basis.dimensions],
|
|
327
|
+
candidateModels: [...input.candidateModels],
|
|
328
|
+
evidence: [...compiled.evidence]
|
|
329
|
+
})
|
|
330
|
+
.pipe(Effect.provideService(FileSystem.FileSystem, fs), Effect.provideService(Path.Path, paths), Effect.mapError((cause) => new EvalServicePublicationError({
|
|
331
|
+
operation: "publish the dimension evidence snapshot",
|
|
332
|
+
detail: detailOf(cause),
|
|
333
|
+
cause
|
|
334
|
+
})));
|
|
335
|
+
return { comparisons, snapshot };
|
|
336
|
+
});
|
|
337
|
+
return EvalService.of({
|
|
338
|
+
validate,
|
|
339
|
+
inspect,
|
|
340
|
+
estimate,
|
|
341
|
+
runComparison,
|
|
342
|
+
qualifyDimensionMatrix
|
|
343
|
+
});
|
|
344
|
+
});
|
|
345
|
+
export const makeEvalServiceLayer = (configuration = {}) => Layer.effect(EvalService, makeEvalService(configuration));
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { test } from "node:test";
|
|
3
|
+
import { compileDimensionEvidenceMatrix, wilsonLowerBound95 } from "../dimension-evidence.js";
|
|
4
|
+
const dimensionIds = [
|
|
5
|
+
"gateway-protocol",
|
|
6
|
+
"eval-routing",
|
|
7
|
+
"account-pooling",
|
|
8
|
+
"typescript-maintenance",
|
|
9
|
+
"release-operations"
|
|
10
|
+
];
|
|
11
|
+
const models = ["openai/model-a", "openai/model-b"];
|
|
12
|
+
const caseIds = ["case-1", "case-2", "case-3", "case-4", "case-5"];
|
|
13
|
+
const judgeModel = "openai/judge";
|
|
14
|
+
function mutable(value) {
|
|
15
|
+
return structuredClone(value);
|
|
16
|
+
}
|
|
17
|
+
const basis = {
|
|
18
|
+
version: 2,
|
|
19
|
+
basisDigest: "definition-set-v2",
|
|
20
|
+
dimensions: dimensionIds.map((id) => ({
|
|
21
|
+
id,
|
|
22
|
+
description: `Requests about ${id}`,
|
|
23
|
+
includes: [`Tasks specifically involving ${id}`],
|
|
24
|
+
excludes: [`Tasks unrelated to ${id}`]
|
|
25
|
+
}))
|
|
26
|
+
};
|
|
27
|
+
function cases(options = {}) {
|
|
28
|
+
return caseIds.map((caseId, index) => {
|
|
29
|
+
const passed = !options.failed?.has(caseId);
|
|
30
|
+
return {
|
|
31
|
+
caseId,
|
|
32
|
+
outcome: passed ? "passed" : "failed",
|
|
33
|
+
measurement: {
|
|
34
|
+
judgeScore: passed ? 0.9 : 0.2,
|
|
35
|
+
...(!options.unpriced?.has(caseId) ? { costUsd: 0.01 + index / 1_000 } : {}),
|
|
36
|
+
...(!options.missingDuration?.has(caseId) ? { durationMs: 100 + index * 10 } : {}),
|
|
37
|
+
inputTokens: 100 + index,
|
|
38
|
+
outputTokens: 20 + index
|
|
39
|
+
}
|
|
40
|
+
};
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
function comparison(dimensionId) {
|
|
44
|
+
return {
|
|
45
|
+
version: 1,
|
|
46
|
+
comparisonId: `comparison-${dimensionId}`,
|
|
47
|
+
profileId: dimensionId,
|
|
48
|
+
suiteDigest: `suite-${dimensionId}`,
|
|
49
|
+
judgeModel,
|
|
50
|
+
startedAt: "2026-08-17T00:00:00.000Z",
|
|
51
|
+
finishedAt: "2026-08-17T00:01:00.000Z",
|
|
52
|
+
models: models.map((model) => ({ model, cases: cases() }))
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
function input() {
|
|
56
|
+
return mutable({
|
|
57
|
+
basis,
|
|
58
|
+
candidateModels: models,
|
|
59
|
+
comparisons: dimensionIds.map((dimensionId) => ({
|
|
60
|
+
dimensionId,
|
|
61
|
+
suiteDigest: `suite-${dimensionId}`,
|
|
62
|
+
judgeModel,
|
|
63
|
+
expectedCaseIds: caseIds,
|
|
64
|
+
comparison: comparison(dimensionId)
|
|
65
|
+
}))
|
|
66
|
+
});
|
|
67
|
+
}
|
|
68
|
+
test("complete comparison results compile into a deterministic model-dimension matrix", () => {
|
|
69
|
+
const first = compileDimensionEvidenceMatrix(input());
|
|
70
|
+
const secondInput = input();
|
|
71
|
+
secondInput.comparisons.reverse();
|
|
72
|
+
secondInput.comparisons.forEach((entry) => {
|
|
73
|
+
entry.comparison.models.reverse();
|
|
74
|
+
entry.comparison.models.forEach((model) => model.cases.reverse());
|
|
75
|
+
});
|
|
76
|
+
const second = compileDimensionEvidenceMatrix(secondInput);
|
|
77
|
+
assert.equal(first.evidence.length, dimensionIds.length * models.length);
|
|
78
|
+
assert.match(first.evidenceDigest, /^[a-f0-9]{64}$/u);
|
|
79
|
+
assert.equal(first.evidenceDigest, second.evidenceDigest);
|
|
80
|
+
assert.deepEqual(first.evidence, second.evidence);
|
|
81
|
+
const cell = first.evidence.find((entry) => entry.dimensionId === "gateway-protocol" && entry.model === "openai/model-a");
|
|
82
|
+
assert.equal(cell?.quality.sampleCount, 5);
|
|
83
|
+
assert.equal(cell?.quality.passRate, 1);
|
|
84
|
+
assert.ok((cell?.quality.lowerConfidenceBound ?? 0) > 0.56);
|
|
85
|
+
assert.ok((cell?.quality.lowerConfidenceBound ?? 1) < 0.57);
|
|
86
|
+
assert.equal(cell?.failureRate, 0);
|
|
87
|
+
assert.equal(cell?.averageJudgeScore, 0.9);
|
|
88
|
+
assert.equal(cell?.p95DurationMs, 140);
|
|
89
|
+
assert.equal(cell?.unpricedCalls, 0);
|
|
90
|
+
assert.equal(cell?.averageCostUsd, 0.012);
|
|
91
|
+
});
|
|
92
|
+
test("partial pricing is explicit and never converted into a known average", () => {
|
|
93
|
+
const value = input();
|
|
94
|
+
value.comparisons[0].comparison.models[0].cases = cases({
|
|
95
|
+
unpriced: new Set(["case-4", "case-5"])
|
|
96
|
+
});
|
|
97
|
+
const cell = compileDimensionEvidenceMatrix(value).evidence.find((entry) => entry.dimensionId === dimensionIds[0] && entry.model === models[0]);
|
|
98
|
+
assert.equal(cell?.unpricedCalls, 2);
|
|
99
|
+
assert.equal(cell?.averageCostUsd, undefined);
|
|
100
|
+
});
|
|
101
|
+
test("partial duration measurements do not produce a misleading percentile", () => {
|
|
102
|
+
const value = input();
|
|
103
|
+
value.comparisons[0].comparison.models[0].cases = cases({
|
|
104
|
+
missingDuration: new Set(["case-5"])
|
|
105
|
+
});
|
|
106
|
+
const cell = compileDimensionEvidenceMatrix(value).evidence.find((entry) => entry.dimensionId === dimensionIds[0] && entry.model === models[0]);
|
|
107
|
+
assert.equal(cell?.p95DurationMs, undefined);
|
|
108
|
+
});
|
|
109
|
+
test("matrix compilation rejects missing and duplicate dimensions or candidates", () => {
|
|
110
|
+
const missingDimension = input();
|
|
111
|
+
missingDimension.comparisons.pop();
|
|
112
|
+
assert.throws(() => compileDimensionEvidenceMatrix(missingDimension), /missing dimension/);
|
|
113
|
+
const duplicateDimension = input();
|
|
114
|
+
duplicateDimension.comparisons[1] = duplicateDimension.comparisons[0];
|
|
115
|
+
assert.throws(() => compileDimensionEvidenceMatrix(duplicateDimension), /duplicate dimension/);
|
|
116
|
+
const missingCandidate = input();
|
|
117
|
+
missingCandidate.comparisons[0].comparison.models.pop();
|
|
118
|
+
assert.throws(() => compileDimensionEvidenceMatrix(missingCandidate), /missing candidate/);
|
|
119
|
+
const unexpectedCandidate = input();
|
|
120
|
+
unexpectedCandidate.comparisons[0].comparison.models[0].model = "openai/unexpected";
|
|
121
|
+
assert.throws(() => compileDimensionEvidenceMatrix(unexpectedCandidate), /unexpected candidate/);
|
|
122
|
+
});
|
|
123
|
+
test("matrix compilation rejects incomplete or ambiguous case evidence", () => {
|
|
124
|
+
const missingCase = input();
|
|
125
|
+
missingCase.comparisons[0].comparison.models[0].cases.pop();
|
|
126
|
+
assert.throws(() => compileDimensionEvidenceMatrix(missingCase), /has 4 cases; expected 5/);
|
|
127
|
+
const duplicateCase = input();
|
|
128
|
+
duplicateCase.comparisons[0].comparison.models[0].cases[4] =
|
|
129
|
+
duplicateCase.comparisons[0].comparison.models[0].cases[0];
|
|
130
|
+
assert.throws(() => compileDimensionEvidenceMatrix(duplicateCase), /duplicate case/);
|
|
131
|
+
const missingJudgeScore = input();
|
|
132
|
+
delete missingJudgeScore.comparisons[0].comparison.models[0].cases[0].measurement.judgeScore;
|
|
133
|
+
assert.throws(() => compileDimensionEvidenceMatrix(missingJudgeScore), /missing a judge score/);
|
|
134
|
+
const cutoff = input();
|
|
135
|
+
cutoff.comparisons[0].comparison.models[0].cases[0].outcome = "cutoff";
|
|
136
|
+
assert.throws(() => compileDimensionEvidenceMatrix(cutoff), /non-terminal case/);
|
|
137
|
+
});
|
|
138
|
+
test("matrix compilation binds profile, suite digest, and judge", () => {
|
|
139
|
+
const wrongProfile = input();
|
|
140
|
+
wrongProfile.comparisons[0].comparison.profileId = "wrong";
|
|
141
|
+
assert.throws(() => compileDimensionEvidenceMatrix(wrongProfile), /does not match dimension/);
|
|
142
|
+
const wrongDigest = input();
|
|
143
|
+
wrongDigest.comparisons[0].comparison.suiteDigest = "wrong";
|
|
144
|
+
assert.throws(() => compileDimensionEvidenceMatrix(wrongDigest), /suite digest does not match/);
|
|
145
|
+
const wrongJudge = input();
|
|
146
|
+
wrongJudge.comparisons[0].comparison.judgeModel = "openai/wrong";
|
|
147
|
+
assert.throws(() => compileDimensionEvidenceMatrix(wrongJudge), /comparison judge does not match/);
|
|
148
|
+
});
|
|
149
|
+
test("Wilson lower confidence bounds are conservative and validate counts", () => {
|
|
150
|
+
assert.equal(wilsonLowerBound95(0, 5), 0);
|
|
151
|
+
assert.ok(wilsonLowerBound95(5, 5) < 1);
|
|
152
|
+
assert.ok(wilsonLowerBound95(19, 20) < 0.95);
|
|
153
|
+
assert.throws(() => wilsonLowerBound95(6, 5), /within the sample/);
|
|
154
|
+
assert.throws(() => wilsonLowerBound95(0, 0), /within the sample/);
|
|
155
|
+
});
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|