@velum-labs/routekit-eval-service 1.1.1 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/service.d.ts +2 -1
- package/dist/service.js +13 -0
- package/dist/test/production-runner.test.js +7 -2
- package/package.json +5 -5
package/dist/service.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { EvalComparisonRequest, EvalComparisonResult, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
|
|
1
|
+
import type { EvalComparisonRequest, EvalComparisonResult, EvalReasoningEffort, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import { EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
|
|
3
3
|
import { EvalEngine } from "@velum-labs/routekit-eval-engine";
|
|
4
4
|
import { Context, Effect, FileSystem, Layer, Path } from "effect";
|
|
@@ -29,6 +29,7 @@ export type DimensionMatrixSuite = {
|
|
|
29
29
|
export type DimensionMatrixQualificationInput = {
|
|
30
30
|
readonly basis: RoutingBasis;
|
|
31
31
|
readonly candidateModels: ReadonlyArray<string>;
|
|
32
|
+
readonly candidateReasoningEfforts?: Readonly<Record<string, EvalReasoningEffort>>;
|
|
32
33
|
readonly classifierModel: string;
|
|
33
34
|
readonly judgeModel: string;
|
|
34
35
|
readonly objective: RoutingObjectivePolicy;
|
package/dist/service.js
CHANGED
|
@@ -40,6 +40,12 @@ const validateConfiguration = (configuration) => {
|
|
|
40
40
|
};
|
|
41
41
|
};
|
|
42
42
|
const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
|
|
43
|
+
const sameReasoningEfforts = (left, right) => {
|
|
44
|
+
const leftEntries = Object.entries(left ?? {});
|
|
45
|
+
const rightEntries = Object.entries(right ?? {});
|
|
46
|
+
return (leftEntries.length === rightEntries.length &&
|
|
47
|
+
leftEntries.every(([model, effort]) => right?.[model] === effort));
|
|
48
|
+
};
|
|
43
49
|
const validateDimensionMatrixInput = (configuration, input) => {
|
|
44
50
|
const validatedConfiguration = validateConfiguration(configuration);
|
|
45
51
|
assertRoutingBasis(input.basis);
|
|
@@ -51,6 +57,9 @@ const validateDimensionMatrixInput = (configuration, input) => {
|
|
|
51
57
|
}
|
|
52
58
|
for (const model of input.candidateModels)
|
|
53
59
|
assertExplicitEvalModel(model, "candidate");
|
|
60
|
+
if (Object.keys(input.candidateReasoningEfforts ?? {}).some((model) => !input.candidateModels.includes(model))) {
|
|
61
|
+
throw new Error("dimension matrix reasoning efforts must belong to candidate models");
|
|
62
|
+
}
|
|
54
63
|
assertExplicitEvalModel(input.classifierModel, "classifier");
|
|
55
64
|
assertExplicitEvalModel(input.judgeModel, "judge");
|
|
56
65
|
const expectedDimensions = new Set(input.basis.dimensions.map((dimension) => dimension.id));
|
|
@@ -79,6 +88,9 @@ const comparisonRequest = (configuration, gatewayUrl, input, suite) => ({
|
|
|
79
88
|
profileId: suite.dimensionId,
|
|
80
89
|
suitePath: suite.suitePath,
|
|
81
90
|
candidateModels: [...input.candidateModels],
|
|
91
|
+
...(input.candidateReasoningEfforts === undefined
|
|
92
|
+
? {}
|
|
93
|
+
: { candidateReasoningEfforts: input.candidateReasoningEfforts }),
|
|
82
94
|
judgeModel: input.judgeModel,
|
|
83
95
|
gatewayUrl,
|
|
84
96
|
...(configuration.full?.concurrency === undefined
|
|
@@ -210,6 +222,7 @@ const loadExecutionManifest = Effect.fn("EvalService.loadExecutionManifest")(fun
|
|
|
210
222
|
})));
|
|
211
223
|
if (manifest.profileId !== request.profileId ||
|
|
212
224
|
!sameStrings(manifest.candidateModels, request.candidateModels) ||
|
|
225
|
+
!sameReasoningEfforts(manifest.candidateReasoningEfforts, request.candidateReasoningEfforts) ||
|
|
213
226
|
manifest.judgeModel !== request.judgeModel) {
|
|
214
227
|
return yield* new EvalServiceValidationError({
|
|
215
228
|
operation: "bind the comparison manifest",
|
|
@@ -194,7 +194,7 @@ test("production runner executes candidate and judge traffic through the live en
|
|
|
194
194
|
'import { setupAgent, setupJudge } from "routekit/eval";',
|
|
195
195
|
'const judge = setupJudge({ agent: setupAgent({ model: "openai/judge" }), minScore: 0.8 });',
|
|
196
196
|
'test("support case", async () => {',
|
|
197
|
-
' const run = await setupAgent({ model: "openai/cheap" }).run({ prompt: "Help", caseId: "support-case" });',
|
|
197
|
+
' const run = await setupAgent({ model: "openai/cheap", parameters: { reasoning: { effort: "low" } } }).run({ prompt: "Help", caseId: "support-case" });',
|
|
198
198
|
" run.toComplete();",
|
|
199
199
|
' await judge.autoEvals({ criteria: "Helpful", prompt: "Help", run });',
|
|
200
200
|
"});"
|
|
@@ -208,7 +208,8 @@ test("production runner executes candidate and judge traffic through the live en
|
|
|
208
208
|
calls.push({
|
|
209
209
|
authorization: incoming.headers.authorization,
|
|
210
210
|
maxOutputTokens: body.max_completion_tokens,
|
|
211
|
-
model: body.model
|
|
211
|
+
model: body.model,
|
|
212
|
+
reasoningEffort: body.reasoning_effort
|
|
212
213
|
});
|
|
213
214
|
const content = body.model === "openai/judge"
|
|
214
215
|
? JSON.stringify({ pass: true, reason: "helpful", score: 0.9 })
|
|
@@ -245,6 +246,10 @@ test("production runner executes candidate and judge traffic through the live en
|
|
|
245
246
|
assert.deepEqual(calls.map(({ model }) => model), ["openai/cheap", "openai/judge"]);
|
|
246
247
|
assert.equal(calls.every(({ authorization }) => authorization === "Bearer parent-only-token"), true);
|
|
247
248
|
assert.equal(calls.every(({ maxOutputTokens }) => maxOutputTokens === 1_024), true);
|
|
249
|
+
assert.deepEqual(calls.map(({ model, reasoningEffort }) => ({ model, reasoningEffort })), [
|
|
250
|
+
{ model: "openai/cheap", reasoningEffort: "low" },
|
|
251
|
+
{ model: "openai/judge", reasoningEffort: undefined }
|
|
252
|
+
]);
|
|
248
253
|
assert.deepEqual(observed, [
|
|
249
254
|
{ phase: "issued", role: "candidate" },
|
|
250
255
|
{
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-service",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.3.0",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -32,10 +32,10 @@
|
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"@effect/platform-node": "4.0.0-rc.108",
|
|
34
34
|
"effect": "4.0.0-rc.108",
|
|
35
|
-
"@velum-labs/routekit-eval-contracts": "1.
|
|
36
|
-
"@velum-labs/routekit-eval-engine": "1.
|
|
37
|
-
"@velum-labs/routekit-eval-store": "1.
|
|
38
|
-
"@velum-labs/routekit-registry": "1.
|
|
35
|
+
"@velum-labs/routekit-eval-contracts": "1.3.0",
|
|
36
|
+
"@velum-labs/routekit-eval-engine": "1.3.0",
|
|
37
|
+
"@velum-labs/routekit-eval-store": "1.3.0",
|
|
38
|
+
"@velum-labs/routekit-registry": "1.3.0"
|
|
39
39
|
},
|
|
40
40
|
"devDependencies": {
|
|
41
41
|
"typescript": "7.0.2"
|