@velum-labs/routekit-eval-service 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/service.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- import type { EvalComparisonRequest, EvalComparisonResult, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
1
+ import type { EvalComparisonRequest, EvalComparisonResult, EvalReasoningEffort, PublishedRoutingActivation, RoutingActivationConstraints, RoutingBasis, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
2
2
  import { EvalRunManifest } from "@velum-labs/routekit-eval-contracts";
3
3
  import { EvalEngine } from "@velum-labs/routekit-eval-engine";
4
4
  import { Context, Effect, FileSystem, Layer, Path } from "effect";
@@ -29,6 +29,7 @@ export type DimensionMatrixSuite = {
29
29
  export type DimensionMatrixQualificationInput = {
30
30
  readonly basis: RoutingBasis;
31
31
  readonly candidateModels: ReadonlyArray<string>;
32
+ readonly candidateReasoningEfforts?: Readonly<Record<string, EvalReasoningEffort>>;
32
33
  readonly classifierModel: string;
33
34
  readonly judgeModel: string;
34
35
  readonly objective: RoutingObjectivePolicy;
package/dist/service.js CHANGED
@@ -40,6 +40,12 @@ const validateConfiguration = (configuration) => {
40
40
  };
41
41
  };
42
42
  const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
43
+ const sameReasoningEfforts = (left, right) => {
44
+ const leftEntries = Object.entries(left ?? {});
45
+ const rightEntries = Object.entries(right ?? {});
46
+ return (leftEntries.length === rightEntries.length &&
47
+ leftEntries.every(([model, effort]) => right?.[model] === effort));
48
+ };
43
49
  const validateDimensionMatrixInput = (configuration, input) => {
44
50
  const validatedConfiguration = validateConfiguration(configuration);
45
51
  assertRoutingBasis(input.basis);
@@ -51,6 +57,9 @@ const validateDimensionMatrixInput = (configuration, input) => {
51
57
  }
52
58
  for (const model of input.candidateModels)
53
59
  assertExplicitEvalModel(model, "candidate");
60
+ if (Object.keys(input.candidateReasoningEfforts ?? {}).some((model) => !input.candidateModels.includes(model))) {
61
+ throw new Error("dimension matrix reasoning efforts must belong to candidate models");
62
+ }
54
63
  assertExplicitEvalModel(input.classifierModel, "classifier");
55
64
  assertExplicitEvalModel(input.judgeModel, "judge");
56
65
  const expectedDimensions = new Set(input.basis.dimensions.map((dimension) => dimension.id));
@@ -79,6 +88,9 @@ const comparisonRequest = (configuration, gatewayUrl, input, suite) => ({
79
88
  profileId: suite.dimensionId,
80
89
  suitePath: suite.suitePath,
81
90
  candidateModels: [...input.candidateModels],
91
+ ...(input.candidateReasoningEfforts === undefined
92
+ ? {}
93
+ : { candidateReasoningEfforts: input.candidateReasoningEfforts }),
82
94
  judgeModel: input.judgeModel,
83
95
  gatewayUrl,
84
96
  ...(configuration.full?.concurrency === undefined
@@ -210,6 +222,7 @@ const loadExecutionManifest = Effect.fn("EvalService.loadExecutionManifest")(fun
210
222
  })));
211
223
  if (manifest.profileId !== request.profileId ||
212
224
  !sameStrings(manifest.candidateModels, request.candidateModels) ||
225
+ !sameReasoningEfforts(manifest.candidateReasoningEfforts, request.candidateReasoningEfforts) ||
213
226
  manifest.judgeModel !== request.judgeModel) {
214
227
  return yield* new EvalServiceValidationError({
215
228
  operation: "bind the comparison manifest",
@@ -194,7 +194,7 @@ test("production runner executes candidate and judge traffic through the live en
194
194
  'import { setupAgent, setupJudge } from "routekit/eval";',
195
195
  'const judge = setupJudge({ agent: setupAgent({ model: "openai/judge" }), minScore: 0.8 });',
196
196
  'test("support case", async () => {',
197
- ' const run = await setupAgent({ model: "openai/cheap" }).run({ prompt: "Help", caseId: "support-case" });',
197
+ ' const run = await setupAgent({ model: "openai/cheap", parameters: { reasoning: { effort: "low" } } }).run({ prompt: "Help", caseId: "support-case" });',
198
198
  " run.toComplete();",
199
199
  ' await judge.autoEvals({ criteria: "Helpful", prompt: "Help", run });',
200
200
  "});"
@@ -208,7 +208,8 @@ test("production runner executes candidate and judge traffic through the live en
208
208
  calls.push({
209
209
  authorization: incoming.headers.authorization,
210
210
  maxOutputTokens: body.max_completion_tokens,
211
- model: body.model
211
+ model: body.model,
212
+ reasoningEffort: body.reasoning_effort
212
213
  });
213
214
  const content = body.model === "openai/judge"
214
215
  ? JSON.stringify({ pass: true, reason: "helpful", score: 0.9 })
@@ -245,6 +246,10 @@ test("production runner executes candidate and judge traffic through the live en
245
246
  assert.deepEqual(calls.map(({ model }) => model), ["openai/cheap", "openai/judge"]);
246
247
  assert.equal(calls.every(({ authorization }) => authorization === "Bearer parent-only-token"), true);
247
248
  assert.equal(calls.every(({ maxOutputTokens }) => maxOutputTokens === 1_024), true);
249
+ assert.deepEqual(calls.map(({ model, reasoningEffort }) => ({ model, reasoningEffort })), [
250
+ { model: "openai/cheap", reasoningEffort: "low" },
251
+ { model: "openai/judge", reasoningEffort: undefined }
252
+ ]);
248
253
  assert.deepEqual(observed, [
249
254
  { phase: "issued", role: "candidate" },
250
255
  {
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-eval-service",
3
3
  "private": false,
4
- "version": "1.2.0",
4
+ "version": "1.3.1",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -32,10 +32,10 @@
32
32
  "dependencies": {
33
33
  "@effect/platform-node": "4.0.0-rc.108",
34
34
  "effect": "4.0.0-rc.108",
35
- "@velum-labs/routekit-eval-contracts": "1.2.0",
36
- "@velum-labs/routekit-eval-engine": "1.2.0",
37
- "@velum-labs/routekit-eval-store": "1.2.0",
38
- "@velum-labs/routekit-registry": "1.2.0"
35
+ "@velum-labs/routekit-eval-contracts": "1.3.1",
36
+ "@velum-labs/routekit-eval-engine": "1.3.1",
37
+ "@velum-labs/routekit-eval-store": "1.3.1",
38
+ "@velum-labs/routekit-registry": "1.3.1"
39
39
  },
40
40
  "devDependencies": {
41
41
  "typescript": "7.0.2"