@velum-labs/routekit-eval-setup 1.1.1 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,18 +1,19 @@
1
1
  import { Context, Effect, Layer, Path } from "effect";
2
2
  import type { EvalProjectArtifactError, EvalSetupInspectionError } from "./errors.js";
3
- import { EvalDimensionContrastError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
3
+ import { EvalDimensionContrastError, type EvalModelCatalogError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
4
4
  import { EvalRepositoryInspector } from "./inspection.js";
5
5
  import { EvalProjectArtifacts } from "./project-artifacts.js";
6
6
  import type { EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProposedDimension, EvalProjectStatus, EvalRunReport } from "./project-contracts.js";
7
7
  import { EvalProjectStore } from "./project-store.js";
8
- export type EvalProjectWorkflowError = EvalProjectStoreError | EvalProjectArtifactError | EvalSetupInspectionError | EvalDimensionContrastError | EvalProjectTransitionError;
8
+ import { EvalModelCatalog } from "./services/model-catalog/service.js";
9
+ export type EvalProjectWorkflowError = EvalProjectStoreError | EvalProjectArtifactError | EvalSetupInspectionError | EvalModelCatalogError | EvalDimensionContrastError | EvalProjectTransitionError;
9
10
  export type EvalProjectWorkflowShape = {
10
11
  readonly setup: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
11
12
  readonly status: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus | undefined, EvalProjectStoreError | EvalProjectArtifactError>;
12
- readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError>;
13
+ readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError | EvalModelCatalogError>;
13
14
  readonly proposeDimensions: (repositoryRoot: string, dimensions: readonly EvalProposedDimension[]) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
14
15
  readonly approveDimensions: (repositoryRoot: string, basisDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
15
- readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
16
+ readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "candidateReasoningEfforts" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
16
17
  readonly approveEvaluations: (repositoryRoot: string, evaluationDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
17
18
  readonly createPlan: (repositoryRoot: string, scope: EvalPlanScope) => Effect.Effect<EvalExecutionPlan, EvalProjectWorkflowError>;
18
19
  readonly startRun: (repositoryRoot: string, planId: string) => Effect.Effect<{
@@ -27,6 +28,6 @@ export type EvalProjectWorkflowShape = {
27
28
  declare const EvalProjectWorkflow_base: Context.ServiceClass<EvalProjectWorkflow, "@velum-labs/routekit-eval-setup/EvalProjectWorkflow", EvalProjectWorkflowShape>;
28
29
  export declare class EvalProjectWorkflow extends EvalProjectWorkflow_base {
29
30
  }
30
- export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
31
- export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
31
+ export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
32
+ export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
32
33
  export {};
@@ -3,9 +3,11 @@ import { RoutingActivationConstraints, RoutingObjectivePolicy } from "@velum-lab
3
3
  import { Clock, Context, Effect, Layer, Path, Schema } from "effect";
4
4
  import { EvalDimensionContrastError, EvalProjectTransitionError } from "./errors.js";
5
5
  import { EvalRepositoryInspector } from "./inspection.js";
6
+ import { candidateReasoningEffortFromAnswer } from "./model-selection.js";
6
7
  import { EvalProjectArtifacts, evaluationProposalDigest, routingBasisDigest } from "./project-artifacts.js";
7
8
  import { EVAL_PROJECT_VERSION, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
8
9
  import { EvalProjectStore } from "./project-store.js";
10
+ import { EvalModelCatalog } from "./services/model-catalog/service.js";
9
11
  const isoNow = Effect.map(Clock.currentTimeMillis, (millis) => new Date(millis).toISOString());
10
12
  const questionForProgress = (progress) => {
11
13
  switch (progress._tag) {
@@ -21,6 +23,12 @@ const questionForProgress = (progress) => {
21
23
  prompt: "Which explicit provider/model IDs may RouteKit route to?",
22
24
  options: []
23
25
  };
26
+ case "CandidateReasoningEffortsRequired":
27
+ return {
28
+ id: "candidate-reasoning-efforts",
29
+ prompt: `Choose an advertised reasoning effort for ${progress.currentCandidateModel}.`,
30
+ options: [...progress.supportedReasoningEfforts[progress.currentCandidateModel]]
31
+ };
24
32
  case "ClassifierModelRequired":
25
33
  return {
26
34
  id: "classifier-model",
@@ -124,7 +132,24 @@ const parseModel = (state, answer, role) => Effect.gen(function* () {
124
132
  const model = yield* nonEmptyAnswer(state, answer);
125
133
  return yield* validateModel(state, model, role);
126
134
  });
135
+ const supportedCandidateReasoningEfforts = (candidates, input) => {
136
+ const supported = {};
137
+ for (const model of candidates) {
138
+ const efforts = [...new Set(input[model] ?? [])];
139
+ if (efforts.length > 0)
140
+ supported[model] = efforts;
141
+ }
142
+ return supported;
143
+ };
144
+ const firstReasoningCandidate = (candidates, supported) => candidates.find((model) => supported[model] !== undefined);
145
+ const nextReasoningCandidate = (candidates, supported, selected) => candidates.find((model) => supported[model] !== undefined && selected[model] === undefined);
127
146
  const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
147
+ const sameReasoningEfforts = (left, right) => {
148
+ const leftEntries = Object.entries(left ?? {});
149
+ const rightEntries = Object.entries(right ?? {});
150
+ return (leftEntries.length === rightEntries.length &&
151
+ leftEntries.every(([model, effort]) => right?.[model] === effort));
152
+ };
128
153
  const sameLedger = (left, right) => left.expectedCalls === right.expectedCalls &&
129
154
  left.observedCalls === right.observedCalls &&
130
155
  left.observedCandidateRows === right.observedCandidateRows &&
@@ -300,6 +325,7 @@ const parseObjective = (state, answer) => Effect.gen(function* () {
300
325
  const configurationFrom = (progress, constraints) => ({
301
326
  workloadDescription: progress.workloadDescription,
302
327
  candidateModels: progress.candidateModels,
328
+ candidateReasoningEfforts: progress.candidateReasoningEfforts ?? {},
303
329
  classifierModel: progress.classifierModel,
304
330
  authorModel: progress.authorModel,
305
331
  judgeModel: progress.judgeModel,
@@ -324,7 +350,7 @@ const parseConstraints = (state, answer) => Effect.gen(function* () {
324
350
  const constraints = yield* Schema.decodeUnknownEffect(RoutingActivationConstraints)(json).pipe(Effect.mapError(() => transitionError(state.stage, "routing constraints contain invalid dimension quality or failure-rate values")));
325
351
  return Object.keys(constraints).length === 0 ? undefined : constraints;
326
352
  });
327
- const advanceSetup = (state, answer, now) => Effect.gen(function* () {
353
+ const advanceSetup = (state, answer, now, discoverReasoningEfforts) => Effect.gen(function* () {
328
354
  const common = {
329
355
  version: state.version,
330
356
  projectId: state.projectId,
@@ -343,16 +369,58 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
343
369
  workloadDescription: yield* nonEmptyAnswer(state, answer)
344
370
  }
345
371
  };
346
- case "CandidateModelsRequired":
372
+ case "CandidateModelsRequired": {
373
+ const candidateModels = yield* parseCandidateModels(state, answer);
374
+ const supportedReasoningEfforts = supportedCandidateReasoningEfforts(candidateModels, yield* discoverReasoningEfforts(candidateModels));
375
+ const currentCandidateModel = firstReasoningCandidate(candidateModels, supportedReasoningEfforts);
347
376
  return {
348
377
  ...common,
349
378
  stage: "setup-required",
350
- progress: {
351
- _tag: "ClassifierModelRequired",
352
- workloadDescription: state.progress.workloadDescription,
353
- candidateModels: yield* parseCandidateModels(state, answer)
354
- }
379
+ progress: currentCandidateModel === undefined
380
+ ? {
381
+ _tag: "ClassifierModelRequired",
382
+ workloadDescription: state.progress.workloadDescription,
383
+ candidateModels,
384
+ candidateReasoningEfforts: {}
385
+ }
386
+ : {
387
+ _tag: "CandidateReasoningEffortsRequired",
388
+ workloadDescription: state.progress.workloadDescription,
389
+ candidateModels,
390
+ candidateReasoningEfforts: {},
391
+ currentCandidateModel,
392
+ supportedReasoningEfforts
393
+ }
355
394
  };
395
+ }
396
+ case "CandidateReasoningEffortsRequired": {
397
+ const supportedReasoningEfforts = state.progress.supportedReasoningEfforts;
398
+ const currentCandidateModel = state.progress.currentCandidateModel;
399
+ const candidateReasoningEfforts = {
400
+ ...state.progress.candidateReasoningEfforts,
401
+ [currentCandidateModel]: yield* Effect.try({
402
+ try: () => candidateReasoningEffortFromAnswer(answer, supportedReasoningEfforts[currentCandidateModel]),
403
+ catch: (cause) => transitionError(state.stage, cause instanceof Error ? cause.message : String(cause))
404
+ })
405
+ };
406
+ const nextCandidateModel = nextReasoningCandidate(state.progress.candidateModels, supportedReasoningEfforts, candidateReasoningEfforts);
407
+ return {
408
+ ...common,
409
+ stage: "setup-required",
410
+ progress: nextCandidateModel === undefined
411
+ ? {
412
+ _tag: "ClassifierModelRequired",
413
+ workloadDescription: state.progress.workloadDescription,
414
+ candidateModels: state.progress.candidateModels,
415
+ candidateReasoningEfforts
416
+ }
417
+ : {
418
+ ...state.progress,
419
+ candidateReasoningEfforts,
420
+ currentCandidateModel: nextCandidateModel
421
+ }
422
+ };
423
+ }
356
424
  case "ClassifierModelRequired":
357
425
  return {
358
426
  ...common,
@@ -361,6 +429,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
361
429
  _tag: "AuthorModelRequired",
362
430
  workloadDescription: state.progress.workloadDescription,
363
431
  candidateModels: state.progress.candidateModels,
432
+ candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
364
433
  classifierModel: yield* parseModel(state, answer, "classifier")
365
434
  }
366
435
  };
@@ -372,6 +441,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
372
441
  _tag: "JudgeModelRequired",
373
442
  workloadDescription: state.progress.workloadDescription,
374
443
  candidateModels: state.progress.candidateModels,
444
+ candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
375
445
  classifierModel: state.progress.classifierModel,
376
446
  authorModel: yield* parseModel(state, answer, "author")
377
447
  }
@@ -384,6 +454,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
384
454
  _tag: "RoutingObjectiveRequired",
385
455
  workloadDescription: state.progress.workloadDescription,
386
456
  candidateModels: state.progress.candidateModels,
457
+ candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
387
458
  classifierModel: state.progress.classifierModel,
388
459
  authorModel: state.progress.authorModel,
389
460
  judgeModel: yield* parseModel(state, answer, "judge")
@@ -398,6 +469,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
398
469
  _tag: "MaximumUnknownWeightRequired",
399
470
  workloadDescription: state.progress.workloadDescription,
400
471
  candidateModels: state.progress.candidateModels,
472
+ candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
401
473
  classifierModel: state.progress.classifierModel,
402
474
  authorModel: state.progress.authorModel,
403
475
  judgeModel: state.progress.judgeModel,
@@ -413,6 +485,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
413
485
  _tag: "RoutingConstraintsRequired",
414
486
  workloadDescription: state.progress.workloadDescription,
415
487
  candidateModels: state.progress.candidateModels,
488
+ candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
416
489
  classifierModel: state.progress.classifierModel,
417
490
  authorModel: state.progress.authorModel,
418
491
  judgeModel: state.progress.judgeModel,
@@ -438,6 +511,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
438
511
  const store = yield* EvalProjectStore;
439
512
  const artifacts = yield* EvalProjectArtifacts;
440
513
  const inspector = yield* EvalRepositoryInspector;
514
+ const modelCatalog = yield* EvalModelCatalog;
441
515
  const paths = yield* Path.Path;
442
516
  const resolveRoot = (repositoryRoot) => paths.resolve(repositoryRoot);
443
517
  const statusOf = (root, state) => Effect.gen(function* () {
@@ -514,7 +588,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
514
588
  if (state.stage !== "setup-required") {
515
589
  return yield* transitionError(state.stage, "project setup has no unanswered question");
516
590
  }
517
- const next = yield* advanceSetup(state, answerText, yield* isoNow);
591
+ const next = yield* advanceSetup(state, answerText, yield* isoNow, modelCatalog.reasoningEfforts);
518
592
  yield* store.save(root, next);
519
593
  return yield* statusOf(root, next);
520
594
  });
@@ -663,6 +737,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
663
737
  version: EVAL_PROJECT_VERSION,
664
738
  basisDigest: state.basisDigest,
665
739
  candidateModels: state.configuration.candidateModels,
740
+ ...(state.configuration.candidateReasoningEfforts === undefined
741
+ ? {}
742
+ : {
743
+ candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
744
+ }),
666
745
  judgeModel: state.configuration.judgeModel,
667
746
  suites: input.suites,
668
747
  decompositionBenchmark: input.decompositionBenchmark,
@@ -764,6 +843,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
764
843
  basisDigest: state.basisDigest,
765
844
  evaluationDigest: state.evaluationDigest,
766
845
  candidateModels: state.configuration.candidateModels,
846
+ ...(state.configuration.candidateReasoningEfforts === undefined
847
+ ? {}
848
+ : {
849
+ candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
850
+ }),
767
851
  classifierModel: state.configuration.classifierModel,
768
852
  authorModel: state.configuration.authorModel,
769
853
  judgeModel: state.configuration.judgeModel,
@@ -792,6 +876,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
792
876
  plan.basisDigest !== state.basisDigest ||
793
877
  plan.evaluationDigest !== state.evaluationDigest ||
794
878
  !sameStrings(plan.candidateModels, state.configuration.candidateModels) ||
879
+ !sameReasoningEfforts(plan.candidateReasoningEfforts, state.configuration.candidateReasoningEfforts) ||
795
880
  plan.classifierModel !== state.configuration.classifierModel ||
796
881
  plan.authorModel !== state.configuration.authorModel ||
797
882
  plan.judgeModel !== state.configuration.judgeModel) {
@@ -1,6 +1,6 @@
1
1
  import type { EvalSetupStage, EvalSetupState } from "@velum-labs/routekit-eval-contracts";
2
2
  import type { RepositoryInspection, SetupQuestion } from "./types.js";
3
- export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection) => SetupQuestion | undefined;
3
+ export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection, reasoningEfforts?: readonly string[]) => SetupQuestion | undefined;
4
4
  export declare const withOpenQuestion: (state: EvalSetupState, inspection?: RepositoryInspection) => {
5
5
  readonly state: EvalSetupState;
6
6
  readonly question?: SetupQuestion;
package/dist/questions.js CHANGED
@@ -2,7 +2,7 @@ const firstThree = (values, fallback) => {
2
2
  const unique = [...new Set(values.filter((value) => value.trim().length > 0))].slice(0, 3);
3
3
  return [unique[0] ?? fallback[0], unique[1] ?? fallback[1], unique[2] ?? fallback[2]];
4
4
  };
5
- export const questionForStage = (stage, inspection) => {
5
+ export const questionForStage = (stage, inspection, reasoningEfforts = []) => {
6
6
  switch (stage) {
7
7
  case "surface":
8
8
  return {
@@ -40,6 +40,12 @@ export const questionForStage = (stage, inspection) => {
40
40
  "Help me find three explicit model IDs"
41
41
  ]
42
42
  };
43
+ case "reasoning-effort":
44
+ return {
45
+ id: stage,
46
+ prompt: "Choose an explicit reasoning effort advertised by the live model catalog.",
47
+ options: [...new Set(reasoningEfforts)]
48
+ };
43
49
  case "spend-approval":
44
50
  return {
45
51
  id: stage,
@@ -0,0 +1,10 @@
1
+ import type { EvalReasoningEffort } from "@velum-labs/routekit-eval-contracts";
2
+ import { Context, type Effect } from "effect";
3
+ import type { EvalModelCatalogError } from "../../errors.js";
4
+ export type EvalModelCatalogShape = {
5
+ readonly reasoningEfforts: (candidateModels: readonly string[]) => Effect.Effect<Readonly<Record<string, readonly EvalReasoningEffort[]>>, EvalModelCatalogError>;
6
+ };
7
+ declare const EvalModelCatalog_base: Context.ServiceClass<EvalModelCatalog, "@velum-labs/routekit-eval-setup/EvalModelCatalog", EvalModelCatalogShape>;
8
+ export declare class EvalModelCatalog extends EvalModelCatalog_base {
9
+ }
10
+ export {};
@@ -0,0 +1,3 @@
1
+ import { Context } from "effect";
2
+ export class EvalModelCatalog extends Context.Service()("@velum-labs/routekit-eval-setup/EvalModelCatalog") {
3
+ }
@@ -1,12 +1,17 @@
1
1
  import assert from "node:assert/strict";
2
2
  import { test } from "node:test";
3
- import { modelSelectionFromAnswer } from "../model-selection.js";
3
+ import { candidateReasoningEffortFromAnswer, modelSelectionFromAnswer } from "../model-selection.js";
4
4
  test("model selection accepts exactly two candidates followed by a distinct judge", () => {
5
5
  assert.deepEqual(modelSelectionFromAnswer("openai/gpt-a anthropic/claude-b google/gemini-judge"), {
6
6
  candidates: ["openai/gpt-a", "anthropic/claude-b"],
7
7
  judgeModel: "google/gemini-judge"
8
8
  });
9
9
  });
10
+ test("reasoning effort selection accepts exactly one advertised opaque value", () => {
11
+ const supported = ["none", "low", "provider-next"];
12
+ assert.equal(candidateReasoningEffortFromAnswer(" provider-next ", supported), "provider-next");
13
+ assert.throws(() => candidateReasoningEffortFromAnswer("max", supported), /must be one of/iu);
14
+ });
10
15
  test("model selection rejects canned answers, duplicates, aliases, and extra ids", () => {
11
16
  assert.throws(() => modelSelectionFromAnswer("Current model, a cheaper candidate, and a stronger candidate"), /exactly three explicit provider\/model ids/iu);
12
17
  assert.throws(() => modelSelectionFromAnswer("openai/a openai/a openai/judge"), /three unique model ids/iu);
@@ -5,16 +5,30 @@ import path from "node:path";
5
5
  import { after, test } from "node:test";
6
6
  import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
7
7
  import { Cause, Effect, Layer } from "effect";
8
- import { EvalDimensionContrastError } from "../errors.js";
8
+ import { EvalDimensionContrastError, EvalModelCatalogError } from "../errors.js";
9
9
  import { EvalRepositoryInspectorLive } from "../inspection.js";
10
10
  import { EvalProjectArtifactsLive } from "../project-artifacts.js";
11
11
  import { scopedExpectedCalls, summarizeEvalRunLedger } from "../project-contracts.js";
12
12
  import { EvalProjectStoreLive } from "../project-store.js";
13
13
  import { EvalProjectWorkflow, EvalProjectWorkflowLive } from "../project-workflow.js";
14
+ import { EvalModelCatalog } from "../services/model-catalog/service.js";
14
15
  const roots = [];
15
16
  after(async () => Promise.all(roots.map((root) => rm(root, { recursive: true, force: true }))));
16
17
  const ProjectDependenciesLive = Layer.mergeAll(EvalProjectStoreLive, EvalProjectArtifactsLive, EvalRepositoryInspectorLive).pipe(Layer.provide(NodeServicesLayer));
17
- const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(NodeServicesLayer));
18
+ const catalogRequests = [];
19
+ const catalogEfforts = (candidateModels) => Effect.sync(() => {
20
+ catalogRequests.push([...candidateModels]);
21
+ return {
22
+ ...(candidateModels.includes("openai/gpt-5.6-luna")
23
+ ? { "openai/gpt-5.6-luna": ["low", "medium", "ultra"] }
24
+ : {}),
25
+ ...(candidateModels.includes("openai/gpt-5.6-terra")
26
+ ? { "openai/gpt-5.6-terra": ["none", "low", "high"] }
27
+ : {})
28
+ };
29
+ });
30
+ const EvalModelCatalogTest = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({ reasoningEfforts: catalogEfforts }));
31
+ const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(EvalModelCatalogTest), Layer.provide(NodeServicesLayer));
18
32
  const makeRepository = async () => {
19
33
  const root = await mkdtemp(path.join(os.tmpdir(), "routekit-eval-project-"));
20
34
  roots.push(root);
@@ -47,22 +61,47 @@ test("setup creates one durable repository project without invoking model-backed
47
61
  });
48
62
  test("answers advance exactly one question and persist a complete compositional configuration", async () => {
49
63
  const root = await makeRepository();
64
+ catalogRequests.length = 0;
50
65
  const statuses = await Effect.runPromise(Effect.gen(function* () {
51
66
  const workflow = yield* EvalProjectWorkflow;
52
67
  yield* workflow.setup(root);
53
68
  const workload = yield* workflow.answer(root, "Customer support and documentation requests");
54
69
  const candidates = yield* workflow.answer(root, "openai/gpt-5.6-luna, openai/gpt-5.6-terra openai/gpt-5.6-sol");
70
+ const lunaEffort = yield* workflow.answer(root, candidates.question.options[1]);
71
+ const terraEffort = yield* workflow.answer(root, lunaEffort.question.options[1]);
55
72
  const classifier = yield* workflow.answer(root, "openai/gpt-5.6-luna");
56
73
  const author = yield* workflow.answer(root, "openai/gpt-5.6-terra");
57
74
  const judge = yield* workflow.answer(root, "openai/gpt-5.6-terra");
58
75
  const objective = yield* workflow.answer(root, '{"kind":"highest-quality"}');
59
76
  const unknown = yield* workflow.answer(root, "0.2");
60
77
  const completed = yield* workflow.answer(root, "{}");
61
- return { workload, candidates, classifier, author, judge, objective, unknown, completed };
78
+ return {
79
+ workload,
80
+ candidates,
81
+ lunaEffort,
82
+ terraEffort,
83
+ classifier,
84
+ author,
85
+ judge,
86
+ objective,
87
+ unknown,
88
+ completed
89
+ };
62
90
  }).pipe(Effect.provide(ProjectWorkflowTestLive)));
63
91
  assert.equal(statuses.workload.question?.id, "candidate-models");
64
92
  assert.equal(statuses.workload.state.revision, 1);
65
- assert.equal(statuses.candidates.question?.id, "classifier-model");
93
+ assert.deepEqual(catalogRequests, [[
94
+ "openai/gpt-5.6-luna",
95
+ "openai/gpt-5.6-terra",
96
+ "openai/gpt-5.6-sol"
97
+ ]]);
98
+ assert.equal(statuses.candidates.question?.id, "candidate-reasoning-efforts");
99
+ assert.match(statuses.candidates.question?.prompt ?? "", /openai\/gpt-5\.6-luna/u);
100
+ assert.deepEqual(statuses.candidates.question?.options, ["low", "medium", "ultra"]);
101
+ assert.equal(statuses.lunaEffort.question?.id, "candidate-reasoning-efforts");
102
+ assert.match(statuses.lunaEffort.question?.prompt ?? "", /openai\/gpt-5\.6-terra/u);
103
+ assert.deepEqual(statuses.lunaEffort.question?.options, ["none", "low", "high"]);
104
+ assert.equal(statuses.terraEffort.question?.id, "classifier-model");
66
105
  assert.equal(statuses.classifier.question?.id, "author-model");
67
106
  assert.equal(statuses.author.question?.id, "judge-model");
68
107
  assert.equal(statuses.judge.question?.id, "routing-objective");
@@ -79,6 +118,10 @@ test("answers advance exactly one question and persist a complete compositional
79
118
  "openai/gpt-5.6-terra",
80
119
  "openai/gpt-5.6-sol"
81
120
  ]);
121
+ assert.deepEqual(statuses.completed.state.configuration.candidateReasoningEfforts, {
122
+ "openai/gpt-5.6-luna": "medium",
123
+ "openai/gpt-5.6-terra": "low"
124
+ });
82
125
  assert.equal(statuses.completed.state.configuration.classifierModel, "openai/gpt-5.6-luna");
83
126
  assert.equal(statuses.completed.state.configuration.authorModel, "openai/gpt-5.6-terra");
84
127
  assert.equal(statuses.completed.state.configuration.judgeModel, "openai/gpt-5.6-terra");
@@ -91,7 +134,7 @@ test("answers advance exactly one question and persist a complete compositional
91
134
  const workflow = yield* EvalProjectWorkflow;
92
135
  return yield* workflow.status(root);
93
136
  }).pipe(Effect.provide(ProjectWorkflowTestLive)));
94
- assert.equal(resumed?.state.revision, 8);
137
+ assert.equal(resumed?.state.revision, 10);
95
138
  assert.equal(resumed?.state.stage, "dimensions-review");
96
139
  });
97
140
  test("invalid answers fail without changing the durable project revision", async () => {
@@ -112,6 +155,30 @@ test("invalid answers fail without changing the durable project revision", async
112
155
  assert.equal(afterDuplicate?.question?.id, "candidate-models");
113
156
  }).pipe(Effect.provide(ProjectWorkflowTestLive)));
114
157
  });
158
+ test("candidate selection fails closed when catalog reasoning metadata is unavailable", async () => {
159
+ const root = await makeRepository();
160
+ const unavailableCatalog = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({
161
+ reasoningEfforts: () => Effect.fail(new EvalModelCatalogError({
162
+ detail: "models.list did not advertise candidate reasoning capabilities"
163
+ }))
164
+ }));
165
+ const workflowLayer = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(unavailableCatalog), Layer.provide(NodeServicesLayer));
166
+ await Effect.runPromise(Effect.gen(function* () {
167
+ const workflow = yield* EvalProjectWorkflow;
168
+ yield* workflow.setup(root);
169
+ yield* workflow.answer(root, "Production workload");
170
+ const answer = yield* Effect.exit(workflow.answer(root, "openai/reasoning openai/plain"));
171
+ assert.equal(answer._tag, "Failure");
172
+ const current = yield* workflow.status(root);
173
+ assert.equal(current?.state.revision, 1);
174
+ assert.equal(current?.state.stage, "setup-required");
175
+ if (current?.state.stage !== "setup-required") {
176
+ assert.fail("expected setup-required state");
177
+ }
178
+ assert.equal(current.state.progress._tag, "CandidateModelsRequired");
179
+ assert.equal(current.question?.id, "candidate-models");
180
+ }).pipe(Effect.provide(workflowLayer)));
181
+ });
115
182
  test("corrupt project state preserves the typed project-store failure", async () => {
116
183
  const root = await makeRepository();
117
184
  await mkdir(path.join(root, ".routekit", "evals"), { recursive: true });
@@ -195,7 +262,11 @@ async function completeProjectSetup(root, candidateModels = "openai/gpt-5.6-luna
195
262
  const workflow = yield* EvalProjectWorkflow;
196
263
  yield* workflow.setup(root);
197
264
  yield* workflow.answer(root, "Production workload");
198
- yield* workflow.answer(root, candidateModels);
265
+ let status = yield* workflow.answer(root, candidateModels);
266
+ while (status.state.stage === "setup-required" &&
267
+ status.state.progress._tag === "CandidateReasoningEffortsRequired") {
268
+ status = yield* workflow.answer(root, status.question.options[0]);
269
+ }
199
270
  yield* workflow.answer(root, "openai/gpt-5.6-luna");
200
271
  yield* workflow.answer(root, "openai/gpt-5.6-terra");
201
272
  yield* workflow.answer(root, "openai/gpt-5.6-terra");
@@ -732,6 +803,10 @@ run.toComplete();
732
803
  assert.equal(compositionManifestAfter, compositionManifestBefore);
733
804
  const dimensionManifest = JSON.parse(dimensionManifestAfter);
734
805
  const compositionManifest = JSON.parse(compositionManifestAfter);
806
+ assert.deepEqual(dimensionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
807
+ assert.deepEqual(compositionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
808
+ assert.match(dimensionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
809
+ assert.match(compositionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
735
810
  assert.equal(dimensionManifest.expectedCallCount, selection.caseIds.length * plan.candidateModels.length * 2);
736
811
  assert.equal(compositionManifest.expectedCallCount, plan.selectedCompositionCaseIds.length * plan.candidateModels.length * 2);
737
812
  });
@@ -8,13 +8,14 @@ test("setup exposes exactly one question for each unresolved stage", () => {
8
8
  "criteria",
9
9
  "constraints",
10
10
  "candidates",
11
+ "reasoning-effort",
11
12
  "spend-approval",
12
13
  "publish"
13
14
  ];
14
15
  for (const stage of stages) {
15
16
  const question = questionForStage(stage);
16
17
  assert.equal(question?.id, stage);
17
- assert.equal(question?.options.length, 3);
18
+ assert.equal(question?.options.length === 3 || stage === "reasoning-effort", true);
18
19
  }
19
20
  assert.equal(questionForStage("completed"), undefined);
20
21
  });
@@ -46,3 +47,8 @@ test("candidate question requires two candidates and a distinct judge", () => {
46
47
  assert.match(question?.prompt ?? "", /exactly three unique provider\/model IDs/iu);
47
48
  assert.doesNotMatch(question?.options.join("\n") ?? "", /save without choosing|current model,/iu);
48
49
  });
50
+ test("reasoning-effort stage exposes an explicit picker without a hidden default", () => {
51
+ const question = questionForStage("reasoning-effort", undefined, ["low", "provider-next"]);
52
+ assert.match(question?.prompt ?? "", /explicit reasoning effort/iu);
53
+ assert.deepEqual(question?.options, ["low", "provider-next"]);
54
+ });
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-eval-setup",
3
3
  "private": false,
4
- "version": "1.1.1",
4
+ "version": "1.3.0",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -22,7 +22,6 @@
22
22
  },
23
23
  "files": [
24
24
  "dist",
25
- "skills",
26
25
  "LICENSE"
27
26
  ],
28
27
  "publishConfig": {
@@ -32,9 +31,9 @@
32
31
  },
33
32
  "dependencies": {
34
33
  "effect": "4.0.0-rc.108",
35
- "@velum-labs/routekit-eval-contracts": "1.1.1",
36
- "@velum-labs/routekit-eval-core": "1.1.1",
37
- "@velum-labs/routekit-runtime": "1.1.1"
34
+ "@velum-labs/routekit-eval-contracts": "1.3.0",
35
+ "@velum-labs/routekit-eval-core": "1.3.0",
36
+ "@velum-labs/routekit-runtime": "1.3.0"
38
37
  },
39
38
  "keywords": [
40
39
  "routekit",
@@ -1 +0,0 @@
1
- export {};
@@ -1,35 +0,0 @@
1
- import assert from "node:assert/strict";
2
- import { readFile } from "node:fs/promises";
3
- import path from "node:path";
4
- import { test } from "node:test";
5
- import { fileURLToPath } from "node:url";
6
- const packageRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..");
7
- test("onboarding skill uses the public CLI and preserves approval boundaries", async () => {
8
- const skill = await readFile(path.join(packageRoot, "skills", "setup-eval-routing", "SKILL.md"), "utf8");
9
- assert.match(skill, /setup-eval-routing/u);
10
- assert.match(skill, /one question per turn/iu);
11
- assert.match(skill, /Never spend or publish silently/u);
12
- assert.match(skill, /public `routekit eval` CLI/u);
13
- assert.match(skill, /\$ROUTEKIT eval --help/u);
14
- for (const command of ["setup", "status", "answer", "validate", "estimate", "run", "publish"]) {
15
- assert.match(skill, new RegExp(`eval ${command}`, "u"));
16
- }
17
- for (const term of [
18
- "routing basis",
19
- "workload dimension",
20
- "request decomposition",
21
- "evidence matrix",
22
- "routing activation"
23
- ]) {
24
- assert.match(skill, new RegExp(term, "iu"));
25
- }
26
- assert.match(skill, /exclusive in-scope request or a distinct near-miss/iu);
27
- assert.match(skill, /product-behavior axes are mixed with repository-change\/process axes/iu);
28
- assert.match(skill, /high weight on almost every ticket/iu);
29
- assert.match(skill, /Unknown weight absorbs the remainder/u);
30
- assert.doesNotMatch(skill, /\beval prepare\b/u);
31
- assert.doesNotMatch(skill, /\barea catalog\b/iu);
32
- assert.doesNotMatch(skill, /Use RouteKit's `EvalSetup` operations/u);
33
- assert.doesNotMatch(skill, /test:e2e:eval-routing/u);
34
- assert.doesNotMatch(skill, /OPENROUTER_API_KEY/u);
35
- });