@velum-labs/routekit-eval-setup 1.1.1 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/effect-api.d.ts +2 -0
- package/dist/effect-api.js +1 -0
- package/dist/errors.d.ts +9 -0
- package/dist/errors.js +5 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.js +2 -1
- package/dist/model-selection.d.ts +2 -0
- package/dist/model-selection.js +7 -0
- package/dist/project-artifacts.js +17 -1
- package/dist/project-authoring.d.ts +1 -1
- package/dist/project-contracts.d.ts +62 -1
- package/dist/project-contracts.js +20 -2
- package/dist/project-workflow.d.ts +7 -6
- package/dist/project-workflow.js +93 -8
- package/dist/questions.d.ts +1 -1
- package/dist/questions.js +7 -1
- package/dist/services/model-catalog/service.d.ts +10 -0
- package/dist/services/model-catalog/service.js +3 -0
- package/dist/test/model-selection.test.js +6 -1
- package/dist/test/project-workflow.test.js +81 -6
- package/dist/test/questions.test.js +7 -1
- package/package.json +4 -5
- package/dist/test/skill.test.d.ts +0 -1
- package/dist/test/skill.test.js +0 -35
- package/skills/setup-eval-routing/SKILL.md +0 -163
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
import { Context, Effect, Layer, Path } from "effect";
|
|
2
2
|
import type { EvalProjectArtifactError, EvalSetupInspectionError } from "./errors.js";
|
|
3
|
-
import { EvalDimensionContrastError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
|
|
3
|
+
import { EvalDimensionContrastError, type EvalModelCatalogError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
|
|
4
4
|
import { EvalRepositoryInspector } from "./inspection.js";
|
|
5
5
|
import { EvalProjectArtifacts } from "./project-artifacts.js";
|
|
6
6
|
import type { EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProposedDimension, EvalProjectStatus, EvalRunReport } from "./project-contracts.js";
|
|
7
7
|
import { EvalProjectStore } from "./project-store.js";
|
|
8
|
-
|
|
8
|
+
import { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
9
|
+
export type EvalProjectWorkflowError = EvalProjectStoreError | EvalProjectArtifactError | EvalSetupInspectionError | EvalModelCatalogError | EvalDimensionContrastError | EvalProjectTransitionError;
|
|
9
10
|
export type EvalProjectWorkflowShape = {
|
|
10
11
|
readonly setup: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
11
12
|
readonly status: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus | undefined, EvalProjectStoreError | EvalProjectArtifactError>;
|
|
12
|
-
readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError>;
|
|
13
|
+
readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError | EvalModelCatalogError>;
|
|
13
14
|
readonly proposeDimensions: (repositoryRoot: string, dimensions: readonly EvalProposedDimension[]) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
14
15
|
readonly approveDimensions: (repositoryRoot: string, basisDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
15
|
-
readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
16
|
+
readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "candidateReasoningEfforts" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
16
17
|
readonly approveEvaluations: (repositoryRoot: string, evaluationDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
17
18
|
readonly createPlan: (repositoryRoot: string, scope: EvalPlanScope) => Effect.Effect<EvalExecutionPlan, EvalProjectWorkflowError>;
|
|
18
19
|
readonly startRun: (repositoryRoot: string, planId: string) => Effect.Effect<{
|
|
@@ -27,6 +28,6 @@ export type EvalProjectWorkflowShape = {
|
|
|
27
28
|
declare const EvalProjectWorkflow_base: Context.ServiceClass<EvalProjectWorkflow, "@velum-labs/routekit-eval-setup/EvalProjectWorkflow", EvalProjectWorkflowShape>;
|
|
28
29
|
export declare class EvalProjectWorkflow extends EvalProjectWorkflow_base {
|
|
29
30
|
}
|
|
30
|
-
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
|
|
31
|
-
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
|
|
31
|
+
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
|
+
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
33
|
export {};
|
package/dist/project-workflow.js
CHANGED
|
@@ -3,9 +3,11 @@ import { RoutingActivationConstraints, RoutingObjectivePolicy } from "@velum-lab
|
|
|
3
3
|
import { Clock, Context, Effect, Layer, Path, Schema } from "effect";
|
|
4
4
|
import { EvalDimensionContrastError, EvalProjectTransitionError } from "./errors.js";
|
|
5
5
|
import { EvalRepositoryInspector } from "./inspection.js";
|
|
6
|
+
import { candidateReasoningEffortFromAnswer } from "./model-selection.js";
|
|
6
7
|
import { EvalProjectArtifacts, evaluationProposalDigest, routingBasisDigest } from "./project-artifacts.js";
|
|
7
8
|
import { EVAL_PROJECT_VERSION, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
8
9
|
import { EvalProjectStore } from "./project-store.js";
|
|
10
|
+
import { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
9
11
|
const isoNow = Effect.map(Clock.currentTimeMillis, (millis) => new Date(millis).toISOString());
|
|
10
12
|
const questionForProgress = (progress) => {
|
|
11
13
|
switch (progress._tag) {
|
|
@@ -21,6 +23,12 @@ const questionForProgress = (progress) => {
|
|
|
21
23
|
prompt: "Which explicit provider/model IDs may RouteKit route to?",
|
|
22
24
|
options: []
|
|
23
25
|
};
|
|
26
|
+
case "CandidateReasoningEffortsRequired":
|
|
27
|
+
return {
|
|
28
|
+
id: "candidate-reasoning-efforts",
|
|
29
|
+
prompt: `Choose an advertised reasoning effort for ${progress.currentCandidateModel}.`,
|
|
30
|
+
options: [...progress.supportedReasoningEfforts[progress.currentCandidateModel]]
|
|
31
|
+
};
|
|
24
32
|
case "ClassifierModelRequired":
|
|
25
33
|
return {
|
|
26
34
|
id: "classifier-model",
|
|
@@ -124,7 +132,24 @@ const parseModel = (state, answer, role) => Effect.gen(function* () {
|
|
|
124
132
|
const model = yield* nonEmptyAnswer(state, answer);
|
|
125
133
|
return yield* validateModel(state, model, role);
|
|
126
134
|
});
|
|
135
|
+
const supportedCandidateReasoningEfforts = (candidates, input) => {
|
|
136
|
+
const supported = {};
|
|
137
|
+
for (const model of candidates) {
|
|
138
|
+
const efforts = [...new Set(input[model] ?? [])];
|
|
139
|
+
if (efforts.length > 0)
|
|
140
|
+
supported[model] = efforts;
|
|
141
|
+
}
|
|
142
|
+
return supported;
|
|
143
|
+
};
|
|
144
|
+
const firstReasoningCandidate = (candidates, supported) => candidates.find((model) => supported[model] !== undefined);
|
|
145
|
+
const nextReasoningCandidate = (candidates, supported, selected) => candidates.find((model) => supported[model] !== undefined && selected[model] === undefined);
|
|
127
146
|
const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
|
|
147
|
+
const sameReasoningEfforts = (left, right) => {
|
|
148
|
+
const leftEntries = Object.entries(left ?? {});
|
|
149
|
+
const rightEntries = Object.entries(right ?? {});
|
|
150
|
+
return (leftEntries.length === rightEntries.length &&
|
|
151
|
+
leftEntries.every(([model, effort]) => right?.[model] === effort));
|
|
152
|
+
};
|
|
128
153
|
const sameLedger = (left, right) => left.expectedCalls === right.expectedCalls &&
|
|
129
154
|
left.observedCalls === right.observedCalls &&
|
|
130
155
|
left.observedCandidateRows === right.observedCandidateRows &&
|
|
@@ -300,6 +325,7 @@ const parseObjective = (state, answer) => Effect.gen(function* () {
|
|
|
300
325
|
const configurationFrom = (progress, constraints) => ({
|
|
301
326
|
workloadDescription: progress.workloadDescription,
|
|
302
327
|
candidateModels: progress.candidateModels,
|
|
328
|
+
candidateReasoningEfforts: progress.candidateReasoningEfforts ?? {},
|
|
303
329
|
classifierModel: progress.classifierModel,
|
|
304
330
|
authorModel: progress.authorModel,
|
|
305
331
|
judgeModel: progress.judgeModel,
|
|
@@ -324,7 +350,7 @@ const parseConstraints = (state, answer) => Effect.gen(function* () {
|
|
|
324
350
|
const constraints = yield* Schema.decodeUnknownEffect(RoutingActivationConstraints)(json).pipe(Effect.mapError(() => transitionError(state.stage, "routing constraints contain invalid dimension quality or failure-rate values")));
|
|
325
351
|
return Object.keys(constraints).length === 0 ? undefined : constraints;
|
|
326
352
|
});
|
|
327
|
-
const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
353
|
+
const advanceSetup = (state, answer, now, discoverReasoningEfforts) => Effect.gen(function* () {
|
|
328
354
|
const common = {
|
|
329
355
|
version: state.version,
|
|
330
356
|
projectId: state.projectId,
|
|
@@ -343,16 +369,58 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
343
369
|
workloadDescription: yield* nonEmptyAnswer(state, answer)
|
|
344
370
|
}
|
|
345
371
|
};
|
|
346
|
-
case "CandidateModelsRequired":
|
|
372
|
+
case "CandidateModelsRequired": {
|
|
373
|
+
const candidateModels = yield* parseCandidateModels(state, answer);
|
|
374
|
+
const supportedReasoningEfforts = supportedCandidateReasoningEfforts(candidateModels, yield* discoverReasoningEfforts(candidateModels));
|
|
375
|
+
const currentCandidateModel = firstReasoningCandidate(candidateModels, supportedReasoningEfforts);
|
|
347
376
|
return {
|
|
348
377
|
...common,
|
|
349
378
|
stage: "setup-required",
|
|
350
|
-
progress:
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
379
|
+
progress: currentCandidateModel === undefined
|
|
380
|
+
? {
|
|
381
|
+
_tag: "ClassifierModelRequired",
|
|
382
|
+
workloadDescription: state.progress.workloadDescription,
|
|
383
|
+
candidateModels,
|
|
384
|
+
candidateReasoningEfforts: {}
|
|
385
|
+
}
|
|
386
|
+
: {
|
|
387
|
+
_tag: "CandidateReasoningEffortsRequired",
|
|
388
|
+
workloadDescription: state.progress.workloadDescription,
|
|
389
|
+
candidateModels,
|
|
390
|
+
candidateReasoningEfforts: {},
|
|
391
|
+
currentCandidateModel,
|
|
392
|
+
supportedReasoningEfforts
|
|
393
|
+
}
|
|
355
394
|
};
|
|
395
|
+
}
|
|
396
|
+
case "CandidateReasoningEffortsRequired": {
|
|
397
|
+
const supportedReasoningEfforts = state.progress.supportedReasoningEfforts;
|
|
398
|
+
const currentCandidateModel = state.progress.currentCandidateModel;
|
|
399
|
+
const candidateReasoningEfforts = {
|
|
400
|
+
...state.progress.candidateReasoningEfforts,
|
|
401
|
+
[currentCandidateModel]: yield* Effect.try({
|
|
402
|
+
try: () => candidateReasoningEffortFromAnswer(answer, supportedReasoningEfforts[currentCandidateModel]),
|
|
403
|
+
catch: (cause) => transitionError(state.stage, cause instanceof Error ? cause.message : String(cause))
|
|
404
|
+
})
|
|
405
|
+
};
|
|
406
|
+
const nextCandidateModel = nextReasoningCandidate(state.progress.candidateModels, supportedReasoningEfforts, candidateReasoningEfforts);
|
|
407
|
+
return {
|
|
408
|
+
...common,
|
|
409
|
+
stage: "setup-required",
|
|
410
|
+
progress: nextCandidateModel === undefined
|
|
411
|
+
? {
|
|
412
|
+
_tag: "ClassifierModelRequired",
|
|
413
|
+
workloadDescription: state.progress.workloadDescription,
|
|
414
|
+
candidateModels: state.progress.candidateModels,
|
|
415
|
+
candidateReasoningEfforts
|
|
416
|
+
}
|
|
417
|
+
: {
|
|
418
|
+
...state.progress,
|
|
419
|
+
candidateReasoningEfforts,
|
|
420
|
+
currentCandidateModel: nextCandidateModel
|
|
421
|
+
}
|
|
422
|
+
};
|
|
423
|
+
}
|
|
356
424
|
case "ClassifierModelRequired":
|
|
357
425
|
return {
|
|
358
426
|
...common,
|
|
@@ -361,6 +429,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
361
429
|
_tag: "AuthorModelRequired",
|
|
362
430
|
workloadDescription: state.progress.workloadDescription,
|
|
363
431
|
candidateModels: state.progress.candidateModels,
|
|
432
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
364
433
|
classifierModel: yield* parseModel(state, answer, "classifier")
|
|
365
434
|
}
|
|
366
435
|
};
|
|
@@ -372,6 +441,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
372
441
|
_tag: "JudgeModelRequired",
|
|
373
442
|
workloadDescription: state.progress.workloadDescription,
|
|
374
443
|
candidateModels: state.progress.candidateModels,
|
|
444
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
375
445
|
classifierModel: state.progress.classifierModel,
|
|
376
446
|
authorModel: yield* parseModel(state, answer, "author")
|
|
377
447
|
}
|
|
@@ -384,6 +454,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
384
454
|
_tag: "RoutingObjectiveRequired",
|
|
385
455
|
workloadDescription: state.progress.workloadDescription,
|
|
386
456
|
candidateModels: state.progress.candidateModels,
|
|
457
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
387
458
|
classifierModel: state.progress.classifierModel,
|
|
388
459
|
authorModel: state.progress.authorModel,
|
|
389
460
|
judgeModel: yield* parseModel(state, answer, "judge")
|
|
@@ -398,6 +469,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
398
469
|
_tag: "MaximumUnknownWeightRequired",
|
|
399
470
|
workloadDescription: state.progress.workloadDescription,
|
|
400
471
|
candidateModels: state.progress.candidateModels,
|
|
472
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
401
473
|
classifierModel: state.progress.classifierModel,
|
|
402
474
|
authorModel: state.progress.authorModel,
|
|
403
475
|
judgeModel: state.progress.judgeModel,
|
|
@@ -413,6 +485,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
413
485
|
_tag: "RoutingConstraintsRequired",
|
|
414
486
|
workloadDescription: state.progress.workloadDescription,
|
|
415
487
|
candidateModels: state.progress.candidateModels,
|
|
488
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
416
489
|
classifierModel: state.progress.classifierModel,
|
|
417
490
|
authorModel: state.progress.authorModel,
|
|
418
491
|
judgeModel: state.progress.judgeModel,
|
|
@@ -438,6 +511,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
438
511
|
const store = yield* EvalProjectStore;
|
|
439
512
|
const artifacts = yield* EvalProjectArtifacts;
|
|
440
513
|
const inspector = yield* EvalRepositoryInspector;
|
|
514
|
+
const modelCatalog = yield* EvalModelCatalog;
|
|
441
515
|
const paths = yield* Path.Path;
|
|
442
516
|
const resolveRoot = (repositoryRoot) => paths.resolve(repositoryRoot);
|
|
443
517
|
const statusOf = (root, state) => Effect.gen(function* () {
|
|
@@ -514,7 +588,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
514
588
|
if (state.stage !== "setup-required") {
|
|
515
589
|
return yield* transitionError(state.stage, "project setup has no unanswered question");
|
|
516
590
|
}
|
|
517
|
-
const next = yield* advanceSetup(state, answerText, yield* isoNow);
|
|
591
|
+
const next = yield* advanceSetup(state, answerText, yield* isoNow, modelCatalog.reasoningEfforts);
|
|
518
592
|
yield* store.save(root, next);
|
|
519
593
|
return yield* statusOf(root, next);
|
|
520
594
|
});
|
|
@@ -663,6 +737,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
663
737
|
version: EVAL_PROJECT_VERSION,
|
|
664
738
|
basisDigest: state.basisDigest,
|
|
665
739
|
candidateModels: state.configuration.candidateModels,
|
|
740
|
+
...(state.configuration.candidateReasoningEfforts === undefined
|
|
741
|
+
? {}
|
|
742
|
+
: {
|
|
743
|
+
candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
|
|
744
|
+
}),
|
|
666
745
|
judgeModel: state.configuration.judgeModel,
|
|
667
746
|
suites: input.suites,
|
|
668
747
|
decompositionBenchmark: input.decompositionBenchmark,
|
|
@@ -764,6 +843,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
764
843
|
basisDigest: state.basisDigest,
|
|
765
844
|
evaluationDigest: state.evaluationDigest,
|
|
766
845
|
candidateModels: state.configuration.candidateModels,
|
|
846
|
+
...(state.configuration.candidateReasoningEfforts === undefined
|
|
847
|
+
? {}
|
|
848
|
+
: {
|
|
849
|
+
candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
|
|
850
|
+
}),
|
|
767
851
|
classifierModel: state.configuration.classifierModel,
|
|
768
852
|
authorModel: state.configuration.authorModel,
|
|
769
853
|
judgeModel: state.configuration.judgeModel,
|
|
@@ -792,6 +876,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
792
876
|
plan.basisDigest !== state.basisDigest ||
|
|
793
877
|
plan.evaluationDigest !== state.evaluationDigest ||
|
|
794
878
|
!sameStrings(plan.candidateModels, state.configuration.candidateModels) ||
|
|
879
|
+
!sameReasoningEfforts(plan.candidateReasoningEfforts, state.configuration.candidateReasoningEfforts) ||
|
|
795
880
|
plan.classifierModel !== state.configuration.classifierModel ||
|
|
796
881
|
plan.authorModel !== state.configuration.authorModel ||
|
|
797
882
|
plan.judgeModel !== state.configuration.judgeModel) {
|
package/dist/questions.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { EvalSetupStage, EvalSetupState } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import type { RepositoryInspection, SetupQuestion } from "./types.js";
|
|
3
|
-
export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection) => SetupQuestion | undefined;
|
|
3
|
+
export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection, reasoningEfforts?: readonly string[]) => SetupQuestion | undefined;
|
|
4
4
|
export declare const withOpenQuestion: (state: EvalSetupState, inspection?: RepositoryInspection) => {
|
|
5
5
|
readonly state: EvalSetupState;
|
|
6
6
|
readonly question?: SetupQuestion;
|
package/dist/questions.js
CHANGED
|
@@ -2,7 +2,7 @@ const firstThree = (values, fallback) => {
|
|
|
2
2
|
const unique = [...new Set(values.filter((value) => value.trim().length > 0))].slice(0, 3);
|
|
3
3
|
return [unique[0] ?? fallback[0], unique[1] ?? fallback[1], unique[2] ?? fallback[2]];
|
|
4
4
|
};
|
|
5
|
-
export const questionForStage = (stage, inspection) => {
|
|
5
|
+
export const questionForStage = (stage, inspection, reasoningEfforts = []) => {
|
|
6
6
|
switch (stage) {
|
|
7
7
|
case "surface":
|
|
8
8
|
return {
|
|
@@ -40,6 +40,12 @@ export const questionForStage = (stage, inspection) => {
|
|
|
40
40
|
"Help me find three explicit model IDs"
|
|
41
41
|
]
|
|
42
42
|
};
|
|
43
|
+
case "reasoning-effort":
|
|
44
|
+
return {
|
|
45
|
+
id: stage,
|
|
46
|
+
prompt: "Choose an explicit reasoning effort advertised by the live model catalog.",
|
|
47
|
+
options: [...new Set(reasoningEfforts)]
|
|
48
|
+
};
|
|
43
49
|
case "spend-approval":
|
|
44
50
|
return {
|
|
45
51
|
id: stage,
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { EvalReasoningEffort } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
+
import { Context, type Effect } from "effect";
|
|
3
|
+
import type { EvalModelCatalogError } from "../../errors.js";
|
|
4
|
+
export type EvalModelCatalogShape = {
|
|
5
|
+
readonly reasoningEfforts: (candidateModels: readonly string[]) => Effect.Effect<Readonly<Record<string, readonly EvalReasoningEffort[]>>, EvalModelCatalogError>;
|
|
6
|
+
};
|
|
7
|
+
declare const EvalModelCatalog_base: Context.ServiceClass<EvalModelCatalog, "@velum-labs/routekit-eval-setup/EvalModelCatalog", EvalModelCatalogShape>;
|
|
8
|
+
export declare class EvalModelCatalog extends EvalModelCatalog_base {
|
|
9
|
+
}
|
|
10
|
+
export {};
|
|
@@ -1,12 +1,17 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
2
|
import { test } from "node:test";
|
|
3
|
-
import { modelSelectionFromAnswer } from "../model-selection.js";
|
|
3
|
+
import { candidateReasoningEffortFromAnswer, modelSelectionFromAnswer } from "../model-selection.js";
|
|
4
4
|
test("model selection accepts exactly two candidates followed by a distinct judge", () => {
|
|
5
5
|
assert.deepEqual(modelSelectionFromAnswer("openai/gpt-a anthropic/claude-b google/gemini-judge"), {
|
|
6
6
|
candidates: ["openai/gpt-a", "anthropic/claude-b"],
|
|
7
7
|
judgeModel: "google/gemini-judge"
|
|
8
8
|
});
|
|
9
9
|
});
|
|
10
|
+
test("reasoning effort selection accepts exactly one advertised opaque value", () => {
|
|
11
|
+
const supported = ["none", "low", "provider-next"];
|
|
12
|
+
assert.equal(candidateReasoningEffortFromAnswer(" provider-next ", supported), "provider-next");
|
|
13
|
+
assert.throws(() => candidateReasoningEffortFromAnswer("max", supported), /must be one of/iu);
|
|
14
|
+
});
|
|
10
15
|
test("model selection rejects canned answers, duplicates, aliases, and extra ids", () => {
|
|
11
16
|
assert.throws(() => modelSelectionFromAnswer("Current model, a cheaper candidate, and a stronger candidate"), /exactly three explicit provider\/model ids/iu);
|
|
12
17
|
assert.throws(() => modelSelectionFromAnswer("openai/a openai/a openai/judge"), /three unique model ids/iu);
|
|
@@ -5,16 +5,30 @@ import path from "node:path";
|
|
|
5
5
|
import { after, test } from "node:test";
|
|
6
6
|
import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
|
|
7
7
|
import { Cause, Effect, Layer } from "effect";
|
|
8
|
-
import { EvalDimensionContrastError } from "../errors.js";
|
|
8
|
+
import { EvalDimensionContrastError, EvalModelCatalogError } from "../errors.js";
|
|
9
9
|
import { EvalRepositoryInspectorLive } from "../inspection.js";
|
|
10
10
|
import { EvalProjectArtifactsLive } from "../project-artifacts.js";
|
|
11
11
|
import { scopedExpectedCalls, summarizeEvalRunLedger } from "../project-contracts.js";
|
|
12
12
|
import { EvalProjectStoreLive } from "../project-store.js";
|
|
13
13
|
import { EvalProjectWorkflow, EvalProjectWorkflowLive } from "../project-workflow.js";
|
|
14
|
+
import { EvalModelCatalog } from "../services/model-catalog/service.js";
|
|
14
15
|
const roots = [];
|
|
15
16
|
after(async () => Promise.all(roots.map((root) => rm(root, { recursive: true, force: true }))));
|
|
16
17
|
const ProjectDependenciesLive = Layer.mergeAll(EvalProjectStoreLive, EvalProjectArtifactsLive, EvalRepositoryInspectorLive).pipe(Layer.provide(NodeServicesLayer));
|
|
17
|
-
const
|
|
18
|
+
const catalogRequests = [];
|
|
19
|
+
const catalogEfforts = (candidateModels) => Effect.sync(() => {
|
|
20
|
+
catalogRequests.push([...candidateModels]);
|
|
21
|
+
return {
|
|
22
|
+
...(candidateModels.includes("openai/gpt-5.6-luna")
|
|
23
|
+
? { "openai/gpt-5.6-luna": ["low", "medium", "ultra"] }
|
|
24
|
+
: {}),
|
|
25
|
+
...(candidateModels.includes("openai/gpt-5.6-terra")
|
|
26
|
+
? { "openai/gpt-5.6-terra": ["none", "low", "high"] }
|
|
27
|
+
: {})
|
|
28
|
+
};
|
|
29
|
+
});
|
|
30
|
+
const EvalModelCatalogTest = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({ reasoningEfforts: catalogEfforts }));
|
|
31
|
+
const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(EvalModelCatalogTest), Layer.provide(NodeServicesLayer));
|
|
18
32
|
const makeRepository = async () => {
|
|
19
33
|
const root = await mkdtemp(path.join(os.tmpdir(), "routekit-eval-project-"));
|
|
20
34
|
roots.push(root);
|
|
@@ -47,22 +61,47 @@ test("setup creates one durable repository project without invoking model-backed
|
|
|
47
61
|
});
|
|
48
62
|
test("answers advance exactly one question and persist a complete compositional configuration", async () => {
|
|
49
63
|
const root = await makeRepository();
|
|
64
|
+
catalogRequests.length = 0;
|
|
50
65
|
const statuses = await Effect.runPromise(Effect.gen(function* () {
|
|
51
66
|
const workflow = yield* EvalProjectWorkflow;
|
|
52
67
|
yield* workflow.setup(root);
|
|
53
68
|
const workload = yield* workflow.answer(root, "Customer support and documentation requests");
|
|
54
69
|
const candidates = yield* workflow.answer(root, "openai/gpt-5.6-luna, openai/gpt-5.6-terra openai/gpt-5.6-sol");
|
|
70
|
+
const lunaEffort = yield* workflow.answer(root, candidates.question.options[1]);
|
|
71
|
+
const terraEffort = yield* workflow.answer(root, lunaEffort.question.options[1]);
|
|
55
72
|
const classifier = yield* workflow.answer(root, "openai/gpt-5.6-luna");
|
|
56
73
|
const author = yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
57
74
|
const judge = yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
58
75
|
const objective = yield* workflow.answer(root, '{"kind":"highest-quality"}');
|
|
59
76
|
const unknown = yield* workflow.answer(root, "0.2");
|
|
60
77
|
const completed = yield* workflow.answer(root, "{}");
|
|
61
|
-
return {
|
|
78
|
+
return {
|
|
79
|
+
workload,
|
|
80
|
+
candidates,
|
|
81
|
+
lunaEffort,
|
|
82
|
+
terraEffort,
|
|
83
|
+
classifier,
|
|
84
|
+
author,
|
|
85
|
+
judge,
|
|
86
|
+
objective,
|
|
87
|
+
unknown,
|
|
88
|
+
completed
|
|
89
|
+
};
|
|
62
90
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
63
91
|
assert.equal(statuses.workload.question?.id, "candidate-models");
|
|
64
92
|
assert.equal(statuses.workload.state.revision, 1);
|
|
65
|
-
assert.
|
|
93
|
+
assert.deepEqual(catalogRequests, [[
|
|
94
|
+
"openai/gpt-5.6-luna",
|
|
95
|
+
"openai/gpt-5.6-terra",
|
|
96
|
+
"openai/gpt-5.6-sol"
|
|
97
|
+
]]);
|
|
98
|
+
assert.equal(statuses.candidates.question?.id, "candidate-reasoning-efforts");
|
|
99
|
+
assert.match(statuses.candidates.question?.prompt ?? "", /openai\/gpt-5\.6-luna/u);
|
|
100
|
+
assert.deepEqual(statuses.candidates.question?.options, ["low", "medium", "ultra"]);
|
|
101
|
+
assert.equal(statuses.lunaEffort.question?.id, "candidate-reasoning-efforts");
|
|
102
|
+
assert.match(statuses.lunaEffort.question?.prompt ?? "", /openai\/gpt-5\.6-terra/u);
|
|
103
|
+
assert.deepEqual(statuses.lunaEffort.question?.options, ["none", "low", "high"]);
|
|
104
|
+
assert.equal(statuses.terraEffort.question?.id, "classifier-model");
|
|
66
105
|
assert.equal(statuses.classifier.question?.id, "author-model");
|
|
67
106
|
assert.equal(statuses.author.question?.id, "judge-model");
|
|
68
107
|
assert.equal(statuses.judge.question?.id, "routing-objective");
|
|
@@ -79,6 +118,10 @@ test("answers advance exactly one question and persist a complete compositional
|
|
|
79
118
|
"openai/gpt-5.6-terra",
|
|
80
119
|
"openai/gpt-5.6-sol"
|
|
81
120
|
]);
|
|
121
|
+
assert.deepEqual(statuses.completed.state.configuration.candidateReasoningEfforts, {
|
|
122
|
+
"openai/gpt-5.6-luna": "medium",
|
|
123
|
+
"openai/gpt-5.6-terra": "low"
|
|
124
|
+
});
|
|
82
125
|
assert.equal(statuses.completed.state.configuration.classifierModel, "openai/gpt-5.6-luna");
|
|
83
126
|
assert.equal(statuses.completed.state.configuration.authorModel, "openai/gpt-5.6-terra");
|
|
84
127
|
assert.equal(statuses.completed.state.configuration.judgeModel, "openai/gpt-5.6-terra");
|
|
@@ -91,7 +134,7 @@ test("answers advance exactly one question and persist a complete compositional
|
|
|
91
134
|
const workflow = yield* EvalProjectWorkflow;
|
|
92
135
|
return yield* workflow.status(root);
|
|
93
136
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
94
|
-
assert.equal(resumed?.state.revision,
|
|
137
|
+
assert.equal(resumed?.state.revision, 10);
|
|
95
138
|
assert.equal(resumed?.state.stage, "dimensions-review");
|
|
96
139
|
});
|
|
97
140
|
test("invalid answers fail without changing the durable project revision", async () => {
|
|
@@ -112,6 +155,30 @@ test("invalid answers fail without changing the durable project revision", async
|
|
|
112
155
|
assert.equal(afterDuplicate?.question?.id, "candidate-models");
|
|
113
156
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
114
157
|
});
|
|
158
|
+
test("candidate selection fails closed when catalog reasoning metadata is unavailable", async () => {
|
|
159
|
+
const root = await makeRepository();
|
|
160
|
+
const unavailableCatalog = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({
|
|
161
|
+
reasoningEfforts: () => Effect.fail(new EvalModelCatalogError({
|
|
162
|
+
detail: "models.list did not advertise candidate reasoning capabilities"
|
|
163
|
+
}))
|
|
164
|
+
}));
|
|
165
|
+
const workflowLayer = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(unavailableCatalog), Layer.provide(NodeServicesLayer));
|
|
166
|
+
await Effect.runPromise(Effect.gen(function* () {
|
|
167
|
+
const workflow = yield* EvalProjectWorkflow;
|
|
168
|
+
yield* workflow.setup(root);
|
|
169
|
+
yield* workflow.answer(root, "Production workload");
|
|
170
|
+
const answer = yield* Effect.exit(workflow.answer(root, "openai/reasoning openai/plain"));
|
|
171
|
+
assert.equal(answer._tag, "Failure");
|
|
172
|
+
const current = yield* workflow.status(root);
|
|
173
|
+
assert.equal(current?.state.revision, 1);
|
|
174
|
+
assert.equal(current?.state.stage, "setup-required");
|
|
175
|
+
if (current?.state.stage !== "setup-required") {
|
|
176
|
+
assert.fail("expected setup-required state");
|
|
177
|
+
}
|
|
178
|
+
assert.equal(current.state.progress._tag, "CandidateModelsRequired");
|
|
179
|
+
assert.equal(current.question?.id, "candidate-models");
|
|
180
|
+
}).pipe(Effect.provide(workflowLayer)));
|
|
181
|
+
});
|
|
115
182
|
test("corrupt project state preserves the typed project-store failure", async () => {
|
|
116
183
|
const root = await makeRepository();
|
|
117
184
|
await mkdir(path.join(root, ".routekit", "evals"), { recursive: true });
|
|
@@ -195,7 +262,11 @@ async function completeProjectSetup(root, candidateModels = "openai/gpt-5.6-luna
|
|
|
195
262
|
const workflow = yield* EvalProjectWorkflow;
|
|
196
263
|
yield* workflow.setup(root);
|
|
197
264
|
yield* workflow.answer(root, "Production workload");
|
|
198
|
-
yield* workflow.answer(root, candidateModels);
|
|
265
|
+
let status = yield* workflow.answer(root, candidateModels);
|
|
266
|
+
while (status.state.stage === "setup-required" &&
|
|
267
|
+
status.state.progress._tag === "CandidateReasoningEffortsRequired") {
|
|
268
|
+
status = yield* workflow.answer(root, status.question.options[0]);
|
|
269
|
+
}
|
|
199
270
|
yield* workflow.answer(root, "openai/gpt-5.6-luna");
|
|
200
271
|
yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
201
272
|
yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
@@ -732,6 +803,10 @@ run.toComplete();
|
|
|
732
803
|
assert.equal(compositionManifestAfter, compositionManifestBefore);
|
|
733
804
|
const dimensionManifest = JSON.parse(dimensionManifestAfter);
|
|
734
805
|
const compositionManifest = JSON.parse(compositionManifestAfter);
|
|
806
|
+
assert.deepEqual(dimensionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
|
|
807
|
+
assert.deepEqual(compositionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
|
|
808
|
+
assert.match(dimensionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
|
|
809
|
+
assert.match(compositionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
|
|
735
810
|
assert.equal(dimensionManifest.expectedCallCount, selection.caseIds.length * plan.candidateModels.length * 2);
|
|
736
811
|
assert.equal(compositionManifest.expectedCallCount, plan.selectedCompositionCaseIds.length * plan.candidateModels.length * 2);
|
|
737
812
|
});
|
|
@@ -8,13 +8,14 @@ test("setup exposes exactly one question for each unresolved stage", () => {
|
|
|
8
8
|
"criteria",
|
|
9
9
|
"constraints",
|
|
10
10
|
"candidates",
|
|
11
|
+
"reasoning-effort",
|
|
11
12
|
"spend-approval",
|
|
12
13
|
"publish"
|
|
13
14
|
];
|
|
14
15
|
for (const stage of stages) {
|
|
15
16
|
const question = questionForStage(stage);
|
|
16
17
|
assert.equal(question?.id, stage);
|
|
17
|
-
assert.equal(question?.options.length,
|
|
18
|
+
assert.equal(question?.options.length === 3 || stage === "reasoning-effort", true);
|
|
18
19
|
}
|
|
19
20
|
assert.equal(questionForStage("completed"), undefined);
|
|
20
21
|
});
|
|
@@ -46,3 +47,8 @@ test("candidate question requires two candidates and a distinct judge", () => {
|
|
|
46
47
|
assert.match(question?.prompt ?? "", /exactly three unique provider\/model IDs/iu);
|
|
47
48
|
assert.doesNotMatch(question?.options.join("\n") ?? "", /save without choosing|current model,/iu);
|
|
48
49
|
});
|
|
50
|
+
test("reasoning-effort stage exposes an explicit picker without a hidden default", () => {
|
|
51
|
+
const question = questionForStage("reasoning-effort", undefined, ["low", "provider-next"]);
|
|
52
|
+
assert.match(question?.prompt ?? "", /explicit reasoning effort/iu);
|
|
53
|
+
assert.deepEqual(question?.options, ["low", "provider-next"]);
|
|
54
|
+
});
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-setup",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.3.0",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -22,7 +22,6 @@
|
|
|
22
22
|
},
|
|
23
23
|
"files": [
|
|
24
24
|
"dist",
|
|
25
|
-
"skills",
|
|
26
25
|
"LICENSE"
|
|
27
26
|
],
|
|
28
27
|
"publishConfig": {
|
|
@@ -32,9 +31,9 @@
|
|
|
32
31
|
},
|
|
33
32
|
"dependencies": {
|
|
34
33
|
"effect": "4.0.0-rc.108",
|
|
35
|
-
"@velum-labs/routekit-eval-contracts": "1.
|
|
36
|
-
"@velum-labs/routekit-eval-core": "1.
|
|
37
|
-
"@velum-labs/routekit-runtime": "1.
|
|
34
|
+
"@velum-labs/routekit-eval-contracts": "1.3.0",
|
|
35
|
+
"@velum-labs/routekit-eval-core": "1.3.0",
|
|
36
|
+
"@velum-labs/routekit-runtime": "1.3.0"
|
|
38
37
|
},
|
|
39
38
|
"keywords": [
|
|
40
39
|
"routekit",
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
export {};
|
package/dist/test/skill.test.js
DELETED
|
@@ -1,35 +0,0 @@
|
|
|
1
|
-
import assert from "node:assert/strict";
|
|
2
|
-
import { readFile } from "node:fs/promises";
|
|
3
|
-
import path from "node:path";
|
|
4
|
-
import { test } from "node:test";
|
|
5
|
-
import { fileURLToPath } from "node:url";
|
|
6
|
-
const packageRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..");
|
|
7
|
-
test("onboarding skill uses the public CLI and preserves approval boundaries", async () => {
|
|
8
|
-
const skill = await readFile(path.join(packageRoot, "skills", "setup-eval-routing", "SKILL.md"), "utf8");
|
|
9
|
-
assert.match(skill, /setup-eval-routing/u);
|
|
10
|
-
assert.match(skill, /one question per turn/iu);
|
|
11
|
-
assert.match(skill, /Never spend or publish silently/u);
|
|
12
|
-
assert.match(skill, /public `routekit eval` CLI/u);
|
|
13
|
-
assert.match(skill, /\$ROUTEKIT eval --help/u);
|
|
14
|
-
for (const command of ["setup", "status", "answer", "validate", "estimate", "run", "publish"]) {
|
|
15
|
-
assert.match(skill, new RegExp(`eval ${command}`, "u"));
|
|
16
|
-
}
|
|
17
|
-
for (const term of [
|
|
18
|
-
"routing basis",
|
|
19
|
-
"workload dimension",
|
|
20
|
-
"request decomposition",
|
|
21
|
-
"evidence matrix",
|
|
22
|
-
"routing activation"
|
|
23
|
-
]) {
|
|
24
|
-
assert.match(skill, new RegExp(term, "iu"));
|
|
25
|
-
}
|
|
26
|
-
assert.match(skill, /exclusive in-scope request or a distinct near-miss/iu);
|
|
27
|
-
assert.match(skill, /product-behavior axes are mixed with repository-change\/process axes/iu);
|
|
28
|
-
assert.match(skill, /high weight on almost every ticket/iu);
|
|
29
|
-
assert.match(skill, /Unknown weight absorbs the remainder/u);
|
|
30
|
-
assert.doesNotMatch(skill, /\beval prepare\b/u);
|
|
31
|
-
assert.doesNotMatch(skill, /\barea catalog\b/iu);
|
|
32
|
-
assert.doesNotMatch(skill, /Use RouteKit's `EvalSetup` operations/u);
|
|
33
|
-
assert.doesNotMatch(skill, /test:e2e:eval-routing/u);
|
|
34
|
-
assert.doesNotMatch(skill, /OPENROUTER_API_KEY/u);
|
|
35
|
-
});
|