@velum-labs/routekit-eval-setup 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/effect-api.d.ts +2 -0
- package/dist/effect-api.js +1 -0
- package/dist/errors.d.ts +9 -0
- package/dist/errors.js +5 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.js +2 -1
- package/dist/model-selection.d.ts +2 -0
- package/dist/model-selection.js +7 -0
- package/dist/project-artifacts.js +17 -1
- package/dist/project-authoring.d.ts +1 -1
- package/dist/project-contracts.d.ts +62 -1
- package/dist/project-contracts.js +20 -2
- package/dist/project-workflow.d.ts +7 -6
- package/dist/project-workflow.js +93 -8
- package/dist/questions.d.ts +1 -1
- package/dist/questions.js +7 -1
- package/dist/services/model-catalog/service.d.ts +10 -0
- package/dist/services/model-catalog/service.js +3 -0
- package/dist/test/model-selection.test.js +6 -1
- package/dist/test/project-workflow.test.js +81 -6
- package/dist/test/questions.test.js +7 -1
- package/package.json +4 -4
package/dist/effect-api.d.ts
CHANGED
|
@@ -15,6 +15,8 @@ export { EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeE
|
|
|
15
15
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
16
16
|
export type { EvalProjectWorkflowError, EvalProjectWorkflowShape } from "./project-workflow.js";
|
|
17
17
|
export { EvalProjectWorkflow, EvalProjectWorkflowLive, makeEvalProjectWorkflow } from "./project-workflow.js";
|
|
18
|
+
export type { EvalModelCatalogShape } from "./services/model-catalog/service.js";
|
|
19
|
+
export { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
18
20
|
export { EvalSetupRunner, EvalSetupRunnerNoop } from "./runner.js";
|
|
19
21
|
export { EvalSetup, EvalSetupLive, makeEvalSetup } from "./service.js";
|
|
20
22
|
export { EvalSetupStateStore, EvalSetupStateStoreLive, initialSetupState, makeFileEvalSetupStateStore } from "./state-store.js";
|
package/dist/effect-api.js
CHANGED
|
@@ -9,6 +9,7 @@ export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDiges
|
|
|
9
9
|
export { EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
10
10
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
11
11
|
export { EvalProjectWorkflow, EvalProjectWorkflowLive, makeEvalProjectWorkflow } from "./project-workflow.js";
|
|
12
|
+
export { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
12
13
|
export { EvalSetupRunner, EvalSetupRunnerNoop } from "./runner.js";
|
|
13
14
|
export { EvalSetup, EvalSetupLive, makeEvalSetup } from "./service.js";
|
|
14
15
|
export { EvalSetupStateStore, EvalSetupStateStoreLive, initialSetupState, makeFileEvalSetupStateStore } from "./state-store.js";
|
package/dist/errors.d.ts
CHANGED
|
@@ -67,6 +67,15 @@ export declare class EvalProjectTransitionError extends EvalProjectTransitionErr
|
|
|
67
67
|
}> {
|
|
68
68
|
get message(): string;
|
|
69
69
|
}
|
|
70
|
+
declare const EvalModelCatalogError_base: new <A extends Record<string, any> = {}>(args: import("effect/Types").VoidIfEmpty<{ readonly [P in keyof A as P extends "_tag" ? never : P]: A[P]; }>) => import("effect/Cause").YieldableError & {
|
|
71
|
+
readonly _tag: "EvalModelCatalogError";
|
|
72
|
+
} & Readonly<A>;
|
|
73
|
+
export declare class EvalModelCatalogError extends EvalModelCatalogError_base<{
|
|
74
|
+
readonly detail: string;
|
|
75
|
+
readonly cause?: unknown;
|
|
76
|
+
}> {
|
|
77
|
+
get message(): string;
|
|
78
|
+
}
|
|
70
79
|
declare const EvalDimensionContrastError_base: new <A extends Record<string, any> = {}>(args: import("effect/Types").VoidIfEmpty<{ readonly [P in keyof A as P extends "_tag" ? never : P]: A[P]; }>) => import("effect/Cause").YieldableError & {
|
|
71
80
|
readonly _tag: "EvalDimensionContrastError";
|
|
72
81
|
} & Readonly<A>;
|
package/dist/errors.js
CHANGED
|
@@ -34,6 +34,11 @@ export class EvalProjectTransitionError extends Data.TaggedError("EvalProjectTra
|
|
|
34
34
|
return `RouteKit Eval project cannot continue from ${this.state}: ${this.detail}`;
|
|
35
35
|
}
|
|
36
36
|
}
|
|
37
|
+
export class EvalModelCatalogError extends Data.TaggedError("EvalModelCatalogError") {
|
|
38
|
+
get message() {
|
|
39
|
+
return `RouteKit Eval could not discover candidate reasoning capabilities: ${this.detail}`;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
37
42
|
export class EvalDimensionContrastError extends Data.TaggedError("EvalDimensionContrastError") {
|
|
38
43
|
get message() {
|
|
39
44
|
return `RouteKit Eval cannot approve routing dimension ${JSON.stringify(this.dimensionId)}: ${this.detail}`;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
export { projectRoutingBasisV3, projectRoutingDimensionV3 } from "./basis-projection.js";
|
|
2
|
-
export { EvalBasisSnapshotError, EvalDimensionContrastError, EvalProjectArtifactError, EvalProjectAuthoringError, EvalProjectStoreError, EvalProjectTransitionError, EvalSetupInspectionError, EvalSetupRunnerError, EvalSetupScaffoldError, EvalSetupStateError, EvalSetupTransitionError } from "./errors.js";
|
|
2
|
+
export { EvalBasisSnapshotError, EvalDimensionContrastError, EvalModelCatalogError, EvalProjectArtifactError, EvalProjectAuthoringError, EvalProjectStoreError, EvalProjectTransitionError, EvalSetupInspectionError, EvalSetupRunnerError, EvalSetupScaffoldError, EvalSetupStateError, EvalSetupTransitionError } from "./errors.js";
|
|
3
|
+
export type { EvalModelCatalogShape } from "./services/model-catalog/service.js";
|
|
4
|
+
export { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
3
5
|
export { captureFrozenRepositorySnapshotV3 } from "./adapters/git-basis-snapshot.js";
|
|
4
6
|
export type { CapturedRepositorySnapshotV3, FrozenRepositorySourceV3 } from "./adapters/git-basis-snapshot.js";
|
|
5
7
|
export { readFrozenRepositorySourcesV3 } from "./adapters/git-basis-snapshot.js";
|
package/dist/index.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
export { projectRoutingBasisV3, projectRoutingDimensionV3 } from "./basis-projection.js";
|
|
2
|
-
export { EvalBasisSnapshotError, EvalDimensionContrastError, EvalProjectArtifactError, EvalProjectAuthoringError, EvalProjectStoreError, EvalProjectTransitionError, EvalSetupInspectionError, EvalSetupRunnerError, EvalSetupScaffoldError, EvalSetupStateError, EvalSetupTransitionError } from "./errors.js";
|
|
2
|
+
export { EvalBasisSnapshotError, EvalDimensionContrastError, EvalModelCatalogError, EvalProjectArtifactError, EvalProjectAuthoringError, EvalProjectStoreError, EvalProjectTransitionError, EvalSetupInspectionError, EvalSetupRunnerError, EvalSetupScaffoldError, EvalSetupStateError, EvalSetupTransitionError } from "./errors.js";
|
|
3
|
+
export { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
3
4
|
export { captureFrozenRepositorySnapshotV3 } from "./adapters/git-basis-snapshot.js";
|
|
4
5
|
export { readFrozenRepositorySourcesV3 } from "./adapters/git-basis-snapshot.js";
|
|
5
6
|
export { authoringRequest, hostDirectory } from "./host-metadata.js";
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import { type EvalReasoningEffort } from "@velum-labs/routekit-eval-contracts";
|
|
1
2
|
export type EvalModelSelection = {
|
|
2
3
|
readonly candidates: readonly [string, string];
|
|
3
4
|
readonly judgeModel: string;
|
|
4
5
|
};
|
|
5
6
|
export declare const assertEvalModelSelection: (candidates: readonly string[], judgeModel: string) => EvalModelSelection;
|
|
6
7
|
export declare const modelSelectionFromAnswer: (answer: string) => EvalModelSelection;
|
|
8
|
+
export declare const candidateReasoningEffortFromAnswer: (answer: string, supported: readonly EvalReasoningEffort[]) => EvalReasoningEffort;
|
package/dist/model-selection.js
CHANGED
|
@@ -35,3 +35,10 @@ export const modelSelectionFromAnswer = (answer) => {
|
|
|
35
35
|
}
|
|
36
36
|
return assertEvalModelSelection(modelIds.slice(0, 2), modelIds[2] ?? "");
|
|
37
37
|
};
|
|
38
|
+
export const candidateReasoningEffortFromAnswer = (answer, supported) => {
|
|
39
|
+
const effort = answer.trim();
|
|
40
|
+
if (!supported.includes(effort)) {
|
|
41
|
+
throw new Error(`reasoning effort must be one of ${supported.join(", ")}`);
|
|
42
|
+
}
|
|
43
|
+
return effort;
|
|
44
|
+
};
|
|
@@ -37,13 +37,20 @@ import manifest from "./routekit.eval-manifest.json" with { type: "json" };
|
|
|
37
37
|
|
|
38
38
|
assert.equal(cases.length, manifest.caseCount);
|
|
39
39
|
assert.deepEqual(cases.map((testCase) => testCase.id), manifest.caseIds);
|
|
40
|
+
const candidateReasoningEfforts = manifest.candidateReasoningEfforts ?? {};
|
|
40
41
|
const judge = setupJudge({
|
|
41
42
|
agent: setupAgent({ model: manifest.judgeModel }),
|
|
42
43
|
minScore: 0.8
|
|
43
44
|
});
|
|
44
45
|
|
|
45
46
|
for (const model of manifest.candidateModels) {
|
|
46
|
-
const
|
|
47
|
+
const effort = candidateReasoningEfforts[model];
|
|
48
|
+
const candidate = setupAgent({
|
|
49
|
+
model,
|
|
50
|
+
...(effort === undefined
|
|
51
|
+
? {}
|
|
52
|
+
: { parameters: { reasoning: { effort } } })
|
|
53
|
+
});
|
|
47
54
|
for (const testCase of cases) {
|
|
48
55
|
test(\`\${model} / \${testCase.id}\`, async () => {
|
|
49
56
|
${renderCandidatePrompt()}
|
|
@@ -89,6 +96,9 @@ function assertEvaluationProposal(proposal) {
|
|
|
89
96
|
version: proposal.version,
|
|
90
97
|
basisDigest: proposal.basisDigest,
|
|
91
98
|
candidateModels: proposal.candidateModels,
|
|
99
|
+
...(proposal.candidateReasoningEfforts === undefined
|
|
100
|
+
? {}
|
|
101
|
+
: { candidateReasoningEfforts: proposal.candidateReasoningEfforts }),
|
|
92
102
|
judgeModel: proposal.judgeModel,
|
|
93
103
|
suites: proposal.suites,
|
|
94
104
|
decompositionBenchmark: proposal.decompositionBenchmark,
|
|
@@ -102,6 +112,9 @@ function assertEvaluationProposal(proposal) {
|
|
|
102
112
|
if (new Set(proposal.candidateModels).size !== proposal.candidateModels.length) {
|
|
103
113
|
throw new Error("evaluation proposal candidate models must be unique");
|
|
104
114
|
}
|
|
115
|
+
if (Object.keys(proposal.candidateReasoningEfforts ?? {}).some((model) => !proposal.candidateModels.includes(model))) {
|
|
116
|
+
throw new Error("evaluation proposal reasoning efforts must belong to candidate models");
|
|
117
|
+
}
|
|
105
118
|
if (proposal.judgeModel.trim().length === 0) {
|
|
106
119
|
throw new Error("evaluation proposal judge must be explicit");
|
|
107
120
|
}
|
|
@@ -285,6 +298,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
285
298
|
version: EVAL_PROJECT_VERSION,
|
|
286
299
|
profileId: suite.dimensionId,
|
|
287
300
|
candidateModels: proposal.candidateModels,
|
|
301
|
+
candidateReasoningEfforts: proposal.candidateReasoningEfforts ?? {},
|
|
288
302
|
judgeModel: proposal.judgeModel,
|
|
289
303
|
caseCount: suite.cases.length,
|
|
290
304
|
caseIds: suite.cases.map((testCase) => testCase.id),
|
|
@@ -324,6 +338,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
324
338
|
version: EVAL_PROJECT_VERSION,
|
|
325
339
|
profileId: selection.dimensionId,
|
|
326
340
|
candidateModels: plan.candidateModels,
|
|
341
|
+
candidateReasoningEfforts: plan.candidateReasoningEfforts ?? {},
|
|
327
342
|
judgeModel: plan.judgeModel,
|
|
328
343
|
caseCount: cases.length,
|
|
329
344
|
caseIds: cases.map((testCase) => testCase.id),
|
|
@@ -346,6 +361,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
346
361
|
version: EVAL_PROJECT_VERSION,
|
|
347
362
|
profileId: "composition",
|
|
348
363
|
candidateModels: plan.candidateModels,
|
|
364
|
+
candidateReasoningEfforts: plan.candidateReasoningEfforts ?? {},
|
|
349
365
|
judgeModel: plan.judgeModel,
|
|
350
366
|
caseCount: compositionCases.length,
|
|
351
367
|
caseIds: compositionCases.map((testCase) => testCase.id),
|
|
@@ -59,7 +59,7 @@ export type EvalProjectAuthorShape = {
|
|
|
59
59
|
readonly sourceInventory: readonly string[];
|
|
60
60
|
readonly configuration: EvalProjectConfiguration;
|
|
61
61
|
readonly basis: RoutingBasis;
|
|
62
|
-
}) => Effect.Effect<Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "judgeModel">, EvalProjectAuthoringError>;
|
|
62
|
+
}) => Effect.Effect<Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "candidateReasoningEfforts" | "judgeModel">, EvalProjectAuthoringError>;
|
|
63
63
|
};
|
|
64
64
|
declare const EvalProjectAuthor_base: Context.ServiceClass<EvalProjectAuthor, "@velum-labs/routekit-eval-setup/EvalProjectAuthor", EvalProjectAuthorShape>;
|
|
65
65
|
export declare class EvalProjectAuthor extends EvalProjectAuthor_base {
|
|
@@ -3,34 +3,45 @@ import { Schema } from "effect";
|
|
|
3
3
|
export declare const EVAL_PROJECT_VERSION: 1;
|
|
4
4
|
export declare const EVAL_PROJECT_V2_VERSION: 2;
|
|
5
5
|
export declare const EvalProjectQuestion: Schema.Struct<{
|
|
6
|
-
readonly id: Schema.Literals<readonly ["workload-description", "candidate-models", "classifier-model", "author-model", "judge-model", "routing-objective", "maximum-unknown-weight", "routing-constraints"]>;
|
|
6
|
+
readonly id: Schema.Literals<readonly ["workload-description", "candidate-models", "candidate-reasoning-efforts", "classifier-model", "author-model", "judge-model", "routing-objective", "maximum-unknown-weight", "routing-constraints"]>;
|
|
7
7
|
readonly prompt: Schema.String;
|
|
8
8
|
readonly options: Schema.$Array<Schema.String>;
|
|
9
9
|
}>;
|
|
10
10
|
export type EvalProjectQuestion = typeof EvalProjectQuestion.Type;
|
|
11
11
|
export declare const EvalProjectSetupProgress: Schema.Union<readonly [Schema.TaggedStruct<"WorkloadDescriptionRequired", {}>, Schema.TaggedStruct<"CandidateModelsRequired", {
|
|
12
12
|
readonly workloadDescription: Schema.String;
|
|
13
|
+
}>, Schema.TaggedStruct<"CandidateReasoningEffortsRequired", {
|
|
14
|
+
readonly workloadDescription: Schema.String;
|
|
15
|
+
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
16
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
17
|
+
readonly currentCandidateModel: Schema.String;
|
|
18
|
+
readonly supportedReasoningEfforts: Schema.$Record<Schema.String, Schema.$Array<Schema.NonEmptyString>>;
|
|
13
19
|
}>, Schema.TaggedStruct<"ClassifierModelRequired", {
|
|
14
20
|
readonly workloadDescription: Schema.String;
|
|
15
21
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
22
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
16
23
|
}>, Schema.TaggedStruct<"AuthorModelRequired", {
|
|
17
24
|
readonly workloadDescription: Schema.String;
|
|
18
25
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
26
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
19
27
|
readonly classifierModel: Schema.String;
|
|
20
28
|
}>, Schema.TaggedStruct<"JudgeModelRequired", {
|
|
21
29
|
readonly workloadDescription: Schema.String;
|
|
22
30
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
31
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
23
32
|
readonly classifierModel: Schema.String;
|
|
24
33
|
readonly authorModel: Schema.String;
|
|
25
34
|
}>, Schema.TaggedStruct<"RoutingObjectiveRequired", {
|
|
26
35
|
readonly workloadDescription: Schema.String;
|
|
27
36
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
37
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
28
38
|
readonly classifierModel: Schema.String;
|
|
29
39
|
readonly authorModel: Schema.String;
|
|
30
40
|
readonly judgeModel: Schema.String;
|
|
31
41
|
}>, Schema.TaggedStruct<"MaximumUnknownWeightRequired", {
|
|
32
42
|
readonly workloadDescription: Schema.String;
|
|
33
43
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
44
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
34
45
|
readonly classifierModel: Schema.String;
|
|
35
46
|
readonly authorModel: Schema.String;
|
|
36
47
|
readonly judgeModel: Schema.String;
|
|
@@ -58,6 +69,7 @@ export declare const EvalProjectSetupProgress: Schema.Union<readonly [Schema.Tag
|
|
|
58
69
|
}>, Schema.TaggedStruct<"RoutingConstraintsRequired", {
|
|
59
70
|
readonly workloadDescription: Schema.String;
|
|
60
71
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
72
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
61
73
|
readonly classifierModel: Schema.String;
|
|
62
74
|
readonly authorModel: Schema.String;
|
|
63
75
|
readonly judgeModel: Schema.String;
|
|
@@ -88,6 +100,7 @@ export type EvalProjectSetupProgress = typeof EvalProjectSetupProgress.Type;
|
|
|
88
100
|
export declare const EvalProjectConfiguration: Schema.Struct<{
|
|
89
101
|
readonly workloadDescription: Schema.String;
|
|
90
102
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
103
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
91
104
|
readonly classifierModel: Schema.String;
|
|
92
105
|
readonly authorModel: Schema.String;
|
|
93
106
|
readonly judgeModel: Schema.String;
|
|
@@ -154,27 +167,38 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
154
167
|
readonly stage: Schema.Literal<"setup-required">;
|
|
155
168
|
readonly progress: Schema.Union<readonly [Schema.TaggedStruct<"WorkloadDescriptionRequired", {}>, Schema.TaggedStruct<"CandidateModelsRequired", {
|
|
156
169
|
readonly workloadDescription: Schema.String;
|
|
170
|
+
}>, Schema.TaggedStruct<"CandidateReasoningEffortsRequired", {
|
|
171
|
+
readonly workloadDescription: Schema.String;
|
|
172
|
+
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
173
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
174
|
+
readonly currentCandidateModel: Schema.String;
|
|
175
|
+
readonly supportedReasoningEfforts: Schema.$Record<Schema.String, Schema.$Array<Schema.NonEmptyString>>;
|
|
157
176
|
}>, Schema.TaggedStruct<"ClassifierModelRequired", {
|
|
158
177
|
readonly workloadDescription: Schema.String;
|
|
159
178
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
179
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
160
180
|
}>, Schema.TaggedStruct<"AuthorModelRequired", {
|
|
161
181
|
readonly workloadDescription: Schema.String;
|
|
162
182
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
183
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
163
184
|
readonly classifierModel: Schema.String;
|
|
164
185
|
}>, Schema.TaggedStruct<"JudgeModelRequired", {
|
|
165
186
|
readonly workloadDescription: Schema.String;
|
|
166
187
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
188
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
167
189
|
readonly classifierModel: Schema.String;
|
|
168
190
|
readonly authorModel: Schema.String;
|
|
169
191
|
}>, Schema.TaggedStruct<"RoutingObjectiveRequired", {
|
|
170
192
|
readonly workloadDescription: Schema.String;
|
|
171
193
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
194
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
172
195
|
readonly classifierModel: Schema.String;
|
|
173
196
|
readonly authorModel: Schema.String;
|
|
174
197
|
readonly judgeModel: Schema.String;
|
|
175
198
|
}>, Schema.TaggedStruct<"MaximumUnknownWeightRequired", {
|
|
176
199
|
readonly workloadDescription: Schema.String;
|
|
177
200
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
201
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
178
202
|
readonly classifierModel: Schema.String;
|
|
179
203
|
readonly authorModel: Schema.String;
|
|
180
204
|
readonly judgeModel: Schema.String;
|
|
@@ -202,6 +226,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
202
226
|
}>, Schema.TaggedStruct<"RoutingConstraintsRequired", {
|
|
203
227
|
readonly workloadDescription: Schema.String;
|
|
204
228
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
229
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
205
230
|
readonly classifierModel: Schema.String;
|
|
206
231
|
readonly authorModel: Schema.String;
|
|
207
232
|
readonly judgeModel: Schema.String;
|
|
@@ -239,6 +264,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
239
264
|
readonly configuration: Schema.Struct<{
|
|
240
265
|
readonly workloadDescription: Schema.String;
|
|
241
266
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
267
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
242
268
|
readonly classifierModel: Schema.String;
|
|
243
269
|
readonly authorModel: Schema.String;
|
|
244
270
|
readonly judgeModel: Schema.String;
|
|
@@ -280,6 +306,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
280
306
|
readonly configuration: Schema.Struct<{
|
|
281
307
|
readonly workloadDescription: Schema.String;
|
|
282
308
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
309
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
283
310
|
readonly classifierModel: Schema.String;
|
|
284
311
|
readonly authorModel: Schema.String;
|
|
285
312
|
readonly judgeModel: Schema.String;
|
|
@@ -322,6 +349,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
322
349
|
readonly configuration: Schema.Struct<{
|
|
323
350
|
readonly workloadDescription: Schema.String;
|
|
324
351
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
352
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
325
353
|
readonly classifierModel: Schema.String;
|
|
326
354
|
readonly authorModel: Schema.String;
|
|
327
355
|
readonly judgeModel: Schema.String;
|
|
@@ -365,6 +393,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
365
393
|
readonly configuration: Schema.Struct<{
|
|
366
394
|
readonly workloadDescription: Schema.String;
|
|
367
395
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
396
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
368
397
|
readonly classifierModel: Schema.String;
|
|
369
398
|
readonly authorModel: Schema.String;
|
|
370
399
|
readonly judgeModel: Schema.String;
|
|
@@ -410,6 +439,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
410
439
|
readonly configuration: Schema.Struct<{
|
|
411
440
|
readonly workloadDescription: Schema.String;
|
|
412
441
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
442
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
413
443
|
readonly classifierModel: Schema.String;
|
|
414
444
|
readonly authorModel: Schema.String;
|
|
415
445
|
readonly judgeModel: Schema.String;
|
|
@@ -455,6 +485,7 @@ export declare const EvalProjectState: Schema.Union<readonly [Schema.Struct<{
|
|
|
455
485
|
readonly configuration: Schema.Struct<{
|
|
456
486
|
readonly workloadDescription: Schema.String;
|
|
457
487
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
488
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
458
489
|
readonly classifierModel: Schema.String;
|
|
459
490
|
readonly authorModel: Schema.String;
|
|
460
491
|
readonly judgeModel: Schema.String;
|
|
@@ -1501,6 +1532,7 @@ export declare const EvalProjectStateV2: Schema.Union<readonly [Schema.Struct<{
|
|
|
1501
1532
|
readonly configuration: Schema.Struct<{
|
|
1502
1533
|
readonly workloadDescription: Schema.String;
|
|
1503
1534
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
1535
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
1504
1536
|
readonly classifierModel: Schema.String;
|
|
1505
1537
|
readonly authorModel: Schema.String;
|
|
1506
1538
|
readonly judgeModel: Schema.String;
|
|
@@ -1699,6 +1731,7 @@ export declare const EvalProjectStateV2: Schema.Union<readonly [Schema.Struct<{
|
|
|
1699
1731
|
readonly configuration: Schema.Struct<{
|
|
1700
1732
|
readonly workloadDescription: Schema.String;
|
|
1701
1733
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
1734
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
1702
1735
|
readonly classifierModel: Schema.String;
|
|
1703
1736
|
readonly authorModel: Schema.String;
|
|
1704
1737
|
readonly judgeModel: Schema.String;
|
|
@@ -1950,6 +1983,7 @@ export declare const EvalProjectStateV2: Schema.Union<readonly [Schema.Struct<{
|
|
|
1950
1983
|
readonly configuration: Schema.Struct<{
|
|
1951
1984
|
readonly workloadDescription: Schema.String;
|
|
1952
1985
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
1986
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
1953
1987
|
readonly classifierModel: Schema.String;
|
|
1954
1988
|
readonly authorModel: Schema.String;
|
|
1955
1989
|
readonly judgeModel: Schema.String;
|
|
@@ -2243,6 +2277,7 @@ export declare const EvalProjectStateV2: Schema.Union<readonly [Schema.Struct<{
|
|
|
2243
2277
|
readonly configuration: Schema.Struct<{
|
|
2244
2278
|
readonly workloadDescription: Schema.String;
|
|
2245
2279
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2280
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2246
2281
|
readonly classifierModel: Schema.String;
|
|
2247
2282
|
readonly authorModel: Schema.String;
|
|
2248
2283
|
readonly judgeModel: Schema.String;
|
|
@@ -2548,6 +2583,7 @@ export declare const EvalProjectStateV2: Schema.Union<readonly [Schema.Struct<{
|
|
|
2548
2583
|
readonly configuration: Schema.Struct<{
|
|
2549
2584
|
readonly workloadDescription: Schema.String;
|
|
2550
2585
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2586
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2551
2587
|
readonly classifierModel: Schema.String;
|
|
2552
2588
|
readonly authorModel: Schema.String;
|
|
2553
2589
|
readonly judgeModel: Schema.String;
|
|
@@ -2584,27 +2620,38 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2584
2620
|
readonly stage: Schema.Literal<"setup-required">;
|
|
2585
2621
|
readonly progress: Schema.Union<readonly [Schema.TaggedStruct<"WorkloadDescriptionRequired", {}>, Schema.TaggedStruct<"CandidateModelsRequired", {
|
|
2586
2622
|
readonly workloadDescription: Schema.String;
|
|
2623
|
+
}>, Schema.TaggedStruct<"CandidateReasoningEffortsRequired", {
|
|
2624
|
+
readonly workloadDescription: Schema.String;
|
|
2625
|
+
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2626
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
2627
|
+
readonly currentCandidateModel: Schema.String;
|
|
2628
|
+
readonly supportedReasoningEfforts: Schema.$Record<Schema.String, Schema.$Array<Schema.NonEmptyString>>;
|
|
2587
2629
|
}>, Schema.TaggedStruct<"ClassifierModelRequired", {
|
|
2588
2630
|
readonly workloadDescription: Schema.String;
|
|
2589
2631
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2632
|
+
readonly candidateReasoningEfforts: Schema.$Record<Schema.String, Schema.NonEmptyString>;
|
|
2590
2633
|
}>, Schema.TaggedStruct<"AuthorModelRequired", {
|
|
2591
2634
|
readonly workloadDescription: Schema.String;
|
|
2592
2635
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2636
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2593
2637
|
readonly classifierModel: Schema.String;
|
|
2594
2638
|
}>, Schema.TaggedStruct<"JudgeModelRequired", {
|
|
2595
2639
|
readonly workloadDescription: Schema.String;
|
|
2596
2640
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2641
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2597
2642
|
readonly classifierModel: Schema.String;
|
|
2598
2643
|
readonly authorModel: Schema.String;
|
|
2599
2644
|
}>, Schema.TaggedStruct<"RoutingObjectiveRequired", {
|
|
2600
2645
|
readonly workloadDescription: Schema.String;
|
|
2601
2646
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2647
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2602
2648
|
readonly classifierModel: Schema.String;
|
|
2603
2649
|
readonly authorModel: Schema.String;
|
|
2604
2650
|
readonly judgeModel: Schema.String;
|
|
2605
2651
|
}>, Schema.TaggedStruct<"MaximumUnknownWeightRequired", {
|
|
2606
2652
|
readonly workloadDescription: Schema.String;
|
|
2607
2653
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2654
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2608
2655
|
readonly classifierModel: Schema.String;
|
|
2609
2656
|
readonly authorModel: Schema.String;
|
|
2610
2657
|
readonly judgeModel: Schema.String;
|
|
@@ -2632,6 +2679,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2632
2679
|
}>, Schema.TaggedStruct<"RoutingConstraintsRequired", {
|
|
2633
2680
|
readonly workloadDescription: Schema.String;
|
|
2634
2681
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2682
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2635
2683
|
readonly classifierModel: Schema.String;
|
|
2636
2684
|
readonly authorModel: Schema.String;
|
|
2637
2685
|
readonly judgeModel: Schema.String;
|
|
@@ -2669,6 +2717,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2669
2717
|
readonly configuration: Schema.Struct<{
|
|
2670
2718
|
readonly workloadDescription: Schema.String;
|
|
2671
2719
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2720
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2672
2721
|
readonly classifierModel: Schema.String;
|
|
2673
2722
|
readonly authorModel: Schema.String;
|
|
2674
2723
|
readonly judgeModel: Schema.String;
|
|
@@ -2710,6 +2759,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2710
2759
|
readonly configuration: Schema.Struct<{
|
|
2711
2760
|
readonly workloadDescription: Schema.String;
|
|
2712
2761
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2762
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2713
2763
|
readonly classifierModel: Schema.String;
|
|
2714
2764
|
readonly authorModel: Schema.String;
|
|
2715
2765
|
readonly judgeModel: Schema.String;
|
|
@@ -2752,6 +2802,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2752
2802
|
readonly configuration: Schema.Struct<{
|
|
2753
2803
|
readonly workloadDescription: Schema.String;
|
|
2754
2804
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2805
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2755
2806
|
readonly classifierModel: Schema.String;
|
|
2756
2807
|
readonly authorModel: Schema.String;
|
|
2757
2808
|
readonly judgeModel: Schema.String;
|
|
@@ -2795,6 +2846,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2795
2846
|
readonly configuration: Schema.Struct<{
|
|
2796
2847
|
readonly workloadDescription: Schema.String;
|
|
2797
2848
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2849
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2798
2850
|
readonly classifierModel: Schema.String;
|
|
2799
2851
|
readonly authorModel: Schema.String;
|
|
2800
2852
|
readonly judgeModel: Schema.String;
|
|
@@ -2840,6 +2892,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2840
2892
|
readonly configuration: Schema.Struct<{
|
|
2841
2893
|
readonly workloadDescription: Schema.String;
|
|
2842
2894
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2895
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2843
2896
|
readonly classifierModel: Schema.String;
|
|
2844
2897
|
readonly authorModel: Schema.String;
|
|
2845
2898
|
readonly judgeModel: Schema.String;
|
|
@@ -2885,6 +2938,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
2885
2938
|
readonly configuration: Schema.Struct<{
|
|
2886
2939
|
readonly workloadDescription: Schema.String;
|
|
2887
2940
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
2941
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
2888
2942
|
readonly classifierModel: Schema.String;
|
|
2889
2943
|
readonly authorModel: Schema.String;
|
|
2890
2944
|
readonly judgeModel: Schema.String;
|
|
@@ -3013,6 +3067,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
3013
3067
|
readonly configuration: Schema.Struct<{
|
|
3014
3068
|
readonly workloadDescription: Schema.String;
|
|
3015
3069
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
3070
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
3016
3071
|
readonly classifierModel: Schema.String;
|
|
3017
3072
|
readonly authorModel: Schema.String;
|
|
3018
3073
|
readonly judgeModel: Schema.String;
|
|
@@ -3211,6 +3266,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
3211
3266
|
readonly configuration: Schema.Struct<{
|
|
3212
3267
|
readonly workloadDescription: Schema.String;
|
|
3213
3268
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
3269
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
3214
3270
|
readonly classifierModel: Schema.String;
|
|
3215
3271
|
readonly authorModel: Schema.String;
|
|
3216
3272
|
readonly judgeModel: Schema.String;
|
|
@@ -3462,6 +3518,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
3462
3518
|
readonly configuration: Schema.Struct<{
|
|
3463
3519
|
readonly workloadDescription: Schema.String;
|
|
3464
3520
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
3521
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
3465
3522
|
readonly classifierModel: Schema.String;
|
|
3466
3523
|
readonly authorModel: Schema.String;
|
|
3467
3524
|
readonly judgeModel: Schema.String;
|
|
@@ -3755,6 +3812,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
3755
3812
|
readonly configuration: Schema.Struct<{
|
|
3756
3813
|
readonly workloadDescription: Schema.String;
|
|
3757
3814
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
3815
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
3758
3816
|
readonly classifierModel: Schema.String;
|
|
3759
3817
|
readonly authorModel: Schema.String;
|
|
3760
3818
|
readonly judgeModel: Schema.String;
|
|
@@ -4060,6 +4118,7 @@ export declare const EvalProjectDocument: Schema.Union<readonly [Schema.Union<re
|
|
|
4060
4118
|
readonly configuration: Schema.Struct<{
|
|
4061
4119
|
readonly workloadDescription: Schema.String;
|
|
4062
4120
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
4121
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
4063
4122
|
readonly classifierModel: Schema.String;
|
|
4064
4123
|
readonly authorModel: Schema.String;
|
|
4065
4124
|
readonly judgeModel: Schema.String;
|
|
@@ -4190,6 +4249,7 @@ export declare const EvalEvaluationProposal: Schema.Struct<{
|
|
|
4190
4249
|
readonly evaluationDigest: Schema.String;
|
|
4191
4250
|
readonly basisDigest: Schema.String;
|
|
4192
4251
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
4252
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
4193
4253
|
readonly judgeModel: Schema.String;
|
|
4194
4254
|
readonly suites: Schema.$Array<Schema.Struct<{
|
|
4195
4255
|
readonly version: Schema.Literal<1>;
|
|
@@ -4262,6 +4322,7 @@ export declare const EvalExecutionPlan: Schema.Struct<{
|
|
|
4262
4322
|
readonly basisDigest: Schema.String;
|
|
4263
4323
|
readonly evaluationDigest: Schema.String;
|
|
4264
4324
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
4325
|
+
readonly candidateReasoningEfforts: Schema.optionalKey<Schema.$Record<Schema.String, Schema.NonEmptyString>>;
|
|
4265
4326
|
readonly classifierModel: Schema.String;
|
|
4266
4327
|
readonly authorModel: Schema.String;
|
|
4267
4328
|
readonly judgeModel: Schema.String;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { DecompositionResult, DimensionScoreLabelV3, DimensionScorePredictionV3, EvalComparisonResult, PublishedRoutingActivation, RoutingBasisApprovalV3, RoutingBasisV3, RoutingActivationConstraints, RequestRoutingRequirements, RoutingObjectivePolicy, WorkloadDimension } from "@velum-labs/routekit-eval-contracts";
|
|
1
|
+
import { DecompositionResult, DimensionScoreLabelV3, DimensionScorePredictionV3, EvalCandidateReasoningEfforts, EvalComparisonResult, EvalReasoningEffort, PublishedRoutingActivation, RoutingBasisApprovalV3, RoutingBasisV3, RoutingActivationConstraints, RequestRoutingRequirements, RoutingObjectivePolicy, WorkloadDimension } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import { Schema } from "effect";
|
|
3
3
|
export const EVAL_PROJECT_VERSION = 1;
|
|
4
4
|
export const EVAL_PROJECT_V2_VERSION = 2;
|
|
@@ -6,6 +6,7 @@ export const EvalProjectQuestion = Schema.Struct({
|
|
|
6
6
|
id: Schema.Literals([
|
|
7
7
|
"workload-description",
|
|
8
8
|
"candidate-models",
|
|
9
|
+
"candidate-reasoning-efforts",
|
|
9
10
|
"classifier-model",
|
|
10
11
|
"author-model",
|
|
11
12
|
"judge-model",
|
|
@@ -20,24 +21,35 @@ const WorkloadDescriptionRequired = Schema.TaggedStruct("WorkloadDescriptionRequ
|
|
|
20
21
|
const CandidateModelsRequired = Schema.TaggedStruct("CandidateModelsRequired", {
|
|
21
22
|
workloadDescription: Schema.String
|
|
22
23
|
});
|
|
24
|
+
const CandidateReasoningEffortsRequired = Schema.TaggedStruct("CandidateReasoningEffortsRequired", {
|
|
25
|
+
workloadDescription: Schema.String,
|
|
26
|
+
candidateModels: Schema.Array(Schema.String),
|
|
27
|
+
candidateReasoningEfforts: EvalCandidateReasoningEfforts,
|
|
28
|
+
currentCandidateModel: Schema.String,
|
|
29
|
+
supportedReasoningEfforts: Schema.Record(Schema.String, Schema.Array(EvalReasoningEffort))
|
|
30
|
+
});
|
|
23
31
|
const ClassifierModelRequired = Schema.TaggedStruct("ClassifierModelRequired", {
|
|
24
32
|
workloadDescription: Schema.String,
|
|
25
|
-
candidateModels: Schema.Array(Schema.String)
|
|
33
|
+
candidateModels: Schema.Array(Schema.String),
|
|
34
|
+
candidateReasoningEfforts: EvalCandidateReasoningEfforts
|
|
26
35
|
});
|
|
27
36
|
const AuthorModelRequired = Schema.TaggedStruct("AuthorModelRequired", {
|
|
28
37
|
workloadDescription: Schema.String,
|
|
29
38
|
candidateModels: Schema.Array(Schema.String),
|
|
39
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
30
40
|
classifierModel: Schema.String
|
|
31
41
|
});
|
|
32
42
|
const JudgeModelRequired = Schema.TaggedStruct("JudgeModelRequired", {
|
|
33
43
|
workloadDescription: Schema.String,
|
|
34
44
|
candidateModels: Schema.Array(Schema.String),
|
|
45
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
35
46
|
classifierModel: Schema.String,
|
|
36
47
|
authorModel: Schema.String
|
|
37
48
|
});
|
|
38
49
|
const RoutingObjectiveRequired = Schema.TaggedStruct("RoutingObjectiveRequired", {
|
|
39
50
|
workloadDescription: Schema.String,
|
|
40
51
|
candidateModels: Schema.Array(Schema.String),
|
|
52
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
41
53
|
classifierModel: Schema.String,
|
|
42
54
|
authorModel: Schema.String,
|
|
43
55
|
judgeModel: Schema.String
|
|
@@ -45,6 +57,7 @@ const RoutingObjectiveRequired = Schema.TaggedStruct("RoutingObjectiveRequired",
|
|
|
45
57
|
const MaximumUnknownWeightRequired = Schema.TaggedStruct("MaximumUnknownWeightRequired", {
|
|
46
58
|
workloadDescription: Schema.String,
|
|
47
59
|
candidateModels: Schema.Array(Schema.String),
|
|
60
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
48
61
|
classifierModel: Schema.String,
|
|
49
62
|
authorModel: Schema.String,
|
|
50
63
|
judgeModel: Schema.String,
|
|
@@ -53,6 +66,7 @@ const MaximumUnknownWeightRequired = Schema.TaggedStruct("MaximumUnknownWeightRe
|
|
|
53
66
|
const RoutingConstraintsRequired = Schema.TaggedStruct("RoutingConstraintsRequired", {
|
|
54
67
|
workloadDescription: Schema.String,
|
|
55
68
|
candidateModels: Schema.Array(Schema.String),
|
|
69
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
56
70
|
classifierModel: Schema.String,
|
|
57
71
|
authorModel: Schema.String,
|
|
58
72
|
judgeModel: Schema.String,
|
|
@@ -62,6 +76,7 @@ const RoutingConstraintsRequired = Schema.TaggedStruct("RoutingConstraintsRequir
|
|
|
62
76
|
export const EvalProjectSetupProgress = Schema.Union([
|
|
63
77
|
WorkloadDescriptionRequired,
|
|
64
78
|
CandidateModelsRequired,
|
|
79
|
+
CandidateReasoningEffortsRequired,
|
|
65
80
|
ClassifierModelRequired,
|
|
66
81
|
AuthorModelRequired,
|
|
67
82
|
JudgeModelRequired,
|
|
@@ -72,6 +87,7 @@ export const EvalProjectSetupProgress = Schema.Union([
|
|
|
72
87
|
export const EvalProjectConfiguration = Schema.Struct({
|
|
73
88
|
workloadDescription: Schema.String,
|
|
74
89
|
candidateModels: Schema.Array(Schema.String),
|
|
90
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
75
91
|
classifierModel: Schema.String,
|
|
76
92
|
authorModel: Schema.String,
|
|
77
93
|
judgeModel: Schema.String,
|
|
@@ -550,6 +566,7 @@ export const EvalEvaluationProposal = Schema.Struct({
|
|
|
550
566
|
evaluationDigest: Schema.String,
|
|
551
567
|
basisDigest: Schema.String,
|
|
552
568
|
candidateModels: Schema.Array(Schema.String),
|
|
569
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
553
570
|
judgeModel: Schema.String,
|
|
554
571
|
suites: Schema.Array(EvalDimensionSuite),
|
|
555
572
|
decompositionBenchmark: EvalDecompositionBenchmark,
|
|
@@ -572,6 +589,7 @@ export const EvalExecutionPlan = Schema.Struct({
|
|
|
572
589
|
basisDigest: Schema.String,
|
|
573
590
|
evaluationDigest: Schema.String,
|
|
574
591
|
candidateModels: Schema.Array(Schema.String),
|
|
592
|
+
candidateReasoningEfforts: Schema.optionalKey(EvalCandidateReasoningEfforts),
|
|
575
593
|
classifierModel: Schema.String,
|
|
576
594
|
authorModel: Schema.String,
|
|
577
595
|
judgeModel: Schema.String,
|
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
import { Context, Effect, Layer, Path } from "effect";
|
|
2
2
|
import type { EvalProjectArtifactError, EvalSetupInspectionError } from "./errors.js";
|
|
3
|
-
import { EvalDimensionContrastError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
|
|
3
|
+
import { EvalDimensionContrastError, type EvalModelCatalogError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
|
|
4
4
|
import { EvalRepositoryInspector } from "./inspection.js";
|
|
5
5
|
import { EvalProjectArtifacts } from "./project-artifacts.js";
|
|
6
6
|
import type { EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProposedDimension, EvalProjectStatus, EvalRunReport } from "./project-contracts.js";
|
|
7
7
|
import { EvalProjectStore } from "./project-store.js";
|
|
8
|
-
|
|
8
|
+
import { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
9
|
+
export type EvalProjectWorkflowError = EvalProjectStoreError | EvalProjectArtifactError | EvalSetupInspectionError | EvalModelCatalogError | EvalDimensionContrastError | EvalProjectTransitionError;
|
|
9
10
|
export type EvalProjectWorkflowShape = {
|
|
10
11
|
readonly setup: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
11
12
|
readonly status: (repositoryRoot: string) => Effect.Effect<EvalProjectStatus | undefined, EvalProjectStoreError | EvalProjectArtifactError>;
|
|
12
|
-
readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError>;
|
|
13
|
+
readonly answer: (repositoryRoot: string, answer: string) => Effect.Effect<EvalProjectStatus, EvalProjectStoreError | EvalProjectArtifactError | EvalProjectTransitionError | EvalModelCatalogError>;
|
|
13
14
|
readonly proposeDimensions: (repositoryRoot: string, dimensions: readonly EvalProposedDimension[]) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
14
15
|
readonly approveDimensions: (repositoryRoot: string, basisDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
15
|
-
readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
16
|
+
readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "candidateReasoningEfforts" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
16
17
|
readonly approveEvaluations: (repositoryRoot: string, evaluationDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
17
18
|
readonly createPlan: (repositoryRoot: string, scope: EvalPlanScope) => Effect.Effect<EvalExecutionPlan, EvalProjectWorkflowError>;
|
|
18
19
|
readonly startRun: (repositoryRoot: string, planId: string) => Effect.Effect<{
|
|
@@ -27,6 +28,6 @@ export type EvalProjectWorkflowShape = {
|
|
|
27
28
|
declare const EvalProjectWorkflow_base: Context.ServiceClass<EvalProjectWorkflow, "@velum-labs/routekit-eval-setup/EvalProjectWorkflow", EvalProjectWorkflowShape>;
|
|
28
29
|
export declare class EvalProjectWorkflow extends EvalProjectWorkflow_base {
|
|
29
30
|
}
|
|
30
|
-
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
|
|
31
|
-
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector>;
|
|
31
|
+
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
|
+
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
33
|
export {};
|
package/dist/project-workflow.js
CHANGED
|
@@ -3,9 +3,11 @@ import { RoutingActivationConstraints, RoutingObjectivePolicy } from "@velum-lab
|
|
|
3
3
|
import { Clock, Context, Effect, Layer, Path, Schema } from "effect";
|
|
4
4
|
import { EvalDimensionContrastError, EvalProjectTransitionError } from "./errors.js";
|
|
5
5
|
import { EvalRepositoryInspector } from "./inspection.js";
|
|
6
|
+
import { candidateReasoningEffortFromAnswer } from "./model-selection.js";
|
|
6
7
|
import { EvalProjectArtifacts, evaluationProposalDigest, routingBasisDigest } from "./project-artifacts.js";
|
|
7
8
|
import { EVAL_PROJECT_VERSION, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
8
9
|
import { EvalProjectStore } from "./project-store.js";
|
|
10
|
+
import { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
9
11
|
const isoNow = Effect.map(Clock.currentTimeMillis, (millis) => new Date(millis).toISOString());
|
|
10
12
|
const questionForProgress = (progress) => {
|
|
11
13
|
switch (progress._tag) {
|
|
@@ -21,6 +23,12 @@ const questionForProgress = (progress) => {
|
|
|
21
23
|
prompt: "Which explicit provider/model IDs may RouteKit route to?",
|
|
22
24
|
options: []
|
|
23
25
|
};
|
|
26
|
+
case "CandidateReasoningEffortsRequired":
|
|
27
|
+
return {
|
|
28
|
+
id: "candidate-reasoning-efforts",
|
|
29
|
+
prompt: `Choose an advertised reasoning effort for ${progress.currentCandidateModel}.`,
|
|
30
|
+
options: [...progress.supportedReasoningEfforts[progress.currentCandidateModel]]
|
|
31
|
+
};
|
|
24
32
|
case "ClassifierModelRequired":
|
|
25
33
|
return {
|
|
26
34
|
id: "classifier-model",
|
|
@@ -124,7 +132,24 @@ const parseModel = (state, answer, role) => Effect.gen(function* () {
|
|
|
124
132
|
const model = yield* nonEmptyAnswer(state, answer);
|
|
125
133
|
return yield* validateModel(state, model, role);
|
|
126
134
|
});
|
|
135
|
+
const supportedCandidateReasoningEfforts = (candidates, input) => {
|
|
136
|
+
const supported = {};
|
|
137
|
+
for (const model of candidates) {
|
|
138
|
+
const efforts = [...new Set(input[model] ?? [])];
|
|
139
|
+
if (efforts.length > 0)
|
|
140
|
+
supported[model] = efforts;
|
|
141
|
+
}
|
|
142
|
+
return supported;
|
|
143
|
+
};
|
|
144
|
+
const firstReasoningCandidate = (candidates, supported) => candidates.find((model) => supported[model] !== undefined);
|
|
145
|
+
const nextReasoningCandidate = (candidates, supported, selected) => candidates.find((model) => supported[model] !== undefined && selected[model] === undefined);
|
|
127
146
|
const sameStrings = (left, right) => left.length === right.length && left.every((value, index) => value === right[index]);
|
|
147
|
+
const sameReasoningEfforts = (left, right) => {
|
|
148
|
+
const leftEntries = Object.entries(left ?? {});
|
|
149
|
+
const rightEntries = Object.entries(right ?? {});
|
|
150
|
+
return (leftEntries.length === rightEntries.length &&
|
|
151
|
+
leftEntries.every(([model, effort]) => right?.[model] === effort));
|
|
152
|
+
};
|
|
128
153
|
const sameLedger = (left, right) => left.expectedCalls === right.expectedCalls &&
|
|
129
154
|
left.observedCalls === right.observedCalls &&
|
|
130
155
|
left.observedCandidateRows === right.observedCandidateRows &&
|
|
@@ -300,6 +325,7 @@ const parseObjective = (state, answer) => Effect.gen(function* () {
|
|
|
300
325
|
const configurationFrom = (progress, constraints) => ({
|
|
301
326
|
workloadDescription: progress.workloadDescription,
|
|
302
327
|
candidateModels: progress.candidateModels,
|
|
328
|
+
candidateReasoningEfforts: progress.candidateReasoningEfforts ?? {},
|
|
303
329
|
classifierModel: progress.classifierModel,
|
|
304
330
|
authorModel: progress.authorModel,
|
|
305
331
|
judgeModel: progress.judgeModel,
|
|
@@ -324,7 +350,7 @@ const parseConstraints = (state, answer) => Effect.gen(function* () {
|
|
|
324
350
|
const constraints = yield* Schema.decodeUnknownEffect(RoutingActivationConstraints)(json).pipe(Effect.mapError(() => transitionError(state.stage, "routing constraints contain invalid dimension quality or failure-rate values")));
|
|
325
351
|
return Object.keys(constraints).length === 0 ? undefined : constraints;
|
|
326
352
|
});
|
|
327
|
-
const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
353
|
+
const advanceSetup = (state, answer, now, discoverReasoningEfforts) => Effect.gen(function* () {
|
|
328
354
|
const common = {
|
|
329
355
|
version: state.version,
|
|
330
356
|
projectId: state.projectId,
|
|
@@ -343,16 +369,58 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
343
369
|
workloadDescription: yield* nonEmptyAnswer(state, answer)
|
|
344
370
|
}
|
|
345
371
|
};
|
|
346
|
-
case "CandidateModelsRequired":
|
|
372
|
+
case "CandidateModelsRequired": {
|
|
373
|
+
const candidateModels = yield* parseCandidateModels(state, answer);
|
|
374
|
+
const supportedReasoningEfforts = supportedCandidateReasoningEfforts(candidateModels, yield* discoverReasoningEfforts(candidateModels));
|
|
375
|
+
const currentCandidateModel = firstReasoningCandidate(candidateModels, supportedReasoningEfforts);
|
|
347
376
|
return {
|
|
348
377
|
...common,
|
|
349
378
|
stage: "setup-required",
|
|
350
|
-
progress:
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
379
|
+
progress: currentCandidateModel === undefined
|
|
380
|
+
? {
|
|
381
|
+
_tag: "ClassifierModelRequired",
|
|
382
|
+
workloadDescription: state.progress.workloadDescription,
|
|
383
|
+
candidateModels,
|
|
384
|
+
candidateReasoningEfforts: {}
|
|
385
|
+
}
|
|
386
|
+
: {
|
|
387
|
+
_tag: "CandidateReasoningEffortsRequired",
|
|
388
|
+
workloadDescription: state.progress.workloadDescription,
|
|
389
|
+
candidateModels,
|
|
390
|
+
candidateReasoningEfforts: {},
|
|
391
|
+
currentCandidateModel,
|
|
392
|
+
supportedReasoningEfforts
|
|
393
|
+
}
|
|
355
394
|
};
|
|
395
|
+
}
|
|
396
|
+
case "CandidateReasoningEffortsRequired": {
|
|
397
|
+
const supportedReasoningEfforts = state.progress.supportedReasoningEfforts;
|
|
398
|
+
const currentCandidateModel = state.progress.currentCandidateModel;
|
|
399
|
+
const candidateReasoningEfforts = {
|
|
400
|
+
...state.progress.candidateReasoningEfforts,
|
|
401
|
+
[currentCandidateModel]: yield* Effect.try({
|
|
402
|
+
try: () => candidateReasoningEffortFromAnswer(answer, supportedReasoningEfforts[currentCandidateModel]),
|
|
403
|
+
catch: (cause) => transitionError(state.stage, cause instanceof Error ? cause.message : String(cause))
|
|
404
|
+
})
|
|
405
|
+
};
|
|
406
|
+
const nextCandidateModel = nextReasoningCandidate(state.progress.candidateModels, supportedReasoningEfforts, candidateReasoningEfforts);
|
|
407
|
+
return {
|
|
408
|
+
...common,
|
|
409
|
+
stage: "setup-required",
|
|
410
|
+
progress: nextCandidateModel === undefined
|
|
411
|
+
? {
|
|
412
|
+
_tag: "ClassifierModelRequired",
|
|
413
|
+
workloadDescription: state.progress.workloadDescription,
|
|
414
|
+
candidateModels: state.progress.candidateModels,
|
|
415
|
+
candidateReasoningEfforts
|
|
416
|
+
}
|
|
417
|
+
: {
|
|
418
|
+
...state.progress,
|
|
419
|
+
candidateReasoningEfforts,
|
|
420
|
+
currentCandidateModel: nextCandidateModel
|
|
421
|
+
}
|
|
422
|
+
};
|
|
423
|
+
}
|
|
356
424
|
case "ClassifierModelRequired":
|
|
357
425
|
return {
|
|
358
426
|
...common,
|
|
@@ -361,6 +429,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
361
429
|
_tag: "AuthorModelRequired",
|
|
362
430
|
workloadDescription: state.progress.workloadDescription,
|
|
363
431
|
candidateModels: state.progress.candidateModels,
|
|
432
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
364
433
|
classifierModel: yield* parseModel(state, answer, "classifier")
|
|
365
434
|
}
|
|
366
435
|
};
|
|
@@ -372,6 +441,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
372
441
|
_tag: "JudgeModelRequired",
|
|
373
442
|
workloadDescription: state.progress.workloadDescription,
|
|
374
443
|
candidateModels: state.progress.candidateModels,
|
|
444
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
375
445
|
classifierModel: state.progress.classifierModel,
|
|
376
446
|
authorModel: yield* parseModel(state, answer, "author")
|
|
377
447
|
}
|
|
@@ -384,6 +454,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
384
454
|
_tag: "RoutingObjectiveRequired",
|
|
385
455
|
workloadDescription: state.progress.workloadDescription,
|
|
386
456
|
candidateModels: state.progress.candidateModels,
|
|
457
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
387
458
|
classifierModel: state.progress.classifierModel,
|
|
388
459
|
authorModel: state.progress.authorModel,
|
|
389
460
|
judgeModel: yield* parseModel(state, answer, "judge")
|
|
@@ -398,6 +469,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
398
469
|
_tag: "MaximumUnknownWeightRequired",
|
|
399
470
|
workloadDescription: state.progress.workloadDescription,
|
|
400
471
|
candidateModels: state.progress.candidateModels,
|
|
472
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
401
473
|
classifierModel: state.progress.classifierModel,
|
|
402
474
|
authorModel: state.progress.authorModel,
|
|
403
475
|
judgeModel: state.progress.judgeModel,
|
|
@@ -413,6 +485,7 @@ const advanceSetup = (state, answer, now) => Effect.gen(function* () {
|
|
|
413
485
|
_tag: "RoutingConstraintsRequired",
|
|
414
486
|
workloadDescription: state.progress.workloadDescription,
|
|
415
487
|
candidateModels: state.progress.candidateModels,
|
|
488
|
+
candidateReasoningEfforts: state.progress.candidateReasoningEfforts,
|
|
416
489
|
classifierModel: state.progress.classifierModel,
|
|
417
490
|
authorModel: state.progress.authorModel,
|
|
418
491
|
judgeModel: state.progress.judgeModel,
|
|
@@ -438,6 +511,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
438
511
|
const store = yield* EvalProjectStore;
|
|
439
512
|
const artifacts = yield* EvalProjectArtifacts;
|
|
440
513
|
const inspector = yield* EvalRepositoryInspector;
|
|
514
|
+
const modelCatalog = yield* EvalModelCatalog;
|
|
441
515
|
const paths = yield* Path.Path;
|
|
442
516
|
const resolveRoot = (repositoryRoot) => paths.resolve(repositoryRoot);
|
|
443
517
|
const statusOf = (root, state) => Effect.gen(function* () {
|
|
@@ -514,7 +588,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
514
588
|
if (state.stage !== "setup-required") {
|
|
515
589
|
return yield* transitionError(state.stage, "project setup has no unanswered question");
|
|
516
590
|
}
|
|
517
|
-
const next = yield* advanceSetup(state, answerText, yield* isoNow);
|
|
591
|
+
const next = yield* advanceSetup(state, answerText, yield* isoNow, modelCatalog.reasoningEfforts);
|
|
518
592
|
yield* store.save(root, next);
|
|
519
593
|
return yield* statusOf(root, next);
|
|
520
594
|
});
|
|
@@ -663,6 +737,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
663
737
|
version: EVAL_PROJECT_VERSION,
|
|
664
738
|
basisDigest: state.basisDigest,
|
|
665
739
|
candidateModels: state.configuration.candidateModels,
|
|
740
|
+
...(state.configuration.candidateReasoningEfforts === undefined
|
|
741
|
+
? {}
|
|
742
|
+
: {
|
|
743
|
+
candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
|
|
744
|
+
}),
|
|
666
745
|
judgeModel: state.configuration.judgeModel,
|
|
667
746
|
suites: input.suites,
|
|
668
747
|
decompositionBenchmark: input.decompositionBenchmark,
|
|
@@ -764,6 +843,11 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
764
843
|
basisDigest: state.basisDigest,
|
|
765
844
|
evaluationDigest: state.evaluationDigest,
|
|
766
845
|
candidateModels: state.configuration.candidateModels,
|
|
846
|
+
...(state.configuration.candidateReasoningEfforts === undefined
|
|
847
|
+
? {}
|
|
848
|
+
: {
|
|
849
|
+
candidateReasoningEfforts: state.configuration.candidateReasoningEfforts
|
|
850
|
+
}),
|
|
767
851
|
classifierModel: state.configuration.classifierModel,
|
|
768
852
|
authorModel: state.configuration.authorModel,
|
|
769
853
|
judgeModel: state.configuration.judgeModel,
|
|
@@ -792,6 +876,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
792
876
|
plan.basisDigest !== state.basisDigest ||
|
|
793
877
|
plan.evaluationDigest !== state.evaluationDigest ||
|
|
794
878
|
!sameStrings(plan.candidateModels, state.configuration.candidateModels) ||
|
|
879
|
+
!sameReasoningEfforts(plan.candidateReasoningEfforts, state.configuration.candidateReasoningEfforts) ||
|
|
795
880
|
plan.classifierModel !== state.configuration.classifierModel ||
|
|
796
881
|
plan.authorModel !== state.configuration.authorModel ||
|
|
797
882
|
plan.judgeModel !== state.configuration.judgeModel) {
|
package/dist/questions.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { EvalSetupStage, EvalSetupState } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import type { RepositoryInspection, SetupQuestion } from "./types.js";
|
|
3
|
-
export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection) => SetupQuestion | undefined;
|
|
3
|
+
export declare const questionForStage: (stage: EvalSetupStage, inspection?: RepositoryInspection, reasoningEfforts?: readonly string[]) => SetupQuestion | undefined;
|
|
4
4
|
export declare const withOpenQuestion: (state: EvalSetupState, inspection?: RepositoryInspection) => {
|
|
5
5
|
readonly state: EvalSetupState;
|
|
6
6
|
readonly question?: SetupQuestion;
|
package/dist/questions.js
CHANGED
|
@@ -2,7 +2,7 @@ const firstThree = (values, fallback) => {
|
|
|
2
2
|
const unique = [...new Set(values.filter((value) => value.trim().length > 0))].slice(0, 3);
|
|
3
3
|
return [unique[0] ?? fallback[0], unique[1] ?? fallback[1], unique[2] ?? fallback[2]];
|
|
4
4
|
};
|
|
5
|
-
export const questionForStage = (stage, inspection) => {
|
|
5
|
+
export const questionForStage = (stage, inspection, reasoningEfforts = []) => {
|
|
6
6
|
switch (stage) {
|
|
7
7
|
case "surface":
|
|
8
8
|
return {
|
|
@@ -40,6 +40,12 @@ export const questionForStage = (stage, inspection) => {
|
|
|
40
40
|
"Help me find three explicit model IDs"
|
|
41
41
|
]
|
|
42
42
|
};
|
|
43
|
+
case "reasoning-effort":
|
|
44
|
+
return {
|
|
45
|
+
id: stage,
|
|
46
|
+
prompt: "Choose an explicit reasoning effort advertised by the live model catalog.",
|
|
47
|
+
options: [...new Set(reasoningEfforts)]
|
|
48
|
+
};
|
|
43
49
|
case "spend-approval":
|
|
44
50
|
return {
|
|
45
51
|
id: stage,
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { EvalReasoningEffort } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
+
import { Context, type Effect } from "effect";
|
|
3
|
+
import type { EvalModelCatalogError } from "../../errors.js";
|
|
4
|
+
export type EvalModelCatalogShape = {
|
|
5
|
+
readonly reasoningEfforts: (candidateModels: readonly string[]) => Effect.Effect<Readonly<Record<string, readonly EvalReasoningEffort[]>>, EvalModelCatalogError>;
|
|
6
|
+
};
|
|
7
|
+
declare const EvalModelCatalog_base: Context.ServiceClass<EvalModelCatalog, "@velum-labs/routekit-eval-setup/EvalModelCatalog", EvalModelCatalogShape>;
|
|
8
|
+
export declare class EvalModelCatalog extends EvalModelCatalog_base {
|
|
9
|
+
}
|
|
10
|
+
export {};
|
|
@@ -1,12 +1,17 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
2
|
import { test } from "node:test";
|
|
3
|
-
import { modelSelectionFromAnswer } from "../model-selection.js";
|
|
3
|
+
import { candidateReasoningEffortFromAnswer, modelSelectionFromAnswer } from "../model-selection.js";
|
|
4
4
|
test("model selection accepts exactly two candidates followed by a distinct judge", () => {
|
|
5
5
|
assert.deepEqual(modelSelectionFromAnswer("openai/gpt-a anthropic/claude-b google/gemini-judge"), {
|
|
6
6
|
candidates: ["openai/gpt-a", "anthropic/claude-b"],
|
|
7
7
|
judgeModel: "google/gemini-judge"
|
|
8
8
|
});
|
|
9
9
|
});
|
|
10
|
+
test("reasoning effort selection accepts exactly one advertised opaque value", () => {
|
|
11
|
+
const supported = ["none", "low", "provider-next"];
|
|
12
|
+
assert.equal(candidateReasoningEffortFromAnswer(" provider-next ", supported), "provider-next");
|
|
13
|
+
assert.throws(() => candidateReasoningEffortFromAnswer("max", supported), /must be one of/iu);
|
|
14
|
+
});
|
|
10
15
|
test("model selection rejects canned answers, duplicates, aliases, and extra ids", () => {
|
|
11
16
|
assert.throws(() => modelSelectionFromAnswer("Current model, a cheaper candidate, and a stronger candidate"), /exactly three explicit provider\/model ids/iu);
|
|
12
17
|
assert.throws(() => modelSelectionFromAnswer("openai/a openai/a openai/judge"), /three unique model ids/iu);
|
|
@@ -5,16 +5,30 @@ import path from "node:path";
|
|
|
5
5
|
import { after, test } from "node:test";
|
|
6
6
|
import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
|
|
7
7
|
import { Cause, Effect, Layer } from "effect";
|
|
8
|
-
import { EvalDimensionContrastError } from "../errors.js";
|
|
8
|
+
import { EvalDimensionContrastError, EvalModelCatalogError } from "../errors.js";
|
|
9
9
|
import { EvalRepositoryInspectorLive } from "../inspection.js";
|
|
10
10
|
import { EvalProjectArtifactsLive } from "../project-artifacts.js";
|
|
11
11
|
import { scopedExpectedCalls, summarizeEvalRunLedger } from "../project-contracts.js";
|
|
12
12
|
import { EvalProjectStoreLive } from "../project-store.js";
|
|
13
13
|
import { EvalProjectWorkflow, EvalProjectWorkflowLive } from "../project-workflow.js";
|
|
14
|
+
import { EvalModelCatalog } from "../services/model-catalog/service.js";
|
|
14
15
|
const roots = [];
|
|
15
16
|
after(async () => Promise.all(roots.map((root) => rm(root, { recursive: true, force: true }))));
|
|
16
17
|
const ProjectDependenciesLive = Layer.mergeAll(EvalProjectStoreLive, EvalProjectArtifactsLive, EvalRepositoryInspectorLive).pipe(Layer.provide(NodeServicesLayer));
|
|
17
|
-
const
|
|
18
|
+
const catalogRequests = [];
|
|
19
|
+
const catalogEfforts = (candidateModels) => Effect.sync(() => {
|
|
20
|
+
catalogRequests.push([...candidateModels]);
|
|
21
|
+
return {
|
|
22
|
+
...(candidateModels.includes("openai/gpt-5.6-luna")
|
|
23
|
+
? { "openai/gpt-5.6-luna": ["low", "medium", "ultra"] }
|
|
24
|
+
: {}),
|
|
25
|
+
...(candidateModels.includes("openai/gpt-5.6-terra")
|
|
26
|
+
? { "openai/gpt-5.6-terra": ["none", "low", "high"] }
|
|
27
|
+
: {})
|
|
28
|
+
};
|
|
29
|
+
});
|
|
30
|
+
const EvalModelCatalogTest = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({ reasoningEfforts: catalogEfforts }));
|
|
31
|
+
const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(EvalModelCatalogTest), Layer.provide(NodeServicesLayer));
|
|
18
32
|
const makeRepository = async () => {
|
|
19
33
|
const root = await mkdtemp(path.join(os.tmpdir(), "routekit-eval-project-"));
|
|
20
34
|
roots.push(root);
|
|
@@ -47,22 +61,47 @@ test("setup creates one durable repository project without invoking model-backed
|
|
|
47
61
|
});
|
|
48
62
|
test("answers advance exactly one question and persist a complete compositional configuration", async () => {
|
|
49
63
|
const root = await makeRepository();
|
|
64
|
+
catalogRequests.length = 0;
|
|
50
65
|
const statuses = await Effect.runPromise(Effect.gen(function* () {
|
|
51
66
|
const workflow = yield* EvalProjectWorkflow;
|
|
52
67
|
yield* workflow.setup(root);
|
|
53
68
|
const workload = yield* workflow.answer(root, "Customer support and documentation requests");
|
|
54
69
|
const candidates = yield* workflow.answer(root, "openai/gpt-5.6-luna, openai/gpt-5.6-terra openai/gpt-5.6-sol");
|
|
70
|
+
const lunaEffort = yield* workflow.answer(root, candidates.question.options[1]);
|
|
71
|
+
const terraEffort = yield* workflow.answer(root, lunaEffort.question.options[1]);
|
|
55
72
|
const classifier = yield* workflow.answer(root, "openai/gpt-5.6-luna");
|
|
56
73
|
const author = yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
57
74
|
const judge = yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
58
75
|
const objective = yield* workflow.answer(root, '{"kind":"highest-quality"}');
|
|
59
76
|
const unknown = yield* workflow.answer(root, "0.2");
|
|
60
77
|
const completed = yield* workflow.answer(root, "{}");
|
|
61
|
-
return {
|
|
78
|
+
return {
|
|
79
|
+
workload,
|
|
80
|
+
candidates,
|
|
81
|
+
lunaEffort,
|
|
82
|
+
terraEffort,
|
|
83
|
+
classifier,
|
|
84
|
+
author,
|
|
85
|
+
judge,
|
|
86
|
+
objective,
|
|
87
|
+
unknown,
|
|
88
|
+
completed
|
|
89
|
+
};
|
|
62
90
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
63
91
|
assert.equal(statuses.workload.question?.id, "candidate-models");
|
|
64
92
|
assert.equal(statuses.workload.state.revision, 1);
|
|
65
|
-
assert.
|
|
93
|
+
assert.deepEqual(catalogRequests, [[
|
|
94
|
+
"openai/gpt-5.6-luna",
|
|
95
|
+
"openai/gpt-5.6-terra",
|
|
96
|
+
"openai/gpt-5.6-sol"
|
|
97
|
+
]]);
|
|
98
|
+
assert.equal(statuses.candidates.question?.id, "candidate-reasoning-efforts");
|
|
99
|
+
assert.match(statuses.candidates.question?.prompt ?? "", /openai\/gpt-5\.6-luna/u);
|
|
100
|
+
assert.deepEqual(statuses.candidates.question?.options, ["low", "medium", "ultra"]);
|
|
101
|
+
assert.equal(statuses.lunaEffort.question?.id, "candidate-reasoning-efforts");
|
|
102
|
+
assert.match(statuses.lunaEffort.question?.prompt ?? "", /openai\/gpt-5\.6-terra/u);
|
|
103
|
+
assert.deepEqual(statuses.lunaEffort.question?.options, ["none", "low", "high"]);
|
|
104
|
+
assert.equal(statuses.terraEffort.question?.id, "classifier-model");
|
|
66
105
|
assert.equal(statuses.classifier.question?.id, "author-model");
|
|
67
106
|
assert.equal(statuses.author.question?.id, "judge-model");
|
|
68
107
|
assert.equal(statuses.judge.question?.id, "routing-objective");
|
|
@@ -79,6 +118,10 @@ test("answers advance exactly one question and persist a complete compositional
|
|
|
79
118
|
"openai/gpt-5.6-terra",
|
|
80
119
|
"openai/gpt-5.6-sol"
|
|
81
120
|
]);
|
|
121
|
+
assert.deepEqual(statuses.completed.state.configuration.candidateReasoningEfforts, {
|
|
122
|
+
"openai/gpt-5.6-luna": "medium",
|
|
123
|
+
"openai/gpt-5.6-terra": "low"
|
|
124
|
+
});
|
|
82
125
|
assert.equal(statuses.completed.state.configuration.classifierModel, "openai/gpt-5.6-luna");
|
|
83
126
|
assert.equal(statuses.completed.state.configuration.authorModel, "openai/gpt-5.6-terra");
|
|
84
127
|
assert.equal(statuses.completed.state.configuration.judgeModel, "openai/gpt-5.6-terra");
|
|
@@ -91,7 +134,7 @@ test("answers advance exactly one question and persist a complete compositional
|
|
|
91
134
|
const workflow = yield* EvalProjectWorkflow;
|
|
92
135
|
return yield* workflow.status(root);
|
|
93
136
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
94
|
-
assert.equal(resumed?.state.revision,
|
|
137
|
+
assert.equal(resumed?.state.revision, 10);
|
|
95
138
|
assert.equal(resumed?.state.stage, "dimensions-review");
|
|
96
139
|
});
|
|
97
140
|
test("invalid answers fail without changing the durable project revision", async () => {
|
|
@@ -112,6 +155,30 @@ test("invalid answers fail without changing the durable project revision", async
|
|
|
112
155
|
assert.equal(afterDuplicate?.question?.id, "candidate-models");
|
|
113
156
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
114
157
|
});
|
|
158
|
+
test("candidate selection fails closed when catalog reasoning metadata is unavailable", async () => {
|
|
159
|
+
const root = await makeRepository();
|
|
160
|
+
const unavailableCatalog = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({
|
|
161
|
+
reasoningEfforts: () => Effect.fail(new EvalModelCatalogError({
|
|
162
|
+
detail: "models.list did not advertise candidate reasoning capabilities"
|
|
163
|
+
}))
|
|
164
|
+
}));
|
|
165
|
+
const workflowLayer = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(unavailableCatalog), Layer.provide(NodeServicesLayer));
|
|
166
|
+
await Effect.runPromise(Effect.gen(function* () {
|
|
167
|
+
const workflow = yield* EvalProjectWorkflow;
|
|
168
|
+
yield* workflow.setup(root);
|
|
169
|
+
yield* workflow.answer(root, "Production workload");
|
|
170
|
+
const answer = yield* Effect.exit(workflow.answer(root, "openai/reasoning openai/plain"));
|
|
171
|
+
assert.equal(answer._tag, "Failure");
|
|
172
|
+
const current = yield* workflow.status(root);
|
|
173
|
+
assert.equal(current?.state.revision, 1);
|
|
174
|
+
assert.equal(current?.state.stage, "setup-required");
|
|
175
|
+
if (current?.state.stage !== "setup-required") {
|
|
176
|
+
assert.fail("expected setup-required state");
|
|
177
|
+
}
|
|
178
|
+
assert.equal(current.state.progress._tag, "CandidateModelsRequired");
|
|
179
|
+
assert.equal(current.question?.id, "candidate-models");
|
|
180
|
+
}).pipe(Effect.provide(workflowLayer)));
|
|
181
|
+
});
|
|
115
182
|
test("corrupt project state preserves the typed project-store failure", async () => {
|
|
116
183
|
const root = await makeRepository();
|
|
117
184
|
await mkdir(path.join(root, ".routekit", "evals"), { recursive: true });
|
|
@@ -195,7 +262,11 @@ async function completeProjectSetup(root, candidateModels = "openai/gpt-5.6-luna
|
|
|
195
262
|
const workflow = yield* EvalProjectWorkflow;
|
|
196
263
|
yield* workflow.setup(root);
|
|
197
264
|
yield* workflow.answer(root, "Production workload");
|
|
198
|
-
yield* workflow.answer(root, candidateModels);
|
|
265
|
+
let status = yield* workflow.answer(root, candidateModels);
|
|
266
|
+
while (status.state.stage === "setup-required" &&
|
|
267
|
+
status.state.progress._tag === "CandidateReasoningEffortsRequired") {
|
|
268
|
+
status = yield* workflow.answer(root, status.question.options[0]);
|
|
269
|
+
}
|
|
199
270
|
yield* workflow.answer(root, "openai/gpt-5.6-luna");
|
|
200
271
|
yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
201
272
|
yield* workflow.answer(root, "openai/gpt-5.6-terra");
|
|
@@ -732,6 +803,10 @@ run.toComplete();
|
|
|
732
803
|
assert.equal(compositionManifestAfter, compositionManifestBefore);
|
|
733
804
|
const dimensionManifest = JSON.parse(dimensionManifestAfter);
|
|
734
805
|
const compositionManifest = JSON.parse(compositionManifestAfter);
|
|
806
|
+
assert.deepEqual(dimensionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
|
|
807
|
+
assert.deepEqual(compositionManifest.candidateReasoningEfforts, plan.candidateReasoningEfforts ?? {});
|
|
808
|
+
assert.match(dimensionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
|
|
809
|
+
assert.match(compositionSuite, /parameters: \{ reasoning: \{ effort \} \}/u);
|
|
735
810
|
assert.equal(dimensionManifest.expectedCallCount, selection.caseIds.length * plan.candidateModels.length * 2);
|
|
736
811
|
assert.equal(compositionManifest.expectedCallCount, plan.selectedCompositionCaseIds.length * plan.candidateModels.length * 2);
|
|
737
812
|
});
|
|
@@ -8,13 +8,14 @@ test("setup exposes exactly one question for each unresolved stage", () => {
|
|
|
8
8
|
"criteria",
|
|
9
9
|
"constraints",
|
|
10
10
|
"candidates",
|
|
11
|
+
"reasoning-effort",
|
|
11
12
|
"spend-approval",
|
|
12
13
|
"publish"
|
|
13
14
|
];
|
|
14
15
|
for (const stage of stages) {
|
|
15
16
|
const question = questionForStage(stage);
|
|
16
17
|
assert.equal(question?.id, stage);
|
|
17
|
-
assert.equal(question?.options.length,
|
|
18
|
+
assert.equal(question?.options.length === 3 || stage === "reasoning-effort", true);
|
|
18
19
|
}
|
|
19
20
|
assert.equal(questionForStage("completed"), undefined);
|
|
20
21
|
});
|
|
@@ -46,3 +47,8 @@ test("candidate question requires two candidates and a distinct judge", () => {
|
|
|
46
47
|
assert.match(question?.prompt ?? "", /exactly three unique provider\/model IDs/iu);
|
|
47
48
|
assert.doesNotMatch(question?.options.join("\n") ?? "", /save without choosing|current model,/iu);
|
|
48
49
|
});
|
|
50
|
+
test("reasoning-effort stage exposes an explicit picker without a hidden default", () => {
|
|
51
|
+
const question = questionForStage("reasoning-effort", undefined, ["low", "provider-next"]);
|
|
52
|
+
assert.match(question?.prompt ?? "", /explicit reasoning effort/iu);
|
|
53
|
+
assert.deepEqual(question?.options, ["low", "provider-next"]);
|
|
54
|
+
});
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-setup",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.3.1",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -31,9 +31,9 @@
|
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"effect": "4.0.0-rc.108",
|
|
34
|
-
"@velum-labs/routekit-eval-contracts": "1.
|
|
35
|
-
"@velum-labs/routekit-eval-core": "1.
|
|
36
|
-
"@velum-labs/routekit-runtime": "1.
|
|
34
|
+
"@velum-labs/routekit-eval-contracts": "1.3.1",
|
|
35
|
+
"@velum-labs/routekit-eval-core": "1.3.1",
|
|
36
|
+
"@velum-labs/routekit-runtime": "1.3.1"
|
|
37
37
|
},
|
|
38
38
|
"keywords": [
|
|
39
39
|
"routekit",
|