@velum-labs/routekit-eval-setup 1.0.22 → 1.0.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,7 +14,7 @@ export type EvalProjectArtifactsShape = {
14
14
  readonly loadEvaluationsApproval: (repositoryRoot: string) => Effect.Effect<EvalArtifactApproval | undefined, EvalProjectArtifactError, never>;
15
15
  readonly saveEvaluationsApproval: (repositoryRoot: string, approval: EvalArtifactApproval) => Effect.Effect<void, EvalProjectArtifactError, never>;
16
16
  readonly savePlan: (repositoryRoot: string, plan: EvalExecutionPlan) => Effect.Effect<void, EvalProjectArtifactError, never>;
17
- readonly materializePlanSuites: (repositoryRoot: string, plan: EvalExecutionPlan, proposal: EvalEvaluationProposal) => Effect.Effect<void, EvalProjectArtifactError, never>;
17
+ readonly materializePlanSuites: (repositoryRoot: string, plan: EvalExecutionPlan, proposal: EvalEvaluationProposal, mode?: "create" | "rematerialize") => Effect.Effect<void, EvalProjectArtifactError, never>;
18
18
  readonly planSuitePath: (repositoryRoot: string, planId: string, dimensionId: string) => Effect.Effect<string, EvalProjectArtifactError, never>;
19
19
  readonly compositionSuitePath: (repositoryRoot: string, planId: string) => Effect.Effect<string, EvalProjectArtifactError, never>;
20
20
  readonly loadPlan: (repositoryRoot: string, planId: string) => Effect.Effect<EvalExecutionPlan | undefined, EvalProjectArtifactError, never>;
@@ -30,16 +30,8 @@ for (const model of manifest.candidateModels) {
30
30
  ? testCase.prompt
31
31
  : [testCase.prompt, "", "Reference material:", "-----", testCase.context, "-----"].join("\\n");
32
32
  const run = await candidate.run({ prompt, caseId: testCase.id });
33
- let candidateCompletionError: unknown;
34
- try {
35
- run.toComplete();
36
- } catch (error) {
37
- candidateCompletionError = error;
38
- }
39
33
  await judge.autoEvals({ criteria: testCase.rubric, prompt, run });
40
- if (candidateCompletionError !== undefined) {
41
- throw candidateCompletionError;
42
- }
34
+ run.toComplete();
43
35
  });
44
36
  }
45
37
  }
@@ -234,7 +226,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
234
226
  loadEvaluationsApproval: (repositoryRoot) => read(repositoryRoot, EVALUATIONS_APPROVAL, Schema.decodeUnknownEffect(EvalArtifactApprovalSchema)),
235
227
  saveEvaluationsApproval: (repositoryRoot, approval) => write(repositoryRoot, EVALUATIONS_APPROVAL, approval),
236
228
  savePlan: (repositoryRoot, plan) => write(repositoryRoot, paths.join("plans", `${plan.planId}.json`), plan, false),
237
- materializePlanSuites: (repositoryRoot, plan, proposal) => Effect.gen(function* () {
229
+ materializePlanSuites: (repositoryRoot, plan, proposal, mode = "create") => Effect.gen(function* () {
238
230
  yield* requireArtifactId("plan", plan.planId);
239
231
  const suites = new Map(proposal.suites.map((suite) => [suite.dimensionId, suite]));
240
232
  for (const selection of plan.selectedCaseIds) {
@@ -254,8 +246,8 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
254
246
  return testCase;
255
247
  });
256
248
  const suiteRoot = paths.join("plans", plan.planId, "dimensions", selection.dimensionId);
257
- yield* writeText(repositoryRoot, paths.join(suiteRoot, `${selection.dimensionId}.eval.ts`), renderDimensionSuite(), false);
258
- yield* write(repositoryRoot, paths.join(suiteRoot, "data", "cases.json"), cases, false);
249
+ yield* writeText(repositoryRoot, paths.join(suiteRoot, `${selection.dimensionId}.eval.ts`), renderDimensionSuite(), mode === "rematerialize");
250
+ yield* write(repositoryRoot, paths.join(suiteRoot, "data", "cases.json"), cases, mode === "rematerialize");
259
251
  yield* write(repositoryRoot, paths.join(suiteRoot, "routekit.eval-manifest.json"), {
260
252
  version: EVAL_PROJECT_VERSION,
261
253
  profileId: selection.dimensionId,
@@ -265,7 +257,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
265
257
  caseIds: cases.map((testCase) => testCase.id),
266
258
  maxOutputTokens: suite.maximumOutputTokens,
267
259
  expectedCallCount: cases.length * plan.candidateModels.length * 2
268
- }, false);
260
+ }, mode === "rematerialize");
269
261
  }
270
262
  const compositionById = new Map(proposal.compositionSuite.cases.map((testCase) => [testCase.id, testCase]));
271
263
  const compositionCases = plan.selectedCompositionCaseIds.map((caseId) => {
@@ -276,8 +268,8 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
276
268
  return testCase;
277
269
  });
278
270
  const compositionRoot = paths.join("plans", plan.planId, "composition");
279
- yield* writeText(repositoryRoot, paths.join(compositionRoot, "composition.eval.ts"), renderDimensionSuite(), false);
280
- yield* write(repositoryRoot, paths.join(compositionRoot, "data", "cases.json"), compositionCases, false);
271
+ yield* writeText(repositoryRoot, paths.join(compositionRoot, "composition.eval.ts"), renderDimensionSuite(), mode === "rematerialize");
272
+ yield* write(repositoryRoot, paths.join(compositionRoot, "data", "cases.json"), compositionCases, mode === "rematerialize");
281
273
  yield* write(repositoryRoot, paths.join(compositionRoot, "routekit.eval-manifest.json"), {
282
274
  version: EVAL_PROJECT_VERSION,
283
275
  profileId: "composition",
@@ -287,7 +279,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
287
279
  caseIds: compositionCases.map((testCase) => testCase.id),
288
280
  maxOutputTokens: proposal.compositionSuite.maximumOutputTokens,
289
281
  expectedCallCount: compositionCases.length * plan.candidateModels.length * 2
290
- }, false);
282
+ }, mode === "rematerialize");
291
283
  }).pipe(Effect.mapError((cause) => cause instanceof EvalProjectArtifactError
292
284
  ? cause
293
285
  : artifactFailure("writing", artifactPath(repositoryRoot, paths.join("plans", plan.planId)), cause))),
@@ -782,6 +782,13 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
782
782
  plan.judgeModel !== state.configuration.judgeModel) {
783
783
  return yield* transitionError(state.stage, "execution plan is stale or does not match the approved project");
784
784
  }
785
+ const proposal = yield* artifacts.loadEvaluationProposal(root);
786
+ if (proposal === undefined ||
787
+ proposal.evaluationDigest !== plan.evaluationDigest ||
788
+ proposal.basisDigest !== plan.basisDigest) {
789
+ return yield* transitionError(state.stage, "the approved evaluation proposal for the execution plan is missing or stale");
790
+ }
791
+ yield* artifacts.materializePlanSuites(root, plan, proposal, "rematerialize");
785
792
  const now = yield* isoNow;
786
793
  const runId = crypto.randomUUID();
787
794
  const next = {
@@ -310,10 +310,9 @@ test("reviewed artifacts are digest-bound and produce an immutable exact-call pl
310
310
  const generatedSuite = await readFile(path.join(root, ".routekit", "evals", "plans", result.plan.planId, "dimensions", firstDimension.id, `${firstDimension.id}.eval.ts`), "utf8");
311
311
  const completionAssertion = generatedSuite.indexOf("run.toComplete();");
312
312
  const judgeCall = generatedSuite.indexOf("await judge.autoEvals");
313
- const completionRethrow = generatedSuite.indexOf("throw candidateCompletionError;");
313
+ assert.ok(judgeCall >= 0);
314
314
  assert.ok(completionAssertion >= 0);
315
- assert.ok(judgeCall > completionAssertion);
316
- assert.ok(completionRethrow > judgeCall);
315
+ assert.ok(completionAssertion > judgeCall);
317
316
  for (const dimension of reviewedDimensions) {
318
317
  assert.equal((await stat(path.join(root, ".routekit", "evals", "dimensions", dimension.id, "suite.json")))
319
318
  .mode & 0o777, 0o600);
@@ -621,6 +620,54 @@ test("failed runs preserve plan revisions for retry while mismatched plans remai
621
620
  assert.equal(rerun.ready.state.evaluationDigest, plan.evaluationDigest);
622
621
  assert.equal(rerun.second.plan.planId, plan.planId);
623
622
  });
623
+ test("startRun rematerializes stale dimension and composition suite templates", async () => {
624
+ const root = await makeRepository();
625
+ const plan = await createPlan(root, "pilot");
626
+ const selection = plan.selectedCaseIds[0];
627
+ const planRoot = path.join(root, ".routekit", "evals", "plans", plan.planId);
628
+ const dimensionRoot = path.join(planRoot, "dimensions", selection.dimensionId);
629
+ const dimensionSuitePath = path.join(dimensionRoot, `${selection.dimensionId}.eval.ts`);
630
+ const compositionRoot = path.join(planRoot, "composition");
631
+ const compositionSuitePath = path.join(compositionRoot, "composition.eval.ts");
632
+ const dimensionManifestPath = path.join(dimensionRoot, "routekit.eval-manifest.json");
633
+ const compositionManifestPath = path.join(compositionRoot, "routekit.eval-manifest.json");
634
+ const staleTemplate = `// generated by RouteKit 1.0.21
635
+ run.toComplete();
636
+ await judge.autoEvals({ criteria: testCase.rubric, prompt, run });
637
+ `;
638
+ await Promise.all([
639
+ writeFile(dimensionSuitePath, staleTemplate),
640
+ writeFile(compositionSuitePath, staleTemplate)
641
+ ]);
642
+ const [dimensionManifestBefore, compositionManifestBefore] = await Promise.all([
643
+ readFile(dimensionManifestPath, "utf8"),
644
+ readFile(compositionManifestPath, "utf8")
645
+ ]);
646
+ await Effect.runPromise(Effect.gen(function* () {
647
+ const workflow = yield* EvalProjectWorkflow;
648
+ yield* workflow.startRun(root, plan.planId);
649
+ }).pipe(Effect.provide(ProjectWorkflowTestLive)));
650
+ const [dimensionSuite, compositionSuite, dimensionManifestAfter, compositionManifestAfter] = await Promise.all([
651
+ readFile(dimensionSuitePath, "utf8"),
652
+ readFile(compositionSuitePath, "utf8"),
653
+ readFile(dimensionManifestPath, "utf8"),
654
+ readFile(compositionManifestPath, "utf8")
655
+ ]);
656
+ for (const suite of [dimensionSuite, compositionSuite]) {
657
+ const completionAssertion = suite.indexOf("run.toComplete();");
658
+ const judgeCall = suite.indexOf("await judge.autoEvals");
659
+ assert.doesNotMatch(suite, /generated by RouteKit 1\.0\.21/u);
660
+ assert.ok(judgeCall >= 0);
661
+ assert.ok(completionAssertion >= 0);
662
+ assert.ok(completionAssertion > judgeCall);
663
+ }
664
+ assert.equal(dimensionManifestAfter, dimensionManifestBefore);
665
+ assert.equal(compositionManifestAfter, compositionManifestBefore);
666
+ const dimensionManifest = JSON.parse(dimensionManifestAfter);
667
+ const compositionManifest = JSON.parse(compositionManifestAfter);
668
+ assert.equal(dimensionManifest.expectedCallCount, selection.caseIds.length * plan.candidateModels.length * 2);
669
+ assert.equal(compositionManifest.expectedCallCount, plan.selectedCompositionCaseIds.length * plan.candidateModels.length * 2);
670
+ });
624
671
  test("failed run reports persist the nested qualification error chain", async () => {
625
672
  const root = await makeRepository();
626
673
  const plan = await createPlan(root, "full");
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-eval-setup",
3
3
  "private": false,
4
- "version": "1.0.22",
4
+ "version": "1.0.24",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -32,8 +32,8 @@
32
32
  },
33
33
  "dependencies": {
34
34
  "effect": "4.0.0-rc.108",
35
- "@velum-labs/routekit-eval-contracts": "1.0.22",
36
- "@velum-labs/routekit-runtime": "1.0.22"
35
+ "@velum-labs/routekit-eval-contracts": "1.0.24",
36
+ "@velum-labs/routekit-runtime": "1.0.24"
37
37
  },
38
38
  "keywords": [
39
39
  "routekit",