@velum-labs/routekit-eval-setup 1.0.21 → 1.0.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,7 +14,7 @@ export type EvalProjectArtifactsShape = {
14
14
  readonly loadEvaluationsApproval: (repositoryRoot: string) => Effect.Effect<EvalArtifactApproval | undefined, EvalProjectArtifactError, never>;
15
15
  readonly saveEvaluationsApproval: (repositoryRoot: string, approval: EvalArtifactApproval) => Effect.Effect<void, EvalProjectArtifactError, never>;
16
16
  readonly savePlan: (repositoryRoot: string, plan: EvalExecutionPlan) => Effect.Effect<void, EvalProjectArtifactError, never>;
17
- readonly materializePlanSuites: (repositoryRoot: string, plan: EvalExecutionPlan, proposal: EvalEvaluationProposal) => Effect.Effect<void, EvalProjectArtifactError, never>;
17
+ readonly materializePlanSuites: (repositoryRoot: string, plan: EvalExecutionPlan, proposal: EvalEvaluationProposal, mode?: "create" | "rematerialize") => Effect.Effect<void, EvalProjectArtifactError, never>;
18
18
  readonly planSuitePath: (repositoryRoot: string, planId: string, dimensionId: string) => Effect.Effect<string, EvalProjectArtifactError, never>;
19
19
  readonly compositionSuitePath: (repositoryRoot: string, planId: string) => Effect.Effect<string, EvalProjectArtifactError, never>;
20
20
  readonly loadPlan: (repositoryRoot: string, planId: string) => Effect.Effect<EvalExecutionPlan | undefined, EvalProjectArtifactError, never>;
@@ -30,8 +30,16 @@ for (const model of manifest.candidateModels) {
30
30
  ? testCase.prompt
31
31
  : [testCase.prompt, "", "Reference material:", "-----", testCase.context, "-----"].join("\\n");
32
32
  const run = await candidate.run({ prompt, caseId: testCase.id });
33
- run.toComplete();
33
+ let candidateCompletionError: unknown;
34
+ try {
35
+ run.toComplete();
36
+ } catch (error) {
37
+ candidateCompletionError = error;
38
+ }
34
39
  await judge.autoEvals({ criteria: testCase.rubric, prompt, run });
40
+ if (candidateCompletionError !== undefined) {
41
+ throw candidateCompletionError;
42
+ }
35
43
  });
36
44
  }
37
45
  }
@@ -226,7 +234,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
226
234
  loadEvaluationsApproval: (repositoryRoot) => read(repositoryRoot, EVALUATIONS_APPROVAL, Schema.decodeUnknownEffect(EvalArtifactApprovalSchema)),
227
235
  saveEvaluationsApproval: (repositoryRoot, approval) => write(repositoryRoot, EVALUATIONS_APPROVAL, approval),
228
236
  savePlan: (repositoryRoot, plan) => write(repositoryRoot, paths.join("plans", `${plan.planId}.json`), plan, false),
229
- materializePlanSuites: (repositoryRoot, plan, proposal) => Effect.gen(function* () {
237
+ materializePlanSuites: (repositoryRoot, plan, proposal, mode = "create") => Effect.gen(function* () {
230
238
  yield* requireArtifactId("plan", plan.planId);
231
239
  const suites = new Map(proposal.suites.map((suite) => [suite.dimensionId, suite]));
232
240
  for (const selection of plan.selectedCaseIds) {
@@ -246,8 +254,8 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
246
254
  return testCase;
247
255
  });
248
256
  const suiteRoot = paths.join("plans", plan.planId, "dimensions", selection.dimensionId);
249
- yield* writeText(repositoryRoot, paths.join(suiteRoot, `${selection.dimensionId}.eval.ts`), renderDimensionSuite(), false);
250
- yield* write(repositoryRoot, paths.join(suiteRoot, "data", "cases.json"), cases, false);
257
+ yield* writeText(repositoryRoot, paths.join(suiteRoot, `${selection.dimensionId}.eval.ts`), renderDimensionSuite(), mode === "rematerialize");
258
+ yield* write(repositoryRoot, paths.join(suiteRoot, "data", "cases.json"), cases, mode === "rematerialize");
251
259
  yield* write(repositoryRoot, paths.join(suiteRoot, "routekit.eval-manifest.json"), {
252
260
  version: EVAL_PROJECT_VERSION,
253
261
  profileId: selection.dimensionId,
@@ -257,7 +265,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
257
265
  caseIds: cases.map((testCase) => testCase.id),
258
266
  maxOutputTokens: suite.maximumOutputTokens,
259
267
  expectedCallCount: cases.length * plan.candidateModels.length * 2
260
- }, false);
268
+ }, mode === "rematerialize");
261
269
  }
262
270
  const compositionById = new Map(proposal.compositionSuite.cases.map((testCase) => [testCase.id, testCase]));
263
271
  const compositionCases = plan.selectedCompositionCaseIds.map((caseId) => {
@@ -268,8 +276,8 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
268
276
  return testCase;
269
277
  });
270
278
  const compositionRoot = paths.join("plans", plan.planId, "composition");
271
- yield* writeText(repositoryRoot, paths.join(compositionRoot, "composition.eval.ts"), renderDimensionSuite(), false);
272
- yield* write(repositoryRoot, paths.join(compositionRoot, "data", "cases.json"), compositionCases, false);
279
+ yield* writeText(repositoryRoot, paths.join(compositionRoot, "composition.eval.ts"), renderDimensionSuite(), mode === "rematerialize");
280
+ yield* write(repositoryRoot, paths.join(compositionRoot, "data", "cases.json"), compositionCases, mode === "rematerialize");
273
281
  yield* write(repositoryRoot, paths.join(compositionRoot, "routekit.eval-manifest.json"), {
274
282
  version: EVAL_PROJECT_VERSION,
275
283
  profileId: "composition",
@@ -279,7 +287,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
279
287
  caseIds: compositionCases.map((testCase) => testCase.id),
280
288
  maxOutputTokens: proposal.compositionSuite.maximumOutputTokens,
281
289
  expectedCallCount: compositionCases.length * plan.candidateModels.length * 2
282
- }, false);
290
+ }, mode === "rematerialize");
283
291
  }).pipe(Effect.mapError((cause) => cause instanceof EvalProjectArtifactError
284
292
  ? cause
285
293
  : artifactFailure("writing", artifactPath(repositoryRoot, paths.join("plans", plan.planId)), cause))),
@@ -782,6 +782,13 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
782
782
  plan.judgeModel !== state.configuration.judgeModel) {
783
783
  return yield* transitionError(state.stage, "execution plan is stale or does not match the approved project");
784
784
  }
785
+ const proposal = yield* artifacts.loadEvaluationProposal(root);
786
+ if (proposal === undefined ||
787
+ proposal.evaluationDigest !== plan.evaluationDigest ||
788
+ proposal.basisDigest !== plan.basisDigest) {
789
+ return yield* transitionError(state.stage, "the approved evaluation proposal for the execution plan is missing or stale");
790
+ }
791
+ yield* artifacts.materializePlanSuites(root, plan, proposal, "rematerialize");
785
792
  const now = yield* isoNow;
786
793
  const runId = crypto.randomUUID();
787
794
  const next = {
@@ -306,6 +306,14 @@ test("reviewed artifacts are digest-bound and produce an immutable exact-call pl
306
306
  const planPath = path.join(root, ".routekit", "evals", "plans", `${result.plan.planId}.json`);
307
307
  assert.equal((await stat(planPath)).mode & 0o777, 0o600);
308
308
  assert.deepEqual(JSON.parse(await readFile(planPath, "utf8")), result.plan);
309
+ const firstDimension = reviewedDimensions[0];
310
+ const generatedSuite = await readFile(path.join(root, ".routekit", "evals", "plans", result.plan.planId, "dimensions", firstDimension.id, `${firstDimension.id}.eval.ts`), "utf8");
311
+ const completionAssertion = generatedSuite.indexOf("run.toComplete();");
312
+ const judgeCall = generatedSuite.indexOf("await judge.autoEvals");
313
+ const completionRethrow = generatedSuite.indexOf("throw candidateCompletionError;");
314
+ assert.ok(completionAssertion >= 0);
315
+ assert.ok(judgeCall > completionAssertion);
316
+ assert.ok(completionRethrow > judgeCall);
309
317
  for (const dimension of reviewedDimensions) {
310
318
  assert.equal((await stat(path.join(root, ".routekit", "evals", "dimensions", dimension.id, "suite.json")))
311
319
  .mode & 0o777, 0o600);
@@ -613,6 +621,55 @@ test("failed runs preserve plan revisions for retry while mismatched plans remai
613
621
  assert.equal(rerun.ready.state.evaluationDigest, plan.evaluationDigest);
614
622
  assert.equal(rerun.second.plan.planId, plan.planId);
615
623
  });
624
+ test("startRun rematerializes stale dimension and composition suite templates", async () => {
625
+ const root = await makeRepository();
626
+ const plan = await createPlan(root, "pilot");
627
+ const selection = plan.selectedCaseIds[0];
628
+ const planRoot = path.join(root, ".routekit", "evals", "plans", plan.planId);
629
+ const dimensionRoot = path.join(planRoot, "dimensions", selection.dimensionId);
630
+ const dimensionSuitePath = path.join(dimensionRoot, `${selection.dimensionId}.eval.ts`);
631
+ const compositionRoot = path.join(planRoot, "composition");
632
+ const compositionSuitePath = path.join(compositionRoot, "composition.eval.ts");
633
+ const dimensionManifestPath = path.join(dimensionRoot, "routekit.eval-manifest.json");
634
+ const compositionManifestPath = path.join(compositionRoot, "routekit.eval-manifest.json");
635
+ const staleTemplate = `// generated by RouteKit 1.0.21
636
+ run.toComplete();
637
+ await judge.autoEvals({ criteria: testCase.rubric, prompt, run });
638
+ `;
639
+ await Promise.all([
640
+ writeFile(dimensionSuitePath, staleTemplate),
641
+ writeFile(compositionSuitePath, staleTemplate)
642
+ ]);
643
+ const [dimensionManifestBefore, compositionManifestBefore] = await Promise.all([
644
+ readFile(dimensionManifestPath, "utf8"),
645
+ readFile(compositionManifestPath, "utf8")
646
+ ]);
647
+ await Effect.runPromise(Effect.gen(function* () {
648
+ const workflow = yield* EvalProjectWorkflow;
649
+ yield* workflow.startRun(root, plan.planId);
650
+ }).pipe(Effect.provide(ProjectWorkflowTestLive)));
651
+ const [dimensionSuite, compositionSuite, dimensionManifestAfter, compositionManifestAfter] = await Promise.all([
652
+ readFile(dimensionSuitePath, "utf8"),
653
+ readFile(compositionSuitePath, "utf8"),
654
+ readFile(dimensionManifestPath, "utf8"),
655
+ readFile(compositionManifestPath, "utf8")
656
+ ]);
657
+ for (const suite of [dimensionSuite, compositionSuite]) {
658
+ const completionAssertion = suite.indexOf("run.toComplete();");
659
+ const judgeCall = suite.indexOf("await judge.autoEvals");
660
+ const completionRethrow = suite.indexOf("throw candidateCompletionError;");
661
+ assert.doesNotMatch(suite, /generated by RouteKit 1\.0\.21/u);
662
+ assert.ok(completionAssertion >= 0);
663
+ assert.ok(judgeCall > completionAssertion);
664
+ assert.ok(completionRethrow > judgeCall);
665
+ }
666
+ assert.equal(dimensionManifestAfter, dimensionManifestBefore);
667
+ assert.equal(compositionManifestAfter, compositionManifestBefore);
668
+ const dimensionManifest = JSON.parse(dimensionManifestAfter);
669
+ const compositionManifest = JSON.parse(compositionManifestAfter);
670
+ assert.equal(dimensionManifest.expectedCallCount, selection.caseIds.length * plan.candidateModels.length * 2);
671
+ assert.equal(compositionManifest.expectedCallCount, plan.selectedCompositionCaseIds.length * plan.candidateModels.length * 2);
672
+ });
616
673
  test("failed run reports persist the nested qualification error chain", async () => {
617
674
  const root = await makeRepository();
618
675
  const plan = await createPlan(root, "full");
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-eval-setup",
3
3
  "private": false,
4
- "version": "1.0.21",
4
+ "version": "1.0.23",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -32,8 +32,8 @@
32
32
  },
33
33
  "dependencies": {
34
34
  "effect": "4.0.0-rc.108",
35
- "@velum-labs/routekit-eval-contracts": "1.0.21",
36
- "@velum-labs/routekit-runtime": "1.0.21"
35
+ "@velum-labs/routekit-eval-contracts": "1.0.23",
36
+ "@velum-labs/routekit-runtime": "1.0.23"
37
37
  },
38
38
  "keywords": [
39
39
  "routekit",