@velum-labs/routekit-eval-setup 1.0.20 → 1.0.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -30,8 +30,16 @@ for (const model of manifest.candidateModels) {
|
|
|
30
30
|
? testCase.prompt
|
|
31
31
|
: [testCase.prompt, "", "Reference material:", "-----", testCase.context, "-----"].join("\\n");
|
|
32
32
|
const run = await candidate.run({ prompt, caseId: testCase.id });
|
|
33
|
-
|
|
33
|
+
let candidateCompletionError: unknown;
|
|
34
|
+
try {
|
|
35
|
+
run.toComplete();
|
|
36
|
+
} catch (error) {
|
|
37
|
+
candidateCompletionError = error;
|
|
38
|
+
}
|
|
34
39
|
await judge.autoEvals({ criteria: testCase.rubric, prompt, run });
|
|
40
|
+
if (candidateCompletionError !== undefined) {
|
|
41
|
+
throw candidateCompletionError;
|
|
42
|
+
}
|
|
35
43
|
});
|
|
36
44
|
}
|
|
37
45
|
}
|
package/dist/project-workflow.js
CHANGED
|
@@ -787,7 +787,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
787
787
|
const next = {
|
|
788
788
|
version: state.version,
|
|
789
789
|
projectId: state.projectId,
|
|
790
|
-
revision: state.revision
|
|
790
|
+
revision: state.revision,
|
|
791
791
|
createdAt: state.createdAt,
|
|
792
792
|
updatedAt: now,
|
|
793
793
|
sourceInventory: state.sourceInventory,
|
|
@@ -879,7 +879,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
879
879
|
const next = {
|
|
880
880
|
version: state.version,
|
|
881
881
|
projectId: state.projectId,
|
|
882
|
-
revision: state.revision
|
|
882
|
+
revision: state.revision,
|
|
883
883
|
createdAt: state.createdAt,
|
|
884
884
|
updatedAt: report.finishedAt,
|
|
885
885
|
sourceInventory: state.sourceInventory,
|
|
@@ -306,6 +306,14 @@ test("reviewed artifacts are digest-bound and produce an immutable exact-call pl
|
|
|
306
306
|
const planPath = path.join(root, ".routekit", "evals", "plans", `${result.plan.planId}.json`);
|
|
307
307
|
assert.equal((await stat(planPath)).mode & 0o777, 0o600);
|
|
308
308
|
assert.deepEqual(JSON.parse(await readFile(planPath, "utf8")), result.plan);
|
|
309
|
+
const firstDimension = reviewedDimensions[0];
|
|
310
|
+
const generatedSuite = await readFile(path.join(root, ".routekit", "evals", "plans", result.plan.planId, "dimensions", firstDimension.id, `${firstDimension.id}.eval.ts`), "utf8");
|
|
311
|
+
const completionAssertion = generatedSuite.indexOf("run.toComplete();");
|
|
312
|
+
const judgeCall = generatedSuite.indexOf("await judge.autoEvals");
|
|
313
|
+
const completionRethrow = generatedSuite.indexOf("throw candidateCompletionError;");
|
|
314
|
+
assert.ok(completionAssertion >= 0);
|
|
315
|
+
assert.ok(judgeCall > completionAssertion);
|
|
316
|
+
assert.ok(completionRethrow > judgeCall);
|
|
309
317
|
for (const dimension of reviewedDimensions) {
|
|
310
318
|
assert.equal((await stat(path.join(root, ".routekit", "evals", "dimensions", dimension.id, "suite.json")))
|
|
311
319
|
.mode & 0o777, 0o600);
|
|
@@ -542,6 +550,77 @@ test("run lifecycle retains complete sanitized evidence and activates only after
|
|
|
542
550
|
const reportPath = path.join(root, ".routekit", "evals", "runs", outcome.started.runId, "report.json");
|
|
543
551
|
assert.equal((await stat(reportPath)).mode & 0o777, 0o600);
|
|
544
552
|
});
|
|
553
|
+
test("failed runs preserve plan revisions for retry while mismatched plans remain stale", async () => {
|
|
554
|
+
const root = await makeRepository();
|
|
555
|
+
const plan = await createPlan(root, "full");
|
|
556
|
+
const mismatchedPlans = [
|
|
557
|
+
{
|
|
558
|
+
...plan,
|
|
559
|
+
planId: `${plan.planId}-project`,
|
|
560
|
+
projectId: "different-project-id"
|
|
561
|
+
},
|
|
562
|
+
{
|
|
563
|
+
...plan,
|
|
564
|
+
planId: `${plan.planId}-evaluation`,
|
|
565
|
+
evaluationDigest: "different-evaluation-digest"
|
|
566
|
+
},
|
|
567
|
+
{
|
|
568
|
+
...plan,
|
|
569
|
+
planId: `${plan.planId}-models`,
|
|
570
|
+
candidateModels: [...plan.candidateModels].reverse()
|
|
571
|
+
}
|
|
572
|
+
];
|
|
573
|
+
await Promise.all(mismatchedPlans.map((mismatchedPlan) => writeFile(path.join(root, ".routekit", "evals", "plans", `${mismatchedPlan.planId}.json`), JSON.stringify(mismatchedPlan))));
|
|
574
|
+
const rerun = await Effect.runPromise(Effect.gen(function* () {
|
|
575
|
+
const workflow = yield* EvalProjectWorkflow;
|
|
576
|
+
for (const mismatchedPlan of mismatchedPlans) {
|
|
577
|
+
const mismatch = yield* Effect.exit(workflow.startRun(root, mismatchedPlan.planId));
|
|
578
|
+
assert.equal(mismatch._tag, "Failure");
|
|
579
|
+
if (mismatch._tag === "Failure") {
|
|
580
|
+
assert.match(String(mismatch.cause), /execution plan is stale/u);
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
const first = yield* workflow.startRun(root, plan.planId);
|
|
584
|
+
const running = yield* workflow.status(root);
|
|
585
|
+
const ready = yield* workflow.failRun(root, {
|
|
586
|
+
version: 1,
|
|
587
|
+
status: "failed",
|
|
588
|
+
runId: first.runId,
|
|
589
|
+
planId: plan.planId,
|
|
590
|
+
projectId: running.state.projectId,
|
|
591
|
+
startedAt: "2026-08-21T03:47:27.000Z",
|
|
592
|
+
finishedAt: "2026-08-21T03:47:48.000Z",
|
|
593
|
+
basisDigest: plan.basisDigest,
|
|
594
|
+
evaluationDigest: plan.evaluationDigest,
|
|
595
|
+
target: {
|
|
596
|
+
kind: "configured",
|
|
597
|
+
identity: "routekit-generation:48",
|
|
598
|
+
publishAllowed: true
|
|
599
|
+
},
|
|
600
|
+
cleanup: { sessionOpened: true, sessionClosed: true },
|
|
601
|
+
comparisons: [],
|
|
602
|
+
ledger: {
|
|
603
|
+
expectedCalls: plan.expectedCallCount,
|
|
604
|
+
observedCalls: 0,
|
|
605
|
+
observedCandidateRows: 0,
|
|
606
|
+
knownInputTokens: 0,
|
|
607
|
+
knownOutputTokens: 0,
|
|
608
|
+
unknownTokenMeasurements: 0,
|
|
609
|
+
knownPricedSubtotalUsd: 0,
|
|
610
|
+
unpricedCalls: 0
|
|
611
|
+
},
|
|
612
|
+
failure: "qualification timed out before observing any calls",
|
|
613
|
+
errors: []
|
|
614
|
+
});
|
|
615
|
+
const second = yield* workflow.startRun(root, plan.planId);
|
|
616
|
+
return { ready, second };
|
|
617
|
+
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
618
|
+
assert.equal(rerun.ready.state.stage, "ready");
|
|
619
|
+
assert.equal(rerun.ready.state.revision, plan.projectRevision);
|
|
620
|
+
assert.equal(rerun.ready.state.basisDigest, plan.basisDigest);
|
|
621
|
+
assert.equal(rerun.ready.state.evaluationDigest, plan.evaluationDigest);
|
|
622
|
+
assert.equal(rerun.second.plan.planId, plan.planId);
|
|
623
|
+
});
|
|
545
624
|
test("failed run reports persist the nested qualification error chain", async () => {
|
|
546
625
|
const root = await makeRepository();
|
|
547
626
|
const plan = await createPlan(root, "full");
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-setup",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.0.
|
|
4
|
+
"version": "1.0.22",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -32,8 +32,8 @@
|
|
|
32
32
|
},
|
|
33
33
|
"dependencies": {
|
|
34
34
|
"effect": "4.0.0-rc.108",
|
|
35
|
-
"@velum-labs/routekit-eval-contracts": "1.0.
|
|
36
|
-
"@velum-labs/routekit-runtime": "1.0.
|
|
35
|
+
"@velum-labs/routekit-eval-contracts": "1.0.22",
|
|
36
|
+
"@velum-labs/routekit-runtime": "1.0.22"
|
|
37
37
|
},
|
|
38
38
|
"keywords": [
|
|
39
39
|
"routekit",
|