@velum-labs/routekit-eval-setup 1.0.16 → 1.0.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts
CHANGED
|
@@ -7,8 +7,8 @@ export type { OriEvalAuthoringApi, OriEvalResult } from "./ori-result.js";
|
|
|
7
7
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
8
8
|
export type { EvalAuthoringCompletion, EvalAuthoringSource, EvalAuthoringTransportShape, EvalProjectAuthorShape } from "./project-authoring.js";
|
|
9
9
|
export { EVAL_AUTHORING_CASES_PER_DIMENSION, EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS, EVAL_AUTHORING_REQUEST_BYTES, EVAL_AUTHORING_SOURCE_BYTES, EVAL_AUTHORING_SOURCE_FILES, EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
10
|
-
export type { EvalClassifierObservation, EvalCompositionCase, EvalCompositionCaseResult, EvalCompositionSuite, EvalDecompositionBenchmark, EvalDecompositionBenchmarkCase, EvalDimensionCase, EvalDimensionSuite, EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProjectArtifactsStatus, EvalProjectStatus, EvalRunCleanup, EvalRunLedger, EvalRunQualification, EvalRunReport, EvalRunTarget } from "./project-contracts.js";
|
|
11
|
-
export { EVAL_PROJECT_VERSION, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectSetupProgress, EvalProjectState, EvalRunCleanup as EvalRunCleanupSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
10
|
+
export type { EvalClassifierObservation, EvalCompositionCase, EvalCompositionCaseResult, EvalCompositionSuite, EvalDecompositionBenchmark, EvalDecompositionBenchmarkCase, EvalDimensionCase, EvalDimensionSuite, EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProjectArtifactsStatus, EvalProjectStatus, EvalRunCleanup, EvalRunFailureError, EvalRunLedger, EvalRunQualification, EvalRunReport, EvalRunTarget } from "./project-contracts.js";
|
|
11
|
+
export { EVAL_PROJECT_VERSION, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectSetupProgress, EvalProjectState, EvalRunCleanup as EvalRunCleanupSchema, EvalRunFailureError as EvalRunFailureErrorSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
12
12
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
13
13
|
export type { EvalProjectWorkflowError, EvalProjectWorkflowShape } from "./project-workflow.js";
|
|
14
14
|
export { EvalProjectWorkflow, EvalProjectWorkflowLive, makeEvalProjectWorkflow } from "./project-workflow.js";
|
package/dist/index.js
CHANGED
|
@@ -4,7 +4,7 @@ export { EvalRepositoryInspector, EvalRepositoryInspectorLive, inspectRepository
|
|
|
4
4
|
export { OriEvalAuthoring, oriAuthoringFromApi } from "./ori-authoring.js";
|
|
5
5
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
6
6
|
export { EVAL_AUTHORING_CASES_PER_DIMENSION, EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS, EVAL_AUTHORING_REQUEST_BYTES, EVAL_AUTHORING_SOURCE_BYTES, EVAL_AUTHORING_SOURCE_FILES, EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
7
|
-
export { EVAL_PROJECT_VERSION, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectSetupProgress, EvalProjectState, EvalRunCleanup as EvalRunCleanupSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
7
|
+
export { EVAL_PROJECT_VERSION, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectSetupProgress, EvalProjectState, EvalRunCleanup as EvalRunCleanupSchema, EvalRunFailureError as EvalRunFailureErrorSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
8
8
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
9
9
|
export { EvalProjectWorkflow, EvalProjectWorkflowLive, makeEvalProjectWorkflow } from "./project-workflow.js";
|
|
10
10
|
export { questionForStage, withOpenQuestion } from "./questions.js";
|
|
@@ -747,6 +747,12 @@ export declare const EvalRunQualification: Schema.Struct<{
|
|
|
747
747
|
}>;
|
|
748
748
|
export type EvalRunQualification = typeof EvalRunQualification.Type;
|
|
749
749
|
export declare function summarizeEvalRunLedger(comparisons: readonly EvalComparisonResultType[], expectedCalls: number, classifierObservations?: readonly EvalClassifierObservation[]): EvalRunLedger;
|
|
750
|
+
export declare const EvalRunFailureError: Schema.Struct<{
|
|
751
|
+
readonly name: Schema.String;
|
|
752
|
+
readonly message: Schema.String;
|
|
753
|
+
readonly stack: Schema.optionalKey<Schema.String>;
|
|
754
|
+
}>;
|
|
755
|
+
export type EvalRunFailureError = typeof EvalRunFailureError.Type;
|
|
750
756
|
/**
|
|
751
757
|
* Sanitized durable result of one immutable execution plan.
|
|
752
758
|
*
|
|
@@ -1078,6 +1084,11 @@ export declare const EvalRunReport: Schema.Union<readonly [Schema.Struct<{
|
|
|
1078
1084
|
}>, Schema.Struct<{
|
|
1079
1085
|
readonly status: Schema.Literal<"failed">;
|
|
1080
1086
|
readonly failure: Schema.String;
|
|
1087
|
+
readonly errors: Schema.optionalKey<Schema.$Array<Schema.Struct<{
|
|
1088
|
+
readonly name: Schema.String;
|
|
1089
|
+
readonly message: Schema.String;
|
|
1090
|
+
readonly stack: Schema.optionalKey<Schema.String>;
|
|
1091
|
+
}>>>;
|
|
1081
1092
|
readonly version: Schema.Literal<1>;
|
|
1082
1093
|
readonly runId: Schema.String;
|
|
1083
1094
|
readonly planId: Schema.String;
|
|
@@ -366,6 +366,11 @@ const EvalRunReportCommon = {
|
|
|
366
366
|
qualification: Schema.optionalKey(EvalRunQualification),
|
|
367
367
|
ledger: EvalRunLedger
|
|
368
368
|
};
|
|
369
|
+
export const EvalRunFailureError = Schema.Struct({
|
|
370
|
+
name: Schema.String,
|
|
371
|
+
message: Schema.String,
|
|
372
|
+
stack: Schema.optionalKey(Schema.String)
|
|
373
|
+
});
|
|
369
374
|
const QualifiedEvalRunReport = Schema.Struct({
|
|
370
375
|
...EvalRunReportCommon,
|
|
371
376
|
status: Schema.Literal("passed"),
|
|
@@ -379,7 +384,8 @@ const CompletedEvalRunReport = Schema.Struct({
|
|
|
379
384
|
const FailedEvalRunReport = Schema.Struct({
|
|
380
385
|
...EvalRunReportCommon,
|
|
381
386
|
status: Schema.Literal("failed"),
|
|
382
|
-
failure: Schema.String
|
|
387
|
+
failure: Schema.String,
|
|
388
|
+
errors: Schema.optionalKey(Schema.Array(EvalRunFailureError))
|
|
383
389
|
});
|
|
384
390
|
/**
|
|
385
391
|
* Sanitized durable result of one immutable execution plan.
|
|
@@ -476,6 +476,77 @@ test("run lifecycle retains complete sanitized evidence and activates only after
|
|
|
476
476
|
const reportPath = path.join(root, ".routekit", "evals", "runs", outcome.started.runId, "report.json");
|
|
477
477
|
assert.equal((await stat(reportPath)).mode & 0o777, 0o600);
|
|
478
478
|
});
|
|
479
|
+
test("failed run reports persist the nested qualification error chain", async () => {
|
|
480
|
+
const root = await makeRepository();
|
|
481
|
+
const plan = await createPlan(root, "full");
|
|
482
|
+
const outcome = await Effect.runPromise(Effect.gen(function* () {
|
|
483
|
+
const workflow = yield* EvalProjectWorkflow;
|
|
484
|
+
const started = yield* workflow.startRun(root, plan.planId);
|
|
485
|
+
const status = yield* workflow.status(root);
|
|
486
|
+
const report = {
|
|
487
|
+
version: 1,
|
|
488
|
+
status: "failed",
|
|
489
|
+
runId: started.runId,
|
|
490
|
+
planId: plan.planId,
|
|
491
|
+
projectId: status.state.projectId,
|
|
492
|
+
startedAt: "2026-08-21T03:47:27.000Z",
|
|
493
|
+
finishedAt: "2026-08-21T03:47:48.000Z",
|
|
494
|
+
basisDigest: plan.basisDigest,
|
|
495
|
+
evaluationDigest: plan.evaluationDigest,
|
|
496
|
+
target: {
|
|
497
|
+
kind: "configured",
|
|
498
|
+
identity: "routekit-generation:48",
|
|
499
|
+
publishAllowed: true
|
|
500
|
+
},
|
|
501
|
+
cleanup: { sessionOpened: true, sessionClosed: true },
|
|
502
|
+
comparisons: [],
|
|
503
|
+
ledger: {
|
|
504
|
+
expectedCalls: plan.expectedCallCount,
|
|
505
|
+
observedCalls: 0,
|
|
506
|
+
observedCandidateRows: 0,
|
|
507
|
+
knownInputTokens: 0,
|
|
508
|
+
knownOutputTokens: 0,
|
|
509
|
+
unknownTokenMeasurements: 0,
|
|
510
|
+
knownPricedSubtotalUsd: 0,
|
|
511
|
+
unpricedCalls: 0
|
|
512
|
+
},
|
|
513
|
+
failure: 'qualification dimension "provider-protocol-translation" failed ' +
|
|
514
|
+
"(per-test timeout 600000ms); observed call ids: none",
|
|
515
|
+
errors: [
|
|
516
|
+
{
|
|
517
|
+
name: "EvalServiceComparisonError",
|
|
518
|
+
message: "RouteKit Eval comparison failed",
|
|
519
|
+
stack: "EvalServiceComparisonError: RouteKit Eval comparison failed"
|
|
520
|
+
},
|
|
521
|
+
{
|
|
522
|
+
name: "RouteKitEvalGatewayBridgeStartError",
|
|
523
|
+
message: "Could not start the scoped RouteKit Eval gateway bridge.",
|
|
524
|
+
stack: "RouteKitEvalGatewayBridgeStartError: Could not start the scoped RouteKit Eval gateway bridge."
|
|
525
|
+
}
|
|
526
|
+
]
|
|
527
|
+
};
|
|
528
|
+
yield* workflow.failRun(root, report);
|
|
529
|
+
return {
|
|
530
|
+
loaded: yield* workflow.result(root, started.runId),
|
|
531
|
+
runId: started.runId
|
|
532
|
+
};
|
|
533
|
+
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
534
|
+
assert.equal(outcome.loaded?.status, "failed");
|
|
535
|
+
assert.deepEqual(outcome.loaded?.status === "failed" ? outcome.loaded.errors : undefined, [
|
|
536
|
+
{
|
|
537
|
+
name: "EvalServiceComparisonError",
|
|
538
|
+
message: "RouteKit Eval comparison failed",
|
|
539
|
+
stack: "EvalServiceComparisonError: RouteKit Eval comparison failed"
|
|
540
|
+
},
|
|
541
|
+
{
|
|
542
|
+
name: "RouteKitEvalGatewayBridgeStartError",
|
|
543
|
+
message: "Could not start the scoped RouteKit Eval gateway bridge.",
|
|
544
|
+
stack: "RouteKitEvalGatewayBridgeStartError: Could not start the scoped RouteKit Eval gateway bridge."
|
|
545
|
+
}
|
|
546
|
+
]);
|
|
547
|
+
const raw = JSON.parse(await readFile(path.join(root, ".routekit", "evals", "runs", outcome.runId, "report.json"), "utf8"));
|
|
548
|
+
assert.equal(raw.errors?.length, 2);
|
|
549
|
+
});
|
|
479
550
|
test("incomplete evidence and external qualification fail closed", async () => {
|
|
480
551
|
const incompleteRoot = await makeRepository();
|
|
481
552
|
const incompletePlan = await createPlan(incompleteRoot, "full");
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-setup",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.0.
|
|
4
|
+
"version": "1.0.17",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -32,8 +32,8 @@
|
|
|
32
32
|
},
|
|
33
33
|
"dependencies": {
|
|
34
34
|
"effect": "4.0.0-rc.108",
|
|
35
|
-
"@velum-labs/routekit-eval-contracts": "1.0.
|
|
36
|
-
"@velum-labs/routekit-runtime": "1.0.
|
|
35
|
+
"@velum-labs/routekit-eval-contracts": "1.0.17",
|
|
36
|
+
"@velum-labs/routekit-runtime": "1.0.17"
|
|
37
37
|
},
|
|
38
38
|
"keywords": [
|
|
39
39
|
"routekit",
|