@velum-labs/routekit-eval-setup 1.3.1 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/effect-api.d.ts +2 -0
- package/dist/effect-api.js +1 -0
- package/dist/eval-event-log.d.ts +89 -0
- package/dist/eval-event-log.js +161 -0
- package/dist/index.d.ts +4 -2
- package/dist/index.js +2 -1
- package/dist/project-artifacts.d.ts +5 -0
- package/dist/project-artifacts.js +23 -15
- package/dist/project-contracts.d.ts +127 -2
- package/dist/project-contracts.js +95 -45
- package/dist/project-workflow.d.ts +4 -3
- package/dist/project-workflow.js +38 -17
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +92 -0
- package/dist/test/project-workflow.test.js +25 -9
- package/package.json +5 -4
package/dist/effect-api.d.ts
CHANGED
|
@@ -10,6 +10,8 @@ export { EvalRepositoryInspector, EvalRepositoryInspectorLive, inspectRepository
|
|
|
10
10
|
export { OriEvalAuthoring, oriAuthoringFromApi } from "./ori-authoring.js";
|
|
11
11
|
export type { OriEvalAuthoringApi, OriEvalResult } from "./ori-result.js";
|
|
12
12
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
13
|
+
export type { EvalCaseCursor, EvalNamedEvent, EvalNamedEventLog, EvalRunCallEvent } from "./eval-event-log.js";
|
|
14
|
+
export { assertEvalRunCanComplete, evalEventLabel, presentEvalEventLog, resumeEvalCaseCursors } from "./eval-event-log.js";
|
|
13
15
|
export type { EvalAuthoringCompletion, EvalAuthoringSource, EvalAuthoringTransportShape, EvalProjectAuthorShape } from "./project-authoring.js";
|
|
14
16
|
export { EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
15
17
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
package/dist/effect-api.js
CHANGED
|
@@ -6,6 +6,7 @@ export { BasisRepair, BasisRepairLive, makeBasisRepair } from "./services/basis-
|
|
|
6
6
|
export { EvalRepositoryInspector, EvalRepositoryInspectorLive, inspectRepository } from "./inspection.js";
|
|
7
7
|
export { OriEvalAuthoring, oriAuthoringFromApi } from "./ori-authoring.js";
|
|
8
8
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
9
|
+
export { assertEvalRunCanComplete, evalEventLabel, presentEvalEventLog, resumeEvalCaseCursors } from "./eval-event-log.js";
|
|
9
10
|
export { EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
10
11
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
11
12
|
export { EvalProjectWorkflow, EvalProjectWorkflowLive, makeEvalProjectWorkflow } from "./project-workflow.js";
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import { Schema } from "effect";
|
|
2
|
+
export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
|
|
3
|
+
readonly name: Schema.Literal<"plan">;
|
|
4
|
+
readonly at: Schema.String;
|
|
5
|
+
readonly runId: Schema.String;
|
|
6
|
+
readonly plannedCalls: Schema.Finite;
|
|
7
|
+
}>, Schema.Struct<{
|
|
8
|
+
readonly at: Schema.String;
|
|
9
|
+
readonly runId: Schema.String;
|
|
10
|
+
readonly dimensionId: Schema.String;
|
|
11
|
+
readonly caseId: Schema.String;
|
|
12
|
+
readonly sequence: Schema.Finite;
|
|
13
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
14
|
+
}>, Schema.Struct<{
|
|
15
|
+
readonly at: Schema.String;
|
|
16
|
+
readonly runId: Schema.String;
|
|
17
|
+
readonly dimensionId: Schema.String;
|
|
18
|
+
readonly caseId: Schema.String;
|
|
19
|
+
readonly sequence: Schema.Finite;
|
|
20
|
+
readonly name: Schema.Literal<"judge#">;
|
|
21
|
+
}>, Schema.Struct<{
|
|
22
|
+
readonly at: Schema.String;
|
|
23
|
+
readonly operationId: Schema.String;
|
|
24
|
+
readonly name: Schema.Literal<"basis">;
|
|
25
|
+
}>, Schema.Struct<{
|
|
26
|
+
readonly at: Schema.String;
|
|
27
|
+
readonly operationId: Schema.String;
|
|
28
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
29
|
+
}>, Schema.Struct<{
|
|
30
|
+
readonly at: Schema.String;
|
|
31
|
+
readonly operationId: Schema.String;
|
|
32
|
+
readonly name: Schema.Literal<"activation">;
|
|
33
|
+
}>]>;
|
|
34
|
+
export type EvalNamedEvent = typeof EvalNamedEvent.Type;
|
|
35
|
+
export declare const EvalNamedEventLog: Schema.$Array<Schema.Union<readonly [Schema.Struct<{
|
|
36
|
+
readonly name: Schema.Literal<"plan">;
|
|
37
|
+
readonly at: Schema.String;
|
|
38
|
+
readonly runId: Schema.String;
|
|
39
|
+
readonly plannedCalls: Schema.Finite;
|
|
40
|
+
}>, Schema.Struct<{
|
|
41
|
+
readonly at: Schema.String;
|
|
42
|
+
readonly runId: Schema.String;
|
|
43
|
+
readonly dimensionId: Schema.String;
|
|
44
|
+
readonly caseId: Schema.String;
|
|
45
|
+
readonly sequence: Schema.Finite;
|
|
46
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
47
|
+
}>, Schema.Struct<{
|
|
48
|
+
readonly at: Schema.String;
|
|
49
|
+
readonly runId: Schema.String;
|
|
50
|
+
readonly dimensionId: Schema.String;
|
|
51
|
+
readonly caseId: Schema.String;
|
|
52
|
+
readonly sequence: Schema.Finite;
|
|
53
|
+
readonly name: Schema.Literal<"judge#">;
|
|
54
|
+
}>, Schema.Struct<{
|
|
55
|
+
readonly at: Schema.String;
|
|
56
|
+
readonly operationId: Schema.String;
|
|
57
|
+
readonly name: Schema.Literal<"basis">;
|
|
58
|
+
}>, Schema.Struct<{
|
|
59
|
+
readonly at: Schema.String;
|
|
60
|
+
readonly operationId: Schema.String;
|
|
61
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
62
|
+
}>, Schema.Struct<{
|
|
63
|
+
readonly at: Schema.String;
|
|
64
|
+
readonly operationId: Schema.String;
|
|
65
|
+
readonly name: Schema.Literal<"activation">;
|
|
66
|
+
}>]>>;
|
|
67
|
+
export type EvalNamedEventLog = typeof EvalNamedEventLog.Type;
|
|
68
|
+
export type EvalRunCallEvent = Extract<EvalNamedEvent, {
|
|
69
|
+
readonly name: "candidate#" | "judge#";
|
|
70
|
+
}>;
|
|
71
|
+
export type EvalCaseCursor = {
|
|
72
|
+
readonly dimensionId: string;
|
|
73
|
+
readonly caseId: string;
|
|
74
|
+
readonly completePairs: number;
|
|
75
|
+
readonly lastEvent?: EvalRunCallEvent;
|
|
76
|
+
readonly hole?: Extract<EvalRunCallEvent, {
|
|
77
|
+
readonly name: "candidate#";
|
|
78
|
+
}>;
|
|
79
|
+
};
|
|
80
|
+
export declare function resumeEvalCaseCursors(log: EvalNamedEventLog): readonly EvalCaseCursor[];
|
|
81
|
+
export declare function evalEventLabel(event: EvalNamedEvent): string;
|
|
82
|
+
export type EvalEventLogPresentation = {
|
|
83
|
+
readonly failed: boolean;
|
|
84
|
+
readonly lines: readonly string[];
|
|
85
|
+
};
|
|
86
|
+
export declare function presentEvalEventLog(log: EvalNamedEventLog, options?: {
|
|
87
|
+
readonly failHoles?: boolean;
|
|
88
|
+
}): EvalEventLogPresentation;
|
|
89
|
+
export declare function assertEvalRunCanComplete(log: EvalNamedEventLog): void;
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
import { Schema } from "effect";
|
|
2
|
+
const NonNegativeInteger = Schema.Finite.pipe(Schema.check(Schema.makeFilter((value) => value >= 0 && Number.isInteger(value) ? undefined : "value must be a non-negative integer")));
|
|
3
|
+
const PositiveInteger = NonNegativeInteger.pipe(Schema.check(Schema.makeFilter((value) => value >= 1 ? undefined : "value must be a positive integer")));
|
|
4
|
+
const RunPlanEvent = Schema.Struct({
|
|
5
|
+
name: Schema.Literal("plan"),
|
|
6
|
+
at: Schema.String,
|
|
7
|
+
runId: Schema.String,
|
|
8
|
+
plannedCalls: NonNegativeInteger
|
|
9
|
+
});
|
|
10
|
+
const RunCallEventFields = {
|
|
11
|
+
at: Schema.String,
|
|
12
|
+
runId: Schema.String,
|
|
13
|
+
dimensionId: Schema.String,
|
|
14
|
+
caseId: Schema.String,
|
|
15
|
+
sequence: PositiveInteger
|
|
16
|
+
};
|
|
17
|
+
const RunCandidateEvent = Schema.Struct({
|
|
18
|
+
name: Schema.Literal("candidate#"),
|
|
19
|
+
...RunCallEventFields
|
|
20
|
+
});
|
|
21
|
+
const RunJudgeEvent = Schema.Struct({
|
|
22
|
+
name: Schema.Literal("judge#"),
|
|
23
|
+
...RunCallEventFields
|
|
24
|
+
});
|
|
25
|
+
const AuthoringEventFields = {
|
|
26
|
+
at: Schema.String,
|
|
27
|
+
operationId: Schema.String
|
|
28
|
+
};
|
|
29
|
+
const AuthoringBasisEvent = Schema.Struct({
|
|
30
|
+
name: Schema.Literal("basis"),
|
|
31
|
+
...AuthoringEventFields
|
|
32
|
+
});
|
|
33
|
+
const AuthoringDimensionsEvent = Schema.Struct({
|
|
34
|
+
name: Schema.Literal("dimensions"),
|
|
35
|
+
...AuthoringEventFields
|
|
36
|
+
});
|
|
37
|
+
const AuthoringActivationEvent = Schema.Struct({
|
|
38
|
+
name: Schema.Literal("activation"),
|
|
39
|
+
...AuthoringEventFields
|
|
40
|
+
});
|
|
41
|
+
export const EvalNamedEvent = Schema.Union([
|
|
42
|
+
RunPlanEvent,
|
|
43
|
+
RunCandidateEvent,
|
|
44
|
+
RunJudgeEvent,
|
|
45
|
+
AuthoringBasisEvent,
|
|
46
|
+
AuthoringDimensionsEvent,
|
|
47
|
+
AuthoringActivationEvent
|
|
48
|
+
]);
|
|
49
|
+
export const EvalNamedEventLog = Schema.Array(EvalNamedEvent);
|
|
50
|
+
const caseKey = (dimensionId, caseId) => `${dimensionId}\u0000${caseId}`;
|
|
51
|
+
export function resumeEvalCaseCursors(log) {
|
|
52
|
+
const cursors = new Map();
|
|
53
|
+
for (const event of log) {
|
|
54
|
+
if (event.name !== "candidate#" && event.name !== "judge#")
|
|
55
|
+
continue;
|
|
56
|
+
const key = caseKey(event.dimensionId, event.caseId);
|
|
57
|
+
const cursor = cursors.get(key) ??
|
|
58
|
+
{
|
|
59
|
+
dimensionId: event.dimensionId,
|
|
60
|
+
caseId: event.caseId,
|
|
61
|
+
candidates: new Map(),
|
|
62
|
+
judges: new Set()
|
|
63
|
+
};
|
|
64
|
+
const firstHoleSequence = (() => {
|
|
65
|
+
let sequence = 1;
|
|
66
|
+
while (cursor.candidates.has(sequence) && cursor.judges.has(sequence)) {
|
|
67
|
+
sequence += 1;
|
|
68
|
+
}
|
|
69
|
+
return sequence;
|
|
70
|
+
})();
|
|
71
|
+
if (cursor.candidates.has(firstHoleSequence) &&
|
|
72
|
+
!cursor.judges.has(firstHoleSequence) &&
|
|
73
|
+
(event.name !== "judge#" || event.sequence !== firstHoleSequence)) {
|
|
74
|
+
cursors.set(key, cursor);
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
if (event.name === "candidate#")
|
|
78
|
+
cursor.candidates.set(event.sequence, event);
|
|
79
|
+
else
|
|
80
|
+
cursor.judges.add(event.sequence);
|
|
81
|
+
cursor.lastEvent = event;
|
|
82
|
+
cursors.set(key, cursor);
|
|
83
|
+
}
|
|
84
|
+
return [...cursors.values()].map((cursor) => {
|
|
85
|
+
let completePairs = 0;
|
|
86
|
+
while (cursor.candidates.has(completePairs + 1) &&
|
|
87
|
+
cursor.judges.has(completePairs + 1)) {
|
|
88
|
+
completePairs += 1;
|
|
89
|
+
}
|
|
90
|
+
const hole = cursor.candidates.get(completePairs + 1);
|
|
91
|
+
return {
|
|
92
|
+
dimensionId: cursor.dimensionId,
|
|
93
|
+
caseId: cursor.caseId,
|
|
94
|
+
completePairs,
|
|
95
|
+
...(hole === undefined
|
|
96
|
+
? cursor.lastEvent === undefined
|
|
97
|
+
? {}
|
|
98
|
+
: { lastEvent: cursor.lastEvent }
|
|
99
|
+
: { lastEvent: hole }),
|
|
100
|
+
...(hole === undefined ? {} : { hole })
|
|
101
|
+
};
|
|
102
|
+
});
|
|
103
|
+
}
|
|
104
|
+
export function evalEventLabel(event) {
|
|
105
|
+
return event.name === "candidate#" || event.name === "judge#"
|
|
106
|
+
? `${event.name}${String(event.sequence)}`
|
|
107
|
+
: event.name;
|
|
108
|
+
}
|
|
109
|
+
const lastPlanEvent = (log) => {
|
|
110
|
+
for (let index = log.length - 1; index >= 0; index -= 1) {
|
|
111
|
+
const event = log[index];
|
|
112
|
+
if (event?.name === "plan")
|
|
113
|
+
return event;
|
|
114
|
+
}
|
|
115
|
+
return undefined;
|
|
116
|
+
};
|
|
117
|
+
export function presentEvalEventLog(log, options = {}) {
|
|
118
|
+
const plan = lastPlanEvent(log);
|
|
119
|
+
if (plan !== undefined && plan.plannedCalls < 1) {
|
|
120
|
+
return {
|
|
121
|
+
failed: true,
|
|
122
|
+
lines: ["run fail no planned calls"]
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
const cursors = resumeEvalCaseCursors(log);
|
|
126
|
+
const failHoles = options.failHoles ?? true;
|
|
127
|
+
const lastEvent = log[log.length - 1];
|
|
128
|
+
const authoringLine = lastEvent?.name === "basis" ||
|
|
129
|
+
lastEvent?.name === "dimensions" ||
|
|
130
|
+
lastEvent?.name === "activation"
|
|
131
|
+
? `authoring ${evalEventLabel(lastEvent)}`
|
|
132
|
+
: undefined;
|
|
133
|
+
const cursorLines = cursors.map((cursor) => {
|
|
134
|
+
const label = cursor.lastEvent === undefined ? "plan" : evalEventLabel(cursor.lastEvent);
|
|
135
|
+
return `${cursor.dimensionId}/${cursor.caseId} ${cursor.hole === undefined || !failHoles
|
|
136
|
+
? label
|
|
137
|
+
: `fail ${label} missing judge#${String(cursor.hole.sequence)}`}`;
|
|
138
|
+
});
|
|
139
|
+
return {
|
|
140
|
+
failed: failHoles && cursors.some((cursor) => cursor.hole !== undefined),
|
|
141
|
+
lines: cursors.length === 0
|
|
142
|
+
? authoringLine === undefined
|
|
143
|
+
? plan === undefined
|
|
144
|
+
? []
|
|
145
|
+
: [`run ${evalEventLabel(plan)}`]
|
|
146
|
+
: [authoringLine]
|
|
147
|
+
: authoringLine === undefined
|
|
148
|
+
? cursorLines
|
|
149
|
+
: [...cursorLines, authoringLine]
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
export function assertEvalRunCanComplete(log) {
|
|
153
|
+
const plan = lastPlanEvent(log);
|
|
154
|
+
if (plan === undefined || plan.plannedCalls < 1) {
|
|
155
|
+
throw new Error("eval run cannot complete without a plan containing at least one call");
|
|
156
|
+
}
|
|
157
|
+
const hole = resumeEvalCaseCursors(log).find((cursor) => cursor.hole !== undefined);
|
|
158
|
+
if (hole?.hole !== undefined) {
|
|
159
|
+
throw new Error(`eval case ${JSON.stringify(hole.dimensionId)}/${JSON.stringify(hole.caseId)} cannot complete after ${evalEventLabel(hole.hole)} without judge#${String(hole.hole.sequence)}`);
|
|
160
|
+
}
|
|
161
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -15,8 +15,10 @@ export type { OriEvalAuthoringApi, OriEvalResult } from "./ori-result.js";
|
|
|
15
15
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
16
16
|
export type { EvalAuthoringCompletion, EvalAuthoringSource, EvalAuthoringTransportShape, EvalProjectAuthorShape } from "./project-authoring.js";
|
|
17
17
|
export { EVAL_AUTHORING_CASES_PER_DIMENSION, EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS, EVAL_AUTHORING_REQUEST_BYTES, EVAL_AUTHORING_SOURCE_BYTES, EVAL_AUTHORING_SOURCE_FILES, EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
18
|
-
export type {
|
|
19
|
-
export {
|
|
18
|
+
export type { EvalCaseCursor, EvalNamedEvent, EvalNamedEventLog, EvalRunCallEvent } from "./eval-event-log.js";
|
|
19
|
+
export { EvalNamedEvent as EvalNamedEventSchema, EvalNamedEventLog as EvalNamedEventLogSchema, assertEvalRunCanComplete, evalEventLabel, presentEvalEventLog, resumeEvalCaseCursors } from "./eval-event-log.js";
|
|
20
|
+
export type { BasisCandidateDatasetV3, BasisCandidatesArtifactsV3, BasisCandidateCheckpointV3, BasisCandidateRecipeV3, BasisCandidateRecordV3, BasisGenerationArtifactsV3, BasisEvaluationProjectionV3, BasisGenerationJobV3, BasisGenerationLedgerV3, BasisGenerationPlanV3, BasisLabelProtocolV3, BasisQualificationArtifactsV3, BasisRepairArtifactsV3, BasisRepairPlanV3, BasisRepairSummaryV3, BasisSelectionReportV3, BasisStructureFindingV3, BasisStructureReportV3, BasisTaskRecordV3, BasisTaskSplitV3, BasisValidationPlanV3, BasisValidationClassifierBindingV3, BasisValidationJobCheckpointV3, BasisValidationResultsArtifactsV3, BasisValidationArtifactsV3, BasisValidationResultV3, EvalClassifierObservation, EvalCompositionCase, EvalCompositionCaseResult, EvalCompositionSuite, EvalCostEstimate, EvalDecompositionBenchmark, EvalDecompositionBenchmarkCase, EvalDimensionCase, EvalDimensionContrast, EvalDimensionSuite, EvalEvaluationProposal, EvalExecutionPlan, EvalPlanScope, EvalProposedDimension, EvalProjectArtifactsStatus, EvalProjectDocument, EvalProjectStateV2, EvalProjectStatus, EvalRoutingBasisProposal, EvalRunCleanup, EvalRunFailureError, EvalRunLedger, EvalRunQualification, EvalRunReport, EvalRunTarget, EvalScopedExpectedCalls, FrozenRepositorySnapshotV3 } from "./project-contracts.js";
|
|
21
|
+
export { EVAL_PROJECT_VERSION, EVAL_PROJECT_V2_VERSION, BasisCandidateDatasetV3 as BasisCandidateDatasetV3Schema, BasisCandidatesArtifactsV3 as BasisCandidatesArtifactsV3Schema, BasisCandidateCheckpointV3 as BasisCandidateCheckpointV3Schema, BasisCandidateRecipeV3 as BasisCandidateRecipeV3Schema, BasisCandidateRecordV3 as BasisCandidateRecordV3Schema, BasisGenerationArtifactsV3 as BasisGenerationArtifactsV3Schema, BasisEvaluationProjectionV3 as BasisEvaluationProjectionV3Schema, BasisGenerationJobV3 as BasisGenerationJobV3Schema, BasisGenerationLedgerV3 as BasisGenerationLedgerV3Schema, BasisGenerationPlanV3 as BasisGenerationPlanV3Schema, BasisLabelProtocolV3 as BasisLabelProtocolV3Schema, BasisQualificationArtifactsV3 as BasisQualificationArtifactsV3Schema, BasisRepairArtifactsV3 as BasisRepairArtifactsV3Schema, BasisRepairPlanV3 as BasisRepairPlanV3Schema, BasisRepairSummaryV3 as BasisRepairSummaryV3Schema, BasisSelectionReportV3 as BasisSelectionReportV3Schema, BasisStructureReportV3 as BasisStructureReportV3Schema, BasisTaskRecordV3 as BasisTaskRecordV3Schema, BasisTaskSplitV3 as BasisTaskSplitV3Schema, BasisValidationPlanV3 as BasisValidationPlanV3Schema, BasisValidationClassifierBindingV3 as BasisValidationClassifierBindingV3Schema, BasisValidationJobCheckpointV3 as BasisValidationJobCheckpointV3Schema, BasisValidationResultsArtifactsV3 as BasisValidationResultsArtifactsV3Schema, BasisValidationArtifactsV3 as BasisValidationArtifactsV3Schema, BasisValidationResultV3 as BasisValidationResultV3Schema, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalCostEstimate as EvalCostEstimateSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionContrast as EvalDimensionContrastSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProposedDimension as EvalProposedDimensionSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectDocument as EvalProjectDocumentSchema, EvalProjectSetupProgress, EvalProjectState, EvalProjectStateV2 as EvalProjectStateV2Schema, EvalRoutingBasisProposal as EvalRoutingBasisProposalSchema, EvalRunCleanup as EvalRunCleanupSchema, EvalRunFailureError as EvalRunFailureErrorSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, FrozenRepositorySnapshotV3 as FrozenRepositorySnapshotV3Schema, estimateEvalPlanCost, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
20
22
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
21
23
|
export type { BasisAuthoringShape } from "./services/basis-authoring/service.js";
|
|
22
24
|
export { BasisAuthoring, BasisAuthoringLive, makeBasisAuthoring } from "./services/basis-authoring/service.js";
|
package/dist/index.js
CHANGED
|
@@ -10,7 +10,8 @@ export { EvalRepositoryInspector, EvalRepositoryInspectorLive, inspectRepository
|
|
|
10
10
|
export { OriEvalAuthoring, oriAuthoringFromApi } from "./ori-authoring.js";
|
|
11
11
|
export { EvalProjectArtifacts, EvalProjectArtifactsLive, evaluationProposalDigest, makeFileEvalProjectArtifacts, routingBasisDigest } from "./project-artifacts.js";
|
|
12
12
|
export { EVAL_AUTHORING_CASES_PER_DIMENSION, EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS, EVAL_AUTHORING_REQUEST_BYTES, EVAL_AUTHORING_SOURCE_BYTES, EVAL_AUTHORING_SOURCE_FILES, EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, makeEvalProjectAuthor, readProjectAuthoringSources, selectProjectAuthoringSourceFiles } from "./project-authoring.js";
|
|
13
|
-
export {
|
|
13
|
+
export { EvalNamedEvent as EvalNamedEventSchema, EvalNamedEventLog as EvalNamedEventLogSchema, assertEvalRunCanComplete, evalEventLabel, presentEvalEventLog, resumeEvalCaseCursors } from "./eval-event-log.js";
|
|
14
|
+
export { EVAL_PROJECT_VERSION, EVAL_PROJECT_V2_VERSION, BasisCandidateDatasetV3 as BasisCandidateDatasetV3Schema, BasisCandidatesArtifactsV3 as BasisCandidatesArtifactsV3Schema, BasisCandidateCheckpointV3 as BasisCandidateCheckpointV3Schema, BasisCandidateRecipeV3 as BasisCandidateRecipeV3Schema, BasisCandidateRecordV3 as BasisCandidateRecordV3Schema, BasisGenerationArtifactsV3 as BasisGenerationArtifactsV3Schema, BasisEvaluationProjectionV3 as BasisEvaluationProjectionV3Schema, BasisGenerationJobV3 as BasisGenerationJobV3Schema, BasisGenerationLedgerV3 as BasisGenerationLedgerV3Schema, BasisGenerationPlanV3 as BasisGenerationPlanV3Schema, BasisLabelProtocolV3 as BasisLabelProtocolV3Schema, BasisQualificationArtifactsV3 as BasisQualificationArtifactsV3Schema, BasisRepairArtifactsV3 as BasisRepairArtifactsV3Schema, BasisRepairPlanV3 as BasisRepairPlanV3Schema, BasisRepairSummaryV3 as BasisRepairSummaryV3Schema, BasisSelectionReportV3 as BasisSelectionReportV3Schema, BasisStructureReportV3 as BasisStructureReportV3Schema, BasisTaskRecordV3 as BasisTaskRecordV3Schema, BasisTaskSplitV3 as BasisTaskSplitV3Schema, BasisValidationPlanV3 as BasisValidationPlanV3Schema, BasisValidationClassifierBindingV3 as BasisValidationClassifierBindingV3Schema, BasisValidationJobCheckpointV3 as BasisValidationJobCheckpointV3Schema, BasisValidationResultsArtifactsV3 as BasisValidationResultsArtifactsV3Schema, BasisValidationArtifactsV3 as BasisValidationArtifactsV3Schema, BasisValidationResultV3 as BasisValidationResultV3Schema, EvalArtifactApproval, EvalCompositionSuite as EvalCompositionSuiteSchema, EvalCostEstimate as EvalCostEstimateSchema, EvalDecompositionBenchmark as EvalDecompositionBenchmarkSchema, EvalDimensionCase as EvalDimensionCaseSchema, EvalDimensionContrast as EvalDimensionContrastSchema, EvalDimensionSuite as EvalDimensionSuiteSchema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalPlanScope as EvalPlanScopeSchema, EvalProposedDimension as EvalProposedDimensionSchema, EvalProjectConfiguration, EvalProjectQuestion, EvalProjectDocument as EvalProjectDocumentSchema, EvalProjectSetupProgress, EvalProjectState, EvalProjectStateV2 as EvalProjectStateV2Schema, EvalRoutingBasisProposal as EvalRoutingBasisProposalSchema, EvalRunCleanup as EvalRunCleanupSchema, EvalRunFailureError as EvalRunFailureErrorSchema, EvalRunLedger as EvalRunLedgerSchema, EvalRunReport as EvalRunReportSchema, EvalRunTarget as EvalRunTargetSchema, FrozenRepositorySnapshotV3 as FrozenRepositorySnapshotV3Schema, estimateEvalPlanCost, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
14
15
|
export { EvalProjectStore, EvalProjectStoreLive, makeFileEvalProjectStore } from "./project-store.js";
|
|
15
16
|
export { BasisAuthoring, BasisAuthoringLive, makeBasisAuthoring } from "./services/basis-authoring/service.js";
|
|
16
17
|
export { BasisOnboarding, BasisOnboardingLive, makeBasisOnboarding } from "./services/basis-onboarding/service.js";
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type RoutingBasis } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import { Context, Effect, FileSystem, Layer, Path } from "effect";
|
|
3
3
|
import { EvalProjectArtifactError } from "./errors.js";
|
|
4
|
+
import { type EvalNamedEventLog } from "./eval-event-log.js";
|
|
4
5
|
import { type EvalArtifactApproval, type BasisCandidatesArtifactsV3, type BasisEvaluationProjectionV3, type BasisCandidateCheckpointV3, type BasisGenerationArtifactsV3, type BasisQualificationArtifactsV3, type BasisRepairArtifactsV3, type BasisValidationJobCheckpointV3, type BasisValidationResultsArtifactsV3, type BasisValidationArtifactsV3, type EvalEvaluationProposal, type EvalExecutionPlan, type EvalRoutingBasisProposal, type EvalRunReport } from "./project-contracts.js";
|
|
5
6
|
export declare function routingBasisDigest(dimensions: RoutingBasis["dimensions"]): string;
|
|
6
7
|
export declare function evaluationProposalDigest(proposal: Omit<EvalEvaluationProposal, "evaluationDigest">): string;
|
|
@@ -47,6 +48,10 @@ export type EvalProjectArtifactsShape = {
|
|
|
47
48
|
readonly saveRunReport: (repositoryRoot: string, report: EvalRunReport) => Effect.Effect<string, EvalProjectArtifactError, never>;
|
|
48
49
|
readonly loadRunReport: (repositoryRoot: string, runId: string) => Effect.Effect<EvalRunReport | undefined, EvalProjectArtifactError, never>;
|
|
49
50
|
readonly listRunReports: (repositoryRoot: string) => Effect.Effect<readonly string[], EvalProjectArtifactError, never>;
|
|
51
|
+
readonly saveRunEventLog: (repositoryRoot: string, runId: string, events: EvalNamedEventLog) => Effect.Effect<void, EvalProjectArtifactError, never>;
|
|
52
|
+
readonly loadRunEventLog: (repositoryRoot: string, runId: string) => Effect.Effect<EvalNamedEventLog, EvalProjectArtifactError, never>;
|
|
53
|
+
readonly saveAuthoringEventLog: (repositoryRoot: string, events: EvalNamedEventLog) => Effect.Effect<void, EvalProjectArtifactError, never>;
|
|
54
|
+
readonly loadAuthoringEventLog: (repositoryRoot: string) => Effect.Effect<EvalNamedEventLog, EvalProjectArtifactError, never>;
|
|
50
55
|
};
|
|
51
56
|
declare const EvalProjectArtifacts_base: Context.ServiceClass<EvalProjectArtifacts, "@velum-labs/routekit-eval-setup/EvalProjectArtifacts", EvalProjectArtifactsShape>;
|
|
52
57
|
export declare class EvalProjectArtifacts extends EvalProjectArtifacts_base {
|
|
@@ -3,6 +3,7 @@ import { assertRoutingBasis } from "@velum-labs/routekit-eval-contracts";
|
|
|
3
3
|
import { writeFileAtomicEffect } from "@velum-labs/routekit-runtime/effect";
|
|
4
4
|
import { Context, Effect, Exit, FileSystem, Layer, Path, Schema } from "effect";
|
|
5
5
|
import { EvalProjectArtifactError } from "./errors.js";
|
|
6
|
+
import { EvalNamedEventLog as EvalNamedEventLogSchema } from "./eval-event-log.js";
|
|
6
7
|
import { EVAL_PROJECT_VERSION, EvalArtifactApproval as EvalArtifactApprovalSchema, BasisCandidatesArtifactsV3 as BasisCandidatesArtifactsV3Schema, BasisEvaluationProjectionV3 as BasisEvaluationProjectionV3Schema, BasisCandidateCheckpointV3 as BasisCandidateCheckpointV3Schema, BasisGenerationArtifactsV3 as BasisGenerationArtifactsV3Schema, BasisQualificationArtifactsV3 as BasisQualificationArtifactsV3Schema, BasisRepairArtifactsV3 as BasisRepairArtifactsV3Schema, BasisValidationJobCheckpointV3 as BasisValidationJobCheckpointV3Schema, BasisValidationResultsArtifactsV3 as BasisValidationResultsArtifactsV3Schema, BasisValidationArtifactsV3 as BasisValidationArtifactsV3Schema, EvalEvaluationProposal as EvalEvaluationProposalSchema, EvalExecutionPlan as EvalExecutionPlanSchema, EvalRoutingBasisProposal as EvalRoutingBasisProposalSchema, EvalRunReport as EvalRunReportSchema } from "./project-contracts.js";
|
|
7
8
|
const BASIS_PROPOSAL = "routing-basis.proposed.json";
|
|
8
9
|
const BASIS_APPROVAL = "routing-basis.approval.json";
|
|
@@ -18,6 +19,21 @@ const BASIS_REPAIR_V3 = "basis/repair.v3.json";
|
|
|
18
19
|
const BASIS_REPAIR_VALIDATION_V3 = "basis/repair-validation.v3.json";
|
|
19
20
|
const BASIS_REPAIR_VALIDATION_RESULTS_V3 = "basis/repair-validation-results.v3.json";
|
|
20
21
|
const ARTIFACT_ID = /^[A-Za-z0-9][A-Za-z0-9_-]{0,127}$/u;
|
|
22
|
+
const repeatedCase = (byId, selectedCaseId) => {
|
|
23
|
+
const exact = byId.get(selectedCaseId);
|
|
24
|
+
if (exact !== undefined)
|
|
25
|
+
return exact;
|
|
26
|
+
const repeated = /^(.*)#[1-9][0-9]*$/u.exec(selectedCaseId);
|
|
27
|
+
const sourceId = repeated?.[1];
|
|
28
|
+
if (sourceId === undefined) {
|
|
29
|
+
throw new Error(`execution plan refers to unknown case ${JSON.stringify(selectedCaseId)}`);
|
|
30
|
+
}
|
|
31
|
+
const testCase = byId.get(sourceId);
|
|
32
|
+
if (testCase === undefined) {
|
|
33
|
+
throw new Error(`execution plan refers to unknown case ${JSON.stringify(selectedCaseId)}`);
|
|
34
|
+
}
|
|
35
|
+
return { ...testCase, id: selectedCaseId };
|
|
36
|
+
};
|
|
21
37
|
const renderCandidatePrompt = () => `const prompt = [
|
|
22
38
|
testCase.prompt,
|
|
23
39
|
"",
|
|
@@ -324,13 +340,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
324
340
|
return yield* artifactFailure("writing", selection.dimensionId, new Error("execution plan refers to an unknown dimension suite"));
|
|
325
341
|
}
|
|
326
342
|
const byId = new Map(suite.cases.map((testCase) => [testCase.id, testCase]));
|
|
327
|
-
const cases = selection.caseIds.map((caseId) =>
|
|
328
|
-
const testCase = byId.get(caseId);
|
|
329
|
-
if (testCase === undefined) {
|
|
330
|
-
throw new Error(`execution plan refers to unknown case ${JSON.stringify(caseId)} in ${JSON.stringify(selection.dimensionId)}`);
|
|
331
|
-
}
|
|
332
|
-
return testCase;
|
|
333
|
-
});
|
|
343
|
+
const cases = selection.caseIds.map((caseId) => repeatedCase(byId, caseId));
|
|
334
344
|
const suiteRoot = paths.join("plans", plan.planId, "dimensions", selection.dimensionId);
|
|
335
345
|
yield* writeText(repositoryRoot, paths.join(suiteRoot, `${selection.dimensionId}.eval.ts`), renderDimensionSuite(), mode === "rematerialize");
|
|
336
346
|
yield* write(repositoryRoot, paths.join(suiteRoot, "data", "cases.json"), cases, mode === "rematerialize");
|
|
@@ -347,13 +357,7 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
347
357
|
}, mode === "rematerialize");
|
|
348
358
|
}
|
|
349
359
|
const compositionById = new Map(proposal.compositionSuite.cases.map((testCase) => [testCase.id, testCase]));
|
|
350
|
-
const compositionCases = plan.selectedCompositionCaseIds.map((caseId) =>
|
|
351
|
-
const testCase = compositionById.get(caseId);
|
|
352
|
-
if (testCase === undefined) {
|
|
353
|
-
throw new Error(`execution plan refers to unknown composition case ${JSON.stringify(caseId)}`);
|
|
354
|
-
}
|
|
355
|
-
return testCase;
|
|
356
|
-
});
|
|
360
|
+
const compositionCases = plan.selectedCompositionCaseIds.map((caseId) => repeatedCase(compositionById, caseId));
|
|
357
361
|
const compositionRoot = paths.join("plans", plan.planId, "composition");
|
|
358
362
|
yield* writeText(repositoryRoot, paths.join(compositionRoot, "composition.eval.ts"), renderDimensionSuite(), mode === "rematerialize");
|
|
359
363
|
yield* write(repositoryRoot, paths.join(compositionRoot, "data", "cases.json"), compositionCases, mode === "rematerialize");
|
|
@@ -435,7 +439,11 @@ export const makeFileEvalProjectArtifacts = Effect.gen(function* () {
|
|
|
435
439
|
reports.push(entry);
|
|
436
440
|
}
|
|
437
441
|
return reports.sort((left, right) => left.localeCompare(right));
|
|
438
|
-
})
|
|
442
|
+
}),
|
|
443
|
+
saveRunEventLog: (repositoryRoot, runId, events) => requireArtifactId("run", runId).pipe(Effect.andThen(Schema.decodeEffect(EvalNamedEventLogSchema)(events).pipe(Effect.mapError((cause) => artifactFailure("writing", artifactPath(repositoryRoot, paths.join("runs", runId, "events.json")), cause)))), Effect.andThen(write(repositoryRoot, paths.join("runs", runId, "events.json"), events))),
|
|
444
|
+
loadRunEventLog: (repositoryRoot, runId) => requireArtifactId("run", runId).pipe(Effect.andThen(read(repositoryRoot, paths.join("runs", runId, "events.json"), Schema.decodeUnknownEffect(EvalNamedEventLogSchema))), Effect.map((events) => events ?? [])),
|
|
445
|
+
saveAuthoringEventLog: (repositoryRoot, events) => Schema.decodeEffect(EvalNamedEventLogSchema)(events).pipe(Effect.mapError((cause) => artifactFailure("writing", artifactPath(repositoryRoot, paths.join("authoring", "events.json")), cause)), Effect.andThen(write(repositoryRoot, paths.join("authoring", "events.json"), events))),
|
|
446
|
+
loadAuthoringEventLog: (repositoryRoot) => read(repositoryRoot, paths.join("authoring", "events.json"), Schema.decodeUnknownEffect(EvalNamedEventLogSchema)).pipe(Effect.map((events) => events ?? []))
|
|
439
447
|
});
|
|
440
448
|
});
|
|
441
449
|
export const EvalProjectArtifactsLive = Layer.effect(EvalProjectArtifacts, makeFileEvalProjectArtifacts);
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { type EvalComparisonResult as EvalComparisonResultType } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
-
import {
|
|
2
|
+
import { type CostServiceShape } from "@velum-labs/routekit-registry";
|
|
3
|
+
import { Effect, Schema } from "effect";
|
|
3
4
|
export declare const EVAL_PROJECT_VERSION: 1;
|
|
4
5
|
export declare const EVAL_PROJECT_V2_VERSION: 2;
|
|
5
6
|
export declare const EvalProjectQuestion: Schema.Struct<{
|
|
@@ -4312,6 +4313,17 @@ export declare const EvalArtifactApproval: Schema.Struct<{
|
|
|
4312
4313
|
export type EvalArtifactApproval = typeof EvalArtifactApproval.Type;
|
|
4313
4314
|
export declare const EvalPlanScope: Schema.Literals<readonly ["pilot", "full"]>;
|
|
4314
4315
|
export type EvalPlanScope = typeof EvalPlanScope.Type;
|
|
4316
|
+
export declare const EvalCostEstimate: Schema.Struct<{
|
|
4317
|
+
readonly callCount: Schema.Finite;
|
|
4318
|
+
readonly maximumInputTokens: Schema.Finite;
|
|
4319
|
+
readonly maximumOutputTokens: Schema.Finite;
|
|
4320
|
+
readonly maximumCostUsd: Schema.optionalKey<Schema.Finite>;
|
|
4321
|
+
readonly knownPricedMaximumUsd: Schema.Finite;
|
|
4322
|
+
readonly unpricedCalls: Schema.Finite;
|
|
4323
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
4324
|
+
readonly unknowns: Schema.$Array<Schema.String>;
|
|
4325
|
+
}>;
|
|
4326
|
+
export type EvalCostEstimate = typeof EvalCostEstimate.Type;
|
|
4315
4327
|
export declare const EvalExecutionPlan: Schema.Struct<{
|
|
4316
4328
|
readonly version: Schema.Literal<1>;
|
|
4317
4329
|
readonly planId: Schema.String;
|
|
@@ -4319,6 +4331,7 @@ export declare const EvalExecutionPlan: Schema.Struct<{
|
|
|
4319
4331
|
readonly projectRevision: Schema.Finite;
|
|
4320
4332
|
readonly createdAt: Schema.String;
|
|
4321
4333
|
readonly scope: Schema.Literals<readonly ["pilot", "full"]>;
|
|
4334
|
+
readonly repetitions: Schema.withDecodingDefaultKey<Schema.Finite, never>;
|
|
4322
4335
|
readonly basisDigest: Schema.String;
|
|
4323
4336
|
readonly evaluationDigest: Schema.String;
|
|
4324
4337
|
readonly candidateModels: Schema.$Array<Schema.String>;
|
|
@@ -4341,6 +4354,16 @@ export declare const EvalExecutionPlan: Schema.Struct<{
|
|
|
4341
4354
|
readonly expectedCandidateCalls: Schema.Finite;
|
|
4342
4355
|
readonly expectedJudgeCalls: Schema.Finite;
|
|
4343
4356
|
readonly expectedCallCount: Schema.Finite;
|
|
4357
|
+
readonly costEstimate: Schema.optionalKey<Schema.Struct<{
|
|
4358
|
+
readonly callCount: Schema.Finite;
|
|
4359
|
+
readonly maximumInputTokens: Schema.Finite;
|
|
4360
|
+
readonly maximumOutputTokens: Schema.Finite;
|
|
4361
|
+
readonly maximumCostUsd: Schema.optionalKey<Schema.Finite>;
|
|
4362
|
+
readonly knownPricedMaximumUsd: Schema.Finite;
|
|
4363
|
+
readonly unpricedCalls: Schema.Finite;
|
|
4364
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
4365
|
+
readonly unknowns: Schema.$Array<Schema.String>;
|
|
4366
|
+
}>>;
|
|
4344
4367
|
}>;
|
|
4345
4368
|
export type EvalExecutionPlan = typeof EvalExecutionPlan.Type;
|
|
4346
4369
|
export type EvalScopedExpectedCalls = {
|
|
@@ -4354,6 +4377,8 @@ export type EvalScopedExpectedCalls = {
|
|
|
4354
4377
|
readonly expectedCallCount: number;
|
|
4355
4378
|
};
|
|
4356
4379
|
export declare function scopedExpectedCalls(plan: Pick<EvalExecutionPlan, "candidateModels" | "selectedCaseIds" | "selectedDecompositionCaseIds" | "selectedCompositionCaseIds">): EvalScopedExpectedCalls;
|
|
4380
|
+
export declare const EVAL_MAXIMUM_INPUT_TOKENS_PER_CALL: number;
|
|
4381
|
+
export declare function estimateEvalPlanCost(plan: Pick<EvalExecutionPlan, "candidateModels" | "candidateReasoningEfforts" | "classifierModel" | "judgeModel" | "maximumOutputTokens" | "expectedDimensionCandidateCalls" | "expectedClassifierCalls" | "expectedCompositionCandidateCalls">, costService: CostServiceShape): Effect.Effect<EvalCostEstimate>;
|
|
4357
4382
|
export declare const EvalRunTarget: Schema.Struct<{
|
|
4358
4383
|
readonly kind: Schema.Literals<readonly ["configured", "external"]>;
|
|
4359
4384
|
readonly identity: Schema.String;
|
|
@@ -4375,6 +4400,7 @@ export declare const EvalRunLedger: Schema.Struct<{
|
|
|
4375
4400
|
readonly unknownTokenMeasurements: Schema.Finite;
|
|
4376
4401
|
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
4377
4402
|
readonly unpricedCalls: Schema.Finite;
|
|
4403
|
+
readonly subscriptionUnitCalls: Schema.withDecodingDefaultKey<Schema.Finite, never>;
|
|
4378
4404
|
}>;
|
|
4379
4405
|
export type EvalRunLedger = typeof EvalRunLedger.Type;
|
|
4380
4406
|
export declare const EvalClassifierObservation: Schema.Struct<{
|
|
@@ -4443,7 +4469,7 @@ export declare const EvalRunQualification: Schema.Struct<{
|
|
|
4443
4469
|
}>;
|
|
4444
4470
|
}>;
|
|
4445
4471
|
export type EvalRunQualification = typeof EvalRunQualification.Type;
|
|
4446
|
-
export declare function summarizeEvalRunLedger(comparisons: readonly EvalComparisonResultType[], expectedCalls: number, classifierObservations?: readonly EvalClassifierObservation[]): EvalRunLedger
|
|
4472
|
+
export declare function summarizeEvalRunLedger(comparisons: readonly EvalComparisonResultType[], expectedCalls: number, costService: CostServiceShape, classifierObservations?: readonly EvalClassifierObservation[], classifierModel?: string): Effect.Effect<EvalRunLedger>;
|
|
4447
4473
|
export declare const EvalRunFailureError: Schema.Struct<{
|
|
4448
4474
|
readonly name: Schema.String;
|
|
4449
4475
|
readonly message: Schema.String;
|
|
@@ -4617,7 +4643,40 @@ export declare const EvalRunReport: Schema.Union<readonly [Schema.Struct<{
|
|
|
4617
4643
|
readonly unknownTokenMeasurements: Schema.Finite;
|
|
4618
4644
|
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
4619
4645
|
readonly unpricedCalls: Schema.Finite;
|
|
4646
|
+
readonly subscriptionUnitCalls: Schema.withDecodingDefaultKey<Schema.Finite, never>;
|
|
4620
4647
|
}>;
|
|
4648
|
+
readonly events: Schema.withDecodingDefaultKey<Schema.optionalKey<Schema.$Array<Schema.Union<readonly [Schema.Struct<{
|
|
4649
|
+
readonly name: Schema.Literal<"plan">;
|
|
4650
|
+
readonly at: Schema.String;
|
|
4651
|
+
readonly runId: Schema.String;
|
|
4652
|
+
readonly plannedCalls: Schema.Finite;
|
|
4653
|
+
}>, Schema.Struct<{
|
|
4654
|
+
readonly at: Schema.String;
|
|
4655
|
+
readonly runId: Schema.String;
|
|
4656
|
+
readonly dimensionId: Schema.String;
|
|
4657
|
+
readonly caseId: Schema.String;
|
|
4658
|
+
readonly sequence: Schema.Finite;
|
|
4659
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
4660
|
+
}>, Schema.Struct<{
|
|
4661
|
+
readonly at: Schema.String;
|
|
4662
|
+
readonly runId: Schema.String;
|
|
4663
|
+
readonly dimensionId: Schema.String;
|
|
4664
|
+
readonly caseId: Schema.String;
|
|
4665
|
+
readonly sequence: Schema.Finite;
|
|
4666
|
+
readonly name: Schema.Literal<"judge#">;
|
|
4667
|
+
}>, Schema.Struct<{
|
|
4668
|
+
readonly at: Schema.String;
|
|
4669
|
+
readonly operationId: Schema.String;
|
|
4670
|
+
readonly name: Schema.Literal<"basis">;
|
|
4671
|
+
}>, Schema.Struct<{
|
|
4672
|
+
readonly at: Schema.String;
|
|
4673
|
+
readonly operationId: Schema.String;
|
|
4674
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
4675
|
+
}>, Schema.Struct<{
|
|
4676
|
+
readonly at: Schema.String;
|
|
4677
|
+
readonly operationId: Schema.String;
|
|
4678
|
+
readonly name: Schema.Literal<"activation">;
|
|
4679
|
+
}>]>>>, never>;
|
|
4621
4680
|
}>, Schema.Struct<{
|
|
4622
4681
|
readonly status: Schema.Literal<"completed">;
|
|
4623
4682
|
readonly activation: Schema.Struct<{
|
|
@@ -4777,7 +4836,40 @@ export declare const EvalRunReport: Schema.Union<readonly [Schema.Struct<{
|
|
|
4777
4836
|
readonly unknownTokenMeasurements: Schema.Finite;
|
|
4778
4837
|
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
4779
4838
|
readonly unpricedCalls: Schema.Finite;
|
|
4839
|
+
readonly subscriptionUnitCalls: Schema.withDecodingDefaultKey<Schema.Finite, never>;
|
|
4780
4840
|
}>;
|
|
4841
|
+
readonly events: Schema.withDecodingDefaultKey<Schema.optionalKey<Schema.$Array<Schema.Union<readonly [Schema.Struct<{
|
|
4842
|
+
readonly name: Schema.Literal<"plan">;
|
|
4843
|
+
readonly at: Schema.String;
|
|
4844
|
+
readonly runId: Schema.String;
|
|
4845
|
+
readonly plannedCalls: Schema.Finite;
|
|
4846
|
+
}>, Schema.Struct<{
|
|
4847
|
+
readonly at: Schema.String;
|
|
4848
|
+
readonly runId: Schema.String;
|
|
4849
|
+
readonly dimensionId: Schema.String;
|
|
4850
|
+
readonly caseId: Schema.String;
|
|
4851
|
+
readonly sequence: Schema.Finite;
|
|
4852
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
4853
|
+
}>, Schema.Struct<{
|
|
4854
|
+
readonly at: Schema.String;
|
|
4855
|
+
readonly runId: Schema.String;
|
|
4856
|
+
readonly dimensionId: Schema.String;
|
|
4857
|
+
readonly caseId: Schema.String;
|
|
4858
|
+
readonly sequence: Schema.Finite;
|
|
4859
|
+
readonly name: Schema.Literal<"judge#">;
|
|
4860
|
+
}>, Schema.Struct<{
|
|
4861
|
+
readonly at: Schema.String;
|
|
4862
|
+
readonly operationId: Schema.String;
|
|
4863
|
+
readonly name: Schema.Literal<"basis">;
|
|
4864
|
+
}>, Schema.Struct<{
|
|
4865
|
+
readonly at: Schema.String;
|
|
4866
|
+
readonly operationId: Schema.String;
|
|
4867
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
4868
|
+
}>, Schema.Struct<{
|
|
4869
|
+
readonly at: Schema.String;
|
|
4870
|
+
readonly operationId: Schema.String;
|
|
4871
|
+
readonly name: Schema.Literal<"activation">;
|
|
4872
|
+
}>]>>>, never>;
|
|
4781
4873
|
}>, Schema.Struct<{
|
|
4782
4874
|
readonly status: Schema.Literal<"failed">;
|
|
4783
4875
|
readonly failure: Schema.String;
|
|
@@ -4887,7 +4979,40 @@ export declare const EvalRunReport: Schema.Union<readonly [Schema.Struct<{
|
|
|
4887
4979
|
readonly unknownTokenMeasurements: Schema.Finite;
|
|
4888
4980
|
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
4889
4981
|
readonly unpricedCalls: Schema.Finite;
|
|
4982
|
+
readonly subscriptionUnitCalls: Schema.withDecodingDefaultKey<Schema.Finite, never>;
|
|
4890
4983
|
}>;
|
|
4984
|
+
readonly events: Schema.withDecodingDefaultKey<Schema.optionalKey<Schema.$Array<Schema.Union<readonly [Schema.Struct<{
|
|
4985
|
+
readonly name: Schema.Literal<"plan">;
|
|
4986
|
+
readonly at: Schema.String;
|
|
4987
|
+
readonly runId: Schema.String;
|
|
4988
|
+
readonly plannedCalls: Schema.Finite;
|
|
4989
|
+
}>, Schema.Struct<{
|
|
4990
|
+
readonly at: Schema.String;
|
|
4991
|
+
readonly runId: Schema.String;
|
|
4992
|
+
readonly dimensionId: Schema.String;
|
|
4993
|
+
readonly caseId: Schema.String;
|
|
4994
|
+
readonly sequence: Schema.Finite;
|
|
4995
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
4996
|
+
}>, Schema.Struct<{
|
|
4997
|
+
readonly at: Schema.String;
|
|
4998
|
+
readonly runId: Schema.String;
|
|
4999
|
+
readonly dimensionId: Schema.String;
|
|
5000
|
+
readonly caseId: Schema.String;
|
|
5001
|
+
readonly sequence: Schema.Finite;
|
|
5002
|
+
readonly name: Schema.Literal<"judge#">;
|
|
5003
|
+
}>, Schema.Struct<{
|
|
5004
|
+
readonly at: Schema.String;
|
|
5005
|
+
readonly operationId: Schema.String;
|
|
5006
|
+
readonly name: Schema.Literal<"basis">;
|
|
5007
|
+
}>, Schema.Struct<{
|
|
5008
|
+
readonly at: Schema.String;
|
|
5009
|
+
readonly operationId: Schema.String;
|
|
5010
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
5011
|
+
}>, Schema.Struct<{
|
|
5012
|
+
readonly at: Schema.String;
|
|
5013
|
+
readonly operationId: Schema.String;
|
|
5014
|
+
readonly name: Schema.Literal<"activation">;
|
|
5015
|
+
}>]>>>, never>;
|
|
4891
5016
|
}>]>;
|
|
4892
5017
|
export type EvalRunReport = typeof EvalRunReport.Type;
|
|
4893
5018
|
export type EvalProjectArtifactsStatus = {
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { DecompositionResult, DimensionScoreLabelV3, DimensionScorePredictionV3, EvalCandidateReasoningEfforts, EvalComparisonResult, EvalReasoningEffort, PublishedRoutingActivation, RoutingBasisApprovalV3, RoutingBasisV3, RoutingActivationConstraints, RequestRoutingRequirements, RoutingObjectivePolicy, WorkloadDimension } from "@velum-labs/routekit-eval-contracts";
|
|
2
|
-
import { Schema } from "effect";
|
|
2
|
+
import { Effect, Schema } from "effect";
|
|
3
|
+
import { EvalNamedEventLog } from "./eval-event-log.js";
|
|
3
4
|
export const EVAL_PROJECT_VERSION = 1;
|
|
4
5
|
export const EVAL_PROJECT_V2_VERSION = 2;
|
|
5
6
|
export const EvalProjectQuestion = Schema.Struct({
|
|
@@ -579,6 +580,16 @@ export const EvalArtifactApproval = Schema.Struct({
|
|
|
579
580
|
approvedAt: Schema.String
|
|
580
581
|
});
|
|
581
582
|
export const EvalPlanScope = Schema.Literals(["pilot", "full"]);
|
|
583
|
+
export const EvalCostEstimate = Schema.Struct({
|
|
584
|
+
callCount: NonNegativeInteger,
|
|
585
|
+
maximumInputTokens: NonNegativeInteger,
|
|
586
|
+
maximumOutputTokens: NonNegativeInteger,
|
|
587
|
+
maximumCostUsd: Schema.optionalKey(NonNegativeFinite),
|
|
588
|
+
knownPricedMaximumUsd: NonNegativeFinite,
|
|
589
|
+
unpricedCalls: NonNegativeInteger,
|
|
590
|
+
subscriptionUnitCalls: NonNegativeInteger,
|
|
591
|
+
unknowns: Schema.Array(Schema.String)
|
|
592
|
+
});
|
|
582
593
|
export const EvalExecutionPlan = Schema.Struct({
|
|
583
594
|
version: Schema.Literal(EVAL_PROJECT_VERSION),
|
|
584
595
|
planId: Schema.String,
|
|
@@ -586,6 +597,7 @@ export const EvalExecutionPlan = Schema.Struct({
|
|
|
586
597
|
projectRevision: NonNegativeInteger,
|
|
587
598
|
createdAt: Schema.String,
|
|
588
599
|
scope: EvalPlanScope,
|
|
600
|
+
repetitions: NonNegativeInteger.pipe(Schema.withDecodingDefaultKey(Effect.succeed(1))),
|
|
589
601
|
basisDigest: Schema.String,
|
|
590
602
|
evaluationDigest: Schema.String,
|
|
591
603
|
candidateModels: Schema.Array(Schema.String),
|
|
@@ -607,7 +619,8 @@ export const EvalExecutionPlan = Schema.Struct({
|
|
|
607
619
|
expectedCompositionJudgeCalls: NonNegativeInteger,
|
|
608
620
|
expectedCandidateCalls: NonNegativeInteger,
|
|
609
621
|
expectedJudgeCalls: NonNegativeInteger,
|
|
610
|
-
expectedCallCount: NonNegativeInteger
|
|
622
|
+
expectedCallCount: NonNegativeInteger,
|
|
623
|
+
costEstimate: Schema.optionalKey(EvalCostEstimate)
|
|
611
624
|
});
|
|
612
625
|
export function scopedExpectedCalls(plan) {
|
|
613
626
|
const candidateModelCount = plan.candidateModels.length;
|
|
@@ -630,6 +643,44 @@ export function scopedExpectedCalls(plan) {
|
|
|
630
643
|
expectedCallCount: expectedCandidateCalls + expectedJudgeCalls + expectedClassifierCalls
|
|
631
644
|
};
|
|
632
645
|
}
|
|
646
|
+
export const EVAL_MAXIMUM_INPUT_TOKENS_PER_CALL = 256 * 1024;
|
|
647
|
+
export function estimateEvalPlanCost(plan, costService) {
|
|
648
|
+
const candidateCaseCalls = plan.expectedDimensionCandidateCalls + plan.expectedCompositionCandidateCalls;
|
|
649
|
+
const callsPerCandidate = plan.candidateModels.length === 0 ? 0 : candidateCaseCalls / plan.candidateModels.length;
|
|
650
|
+
const calls = [
|
|
651
|
+
...plan.candidateModels.map((model) => ({
|
|
652
|
+
model,
|
|
653
|
+
callCount: callsPerCandidate,
|
|
654
|
+
maximumInputTokens: EVAL_MAXIMUM_INPUT_TOKENS_PER_CALL,
|
|
655
|
+
maximumOutputTokens: plan.maximumOutputTokens,
|
|
656
|
+
...(plan.candidateReasoningEfforts?.[model] === undefined
|
|
657
|
+
? {}
|
|
658
|
+
: { reasoningEffort: plan.candidateReasoningEfforts[model] })
|
|
659
|
+
})),
|
|
660
|
+
{
|
|
661
|
+
model: plan.judgeModel,
|
|
662
|
+
callCount: candidateCaseCalls,
|
|
663
|
+
maximumInputTokens: EVAL_MAXIMUM_INPUT_TOKENS_PER_CALL,
|
|
664
|
+
maximumOutputTokens: plan.maximumOutputTokens
|
|
665
|
+
},
|
|
666
|
+
{
|
|
667
|
+
model: plan.classifierModel,
|
|
668
|
+
callCount: plan.expectedClassifierCalls,
|
|
669
|
+
maximumInputTokens: EVAL_MAXIMUM_INPUT_TOKENS_PER_CALL,
|
|
670
|
+
maximumOutputTokens: plan.maximumOutputTokens
|
|
671
|
+
}
|
|
672
|
+
];
|
|
673
|
+
return costService.estimate(calls).pipe(Effect.map((estimate) => ({
|
|
674
|
+
callCount: estimate.callCount,
|
|
675
|
+
maximumInputTokens: estimate.knownInputTokens,
|
|
676
|
+
maximumOutputTokens: estimate.knownOutputTokens,
|
|
677
|
+
...(estimate.maximumCostUsd === undefined ? {} : { maximumCostUsd: estimate.maximumCostUsd }),
|
|
678
|
+
knownPricedMaximumUsd: estimate.knownPricedMaximumUsd,
|
|
679
|
+
unpricedCalls: estimate.unpricedCalls + estimate.upstreamManagedCalls + estimate.unknownBillingCalls,
|
|
680
|
+
subscriptionUnitCalls: estimate.subscriptionUnitCalls,
|
|
681
|
+
unknowns: estimate.unknowns
|
|
682
|
+
})));
|
|
683
|
+
}
|
|
633
684
|
export const EvalRunTarget = Schema.Struct({
|
|
634
685
|
kind: Schema.Literals(["configured", "external"]),
|
|
635
686
|
identity: Schema.String,
|
|
@@ -648,7 +699,8 @@ export const EvalRunLedger = Schema.Struct({
|
|
|
648
699
|
knownOutputTokens: NonNegativeInteger,
|
|
649
700
|
unknownTokenMeasurements: NonNegativeInteger,
|
|
650
701
|
knownPricedSubtotalUsd: NonNegativeFinite,
|
|
651
|
-
unpricedCalls: NonNegativeInteger
|
|
702
|
+
unpricedCalls: NonNegativeInteger,
|
|
703
|
+
subscriptionUnitCalls: NonNegativeInteger.pipe(Schema.withDecodingDefaultKey(Effect.succeed(0)))
|
|
652
704
|
});
|
|
653
705
|
export const EvalClassifierObservation = Schema.Struct({
|
|
654
706
|
caseId: Schema.String,
|
|
@@ -687,14 +739,10 @@ export const EvalRunQualification = Schema.Struct({
|
|
|
687
739
|
cases: Schema.Array(EvalCompositionCaseResult)
|
|
688
740
|
})
|
|
689
741
|
});
|
|
690
|
-
export function summarizeEvalRunLedger(comparisons, expectedCalls, classifierObservations = []) {
|
|
742
|
+
export function summarizeEvalRunLedger(comparisons, expectedCalls, costService, classifierObservations = [], classifierModel = "unknown") {
|
|
691
743
|
let observedCalls = 0;
|
|
692
744
|
let observedCandidateRows = 0;
|
|
693
|
-
|
|
694
|
-
let knownOutputTokens = 0;
|
|
695
|
-
let unknownTokenMeasurements = 0;
|
|
696
|
-
let knownPricedSubtotalUsd = 0;
|
|
697
|
-
let unpricedCalls = 0;
|
|
745
|
+
const actualCalls = [];
|
|
698
746
|
for (const comparison of comparisons) {
|
|
699
747
|
const calls = comparison.calls ??
|
|
700
748
|
comparison.models.flatMap((model) => model.cases.map((testCase) => ({
|
|
@@ -707,49 +755,50 @@ export function summarizeEvalRunLedger(comparisons, expectedCalls, classifierObs
|
|
|
707
755
|
observedCalls += 1;
|
|
708
756
|
if (call.role === "candidate")
|
|
709
757
|
observedCandidateRows += 1;
|
|
710
|
-
|
|
711
|
-
call.
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
}
|
|
758
|
+
actualCalls.push({
|
|
759
|
+
model: call.model,
|
|
760
|
+
usage: {
|
|
761
|
+
...(call.measurement.inputTokens === undefined
|
|
762
|
+
? {}
|
|
763
|
+
: { inputTokens: call.measurement.inputTokens }),
|
|
764
|
+
...(call.measurement.outputTokens === undefined
|
|
765
|
+
? {}
|
|
766
|
+
: { outputTokens: call.measurement.outputTokens })
|
|
767
|
+
},
|
|
768
|
+
...(call.measurement.costUsd === undefined
|
|
769
|
+
? {}
|
|
770
|
+
: { providerCostUsd: call.measurement.costUsd })
|
|
771
|
+
});
|
|
724
772
|
}
|
|
725
773
|
}
|
|
726
774
|
for (const observation of classifierObservations) {
|
|
727
775
|
observedCalls += 1;
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
}
|
|
776
|
+
actualCalls.push({
|
|
777
|
+
model: classifierModel,
|
|
778
|
+
usage: {
|
|
779
|
+
...(observation.measurement.inputTokens === undefined
|
|
780
|
+
? {}
|
|
781
|
+
: { inputTokens: observation.measurement.inputTokens }),
|
|
782
|
+
...(observation.measurement.outputTokens === undefined
|
|
783
|
+
? {}
|
|
784
|
+
: { outputTokens: observation.measurement.outputTokens })
|
|
785
|
+
},
|
|
786
|
+
...(observation.measurement.costUsd === undefined
|
|
787
|
+
? {}
|
|
788
|
+
: { providerCostUsd: observation.measurement.costUsd })
|
|
789
|
+
});
|
|
742
790
|
}
|
|
743
|
-
return {
|
|
791
|
+
return costService.actuals(actualCalls).pipe(Effect.map((actuals) => ({
|
|
744
792
|
expectedCalls,
|
|
745
793
|
observedCalls,
|
|
746
794
|
observedCandidateRows,
|
|
747
|
-
knownInputTokens,
|
|
748
|
-
knownOutputTokens,
|
|
749
|
-
unknownTokenMeasurements,
|
|
750
|
-
knownPricedSubtotalUsd,
|
|
751
|
-
unpricedCalls
|
|
752
|
-
|
|
795
|
+
knownInputTokens: actuals.knownInputTokens,
|
|
796
|
+
knownOutputTokens: actuals.knownOutputTokens,
|
|
797
|
+
unknownTokenMeasurements: actuals.unknownTokenCalls,
|
|
798
|
+
knownPricedSubtotalUsd: actuals.knownPricedSubtotalUsd,
|
|
799
|
+
unpricedCalls: actuals.unpricedCalls + actuals.upstreamManagedCalls + actuals.unknownBillingCalls,
|
|
800
|
+
subscriptionUnitCalls: actuals.subscriptionUnitCalls
|
|
801
|
+
})));
|
|
753
802
|
}
|
|
754
803
|
const EvalRunReportCommon = {
|
|
755
804
|
version: Schema.Literal(EVAL_PROJECT_VERSION),
|
|
@@ -764,7 +813,8 @@ const EvalRunReportCommon = {
|
|
|
764
813
|
cleanup: EvalRunCleanup,
|
|
765
814
|
comparisons: Schema.Array(EvalComparisonResult),
|
|
766
815
|
qualification: Schema.optionalKey(EvalRunQualification),
|
|
767
|
-
ledger: EvalRunLedger
|
|
816
|
+
ledger: EvalRunLedger,
|
|
817
|
+
events: Schema.optionalKey(EvalNamedEventLog).pipe(Schema.withDecodingDefaultKey(Effect.succeed([])))
|
|
768
818
|
};
|
|
769
819
|
export const EvalRunFailureError = Schema.Struct({
|
|
770
820
|
name: Schema.String,
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { CostService } from "@velum-labs/routekit-registry";
|
|
1
2
|
import { Context, Effect, Layer, Path } from "effect";
|
|
2
3
|
import type { EvalProjectArtifactError, EvalSetupInspectionError } from "./errors.js";
|
|
3
4
|
import { EvalDimensionContrastError, type EvalModelCatalogError, type EvalProjectStoreError, EvalProjectTransitionError } from "./errors.js";
|
|
@@ -15,7 +16,7 @@ export type EvalProjectWorkflowShape = {
|
|
|
15
16
|
readonly approveDimensions: (repositoryRoot: string, basisDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
16
17
|
readonly proposeEvaluations: (repositoryRoot: string, proposal: Omit<EvalEvaluationProposal, "version" | "evaluationDigest" | "basisDigest" | "candidateModels" | "candidateReasoningEfforts" | "judgeModel">) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
17
18
|
readonly approveEvaluations: (repositoryRoot: string, evaluationDigest: string) => Effect.Effect<EvalProjectStatus, EvalProjectWorkflowError>;
|
|
18
|
-
readonly createPlan: (repositoryRoot: string, scope: EvalPlanScope) => Effect.Effect<EvalExecutionPlan, EvalProjectWorkflowError>;
|
|
19
|
+
readonly createPlan: (repositoryRoot: string, scope: EvalPlanScope, repetitions?: number) => Effect.Effect<EvalExecutionPlan, EvalProjectWorkflowError>;
|
|
19
20
|
readonly startRun: (repositoryRoot: string, planId: string) => Effect.Effect<{
|
|
20
21
|
readonly plan: EvalExecutionPlan;
|
|
21
22
|
readonly runId: string;
|
|
@@ -28,6 +29,6 @@ export type EvalProjectWorkflowShape = {
|
|
|
28
29
|
declare const EvalProjectWorkflow_base: Context.ServiceClass<EvalProjectWorkflow, "@velum-labs/routekit-eval-setup/EvalProjectWorkflow", EvalProjectWorkflowShape>;
|
|
29
30
|
export declare class EvalProjectWorkflow extends EvalProjectWorkflow_base {
|
|
30
31
|
}
|
|
31
|
-
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
|
-
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog>;
|
|
32
|
+
export declare const makeEvalProjectWorkflow: Effect.Effect<EvalProjectWorkflowShape, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog | CostService>;
|
|
33
|
+
export declare const EvalProjectWorkflowLive: Layer.Layer<EvalProjectWorkflow, never, EvalProjectArtifacts | Path.Path | EvalProjectStore | EvalRepositoryInspector | EvalModelCatalog | CostService>;
|
|
33
34
|
export {};
|
package/dist/project-workflow.js
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
import { assertDecompositionResult, assertPublishedRoutingActivation, assertRoutingObjectivePolicy, isForbiddenEvalModel } from "@velum-labs/routekit-eval-contracts";
|
|
2
2
|
import { RoutingActivationConstraints, RoutingObjectivePolicy } from "@velum-labs/routekit-eval-contracts";
|
|
3
|
+
import { CostService } from "@velum-labs/routekit-registry";
|
|
3
4
|
import { Clock, Context, Effect, Layer, Path, Schema } from "effect";
|
|
4
5
|
import { EvalDimensionContrastError, EvalProjectTransitionError } from "./errors.js";
|
|
5
6
|
import { EvalRepositoryInspector } from "./inspection.js";
|
|
6
7
|
import { candidateReasoningEffortFromAnswer } from "./model-selection.js";
|
|
7
8
|
import { EvalProjectArtifacts, evaluationProposalDigest, routingBasisDigest } from "./project-artifacts.js";
|
|
8
|
-
import { EVAL_PROJECT_VERSION, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
9
|
+
import { EVAL_PROJECT_VERSION, estimateEvalPlanCost, scopedExpectedCalls, summarizeEvalRunLedger } from "./project-contracts.js";
|
|
9
10
|
import { EvalProjectStore } from "./project-store.js";
|
|
10
11
|
import { EvalModelCatalog } from "./services/model-catalog/service.js";
|
|
11
12
|
const isoNow = Effect.map(Clock.currentTimeMillis, (millis) => new Date(millis).toISOString());
|
|
@@ -157,8 +158,9 @@ const sameLedger = (left, right) => left.expectedCalls === right.expectedCalls &
|
|
|
157
158
|
left.knownOutputTokens === right.knownOutputTokens &&
|
|
158
159
|
left.unknownTokenMeasurements === right.unknownTokenMeasurements &&
|
|
159
160
|
left.knownPricedSubtotalUsd === right.knownPricedSubtotalUsd &&
|
|
160
|
-
left.unpricedCalls === right.unpricedCalls
|
|
161
|
-
|
|
161
|
+
left.unpricedCalls === right.unpricedCalls &&
|
|
162
|
+
left.subscriptionUnitCalls === right.subscriptionUnitCalls;
|
|
163
|
+
const validateQualifiedReport = (state, plan, proposal, report, costService) => Effect.gen(function* () {
|
|
162
164
|
assertPublishedRoutingActivation(report.activation, {
|
|
163
165
|
dimensionIds: plan.selectedCaseIds.map((entry) => entry.dimensionId)
|
|
164
166
|
});
|
|
@@ -302,13 +304,13 @@ const validateQualifiedReport = (state, plan, proposal, report) => {
|
|
|
302
304
|
}
|
|
303
305
|
}
|
|
304
306
|
const expected = scopedExpectedCalls(plan);
|
|
305
|
-
const ledger = summarizeEvalRunLedger(report.comparisons, expected.expectedCallCount, qualification?.decomposition.observations ?? []);
|
|
307
|
+
const ledger = yield* summarizeEvalRunLedger(report.comparisons, expected.expectedCallCount, costService, qualification?.decomposition.observations ?? [], plan.classifierModel);
|
|
306
308
|
if (ledger.observedCalls !== expected.expectedCallCount ||
|
|
307
309
|
ledger.observedCandidateRows !== expected.expectedCandidateCalls ||
|
|
308
310
|
!sameLedger(ledger, report.ledger)) {
|
|
309
311
|
throw new Error("qualification ledger does not match the completed evidence");
|
|
310
312
|
}
|
|
311
|
-
};
|
|
313
|
+
});
|
|
312
314
|
const parseObjective = (state, answer) => Effect.gen(function* () {
|
|
313
315
|
const text = yield* nonEmptyAnswer(state, answer);
|
|
314
316
|
const json = yield* Effect.try({
|
|
@@ -512,6 +514,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
512
514
|
const artifacts = yield* EvalProjectArtifacts;
|
|
513
515
|
const inspector = yield* EvalRepositoryInspector;
|
|
514
516
|
const modelCatalog = yield* EvalModelCatalog;
|
|
517
|
+
const costService = yield* CostService;
|
|
515
518
|
const paths = yield* Path.Path;
|
|
516
519
|
const resolveRoot = (repositoryRoot) => paths.resolve(repositoryRoot);
|
|
517
520
|
const statusOf = (root, state) => Effect.gen(function* () {
|
|
@@ -790,7 +793,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
790
793
|
yield* store.save(root, next);
|
|
791
794
|
return yield* statusOf(root, next);
|
|
792
795
|
});
|
|
793
|
-
const createPlan = (repositoryRoot, scope) => Effect.gen(function* () {
|
|
796
|
+
const createPlan = (repositoryRoot, scope, repetitions = 1) => Effect.gen(function* () {
|
|
794
797
|
const root = resolveRoot(repositoryRoot);
|
|
795
798
|
const state = yield* loadRequired(root);
|
|
796
799
|
if (state.stage !== "ready" && state.stage !== "qualified") {
|
|
@@ -805,6 +808,9 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
805
808
|
proposal.basisDigest !== state.basisDigest) {
|
|
806
809
|
return yield* transitionError(state.stage, "approved evaluation artifacts are stale");
|
|
807
810
|
}
|
|
811
|
+
if (!Number.isSafeInteger(repetitions) || repetitions < 1) {
|
|
812
|
+
return yield* transitionError(state.stage, "eval repetitions N must be at least 1");
|
|
813
|
+
}
|
|
808
814
|
if (scope === "full") {
|
|
809
815
|
const shortSuite = proposal.suites.find((suite) => suite.cases.length < 20);
|
|
810
816
|
if (shortSuite !== undefined ||
|
|
@@ -817,16 +823,22 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
817
823
|
}
|
|
818
824
|
const selectedCaseIds = proposal.suites.map((suite) => ({
|
|
819
825
|
dimensionId: suite.dimensionId,
|
|
820
|
-
caseIds: suite.cases
|
|
826
|
+
caseIds: Array.from({ length: repetitions }, (_, repetition) => suite.cases
|
|
821
827
|
.slice(0, scope === "pilot" ? 5 : undefined)
|
|
822
|
-
.map((testCase) =>
|
|
828
|
+
.map((testCase) => repetitions === 1
|
|
829
|
+
? testCase.id
|
|
830
|
+
: `${testCase.id}#${String(repetition + 1)}`)).flat()
|
|
823
831
|
}));
|
|
824
|
-
const selectedDecompositionCaseIds = proposal.decompositionBenchmark.cases
|
|
832
|
+
const selectedDecompositionCaseIds = Array.from({ length: repetitions }, (_, repetition) => proposal.decompositionBenchmark.cases
|
|
825
833
|
.slice(0, scope === "pilot" ? 5 : undefined)
|
|
826
|
-
.map((testCase) =>
|
|
827
|
-
|
|
834
|
+
.map((testCase) => repetitions === 1
|
|
835
|
+
? testCase.id
|
|
836
|
+
: `${testCase.id}#${String(repetition + 1)}`)).flat();
|
|
837
|
+
const selectedCompositionCaseIds = Array.from({ length: repetitions }, (_, repetition) => proposal.compositionSuite.cases
|
|
828
838
|
.slice(0, scope === "pilot" ? 5 : undefined)
|
|
829
|
-
.map((testCase) =>
|
|
839
|
+
.map((testCase) => repetitions === 1
|
|
840
|
+
? testCase.id
|
|
841
|
+
: `${testCase.id}#${String(repetition + 1)}`)).flat();
|
|
830
842
|
const expected = scopedExpectedCalls({
|
|
831
843
|
candidateModels: proposal.candidateModels,
|
|
832
844
|
selectedCaseIds,
|
|
@@ -840,6 +852,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
840
852
|
projectRevision: state.revision,
|
|
841
853
|
createdAt: yield* isoNow,
|
|
842
854
|
scope,
|
|
855
|
+
repetitions,
|
|
843
856
|
basisDigest: state.basisDigest,
|
|
844
857
|
evaluationDigest: state.evaluationDigest,
|
|
845
858
|
candidateModels: state.configuration.candidateModels,
|
|
@@ -855,7 +868,15 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
855
868
|
selectedDecompositionCaseIds,
|
|
856
869
|
selectedCompositionCaseIds,
|
|
857
870
|
maximumOutputTokens: Math.max(proposal.compositionSuite.maximumOutputTokens, ...proposal.suites.map((suite) => suite.maximumOutputTokens)),
|
|
858
|
-
...expected
|
|
871
|
+
...expected,
|
|
872
|
+
costEstimate: yield* estimateEvalPlanCost({
|
|
873
|
+
candidateModels: state.configuration.candidateModels,
|
|
874
|
+
candidateReasoningEfforts: state.configuration.candidateReasoningEfforts,
|
|
875
|
+
classifierModel: state.configuration.classifierModel,
|
|
876
|
+
judgeModel: state.configuration.judgeModel,
|
|
877
|
+
maximumOutputTokens: Math.max(proposal.compositionSuite.maximumOutputTokens, ...proposal.suites.map((suite) => suite.maximumOutputTokens)),
|
|
878
|
+
...expected
|
|
879
|
+
}, costService)
|
|
859
880
|
};
|
|
860
881
|
yield* artifacts.materializePlanSuites(root, plan, proposal);
|
|
861
882
|
yield* artifacts.savePlan(root, plan);
|
|
@@ -877,6 +898,9 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
877
898
|
plan.evaluationDigest !== state.evaluationDigest ||
|
|
878
899
|
!sameStrings(plan.candidateModels, state.configuration.candidateModels) ||
|
|
879
900
|
!sameReasoningEfforts(plan.candidateReasoningEfforts, state.configuration.candidateReasoningEfforts) ||
|
|
901
|
+
(plan.costEstimate !== undefined &&
|
|
902
|
+
JSON.stringify(plan.costEstimate) !==
|
|
903
|
+
JSON.stringify(yield* estimateEvalPlanCost(plan, costService))) ||
|
|
880
904
|
plan.classifierModel !== state.configuration.classifierModel ||
|
|
881
905
|
plan.authorModel !== state.configuration.authorModel ||
|
|
882
906
|
plan.judgeModel !== state.configuration.judgeModel) {
|
|
@@ -940,10 +964,7 @@ export const makeEvalProjectWorkflow = Effect.gen(function* () {
|
|
|
940
964
|
(plan.scope === "pilot" && report.status !== "completed")) {
|
|
941
965
|
return yield* transitionError(state.stage, "only full plans may qualify routing activation; pilot plans are completed diagnostics");
|
|
942
966
|
}
|
|
943
|
-
yield* Effect.
|
|
944
|
-
try: () => validateQualifiedReport(state, plan, proposal, report),
|
|
945
|
-
catch: (cause) => transitionError(state.stage, cause instanceof Error ? cause.message : "qualification report is invalid")
|
|
946
|
-
});
|
|
967
|
+
yield* validateQualifiedReport(state, plan, proposal, report, costService).pipe(Effect.catchDefect((cause) => Effect.fail(transitionError(state.stage, cause instanceof Error ? cause.message : "qualification report is invalid"))));
|
|
947
968
|
const reportPath = yield* artifacts.saveRunReport(root, report);
|
|
948
969
|
const common = {
|
|
949
970
|
version: state.version,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { test } from "node:test";
|
|
3
|
+
import { Schema } from "effect";
|
|
4
|
+
import { EvalNamedEventLog, assertEvalRunCanComplete, presentEvalEventLog, resumeEvalCaseCursors } from "../eval-event-log.js";
|
|
5
|
+
const at = "2026-08-26T00:00:00.000Z";
|
|
6
|
+
const call = (name, caseId, sequence, dimensionId = "dimension") => ({
|
|
7
|
+
name,
|
|
8
|
+
at,
|
|
9
|
+
runId: "run",
|
|
10
|
+
dimensionId,
|
|
11
|
+
caseId,
|
|
12
|
+
sequence
|
|
13
|
+
});
|
|
14
|
+
test("N=4 candidate-only activity stays per-case and fails before completion", () => {
|
|
15
|
+
const events = [
|
|
16
|
+
{ name: "plan", at, runId: "run", plannedCalls: 80 },
|
|
17
|
+
...Array.from({ length: 40 }, (_, index) => call("candidate#", `case-${String((index % 4) + 1)}`, Math.floor(index / 4) + 1))
|
|
18
|
+
];
|
|
19
|
+
const presentation = presentEvalEventLog(events);
|
|
20
|
+
assert.equal(presentation.failed, true);
|
|
21
|
+
assert.equal(presentation.lines.length, 4);
|
|
22
|
+
assert.ok(presentation.lines.every((line) => /fail candidate#1 missing judge#1/u.test(line)));
|
|
23
|
+
assert.doesNotMatch(presentation.lines.join("\n"), /40\/80|candidate#40/u);
|
|
24
|
+
assert.throws(() => assertEvalRunCanComplete(events), /cannot complete after candidate#1/u);
|
|
25
|
+
});
|
|
26
|
+
test("resume cursor stops at the first unmatched candidate and does not resume past the hole", () => {
|
|
27
|
+
const events = [
|
|
28
|
+
{ name: "plan", at, runId: "run", plannedCalls: 8 },
|
|
29
|
+
call("candidate#", "case-a", 1),
|
|
30
|
+
call("judge#", "case-a", 1),
|
|
31
|
+
call("candidate#", "case-a", 2),
|
|
32
|
+
call("candidate#", "case-a", 3),
|
|
33
|
+
call("judge#", "case-a", 3)
|
|
34
|
+
];
|
|
35
|
+
assert.deepEqual(resumeEvalCaseCursors(events), [
|
|
36
|
+
{
|
|
37
|
+
dimensionId: "dimension",
|
|
38
|
+
caseId: "case-a",
|
|
39
|
+
completePairs: 1,
|
|
40
|
+
lastEvent: call("candidate#", "case-a", 2),
|
|
41
|
+
hole: call("candidate#", "case-a", 2)
|
|
42
|
+
}
|
|
43
|
+
]);
|
|
44
|
+
});
|
|
45
|
+
test("zero planned calls is a failed pulse, not progress", () => {
|
|
46
|
+
const events = [{ name: "plan", at, runId: "run", plannedCalls: 0 }];
|
|
47
|
+
assert.deepEqual(presentEvalEventLog(events), {
|
|
48
|
+
failed: true,
|
|
49
|
+
lines: ["run fail no planned calls"]
|
|
50
|
+
});
|
|
51
|
+
assert.throws(() => assertEvalRunCanComplete(events), /at least one call/u);
|
|
52
|
+
});
|
|
53
|
+
test("CI JSON round-trips the exact same named event log", () => {
|
|
54
|
+
const events = [
|
|
55
|
+
{ name: "basis", at, operationId: "author-basis" },
|
|
56
|
+
{ name: "dimensions", at, operationId: "author-dimensions" },
|
|
57
|
+
{ name: "activation", at, operationId: "author-activation" },
|
|
58
|
+
{ name: "plan", at, runId: "run", plannedCalls: 2 },
|
|
59
|
+
call("candidate#", "case-a", 1),
|
|
60
|
+
call("judge#", "case-a", 1)
|
|
61
|
+
];
|
|
62
|
+
const decoded = Schema.decodeUnknownSync(EvalNamedEventLog)(JSON.parse(JSON.stringify(events)));
|
|
63
|
+
assert.deepEqual(decoded, events);
|
|
64
|
+
assert.doesNotThrow(() => assertEvalRunCanComplete(decoded));
|
|
65
|
+
assert.deepEqual([...new Set(decoded.map((event) => event.name))], ["basis", "dimensions", "activation", "plan", "candidate#", "judge#"]);
|
|
66
|
+
});
|
|
67
|
+
test("authoring heartbeat is the last named artifact event", () => {
|
|
68
|
+
const events = [
|
|
69
|
+
{ name: "basis", at, operationId: "author-basis" },
|
|
70
|
+
{ name: "dimensions", at, operationId: "author-dimensions" }
|
|
71
|
+
];
|
|
72
|
+
assert.deepEqual(presentEvalEventLog(events, { failHoles: false }), {
|
|
73
|
+
failed: false,
|
|
74
|
+
lines: ["authoring dimensions"]
|
|
75
|
+
});
|
|
76
|
+
});
|
|
77
|
+
test("live presentation keeps a candidate hole in flight and failure presentation names it", () => {
|
|
78
|
+
const events = [
|
|
79
|
+
{ name: "plan", at, runId: "run", plannedCalls: 2 },
|
|
80
|
+
call("candidate#", "case-a", 1)
|
|
81
|
+
];
|
|
82
|
+
const live = presentEvalEventLog(events, { failHoles: false });
|
|
83
|
+
assert.deepEqual(live, {
|
|
84
|
+
failed: false,
|
|
85
|
+
lines: ["dimension/case-a candidate#1"]
|
|
86
|
+
});
|
|
87
|
+
assert.doesNotMatch(live.lines.join("\n"), /missing judge#/u);
|
|
88
|
+
assert.deepEqual(presentEvalEventLog(events, { failHoles: true }), {
|
|
89
|
+
failed: true,
|
|
90
|
+
lines: ["dimension/case-a fail candidate#1 missing judge#1"]
|
|
91
|
+
});
|
|
92
|
+
});
|
|
@@ -4,11 +4,12 @@ import os from "node:os";
|
|
|
4
4
|
import path from "node:path";
|
|
5
5
|
import { after, test } from "node:test";
|
|
6
6
|
import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
|
|
7
|
+
import { CostServiceLive, makeCostService } from "@velum-labs/routekit-registry";
|
|
7
8
|
import { Cause, Effect, Layer } from "effect";
|
|
8
9
|
import { EvalDimensionContrastError, EvalModelCatalogError } from "../errors.js";
|
|
9
10
|
import { EvalRepositoryInspectorLive } from "../inspection.js";
|
|
10
11
|
import { EvalProjectArtifactsLive } from "../project-artifacts.js";
|
|
11
|
-
import { scopedExpectedCalls, summarizeEvalRunLedger } from "../project-contracts.js";
|
|
12
|
+
import { estimateEvalPlanCost, scopedExpectedCalls, summarizeEvalRunLedger } from "../project-contracts.js";
|
|
12
13
|
import { EvalProjectStoreLive } from "../project-store.js";
|
|
13
14
|
import { EvalProjectWorkflow, EvalProjectWorkflowLive } from "../project-workflow.js";
|
|
14
15
|
import { EvalModelCatalog } from "../services/model-catalog/service.js";
|
|
@@ -28,7 +29,8 @@ const catalogEfforts = (candidateModels) => Effect.sync(() => {
|
|
|
28
29
|
};
|
|
29
30
|
});
|
|
30
31
|
const EvalModelCatalogTest = Layer.succeed(EvalModelCatalog, EvalModelCatalog.of({ reasoningEfforts: catalogEfforts }));
|
|
31
|
-
const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(EvalModelCatalogTest), Layer.provide(NodeServicesLayer));
|
|
32
|
+
const ProjectWorkflowTestLive = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(EvalModelCatalogTest), Layer.provide(CostServiceLive), Layer.provide(NodeServicesLayer));
|
|
33
|
+
const testCostService = makeCostService();
|
|
32
34
|
const makeRepository = async () => {
|
|
33
35
|
const root = await mkdtemp(path.join(os.tmpdir(), "routekit-eval-project-"));
|
|
34
36
|
roots.push(root);
|
|
@@ -162,7 +164,7 @@ test("candidate selection fails closed when catalog reasoning metadata is unavai
|
|
|
162
164
|
detail: "models.list did not advertise candidate reasoning capabilities"
|
|
163
165
|
}))
|
|
164
166
|
}));
|
|
165
|
-
const workflowLayer = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(unavailableCatalog), Layer.provide(NodeServicesLayer));
|
|
167
|
+
const workflowLayer = EvalProjectWorkflowLive.pipe(Layer.provide(ProjectDependenciesLive), Layer.provide(unavailableCatalog), Layer.provide(CostServiceLive), Layer.provide(NodeServicesLayer));
|
|
166
168
|
await Effect.runPromise(Effect.gen(function* () {
|
|
167
169
|
const workflow = yield* EvalProjectWorkflow;
|
|
168
170
|
yield* workflow.setup(root);
|
|
@@ -356,7 +358,8 @@ test("reviewed artifacts are digest-bound and produce an immutable exact-call pl
|
|
|
356
358
|
assert.equal(ready.state.stage, "ready");
|
|
357
359
|
assert.equal(ready.nextAction, "run");
|
|
358
360
|
const plan = yield* workflow.createPlan(root, "pilot");
|
|
359
|
-
|
|
361
|
+
const repeatedPlan = yield* workflow.createPlan(root, "pilot", 2);
|
|
362
|
+
return { basisDigest, evaluationDigest, plan, repeatedPlan };
|
|
360
363
|
}).pipe(Effect.provide(ProjectWorkflowTestLive)));
|
|
361
364
|
const storedBasis = JSON.parse(await readFile(path.join(root, ".routekit", "evals", "routing-basis.proposed.json"), "utf8"));
|
|
362
365
|
assert.equal(storedBasis.dimensionContrasts.length, reviewedDimensions.length);
|
|
@@ -374,6 +377,12 @@ test("reviewed artifacts are digest-bound and produce an immutable exact-call pl
|
|
|
374
377
|
assert.equal(result.plan.expectedJudgeCalls, 90);
|
|
375
378
|
assert.equal(result.plan.expectedCallCount, 185);
|
|
376
379
|
assert.equal(result.plan.maximumOutputTokens, 256);
|
|
380
|
+
assert.equal(result.plan.repetitions, 1);
|
|
381
|
+
assert.equal(result.repeatedPlan.repetitions, 2);
|
|
382
|
+
assert.equal(result.repeatedPlan.expectedCallCount, result.plan.expectedCallCount * 2);
|
|
383
|
+
assert.equal(result.plan.costEstimate !== undefined, true);
|
|
384
|
+
assert.equal(result.repeatedPlan.costEstimate !== undefined, true);
|
|
385
|
+
assert.equal(result.repeatedPlan.costEstimate?.knownPricedMaximumUsd, (result.plan.costEstimate?.knownPricedMaximumUsd ?? 0) * 2);
|
|
377
386
|
const planPath = path.join(root, ".routekit", "evals", "plans", `${result.plan.planId}.json`);
|
|
378
387
|
assert.equal((await stat(planPath)).mode & 0o777, 0o600);
|
|
379
388
|
assert.deepEqual(JSON.parse(await readFile(planPath, "utf8")), result.plan);
|
|
@@ -566,7 +575,7 @@ const qualifiedReport = (plan, runId, projectId, publishAllowed = true) => {
|
|
|
566
575
|
}))
|
|
567
576
|
}
|
|
568
577
|
},
|
|
569
|
-
ledger: summarizeEvalRunLedger(comparisons, plan.expectedCallCount, observations),
|
|
578
|
+
ledger: Effect.runSync(summarizeEvalRunLedger(comparisons, plan.expectedCallCount, testCostService, observations)),
|
|
570
579
|
activation: {
|
|
571
580
|
version: 2,
|
|
572
581
|
generatedAt: "2026-08-18T00:02:00.000Z",
|
|
@@ -633,8 +642,9 @@ test("finishRun derives ledger expectations from the selected case slice", async
|
|
|
633
642
|
const root = await makeRepository();
|
|
634
643
|
const created = await createPlan(root, "full", "openai/gpt-5.6-luna openai/gpt-5.6-terra");
|
|
635
644
|
const selectedDimension = created.selectedCaseIds[0];
|
|
636
|
-
const
|
|
645
|
+
const stalePlanInput = {
|
|
637
646
|
...created,
|
|
647
|
+
repetitions: 1,
|
|
638
648
|
selectedCaseIds: [
|
|
639
649
|
{
|
|
640
650
|
dimensionId: selectedDimension.dimensionId,
|
|
@@ -652,6 +662,10 @@ test("finishRun derives ledger expectations from the selected case slice", async
|
|
|
652
662
|
expectedJudgeCalls: 3_400,
|
|
653
663
|
expectedCallCount: 6_820
|
|
654
664
|
};
|
|
665
|
+
const stalePlan = {
|
|
666
|
+
...stalePlanInput,
|
|
667
|
+
costEstimate: Effect.runSync(estimateEvalPlanCost(stalePlanInput, testCostService))
|
|
668
|
+
};
|
|
655
669
|
await writeFile(path.join(root, ".routekit", "evals", "plans", `${created.planId}.json`), `${JSON.stringify(stalePlan)}\n`);
|
|
656
670
|
const outcome = await Effect.runPromise(Effect.gen(function* () {
|
|
657
671
|
const workflow = yield* EvalProjectWorkflow;
|
|
@@ -671,7 +685,7 @@ test("finishRun derives ledger expectations from the selected case slice", async
|
|
|
671
685
|
});
|
|
672
686
|
const scopedReport = {
|
|
673
687
|
...report,
|
|
674
|
-
ledger: summarizeEvalRunLedger(report.comparisons, expected.expectedCallCount)
|
|
688
|
+
ledger: yield* summarizeEvalRunLedger(report.comparisons, expected.expectedCallCount, testCostService)
|
|
675
689
|
};
|
|
676
690
|
const finished = yield* workflow.finishRun(root, scopedReport);
|
|
677
691
|
return { finished, report: scopedReport };
|
|
@@ -738,7 +752,8 @@ test("failed runs preserve plan revisions for retry while mismatched plans remai
|
|
|
738
752
|
knownOutputTokens: 0,
|
|
739
753
|
unknownTokenMeasurements: 0,
|
|
740
754
|
knownPricedSubtotalUsd: 0,
|
|
741
|
-
unpricedCalls: 0
|
|
755
|
+
unpricedCalls: 0,
|
|
756
|
+
subscriptionUnitCalls: 0
|
|
742
757
|
},
|
|
743
758
|
failure: "qualification timed out before observing any calls",
|
|
744
759
|
errors: []
|
|
@@ -842,7 +857,8 @@ test("failed run reports persist the nested qualification error chain", async ()
|
|
|
842
857
|
knownOutputTokens: 0,
|
|
843
858
|
unknownTokenMeasurements: 0,
|
|
844
859
|
knownPricedSubtotalUsd: 0,
|
|
845
|
-
unpricedCalls: 0
|
|
860
|
+
unpricedCalls: 0,
|
|
861
|
+
subscriptionUnitCalls: 0
|
|
846
862
|
},
|
|
847
863
|
failure: 'qualification dimension "provider-protocol-translation" failed ' +
|
|
848
864
|
"(per-test timeout 600000ms); observed call ids: none",
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-eval-setup",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.4.0",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -31,9 +31,10 @@
|
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"effect": "4.0.0-rc.108",
|
|
34
|
-
"@velum-labs/routekit-eval-contracts": "1.
|
|
35
|
-
"@velum-labs/routekit-eval-core": "1.
|
|
36
|
-
"@velum-labs/routekit-
|
|
34
|
+
"@velum-labs/routekit-eval-contracts": "1.4.0",
|
|
35
|
+
"@velum-labs/routekit-eval-core": "1.4.0",
|
|
36
|
+
"@velum-labs/routekit-registry": "1.4.0",
|
|
37
|
+
"@velum-labs/routekit-runtime": "1.4.0"
|
|
37
38
|
},
|
|
38
39
|
"keywords": [
|
|
39
40
|
"routekit",
|