raindrop-ai 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-EWHJ2YBW.mjs → chunk-RIL2CV7Q.mjs} +1 -2
- package/dist/{chunk-EFATLUNL.mjs → chunk-UY6537WW.mjs} +359 -357
- package/dist/{chunk-22CNYHSA.mjs → chunk-VBZXQNJY.mjs} +101 -699
- package/dist/evals/cli.js +667 -1803
- package/dist/evals/cli.mjs +3 -3
- package/dist/{index-gf1epf_P.d.ts → index-DZB7DoZS.d.mts} +33354 -32553
- package/dist/{index-gf1epf_P.d.mts → index-DZB7DoZS.d.ts} +33354 -32553
- package/dist/index.d.mts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +868 -1451
- package/dist/index.mjs +5 -3
- package/dist/{portable-runtime-6X2HHB4Y.mjs → portable-runtime-F7W4STGH.mjs} +1 -1
- package/dist/tracing/index.d.mts +1 -1
- package/dist/tracing/index.d.ts +1 -1
- package/dist/tracing/index.js +1 -1
- package/dist/tracing/index.mjs +1 -1
- package/package.json +2 -2
|
@@ -1,77 +1,7 @@
|
|
|
1
|
-
// src/evals/
|
|
2
|
-
import { createHash } from "crypto";
|
|
1
|
+
// src/evals/generated/manifest.ts
|
|
3
2
|
import { z } from "zod";
|
|
4
|
-
var EvalScoreSchema = z.union([
|
|
5
|
-
z.literal(1),
|
|
6
|
-
z.literal(2),
|
|
7
|
-
z.literal(3),
|
|
8
|
-
z.literal(4),
|
|
9
|
-
z.literal(5)
|
|
10
|
-
]);
|
|
11
|
-
var EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
|
|
12
|
-
var DatasetRowInputSchema = z.object({
|
|
13
|
-
id: z.string().trim().min(1),
|
|
14
|
-
name: z.string().optional(),
|
|
15
|
-
input: z.string().nullable(),
|
|
16
|
-
output: z.string().nullable().default(null),
|
|
17
|
-
properties: z.record(z.string()).default({})
|
|
18
|
-
});
|
|
19
|
-
function defineDataset(dataset) {
|
|
20
|
-
var _a, _b;
|
|
21
|
-
const id = dataset.id.trim();
|
|
22
|
-
const name = dataset.name.trim();
|
|
23
|
-
const rows = dataset.rows.map((row) => {
|
|
24
|
-
var _a2;
|
|
25
|
-
const parsed = DatasetRowInputSchema.parse(row);
|
|
26
|
-
return {
|
|
27
|
-
id: parsed.id,
|
|
28
|
-
name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
|
|
29
|
-
input: parsed.input,
|
|
30
|
-
output: parsed.output,
|
|
31
|
-
properties: parsed.properties
|
|
32
|
-
};
|
|
33
|
-
});
|
|
34
|
-
const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : createHash("sha256").update(
|
|
35
|
-
JSON.stringify(
|
|
36
|
-
rows.map((row) => ({
|
|
37
|
-
...row,
|
|
38
|
-
properties: Object.fromEntries(
|
|
39
|
-
Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
|
|
40
|
-
)
|
|
41
|
-
}))
|
|
42
|
-
)
|
|
43
|
-
).digest("hex");
|
|
44
|
-
if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
|
|
45
|
-
if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
|
|
46
|
-
if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
|
|
47
|
-
if (dataset.rows.length === 0) throw new Error("Raindrop eval datasets require at least one row");
|
|
48
|
-
const ids = rows.map((entry) => entry.id);
|
|
49
|
-
if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
|
|
50
|
-
if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
|
|
51
|
-
if (dataset.remote) {
|
|
52
|
-
if (!dataset.remote.reference.trim())
|
|
53
|
-
throw new Error("Raindrop remote dataset reference cannot be empty");
|
|
54
|
-
if (!/^[a-f0-9]{64}$/.test(dataset.remote.fingerprint)) {
|
|
55
|
-
throw new Error("Raindrop remote dataset fingerprint must be a SHA-256 digest");
|
|
56
|
-
}
|
|
57
|
-
}
|
|
58
|
-
const defined = {
|
|
59
|
-
...dataset,
|
|
60
|
-
id,
|
|
61
|
-
name,
|
|
62
|
-
version,
|
|
63
|
-
rows,
|
|
64
|
-
[EVAL_DATASET]: true
|
|
65
|
-
};
|
|
66
|
-
Object.defineProperty(defined, EVAL_DATASET, { enumerable: false });
|
|
67
|
-
return defined;
|
|
68
|
-
}
|
|
69
|
-
function isEvalDataset(value) {
|
|
70
|
-
return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
|
|
71
|
-
}
|
|
72
3
|
var BooleanThresholdSchema = z.object({ equals: z.boolean() }).strict();
|
|
73
4
|
var FiniteThresholdSchema = z.number().finite();
|
|
74
|
-
var EvalOutputSchema = z.enum(["boolean", "score", "number"]);
|
|
75
5
|
var NumericThresholdSchema = z.union([
|
|
76
6
|
z.object({ gte: FiniteThresholdSchema, lte: FiniteThresholdSchema.optional() }).strict(),
|
|
77
7
|
z.object({ lte: FiniteThresholdSchema }).strict()
|
|
@@ -79,194 +9,32 @@ var NumericThresholdSchema = z.union([
|
|
|
79
9
|
(threshold) => !("gte" in threshold) || threshold.lte === void 0 || threshold.gte <= threshold.lte,
|
|
80
10
|
{ message: "Lower threshold cannot exceed upper threshold" }
|
|
81
11
|
);
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
if (!intent || intent.length > 1e4) {
|
|
98
|
-
throw new Error(
|
|
99
|
-
`Raindrop evaluator program ${slug} intent must be between 1 and 10000 characters`
|
|
100
|
-
);
|
|
101
|
-
}
|
|
102
|
-
if (program.description != null && program.description.trim().length > 2e3) {
|
|
103
|
-
throw new Error(`Raindrop evaluator program ${slug} description exceeds 2000 characters`);
|
|
104
|
-
}
|
|
105
|
-
const rules = (_b = (_a = program.rules) == null ? void 0 : _a.map((rule) => rule.trim())) != null ? _b : [];
|
|
106
|
-
if (rules.length > 12 || rules.some((rule) => !rule || rule.length > 2e3)) {
|
|
107
|
-
throw new Error(`Raindrop evaluator program ${slug} rules are invalid`);
|
|
108
|
-
}
|
|
109
|
-
if (program.expected && (!z.string().uuid().safeParse(program.expected.evalId).success || !Number.isInteger(program.expected.programVersion) || program.expected.programVersion <= 0)) {
|
|
110
|
-
throw new Error(`Raindrop evaluator program ${slug} expected identity is invalid`);
|
|
111
|
-
}
|
|
112
|
-
return {
|
|
113
|
-
...program,
|
|
114
|
-
kind: "program",
|
|
115
|
-
scope: z.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
|
|
116
|
-
slug,
|
|
117
|
-
name,
|
|
118
|
-
source,
|
|
119
|
-
intent,
|
|
120
|
-
description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
|
|
121
|
-
rules
|
|
122
|
-
};
|
|
123
|
-
}
|
|
124
|
-
function defineLocalEvaluator(evaluator) {
|
|
125
|
-
var _a;
|
|
126
|
-
return { ...evaluator, scope: z.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
|
|
127
|
-
}
|
|
128
|
-
var EVAL_DEFINITION = /* @__PURE__ */ Symbol.for("raindrop.evalDefinition");
|
|
129
|
-
function defineEvalSuite(definition) {
|
|
130
|
-
const name = definition.name.trim();
|
|
131
|
-
if (name.length === 0) throw new Error("Raindrop eval suite names cannot be empty");
|
|
132
|
-
const dataset = typeof definition.dataset === "string" ? definition.dataset.trim() : defineDataset(definition.dataset);
|
|
133
|
-
if (typeof dataset === "string" && dataset.length === 0) {
|
|
134
|
-
throw new Error("Raindrop eval suite datasets cannot be empty");
|
|
135
|
-
}
|
|
136
|
-
if (definition.datasetVersionId !== void 0) {
|
|
137
|
-
z.string().uuid().parse(definition.datasetVersionId);
|
|
138
|
-
if (typeof dataset !== "string")
|
|
139
|
-
throw new Error("A dataset version pin requires a published dataset slug");
|
|
140
|
-
}
|
|
141
|
-
if (definition.agent !== void 0 && definition.run !== void 0) {
|
|
142
|
-
throw new Error("An eval suite accepts either agent or run, not both");
|
|
143
|
-
}
|
|
144
|
-
if (definition.agent === void 0 && typeof definition.run !== "function") {
|
|
145
|
-
throw new Error("Raindrop eval suite run must be a function");
|
|
146
|
-
}
|
|
147
|
-
if (!Array.isArray(definition.evaluators) || definition.evaluators.length === 0) {
|
|
148
|
-
throw new Error("Raindrop eval suites require at least one evaluator");
|
|
149
|
-
}
|
|
150
|
-
validateOptionalNumber("concurrency", definition.concurrency, true, false);
|
|
151
|
-
validateOptionalNumber("traceWaitMs", definition.traceWaitMs, false, true);
|
|
152
|
-
validateOptionalNumber("evalWaitMs", definition.evalWaitMs, false, true);
|
|
153
|
-
validateOptionalNumber("evalPollIntervalMs", definition.evalPollIntervalMs, false, false);
|
|
154
|
-
const evaluators = definition.evaluators.map((evaluator) => validateEvaluator(evaluator));
|
|
155
|
-
const evaluatorNames = evaluators.map((entry) => evaluatorName(entry));
|
|
156
|
-
if (new Set(evaluatorNames).size !== evaluatorNames.length) {
|
|
157
|
-
throw new Error("Raindrop eval suite evaluators must be unique");
|
|
158
|
-
}
|
|
159
|
-
const branded = {
|
|
160
|
-
...definition,
|
|
161
|
-
[EVAL_DEFINITION]: true,
|
|
162
|
-
name,
|
|
163
|
-
dataset,
|
|
164
|
-
evaluators
|
|
165
|
-
};
|
|
166
|
-
Object.defineProperty(branded, EVAL_DEFINITION, { enumerable: false });
|
|
167
|
-
return branded;
|
|
168
|
-
}
|
|
169
|
-
function isEvalSuiteDefinition(value) {
|
|
170
|
-
return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
|
|
171
|
-
}
|
|
172
|
-
function isPublishedEvalSuite(definition) {
|
|
173
|
-
return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
|
|
174
|
-
({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
|
|
175
|
-
);
|
|
176
|
-
}
|
|
177
|
-
function evaluatorName(evaluator) {
|
|
178
|
-
return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
|
|
179
|
-
}
|
|
180
|
-
function evaluatorOutput(evaluator) {
|
|
181
|
-
return typeof evaluator.evaluator === "string" ? void 0 : evaluator.evaluator.output;
|
|
182
|
-
}
|
|
183
|
-
function validateEvaluatorOutput(evaluator, rawOutput) {
|
|
184
|
-
const name = evaluatorName(evaluator);
|
|
185
|
-
const output = EvalOutputSchema.parse(rawOutput);
|
|
186
|
-
const localOutput = evaluatorOutput(evaluator);
|
|
187
|
-
if (localOutput !== void 0 && localOutput !== output) {
|
|
188
|
-
throw new Error(`Raindrop local evaluator ${name} expected ${localOutput}, received ${output}`);
|
|
189
|
-
}
|
|
190
|
-
switch (output) {
|
|
191
|
-
case "boolean":
|
|
192
|
-
parseBooleanThreshold(evaluator.threshold);
|
|
193
|
-
break;
|
|
194
|
-
case "score":
|
|
195
|
-
parseNumericThreshold(name, "score", evaluator.threshold);
|
|
196
|
-
break;
|
|
197
|
-
case "number":
|
|
198
|
-
parseNumericThreshold(name, "number", evaluator.threshold);
|
|
199
|
-
break;
|
|
200
|
-
}
|
|
201
|
-
}
|
|
202
|
-
function validateEvaluator(evaluator) {
|
|
203
|
-
if (typeof evaluator !== "object" || evaluator === null || !("evaluator" in evaluator)) {
|
|
204
|
-
throw new Error("Raindrop eval evaluator entries must name an evaluator");
|
|
205
|
-
}
|
|
206
|
-
if ("output" in evaluator) {
|
|
207
|
-
throw new Error(
|
|
208
|
-
"Raindrop eval evaluator entries do not accept output; the evaluator definition owns it"
|
|
209
|
-
);
|
|
210
|
-
}
|
|
211
|
-
if (typeof evaluator.evaluator !== "string" && (typeof evaluator.evaluator !== "object" || evaluator.evaluator === null || typeof evaluator.evaluator.slug !== "string" || typeof evaluator.evaluator.name !== "string")) {
|
|
212
|
-
throw new Error("Raindrop local eval evaluators require a slug, name, output, and judge");
|
|
213
|
-
}
|
|
214
|
-
const name = evaluatorName(evaluator);
|
|
215
|
-
if (name.length === 0) throw new Error("Raindrop eval evaluator names cannot be empty");
|
|
216
|
-
if (typeof evaluator.evaluator === "string") {
|
|
217
|
-
const threshold = evaluator.threshold;
|
|
218
|
-
if (threshold !== void 0) {
|
|
219
|
-
const boolean = BooleanThresholdSchema.safeParse(threshold);
|
|
220
|
-
const numeric = NumericThresholdSchema.safeParse(threshold);
|
|
221
|
-
if (!boolean.success && !numeric.success) {
|
|
222
|
-
throw new Error(`Raindrop hosted evaluator ${name} has an invalid threshold`);
|
|
223
|
-
}
|
|
224
|
-
}
|
|
225
|
-
return { ...evaluator, evaluator: name };
|
|
226
|
-
}
|
|
227
|
-
if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "portable") {
|
|
228
|
-
validateEvaluatorOutput(evaluator, evaluator.evaluator.output);
|
|
229
|
-
return evaluator;
|
|
230
|
-
}
|
|
231
|
-
if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "program") {
|
|
232
|
-
const program = defineEvaluatorProgram(evaluator.evaluator);
|
|
233
|
-
validateEvaluatorOutput({ ...evaluator, evaluator: program }, program.output);
|
|
234
|
-
return { ...evaluator, evaluator: program };
|
|
235
|
-
}
|
|
236
|
-
if (typeof evaluator.evaluator.judge !== "function") {
|
|
237
|
-
throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
|
|
238
|
-
}
|
|
239
|
-
if (evaluator.evaluator.scope !== "row") {
|
|
240
|
-
throw new Error(`Raindrop local evaluator ${name} scope must be row`);
|
|
241
|
-
}
|
|
242
|
-
const output = EvalOutputSchema.parse(evaluator.evaluator.output);
|
|
243
|
-
validateEvaluatorOutput(evaluator, output);
|
|
244
|
-
return evaluator;
|
|
245
|
-
}
|
|
246
|
-
function validateOptionalNumber(name, value, integer, allowZero) {
|
|
247
|
-
if (value === void 0) return;
|
|
248
|
-
const valid = Number.isFinite(value) && (!integer || Number.isInteger(value)) && (allowZero ? value >= 0 : value > 0);
|
|
249
|
-
if (!valid) {
|
|
250
|
-
const kind = integer ? "a positive integer" : allowZero ? "non-negative" : "positive";
|
|
251
|
-
throw new Error(`Raindrop eval ${name} must be ${kind}`);
|
|
252
|
-
}
|
|
253
|
-
}
|
|
254
|
-
function parseBooleanThreshold(threshold) {
|
|
255
|
-
return threshold === void 0 ? void 0 : BooleanThresholdSchema.parse(threshold);
|
|
256
|
-
}
|
|
257
|
-
function parseNumericThreshold(name, output, threshold) {
|
|
258
|
-
const parsed = threshold === void 0 ? void 0 : NumericThresholdSchema.parse(threshold);
|
|
259
|
-
if (output === "score" && parsed) {
|
|
260
|
-
const lower = "gte" in parsed ? parsed.gte : void 0;
|
|
261
|
-
const inRange = (lower === void 0 || lower >= 1 && lower <= 5) && (parsed.lte === void 0 || parsed.lte >= 1 && parsed.lte <= 5);
|
|
262
|
-
if (!inRange)
|
|
263
|
-
throw new Error(`Raindrop score evaluator ${name} thresholds must be from 1 to 5`);
|
|
264
|
-
}
|
|
265
|
-
return parsed;
|
|
266
|
-
}
|
|
12
|
+
var EvalSuiteManifestSchema = z.object({
|
|
13
|
+
name: z.string().trim().min(1),
|
|
14
|
+
dataset: z.string().trim().min(1),
|
|
15
|
+
datasetVersionId: z.string().uuid().optional(),
|
|
16
|
+
evaluators: z.array(
|
|
17
|
+
z.object({
|
|
18
|
+
evaluator: z.string().trim().min(1),
|
|
19
|
+
threshold: z.union([BooleanThresholdSchema, NumericThresholdSchema]).optional()
|
|
20
|
+
}).strict()
|
|
21
|
+
).min(1),
|
|
22
|
+
concurrency: z.number().int().positive().optional(),
|
|
23
|
+
traceWaitMs: z.number().nonnegative().optional(),
|
|
24
|
+
evalWaitMs: z.number().nonnegative().optional(),
|
|
25
|
+
evalPollIntervalMs: z.number().positive().optional()
|
|
26
|
+
}).strict();
|
|
267
27
|
|
|
268
28
|
// src/evals/generated/trace.ts
|
|
269
29
|
import z2 from "zod";
|
|
30
|
+
var JsonValueSchema = z2.lazy(() => z2.union([
|
|
31
|
+
z2.string(),
|
|
32
|
+
z2.number(),
|
|
33
|
+
z2.boolean(),
|
|
34
|
+
z2.null(),
|
|
35
|
+
z2.array(JsonValueSchema),
|
|
36
|
+
z2.record(JsonValueSchema)
|
|
37
|
+
]));
|
|
270
38
|
var stringToJSONSchema = z2.string().transform((str, ctx) => {
|
|
271
39
|
try {
|
|
272
40
|
return JSON.parse(str);
|
|
@@ -635,14 +403,20 @@ var ReplayTraceSnapshotSpanSchema = TracesListPipe.output.extend({
|
|
|
635
403
|
duration_ns: NanosecondsStringSchema,
|
|
636
404
|
_inserted_at: z2.string().min(1)
|
|
637
405
|
}).strict();
|
|
406
|
+
var EvalDatasetCaseIdSchema = z2.string().min(1).max(512).refine((value) => value === value.trim(), "Case id cannot have surrounding whitespace");
|
|
407
|
+
var EvalCaseExpectationSchema = z2.object({
|
|
408
|
+
description: z2.string().trim().min(1).max(1e4),
|
|
409
|
+
basis: z2.enum(["human", "policy", "code", "inferred"]).optional(),
|
|
410
|
+
reference: z2.string().trim().min(1).max(2e3).optional()
|
|
411
|
+
}).strict();
|
|
638
412
|
var TraceSchema = z2.object({
|
|
639
413
|
origin: z2.enum(["production", "dataset", "simulation", "replay"]),
|
|
640
414
|
event: AIAnalyticsEventSchema,
|
|
641
|
-
entries: z2.array(RichInteractionEntrySchema)
|
|
415
|
+
entries: z2.array(RichInteractionEntrySchema),
|
|
416
|
+
caseContext: z2.object({ caseId: EvalDatasetCaseIdSchema, expectation: EvalCaseExpectationSchema.optional() }).optional()
|
|
642
417
|
});
|
|
643
418
|
var CapturedTraceSpanSchema = ReplayTraceSnapshotSpanSchema;
|
|
644
419
|
var EVAL_DATASET_MAX_SPANS_PER_TRACE = 500;
|
|
645
|
-
var EvalDatasetCaseIdSchema = z2.string().min(1).max(512).refine((value) => value === value.trim(), "Case id cannot have surrounding whitespace");
|
|
646
420
|
var CapturedTraceSnapshotSchema = z2.object({
|
|
647
421
|
event: AIAnalyticsEventSchema,
|
|
648
422
|
spans: z2.array(CapturedTraceSpanSchema).min(1).max(EVAL_DATASET_MAX_SPANS_PER_TRACE)
|
|
@@ -712,15 +486,6 @@ var CapturedTraceSnapshotSchema = z2.object({
|
|
|
712
486
|
path: ["spans"]
|
|
713
487
|
});
|
|
714
488
|
}
|
|
715
|
-
for (const [index, span] of spans.entries()) {
|
|
716
|
-
if (span.parent_span_id && !spanIds.has(span.parent_span_id)) {
|
|
717
|
-
context.addIssue({
|
|
718
|
-
code: z2.ZodIssueCode.custom,
|
|
719
|
-
message: `Missing parent span ${span.parent_span_id}`,
|
|
720
|
-
path: ["spans", index, "parent_span_id"]
|
|
721
|
-
});
|
|
722
|
-
}
|
|
723
|
-
}
|
|
724
489
|
const parentBySpanId = new Map(spans.map((span) => [span.span_id, span.parent_span_id]));
|
|
725
490
|
for (const [index, span] of spans.entries()) {
|
|
726
491
|
const path = /* @__PURE__ */ new Set();
|
|
@@ -745,8 +510,9 @@ var EvalCaseTraceSchema = z2.object({
|
|
|
745
510
|
});
|
|
746
511
|
var EvalCaseTracePairSchema = z2.object({
|
|
747
512
|
caseId: EvalDatasetCaseIdSchema,
|
|
513
|
+
output: JsonValueSchema.optional(),
|
|
748
514
|
candidate: EvalCaseTraceSchema.extend({
|
|
749
|
-
trace: TraceSchema.extend({ origin: z2.
|
|
515
|
+
trace: TraceSchema.extend({ origin: z2.enum(["replay", "dataset"]) })
|
|
750
516
|
}),
|
|
751
517
|
reference: EvalCaseTraceSchema.extend({
|
|
752
518
|
trace: TraceSchema.extend({ origin: z2.literal("dataset") })
|
|
@@ -1012,37 +778,6 @@ function traceStepsToEntries(steps) {
|
|
|
1012
778
|
function projectWorkshopTrace(input) {
|
|
1013
779
|
return createTrace(input.origin, input.event, traceStepsToEntries(buildTraceSteps([...input.spans], { includePayloads: true })).map(boundEvalEntry));
|
|
1014
780
|
}
|
|
1015
|
-
function projectCapturedEvent(input) {
|
|
1016
|
-
var _a, _b, _c, _d, _e, _f;
|
|
1017
|
-
const ordered = [...input.spans].sort((left, right) => compareNanoseconds(left.start_unix_ns, right.start_unix_ns));
|
|
1018
|
-
const eventSpan = ordered.find((span) => span.span_id === `evt_${input.eventId}`);
|
|
1019
|
-
const llmSpans = ordered.filter((span) => span.span_type === "LLM_GENERATION" || span.span_type === "LLM_GENERATION_STREAM");
|
|
1020
|
-
const inputSpan = ((_a = eventSpan == null ? void 0 : eventSpan.input_payload) == null ? void 0 : _a.trim()) ? eventSpan : llmSpans.find((span) => {
|
|
1021
|
-
var _a2;
|
|
1022
|
-
return Boolean((_a2 = span.input_payload) == null ? void 0 : _a2.trim());
|
|
1023
|
-
});
|
|
1024
|
-
const reverseLlmSpans = [...llmSpans].reverse();
|
|
1025
|
-
const outputSpan = (_c = ((_b = eventSpan == null ? void 0 : eventSpan.output_payload) == null ? void 0 : _b.trim()) ? eventSpan : void 0) != null ? _c : reverseLlmSpans.find((span) => {
|
|
1026
|
-
var _a2;
|
|
1027
|
-
return Boolean((_a2 = span.output_payload) == null ? void 0 : _a2.trim()) && extractFinishReason(span) !== "tool-calls";
|
|
1028
|
-
});
|
|
1029
|
-
const modelSpan = (eventSpan == null ? void 0 : eventSpan.model) ? eventSpan : reverseLlmSpans.find((span) => Boolean(span.model));
|
|
1030
|
-
return AIAnalyticsEventSchema.parse({
|
|
1031
|
-
id: input.eventId,
|
|
1032
|
-
_raindropDatabaseId: input.internalEventId,
|
|
1033
|
-
name: input.name,
|
|
1034
|
-
timestamp: input.timestamp,
|
|
1035
|
-
receivedAt: input.receivedAt,
|
|
1036
|
-
userId: input.userId,
|
|
1037
|
-
aiData: {
|
|
1038
|
-
input: (_d = inputSpan == null ? void 0 : inputSpan.input_payload) != null ? _d : null,
|
|
1039
|
-
output: (_e = outputSpan == null ? void 0 : outputSpan.output_payload) != null ? _e : null,
|
|
1040
|
-
model: (_f = modelSpan == null ? void 0 : modelSpan.model) != null ? _f : null,
|
|
1041
|
-
convoId: input.convoId
|
|
1042
|
-
},
|
|
1043
|
-
properties: input.properties
|
|
1044
|
-
});
|
|
1045
|
-
}
|
|
1046
781
|
function traceOutput(trace) {
|
|
1047
782
|
var _a;
|
|
1048
783
|
return (_a = trace.event.aiData.output) != null ? _a : null;
|
|
@@ -1050,64 +785,329 @@ function traceOutput(trace) {
|
|
|
1050
785
|
function traceToolCalls(trace, name) {
|
|
1051
786
|
return trace.entries.filter((entry) => entry.type === "tool_call" && (name === void 0 || entry.name === name));
|
|
1052
787
|
}
|
|
1053
|
-
var TRACE_PROJECTION_SHA256 = "
|
|
788
|
+
var TRACE_PROJECTION_SHA256 = "22c666c6ace47d74d3f7dc477e68b27547257b319ba4014a4a32fbf977baac72";
|
|
1054
789
|
|
|
1055
|
-
// src/evals/
|
|
790
|
+
// src/evals/definition.ts
|
|
791
|
+
import { createHash } from "crypto";
|
|
1056
792
|
import { z as z3 } from "zod";
|
|
1057
|
-
var
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
793
|
+
var EvalScoreSchema = z3.union([
|
|
794
|
+
z3.literal(1),
|
|
795
|
+
z3.literal(2),
|
|
796
|
+
z3.literal(3),
|
|
797
|
+
z3.literal(4),
|
|
798
|
+
z3.literal(5)
|
|
799
|
+
]);
|
|
800
|
+
var EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
|
|
801
|
+
var DatasetRowInputSchema = z3.object({
|
|
802
|
+
id: z3.string().trim().min(1),
|
|
803
|
+
name: z3.string().optional(),
|
|
804
|
+
input: z3.string().nullable(),
|
|
805
|
+
output: JsonValueSchema.default(null),
|
|
806
|
+
properties: z3.record(z3.string()).default({})
|
|
807
|
+
});
|
|
808
|
+
function defineDataset(dataset) {
|
|
809
|
+
var _a, _b;
|
|
810
|
+
const id = dataset.id.trim();
|
|
811
|
+
const name = dataset.name.trim();
|
|
812
|
+
const rows = dataset.rows.map((row) => {
|
|
813
|
+
var _a2;
|
|
814
|
+
const parsed = DatasetRowInputSchema.parse(row);
|
|
815
|
+
return {
|
|
816
|
+
id: parsed.id,
|
|
817
|
+
name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
|
|
818
|
+
input: parsed.input,
|
|
819
|
+
output: parsed.output,
|
|
820
|
+
properties: parsed.properties
|
|
821
|
+
};
|
|
822
|
+
});
|
|
823
|
+
const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : createHash("sha256").update(
|
|
824
|
+
JSON.stringify(
|
|
825
|
+
rows.map((row) => ({
|
|
826
|
+
...row,
|
|
827
|
+
properties: Object.fromEntries(
|
|
828
|
+
Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
|
|
829
|
+
)
|
|
830
|
+
}))
|
|
831
|
+
)
|
|
832
|
+
).digest("hex");
|
|
833
|
+
if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
|
|
834
|
+
if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
|
|
835
|
+
if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
|
|
836
|
+
if (dataset.rows.length === 0) throw new Error("Raindrop eval datasets require at least one row");
|
|
837
|
+
const ids = rows.map((entry) => entry.id);
|
|
838
|
+
if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
|
|
839
|
+
if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
|
|
840
|
+
if (dataset.remote) {
|
|
841
|
+
if (!dataset.remote.reference.trim())
|
|
842
|
+
throw new Error("Raindrop remote dataset reference cannot be empty");
|
|
843
|
+
if (!/^[a-f0-9]{64}$/.test(dataset.remote.fingerprint)) {
|
|
844
|
+
throw new Error("Raindrop remote dataset fingerprint must be a SHA-256 digest");
|
|
845
|
+
}
|
|
846
|
+
}
|
|
847
|
+
const defined = {
|
|
848
|
+
...dataset,
|
|
849
|
+
id,
|
|
850
|
+
name,
|
|
851
|
+
version,
|
|
852
|
+
rows,
|
|
853
|
+
[EVAL_DATASET]: true
|
|
854
|
+
};
|
|
855
|
+
Object.defineProperty(defined, EVAL_DATASET, { enumerable: false });
|
|
856
|
+
return defined;
|
|
857
|
+
}
|
|
858
|
+
function isEvalDataset(value) {
|
|
859
|
+
return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
|
|
860
|
+
}
|
|
861
|
+
var EvalOutputSchema = z3.enum(["boolean", "score", "number"]);
|
|
862
|
+
function defineEvaluatorProgram(program) {
|
|
863
|
+
var _a, _b, _c, _d;
|
|
864
|
+
const slug = program.slug.trim();
|
|
865
|
+
const name = program.name.trim();
|
|
866
|
+
const source = program.source.trim();
|
|
867
|
+
const intent = program.intent.trim();
|
|
868
|
+
if (!/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(slug) || slug.length > 64) {
|
|
869
|
+
throw new Error(
|
|
870
|
+
"Raindrop evaluator program slugs must use lowercase letters, numbers, and single hyphens"
|
|
871
|
+
);
|
|
872
|
+
}
|
|
873
|
+
if (!name || name.length > 120) {
|
|
874
|
+
throw new Error("Raindrop evaluator program names must be between 1 and 120 characters");
|
|
875
|
+
}
|
|
876
|
+
if (!source) throw new Error(`Raindrop evaluator program ${slug} source cannot be empty`);
|
|
877
|
+
if (!intent || intent.length > 1e4) {
|
|
878
|
+
throw new Error(
|
|
879
|
+
`Raindrop evaluator program ${slug} intent must be between 1 and 10000 characters`
|
|
880
|
+
);
|
|
881
|
+
}
|
|
882
|
+
if (program.description != null && program.description.trim().length > 2e3) {
|
|
883
|
+
throw new Error(`Raindrop evaluator program ${slug} description exceeds 2000 characters`);
|
|
884
|
+
}
|
|
885
|
+
const rules = (_b = (_a = program.rules) == null ? void 0 : _a.map((rule) => rule.trim())) != null ? _b : [];
|
|
886
|
+
if (rules.length > 12 || rules.some((rule) => !rule || rule.length > 2e3)) {
|
|
887
|
+
throw new Error(`Raindrop evaluator program ${slug} rules are invalid`);
|
|
888
|
+
}
|
|
889
|
+
if (program.expected && (!z3.string().uuid().safeParse(program.expected.evalId).success || !Number.isInteger(program.expected.programVersion) || program.expected.programVersion <= 0)) {
|
|
890
|
+
throw new Error(`Raindrop evaluator program ${slug} expected identity is invalid`);
|
|
891
|
+
}
|
|
892
|
+
return {
|
|
893
|
+
...program,
|
|
894
|
+
kind: "program",
|
|
895
|
+
scope: z3.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
|
|
896
|
+
slug,
|
|
897
|
+
name,
|
|
898
|
+
source,
|
|
899
|
+
intent,
|
|
900
|
+
description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
|
|
901
|
+
rules
|
|
902
|
+
};
|
|
903
|
+
}
|
|
904
|
+
function defineLocalEvaluator(evaluator) {
|
|
905
|
+
var _a;
|
|
906
|
+
return { ...evaluator, scope: z3.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
|
|
907
|
+
}
|
|
908
|
+
var EVAL_DEFINITION = /* @__PURE__ */ Symbol.for("raindrop.evalDefinition");
|
|
909
|
+
function defineEvalSuite(definition, binding) {
|
|
910
|
+
if (binding) return createEvalSuite({ ...EvalSuiteManifestSchema.parse(definition), ...binding });
|
|
911
|
+
if (!("run" in definition) && !("agent" in definition)) {
|
|
912
|
+
throw new Error("Bind the eval manifest to your local run function or agent");
|
|
913
|
+
}
|
|
914
|
+
return createEvalSuite(definition);
|
|
915
|
+
}
|
|
916
|
+
function createEvalSuite(definition) {
|
|
917
|
+
const name = definition.name.trim();
|
|
918
|
+
if (name.length === 0) throw new Error("Raindrop eval suite names cannot be empty");
|
|
919
|
+
const dataset = typeof definition.dataset === "string" ? definition.dataset.trim() : defineDataset(definition.dataset);
|
|
920
|
+
if (typeof dataset === "string" && dataset.length === 0) {
|
|
921
|
+
throw new Error("Raindrop eval suite datasets cannot be empty");
|
|
922
|
+
}
|
|
923
|
+
if (definition.datasetVersionId !== void 0) {
|
|
924
|
+
z3.string().uuid().parse(definition.datasetVersionId);
|
|
925
|
+
if (typeof dataset !== "string")
|
|
926
|
+
throw new Error("A dataset version pin requires a published dataset slug");
|
|
927
|
+
}
|
|
928
|
+
if (definition.agent !== void 0 && definition.run !== void 0) {
|
|
929
|
+
throw new Error("An eval suite accepts either agent or run, not both");
|
|
930
|
+
}
|
|
931
|
+
if (definition.agent === void 0 && typeof definition.run !== "function") {
|
|
932
|
+
throw new Error("Raindrop eval suite run must be a function");
|
|
933
|
+
}
|
|
934
|
+
if (!Array.isArray(definition.evaluators) || definition.evaluators.length === 0) {
|
|
935
|
+
throw new Error("Raindrop eval suites require at least one evaluator");
|
|
936
|
+
}
|
|
937
|
+
validateOptionalNumber("concurrency", definition.concurrency, true, false);
|
|
938
|
+
validateOptionalNumber("traceWaitMs", definition.traceWaitMs, false, true);
|
|
939
|
+
validateOptionalNumber("evalWaitMs", definition.evalWaitMs, false, true);
|
|
940
|
+
validateOptionalNumber("evalPollIntervalMs", definition.evalPollIntervalMs, false, false);
|
|
941
|
+
const evaluators = definition.evaluators.map((evaluator) => validateEvaluator(evaluator));
|
|
942
|
+
const evaluatorNames = evaluators.map((entry) => evaluatorName(entry));
|
|
943
|
+
if (new Set(evaluatorNames).size !== evaluatorNames.length) {
|
|
944
|
+
throw new Error("Raindrop eval suite evaluators must be unique");
|
|
945
|
+
}
|
|
946
|
+
const branded = {
|
|
947
|
+
...definition,
|
|
948
|
+
[EVAL_DEFINITION]: true,
|
|
949
|
+
name,
|
|
950
|
+
dataset,
|
|
951
|
+
evaluators
|
|
952
|
+
};
|
|
953
|
+
Object.defineProperty(branded, EVAL_DEFINITION, { enumerable: false });
|
|
954
|
+
return branded;
|
|
955
|
+
}
|
|
956
|
+
function isEvalSuiteDefinition(value) {
|
|
957
|
+
return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
|
|
958
|
+
}
|
|
959
|
+
function isPublishedEvalSuite(definition) {
|
|
960
|
+
return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
|
|
961
|
+
({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
|
|
962
|
+
);
|
|
963
|
+
}
|
|
964
|
+
function evaluatorName(evaluator) {
|
|
965
|
+
return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
|
|
966
|
+
}
|
|
967
|
+
function evaluatorOutput(evaluator) {
|
|
968
|
+
return typeof evaluator.evaluator === "string" ? void 0 : evaluator.evaluator.output;
|
|
969
|
+
}
|
|
970
|
+
function validateEvaluatorOutput(evaluator, rawOutput) {
|
|
971
|
+
const name = evaluatorName(evaluator);
|
|
972
|
+
const output = EvalOutputSchema.parse(rawOutput);
|
|
973
|
+
const localOutput = evaluatorOutput(evaluator);
|
|
974
|
+
if (localOutput !== void 0 && localOutput !== output) {
|
|
975
|
+
throw new Error(`Raindrop local evaluator ${name} expected ${localOutput}, received ${output}`);
|
|
976
|
+
}
|
|
977
|
+
switch (output) {
|
|
978
|
+
case "boolean":
|
|
979
|
+
parseBooleanThreshold(evaluator.threshold);
|
|
980
|
+
break;
|
|
981
|
+
case "score":
|
|
982
|
+
parseNumericThreshold(name, "score", evaluator.threshold);
|
|
983
|
+
break;
|
|
984
|
+
case "number":
|
|
985
|
+
parseNumericThreshold(name, "number", evaluator.threshold);
|
|
986
|
+
break;
|
|
987
|
+
}
|
|
988
|
+
}
|
|
989
|
+
function validateEvaluator(evaluator) {
|
|
990
|
+
if (typeof evaluator !== "object" || evaluator === null || !("evaluator" in evaluator)) {
|
|
991
|
+
throw new Error("Raindrop eval evaluator entries must name an evaluator");
|
|
992
|
+
}
|
|
993
|
+
if ("output" in evaluator) {
|
|
994
|
+
throw new Error(
|
|
995
|
+
"Raindrop eval evaluator entries do not accept output; the evaluator definition owns it"
|
|
996
|
+
);
|
|
997
|
+
}
|
|
998
|
+
if (typeof evaluator.evaluator !== "string" && (typeof evaluator.evaluator !== "object" || evaluator.evaluator === null || typeof evaluator.evaluator.slug !== "string" || typeof evaluator.evaluator.name !== "string")) {
|
|
999
|
+
throw new Error("Raindrop local eval evaluators require a slug, name, output, and judge");
|
|
1000
|
+
}
|
|
1001
|
+
const name = evaluatorName(evaluator);
|
|
1002
|
+
if (name.length === 0) throw new Error("Raindrop eval evaluator names cannot be empty");
|
|
1003
|
+
if (typeof evaluator.evaluator === "string") {
|
|
1004
|
+
const threshold = evaluator.threshold;
|
|
1005
|
+
if (threshold !== void 0) {
|
|
1006
|
+
const boolean = BooleanThresholdSchema.safeParse(threshold);
|
|
1007
|
+
const numeric = NumericThresholdSchema.safeParse(threshold);
|
|
1008
|
+
if (!boolean.success && !numeric.success) {
|
|
1009
|
+
throw new Error(`Raindrop hosted evaluator ${name} has an invalid threshold`);
|
|
1010
|
+
}
|
|
1011
|
+
}
|
|
1012
|
+
return { ...evaluator, evaluator: name };
|
|
1013
|
+
}
|
|
1014
|
+
if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "portable") {
|
|
1015
|
+
validateEvaluatorOutput(evaluator, evaluator.evaluator.output);
|
|
1016
|
+
return evaluator;
|
|
1017
|
+
}
|
|
1018
|
+
if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "program") {
|
|
1019
|
+
const program = defineEvaluatorProgram(evaluator.evaluator);
|
|
1020
|
+
validateEvaluatorOutput({ ...evaluator, evaluator: program }, program.output);
|
|
1021
|
+
return { ...evaluator, evaluator: program };
|
|
1022
|
+
}
|
|
1023
|
+
if (typeof evaluator.evaluator.judge !== "function") {
|
|
1024
|
+
throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
|
|
1025
|
+
}
|
|
1026
|
+
if (evaluator.evaluator.scope !== "row") {
|
|
1027
|
+
throw new Error(`Raindrop local evaluator ${name} scope must be row`);
|
|
1028
|
+
}
|
|
1029
|
+
const output = EvalOutputSchema.parse(evaluator.evaluator.output);
|
|
1030
|
+
validateEvaluatorOutput(evaluator, output);
|
|
1031
|
+
return evaluator;
|
|
1032
|
+
}
|
|
1033
|
+
function validateOptionalNumber(name, value, integer, allowZero) {
|
|
1034
|
+
if (value === void 0) return;
|
|
1035
|
+
const valid = Number.isFinite(value) && (!integer || Number.isInteger(value)) && (allowZero ? value >= 0 : value > 0);
|
|
1036
|
+
if (!valid) {
|
|
1037
|
+
const kind = integer ? "a positive integer" : allowZero ? "non-negative" : "positive";
|
|
1038
|
+
throw new Error(`Raindrop eval ${name} must be ${kind}`);
|
|
1039
|
+
}
|
|
1040
|
+
}
|
|
1041
|
+
function parseBooleanThreshold(threshold) {
|
|
1042
|
+
return threshold === void 0 ? void 0 : BooleanThresholdSchema.parse(threshold);
|
|
1043
|
+
}
|
|
1044
|
+
function parseNumericThreshold(name, output, threshold) {
|
|
1045
|
+
const parsed = threshold === void 0 ? void 0 : NumericThresholdSchema.parse(threshold);
|
|
1046
|
+
if (output === "score" && parsed) {
|
|
1047
|
+
const lower = "gte" in parsed ? parsed.gte : void 0;
|
|
1048
|
+
const inRange = (lower === void 0 || lower >= 1 && lower <= 5) && (parsed.lte === void 0 || parsed.lte >= 1 && parsed.lte <= 5);
|
|
1049
|
+
if (!inRange)
|
|
1050
|
+
throw new Error(`Raindrop score evaluator ${name} thresholds must be from 1 to 5`);
|
|
1051
|
+
}
|
|
1052
|
+
return parsed;
|
|
1053
|
+
}
|
|
1054
|
+
|
|
1055
|
+
// src/evals/portable.ts
|
|
1056
|
+
import { z as z4 } from "zod";
|
|
1057
|
+
var PortableEvalArtifactSchema = z4.object({
|
|
1058
|
+
format: z4.literal("raindrop.eval-program"),
|
|
1059
|
+
formatVersion: z4.literal(1),
|
|
1060
|
+
runtimeAbi: z4.literal("raindrop-eval-v2"),
|
|
1061
|
+
evaluator: z4.object({
|
|
1062
|
+
id: z4.string().uuid(),
|
|
1063
|
+
slug: z4.string().min(1),
|
|
1064
|
+
programVersion: z4.number().int().positive(),
|
|
1065
|
+
executionMode: z4.enum(["deterministic", "judge"]),
|
|
1066
|
+
outputType: z4.enum(["boolean", "score", "number"]),
|
|
1067
|
+
scope: z4.literal("batch")
|
|
1068
1068
|
}).strict(),
|
|
1069
|
-
traceProjectionVersion:
|
|
1070
|
-
traceProjectionSha256:
|
|
1071
|
-
source:
|
|
1072
|
-
sha256:
|
|
1069
|
+
traceProjectionVersion: z4.literal(1),
|
|
1070
|
+
traceProjectionSha256: z4.string().regex(/^[a-f0-9]{64}$/),
|
|
1071
|
+
source: z4.string().min(1),
|
|
1072
|
+
sha256: z4.string().regex(/^[a-f0-9]{64}$/)
|
|
1073
1073
|
}).strict();
|
|
1074
|
-
var PortableEvalResultSchema =
|
|
1075
|
-
outcomes:
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
state:
|
|
1079
|
-
traceId:
|
|
1080
|
-
pass:
|
|
1081
|
-
note:
|
|
1074
|
+
var PortableEvalResultSchema = z4.object({
|
|
1075
|
+
outcomes: z4.array(
|
|
1076
|
+
z4.union([
|
|
1077
|
+
z4.object({
|
|
1078
|
+
state: z4.literal("graded"),
|
|
1079
|
+
traceId: z4.string(),
|
|
1080
|
+
pass: z4.boolean(),
|
|
1081
|
+
note: z4.string().optional()
|
|
1082
1082
|
}),
|
|
1083
|
-
|
|
1084
|
-
state:
|
|
1085
|
-
traceId:
|
|
1083
|
+
z4.object({
|
|
1084
|
+
state: z4.literal("graded"),
|
|
1085
|
+
traceId: z4.string(),
|
|
1086
1086
|
score: EvalScoreSchema,
|
|
1087
|
-
note:
|
|
1087
|
+
note: z4.string().optional()
|
|
1088
1088
|
}),
|
|
1089
|
-
|
|
1090
|
-
state:
|
|
1091
|
-
traceId:
|
|
1092
|
-
value:
|
|
1093
|
-
note:
|
|
1089
|
+
z4.object({
|
|
1090
|
+
state: z4.literal("graded"),
|
|
1091
|
+
traceId: z4.string(),
|
|
1092
|
+
value: z4.number().finite(),
|
|
1093
|
+
note: z4.string().optional()
|
|
1094
1094
|
}),
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
state:
|
|
1098
|
-
traceId:
|
|
1099
|
-
reason:
|
|
1095
|
+
z4.object({ state: z4.literal("errored"), traceId: z4.string(), message: z4.string() }),
|
|
1096
|
+
z4.object({
|
|
1097
|
+
state: z4.literal("ungraded"),
|
|
1098
|
+
traceId: z4.string(),
|
|
1099
|
+
reason: z4.enum(["not sampled", "trace unavailable", "content unreadable"])
|
|
1100
1100
|
})
|
|
1101
1101
|
])
|
|
1102
1102
|
),
|
|
1103
|
-
failures:
|
|
1104
|
-
stats:
|
|
1105
|
-
traceCount:
|
|
1106
|
-
hydratedCount:
|
|
1107
|
-
judgeCalls:
|
|
1108
|
-
durationMs:
|
|
1109
|
-
skippedRows:
|
|
1110
|
-
skippedReason:
|
|
1103
|
+
failures: z4.array(z4.string()),
|
|
1104
|
+
stats: z4.object({
|
|
1105
|
+
traceCount: z4.number().int().nonnegative(),
|
|
1106
|
+
hydratedCount: z4.number().int().nonnegative(),
|
|
1107
|
+
judgeCalls: z4.number().int().nonnegative(),
|
|
1108
|
+
durationMs: z4.number().nonnegative(),
|
|
1109
|
+
skippedRows: z4.number().int().nonnegative(),
|
|
1110
|
+
skippedReason: z4.literal("no trace").nullable()
|
|
1111
1111
|
})
|
|
1112
1112
|
});
|
|
1113
1113
|
function parsePortableEvalResult(value) {
|
|
@@ -1138,7 +1138,7 @@ async function evaluatePortableEvaluator(evaluator, input) {
|
|
|
1138
1138
|
const artifact = await verifyPortableEvalArtifact(evaluator.artifact, {
|
|
1139
1139
|
expectedSha256: evaluator.expectedSha256
|
|
1140
1140
|
});
|
|
1141
|
-
const { runPortableEvaluator } = await import("./portable-runtime-
|
|
1141
|
+
const { runPortableEvaluator } = await import("./portable-runtime-F7W4STGH.mjs");
|
|
1142
1142
|
const cases = ((_a = input.rows) != null ? _a : []).map(({ rowId, ...pair }) => ({ ...pair, caseId: rowId }));
|
|
1143
1143
|
if (cases.length > 0 && (cases.length !== input.traces.length || cases.some(
|
|
1144
1144
|
(entry, index) => {
|
|
@@ -1218,11 +1218,21 @@ function canonicalJson(value) {
|
|
|
1218
1218
|
}
|
|
1219
1219
|
|
|
1220
1220
|
export {
|
|
1221
|
+
BooleanThresholdSchema,
|
|
1222
|
+
NumericThresholdSchema,
|
|
1223
|
+
EvalSuiteManifestSchema,
|
|
1224
|
+
JsonValueSchema,
|
|
1225
|
+
EvalDatasetCaseIdSchema,
|
|
1226
|
+
EvalCaseExpectationSchema,
|
|
1227
|
+
TraceSchema,
|
|
1228
|
+
CapturedTraceSnapshotSchema,
|
|
1229
|
+
EvalCaseTracePairSchema,
|
|
1230
|
+
projectWorkshopTrace,
|
|
1231
|
+
traceOutput,
|
|
1232
|
+
traceToolCalls,
|
|
1221
1233
|
EvalScoreSchema,
|
|
1222
1234
|
defineDataset,
|
|
1223
1235
|
isEvalDataset,
|
|
1224
|
-
BooleanThresholdSchema,
|
|
1225
|
-
NumericThresholdSchema,
|
|
1226
1236
|
defineEvaluatorProgram,
|
|
1227
1237
|
defineLocalEvaluator,
|
|
1228
1238
|
defineEvalSuite,
|
|
@@ -1231,14 +1241,6 @@ export {
|
|
|
1231
1241
|
evaluatorName,
|
|
1232
1242
|
evaluatorOutput,
|
|
1233
1243
|
validateEvaluatorOutput,
|
|
1234
|
-
TraceSchema,
|
|
1235
|
-
EvalDatasetCaseIdSchema,
|
|
1236
|
-
CapturedTraceSnapshotSchema,
|
|
1237
|
-
EvalCaseTracePairSchema,
|
|
1238
|
-
projectWorkshopTrace,
|
|
1239
|
-
projectCapturedEvent,
|
|
1240
|
-
traceOutput,
|
|
1241
|
-
traceToolCalls,
|
|
1242
1244
|
parsePortableEvalResult,
|
|
1243
1245
|
importPortableEvaluator,
|
|
1244
1246
|
evaluatePortableEvaluator,
|