raindrop-ai 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,77 +1,7 @@
1
- // src/evals/definition.ts
2
- import { createHash } from "crypto";
1
+ // src/evals/generated/manifest.ts
3
2
  import { z } from "zod";
4
- var EvalScoreSchema = z.union([
5
- z.literal(1),
6
- z.literal(2),
7
- z.literal(3),
8
- z.literal(4),
9
- z.literal(5)
10
- ]);
11
- var EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
12
- var DatasetRowInputSchema = z.object({
13
- id: z.string().trim().min(1),
14
- name: z.string().optional(),
15
- input: z.string().nullable(),
16
- output: z.string().nullable().default(null),
17
- properties: z.record(z.string()).default({})
18
- });
19
- function defineDataset(dataset) {
20
- var _a, _b;
21
- const id = dataset.id.trim();
22
- const name = dataset.name.trim();
23
- const rows = dataset.rows.map((row) => {
24
- var _a2;
25
- const parsed = DatasetRowInputSchema.parse(row);
26
- return {
27
- id: parsed.id,
28
- name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
29
- input: parsed.input,
30
- output: parsed.output,
31
- properties: parsed.properties
32
- };
33
- });
34
- const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : createHash("sha256").update(
35
- JSON.stringify(
36
- rows.map((row) => ({
37
- ...row,
38
- properties: Object.fromEntries(
39
- Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
40
- )
41
- }))
42
- )
43
- ).digest("hex");
44
- if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
45
- if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
46
- if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
47
- if (dataset.rows.length === 0) throw new Error("Raindrop eval datasets require at least one row");
48
- const ids = rows.map((entry) => entry.id);
49
- if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
50
- if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
51
- if (dataset.remote) {
52
- if (!dataset.remote.reference.trim())
53
- throw new Error("Raindrop remote dataset reference cannot be empty");
54
- if (!/^[a-f0-9]{64}$/.test(dataset.remote.fingerprint)) {
55
- throw new Error("Raindrop remote dataset fingerprint must be a SHA-256 digest");
56
- }
57
- }
58
- const defined = {
59
- ...dataset,
60
- id,
61
- name,
62
- version,
63
- rows,
64
- [EVAL_DATASET]: true
65
- };
66
- Object.defineProperty(defined, EVAL_DATASET, { enumerable: false });
67
- return defined;
68
- }
69
- function isEvalDataset(value) {
70
- return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
71
- }
72
3
  var BooleanThresholdSchema = z.object({ equals: z.boolean() }).strict();
73
4
  var FiniteThresholdSchema = z.number().finite();
74
- var EvalOutputSchema = z.enum(["boolean", "score", "number"]);
75
5
  var NumericThresholdSchema = z.union([
76
6
  z.object({ gte: FiniteThresholdSchema, lte: FiniteThresholdSchema.optional() }).strict(),
77
7
  z.object({ lte: FiniteThresholdSchema }).strict()
@@ -79,194 +9,32 @@ var NumericThresholdSchema = z.union([
79
9
  (threshold) => !("gte" in threshold) || threshold.lte === void 0 || threshold.gte <= threshold.lte,
80
10
  { message: "Lower threshold cannot exceed upper threshold" }
81
11
  );
82
- function defineEvaluatorProgram(program) {
83
- var _a, _b, _c, _d;
84
- const slug = program.slug.trim();
85
- const name = program.name.trim();
86
- const source = program.source.trim();
87
- const intent = program.intent.trim();
88
- if (!/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(slug) || slug.length > 64) {
89
- throw new Error(
90
- "Raindrop evaluator program slugs must use lowercase letters, numbers, and single hyphens"
91
- );
92
- }
93
- if (!name || name.length > 120) {
94
- throw new Error("Raindrop evaluator program names must be between 1 and 120 characters");
95
- }
96
- if (!source) throw new Error(`Raindrop evaluator program ${slug} source cannot be empty`);
97
- if (!intent || intent.length > 1e4) {
98
- throw new Error(
99
- `Raindrop evaluator program ${slug} intent must be between 1 and 10000 characters`
100
- );
101
- }
102
- if (program.description != null && program.description.trim().length > 2e3) {
103
- throw new Error(`Raindrop evaluator program ${slug} description exceeds 2000 characters`);
104
- }
105
- const rules = (_b = (_a = program.rules) == null ? void 0 : _a.map((rule) => rule.trim())) != null ? _b : [];
106
- if (rules.length > 12 || rules.some((rule) => !rule || rule.length > 2e3)) {
107
- throw new Error(`Raindrop evaluator program ${slug} rules are invalid`);
108
- }
109
- if (program.expected && (!z.string().uuid().safeParse(program.expected.evalId).success || !Number.isInteger(program.expected.programVersion) || program.expected.programVersion <= 0)) {
110
- throw new Error(`Raindrop evaluator program ${slug} expected identity is invalid`);
111
- }
112
- return {
113
- ...program,
114
- kind: "program",
115
- scope: z.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
116
- slug,
117
- name,
118
- source,
119
- intent,
120
- description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
121
- rules
122
- };
123
- }
124
- function defineLocalEvaluator(evaluator) {
125
- var _a;
126
- return { ...evaluator, scope: z.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
127
- }
128
- var EVAL_DEFINITION = /* @__PURE__ */ Symbol.for("raindrop.evalDefinition");
129
- function defineEvalSuite(definition) {
130
- const name = definition.name.trim();
131
- if (name.length === 0) throw new Error("Raindrop eval suite names cannot be empty");
132
- const dataset = typeof definition.dataset === "string" ? definition.dataset.trim() : defineDataset(definition.dataset);
133
- if (typeof dataset === "string" && dataset.length === 0) {
134
- throw new Error("Raindrop eval suite datasets cannot be empty");
135
- }
136
- if (definition.datasetVersionId !== void 0) {
137
- z.string().uuid().parse(definition.datasetVersionId);
138
- if (typeof dataset !== "string")
139
- throw new Error("A dataset version pin requires a published dataset slug");
140
- }
141
- if (definition.agent !== void 0 && definition.run !== void 0) {
142
- throw new Error("An eval suite accepts either agent or run, not both");
143
- }
144
- if (definition.agent === void 0 && typeof definition.run !== "function") {
145
- throw new Error("Raindrop eval suite run must be a function");
146
- }
147
- if (!Array.isArray(definition.evaluators) || definition.evaluators.length === 0) {
148
- throw new Error("Raindrop eval suites require at least one evaluator");
149
- }
150
- validateOptionalNumber("concurrency", definition.concurrency, true, false);
151
- validateOptionalNumber("traceWaitMs", definition.traceWaitMs, false, true);
152
- validateOptionalNumber("evalWaitMs", definition.evalWaitMs, false, true);
153
- validateOptionalNumber("evalPollIntervalMs", definition.evalPollIntervalMs, false, false);
154
- const evaluators = definition.evaluators.map((evaluator) => validateEvaluator(evaluator));
155
- const evaluatorNames = evaluators.map((entry) => evaluatorName(entry));
156
- if (new Set(evaluatorNames).size !== evaluatorNames.length) {
157
- throw new Error("Raindrop eval suite evaluators must be unique");
158
- }
159
- const branded = {
160
- ...definition,
161
- [EVAL_DEFINITION]: true,
162
- name,
163
- dataset,
164
- evaluators
165
- };
166
- Object.defineProperty(branded, EVAL_DEFINITION, { enumerable: false });
167
- return branded;
168
- }
169
- function isEvalSuiteDefinition(value) {
170
- return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
171
- }
172
- function isPublishedEvalSuite(definition) {
173
- return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
174
- ({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
175
- );
176
- }
177
- function evaluatorName(evaluator) {
178
- return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
179
- }
180
- function evaluatorOutput(evaluator) {
181
- return typeof evaluator.evaluator === "string" ? void 0 : evaluator.evaluator.output;
182
- }
183
- function validateEvaluatorOutput(evaluator, rawOutput) {
184
- const name = evaluatorName(evaluator);
185
- const output = EvalOutputSchema.parse(rawOutput);
186
- const localOutput = evaluatorOutput(evaluator);
187
- if (localOutput !== void 0 && localOutput !== output) {
188
- throw new Error(`Raindrop local evaluator ${name} expected ${localOutput}, received ${output}`);
189
- }
190
- switch (output) {
191
- case "boolean":
192
- parseBooleanThreshold(evaluator.threshold);
193
- break;
194
- case "score":
195
- parseNumericThreshold(name, "score", evaluator.threshold);
196
- break;
197
- case "number":
198
- parseNumericThreshold(name, "number", evaluator.threshold);
199
- break;
200
- }
201
- }
202
- function validateEvaluator(evaluator) {
203
- if (typeof evaluator !== "object" || evaluator === null || !("evaluator" in evaluator)) {
204
- throw new Error("Raindrop eval evaluator entries must name an evaluator");
205
- }
206
- if ("output" in evaluator) {
207
- throw new Error(
208
- "Raindrop eval evaluator entries do not accept output; the evaluator definition owns it"
209
- );
210
- }
211
- if (typeof evaluator.evaluator !== "string" && (typeof evaluator.evaluator !== "object" || evaluator.evaluator === null || typeof evaluator.evaluator.slug !== "string" || typeof evaluator.evaluator.name !== "string")) {
212
- throw new Error("Raindrop local eval evaluators require a slug, name, output, and judge");
213
- }
214
- const name = evaluatorName(evaluator);
215
- if (name.length === 0) throw new Error("Raindrop eval evaluator names cannot be empty");
216
- if (typeof evaluator.evaluator === "string") {
217
- const threshold = evaluator.threshold;
218
- if (threshold !== void 0) {
219
- const boolean = BooleanThresholdSchema.safeParse(threshold);
220
- const numeric = NumericThresholdSchema.safeParse(threshold);
221
- if (!boolean.success && !numeric.success) {
222
- throw new Error(`Raindrop hosted evaluator ${name} has an invalid threshold`);
223
- }
224
- }
225
- return { ...evaluator, evaluator: name };
226
- }
227
- if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "portable") {
228
- validateEvaluatorOutput(evaluator, evaluator.evaluator.output);
229
- return evaluator;
230
- }
231
- if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "program") {
232
- const program = defineEvaluatorProgram(evaluator.evaluator);
233
- validateEvaluatorOutput({ ...evaluator, evaluator: program }, program.output);
234
- return { ...evaluator, evaluator: program };
235
- }
236
- if (typeof evaluator.evaluator.judge !== "function") {
237
- throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
238
- }
239
- if (evaluator.evaluator.scope !== "row") {
240
- throw new Error(`Raindrop local evaluator ${name} scope must be row`);
241
- }
242
- const output = EvalOutputSchema.parse(evaluator.evaluator.output);
243
- validateEvaluatorOutput(evaluator, output);
244
- return evaluator;
245
- }
246
- function validateOptionalNumber(name, value, integer, allowZero) {
247
- if (value === void 0) return;
248
- const valid = Number.isFinite(value) && (!integer || Number.isInteger(value)) && (allowZero ? value >= 0 : value > 0);
249
- if (!valid) {
250
- const kind = integer ? "a positive integer" : allowZero ? "non-negative" : "positive";
251
- throw new Error(`Raindrop eval ${name} must be ${kind}`);
252
- }
253
- }
254
- function parseBooleanThreshold(threshold) {
255
- return threshold === void 0 ? void 0 : BooleanThresholdSchema.parse(threshold);
256
- }
257
- function parseNumericThreshold(name, output, threshold) {
258
- const parsed = threshold === void 0 ? void 0 : NumericThresholdSchema.parse(threshold);
259
- if (output === "score" && parsed) {
260
- const lower = "gte" in parsed ? parsed.gte : void 0;
261
- const inRange = (lower === void 0 || lower >= 1 && lower <= 5) && (parsed.lte === void 0 || parsed.lte >= 1 && parsed.lte <= 5);
262
- if (!inRange)
263
- throw new Error(`Raindrop score evaluator ${name} thresholds must be from 1 to 5`);
264
- }
265
- return parsed;
266
- }
12
+ var EvalSuiteManifestSchema = z.object({
13
+ name: z.string().trim().min(1),
14
+ dataset: z.string().trim().min(1),
15
+ datasetVersionId: z.string().uuid().optional(),
16
+ evaluators: z.array(
17
+ z.object({
18
+ evaluator: z.string().trim().min(1),
19
+ threshold: z.union([BooleanThresholdSchema, NumericThresholdSchema]).optional()
20
+ }).strict()
21
+ ).min(1),
22
+ concurrency: z.number().int().positive().optional(),
23
+ traceWaitMs: z.number().nonnegative().optional(),
24
+ evalWaitMs: z.number().nonnegative().optional(),
25
+ evalPollIntervalMs: z.number().positive().optional()
26
+ }).strict();
267
27
 
268
28
  // src/evals/generated/trace.ts
269
29
  import z2 from "zod";
30
+ var JsonValueSchema = z2.lazy(() => z2.union([
31
+ z2.string(),
32
+ z2.number(),
33
+ z2.boolean(),
34
+ z2.null(),
35
+ z2.array(JsonValueSchema),
36
+ z2.record(JsonValueSchema)
37
+ ]));
270
38
  var stringToJSONSchema = z2.string().transform((str, ctx) => {
271
39
  try {
272
40
  return JSON.parse(str);
@@ -635,14 +403,20 @@ var ReplayTraceSnapshotSpanSchema = TracesListPipe.output.extend({
635
403
  duration_ns: NanosecondsStringSchema,
636
404
  _inserted_at: z2.string().min(1)
637
405
  }).strict();
406
+ var EvalDatasetCaseIdSchema = z2.string().min(1).max(512).refine((value) => value === value.trim(), "Case id cannot have surrounding whitespace");
407
+ var EvalCaseExpectationSchema = z2.object({
408
+ description: z2.string().trim().min(1).max(1e4),
409
+ basis: z2.enum(["human", "policy", "code", "inferred"]).optional(),
410
+ reference: z2.string().trim().min(1).max(2e3).optional()
411
+ }).strict();
638
412
  var TraceSchema = z2.object({
639
413
  origin: z2.enum(["production", "dataset", "simulation", "replay"]),
640
414
  event: AIAnalyticsEventSchema,
641
- entries: z2.array(RichInteractionEntrySchema)
415
+ entries: z2.array(RichInteractionEntrySchema),
416
+ caseContext: z2.object({ caseId: EvalDatasetCaseIdSchema, expectation: EvalCaseExpectationSchema.optional() }).optional()
642
417
  });
643
418
  var CapturedTraceSpanSchema = ReplayTraceSnapshotSpanSchema;
644
419
  var EVAL_DATASET_MAX_SPANS_PER_TRACE = 500;
645
- var EvalDatasetCaseIdSchema = z2.string().min(1).max(512).refine((value) => value === value.trim(), "Case id cannot have surrounding whitespace");
646
420
  var CapturedTraceSnapshotSchema = z2.object({
647
421
  event: AIAnalyticsEventSchema,
648
422
  spans: z2.array(CapturedTraceSpanSchema).min(1).max(EVAL_DATASET_MAX_SPANS_PER_TRACE)
@@ -712,15 +486,6 @@ var CapturedTraceSnapshotSchema = z2.object({
712
486
  path: ["spans"]
713
487
  });
714
488
  }
715
- for (const [index, span] of spans.entries()) {
716
- if (span.parent_span_id && !spanIds.has(span.parent_span_id)) {
717
- context.addIssue({
718
- code: z2.ZodIssueCode.custom,
719
- message: `Missing parent span ${span.parent_span_id}`,
720
- path: ["spans", index, "parent_span_id"]
721
- });
722
- }
723
- }
724
489
  const parentBySpanId = new Map(spans.map((span) => [span.span_id, span.parent_span_id]));
725
490
  for (const [index, span] of spans.entries()) {
726
491
  const path = /* @__PURE__ */ new Set();
@@ -745,8 +510,9 @@ var EvalCaseTraceSchema = z2.object({
745
510
  });
746
511
  var EvalCaseTracePairSchema = z2.object({
747
512
  caseId: EvalDatasetCaseIdSchema,
513
+ output: JsonValueSchema.optional(),
748
514
  candidate: EvalCaseTraceSchema.extend({
749
- trace: TraceSchema.extend({ origin: z2.literal("replay") })
515
+ trace: TraceSchema.extend({ origin: z2.enum(["replay", "dataset"]) })
750
516
  }),
751
517
  reference: EvalCaseTraceSchema.extend({
752
518
  trace: TraceSchema.extend({ origin: z2.literal("dataset") })
@@ -1012,37 +778,6 @@ function traceStepsToEntries(steps) {
1012
778
  function projectWorkshopTrace(input) {
1013
779
  return createTrace(input.origin, input.event, traceStepsToEntries(buildTraceSteps([...input.spans], { includePayloads: true })).map(boundEvalEntry));
1014
780
  }
1015
- function projectCapturedEvent(input) {
1016
- var _a, _b, _c, _d, _e, _f;
1017
- const ordered = [...input.spans].sort((left, right) => compareNanoseconds(left.start_unix_ns, right.start_unix_ns));
1018
- const eventSpan = ordered.find((span) => span.span_id === `evt_${input.eventId}`);
1019
- const llmSpans = ordered.filter((span) => span.span_type === "LLM_GENERATION" || span.span_type === "LLM_GENERATION_STREAM");
1020
- const inputSpan = ((_a = eventSpan == null ? void 0 : eventSpan.input_payload) == null ? void 0 : _a.trim()) ? eventSpan : llmSpans.find((span) => {
1021
- var _a2;
1022
- return Boolean((_a2 = span.input_payload) == null ? void 0 : _a2.trim());
1023
- });
1024
- const reverseLlmSpans = [...llmSpans].reverse();
1025
- const outputSpan = (_c = ((_b = eventSpan == null ? void 0 : eventSpan.output_payload) == null ? void 0 : _b.trim()) ? eventSpan : void 0) != null ? _c : reverseLlmSpans.find((span) => {
1026
- var _a2;
1027
- return Boolean((_a2 = span.output_payload) == null ? void 0 : _a2.trim()) && extractFinishReason(span) !== "tool-calls";
1028
- });
1029
- const modelSpan = (eventSpan == null ? void 0 : eventSpan.model) ? eventSpan : reverseLlmSpans.find((span) => Boolean(span.model));
1030
- return AIAnalyticsEventSchema.parse({
1031
- id: input.eventId,
1032
- _raindropDatabaseId: input.internalEventId,
1033
- name: input.name,
1034
- timestamp: input.timestamp,
1035
- receivedAt: input.receivedAt,
1036
- userId: input.userId,
1037
- aiData: {
1038
- input: (_d = inputSpan == null ? void 0 : inputSpan.input_payload) != null ? _d : null,
1039
- output: (_e = outputSpan == null ? void 0 : outputSpan.output_payload) != null ? _e : null,
1040
- model: (_f = modelSpan == null ? void 0 : modelSpan.model) != null ? _f : null,
1041
- convoId: input.convoId
1042
- },
1043
- properties: input.properties
1044
- });
1045
- }
1046
781
  function traceOutput(trace) {
1047
782
  var _a;
1048
783
  return (_a = trace.event.aiData.output) != null ? _a : null;
@@ -1050,64 +785,329 @@ function traceOutput(trace) {
1050
785
  function traceToolCalls(trace, name) {
1051
786
  return trace.entries.filter((entry) => entry.type === "tool_call" && (name === void 0 || entry.name === name));
1052
787
  }
1053
- var TRACE_PROJECTION_SHA256 = "7d12eb03150b8f203e4404622f047676f50b3aa1e5442f0ed42d77d6002a33a6";
788
+ var TRACE_PROJECTION_SHA256 = "22c666c6ace47d74d3f7dc477e68b27547257b319ba4014a4a32fbf977baac72";
1054
789
 
1055
- // src/evals/portable.ts
790
+ // src/evals/definition.ts
791
+ import { createHash } from "crypto";
1056
792
  import { z as z3 } from "zod";
1057
- var PortableEvalArtifactSchema = z3.object({
1058
- format: z3.literal("raindrop.eval-program"),
1059
- formatVersion: z3.literal(1),
1060
- runtimeAbi: z3.literal("raindrop-eval-v2"),
1061
- evaluator: z3.object({
1062
- id: z3.string().uuid(),
1063
- slug: z3.string().min(1),
1064
- programVersion: z3.number().int().positive(),
1065
- executionMode: z3.enum(["deterministic", "judge"]),
1066
- outputType: z3.enum(["boolean", "score", "number"]),
1067
- scope: z3.literal("batch")
793
+ var EvalScoreSchema = z3.union([
794
+ z3.literal(1),
795
+ z3.literal(2),
796
+ z3.literal(3),
797
+ z3.literal(4),
798
+ z3.literal(5)
799
+ ]);
800
+ var EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
801
+ var DatasetRowInputSchema = z3.object({
802
+ id: z3.string().trim().min(1),
803
+ name: z3.string().optional(),
804
+ input: z3.string().nullable(),
805
+ output: JsonValueSchema.default(null),
806
+ properties: z3.record(z3.string()).default({})
807
+ });
808
+ function defineDataset(dataset) {
809
+ var _a, _b;
810
+ const id = dataset.id.trim();
811
+ const name = dataset.name.trim();
812
+ const rows = dataset.rows.map((row) => {
813
+ var _a2;
814
+ const parsed = DatasetRowInputSchema.parse(row);
815
+ return {
816
+ id: parsed.id,
817
+ name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
818
+ input: parsed.input,
819
+ output: parsed.output,
820
+ properties: parsed.properties
821
+ };
822
+ });
823
+ const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : createHash("sha256").update(
824
+ JSON.stringify(
825
+ rows.map((row) => ({
826
+ ...row,
827
+ properties: Object.fromEntries(
828
+ Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
829
+ )
830
+ }))
831
+ )
832
+ ).digest("hex");
833
+ if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
834
+ if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
835
+ if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
836
+ if (dataset.rows.length === 0) throw new Error("Raindrop eval datasets require at least one row");
837
+ const ids = rows.map((entry) => entry.id);
838
+ if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
839
+ if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
840
+ if (dataset.remote) {
841
+ if (!dataset.remote.reference.trim())
842
+ throw new Error("Raindrop remote dataset reference cannot be empty");
843
+ if (!/^[a-f0-9]{64}$/.test(dataset.remote.fingerprint)) {
844
+ throw new Error("Raindrop remote dataset fingerprint must be a SHA-256 digest");
845
+ }
846
+ }
847
+ const defined = {
848
+ ...dataset,
849
+ id,
850
+ name,
851
+ version,
852
+ rows,
853
+ [EVAL_DATASET]: true
854
+ };
855
+ Object.defineProperty(defined, EVAL_DATASET, { enumerable: false });
856
+ return defined;
857
+ }
858
+ function isEvalDataset(value) {
859
+ return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
860
+ }
861
+ var EvalOutputSchema = z3.enum(["boolean", "score", "number"]);
862
+ function defineEvaluatorProgram(program) {
863
+ var _a, _b, _c, _d;
864
+ const slug = program.slug.trim();
865
+ const name = program.name.trim();
866
+ const source = program.source.trim();
867
+ const intent = program.intent.trim();
868
+ if (!/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(slug) || slug.length > 64) {
869
+ throw new Error(
870
+ "Raindrop evaluator program slugs must use lowercase letters, numbers, and single hyphens"
871
+ );
872
+ }
873
+ if (!name || name.length > 120) {
874
+ throw new Error("Raindrop evaluator program names must be between 1 and 120 characters");
875
+ }
876
+ if (!source) throw new Error(`Raindrop evaluator program ${slug} source cannot be empty`);
877
+ if (!intent || intent.length > 1e4) {
878
+ throw new Error(
879
+ `Raindrop evaluator program ${slug} intent must be between 1 and 10000 characters`
880
+ );
881
+ }
882
+ if (program.description != null && program.description.trim().length > 2e3) {
883
+ throw new Error(`Raindrop evaluator program ${slug} description exceeds 2000 characters`);
884
+ }
885
+ const rules = (_b = (_a = program.rules) == null ? void 0 : _a.map((rule) => rule.trim())) != null ? _b : [];
886
+ if (rules.length > 12 || rules.some((rule) => !rule || rule.length > 2e3)) {
887
+ throw new Error(`Raindrop evaluator program ${slug} rules are invalid`);
888
+ }
889
+ if (program.expected && (!z3.string().uuid().safeParse(program.expected.evalId).success || !Number.isInteger(program.expected.programVersion) || program.expected.programVersion <= 0)) {
890
+ throw new Error(`Raindrop evaluator program ${slug} expected identity is invalid`);
891
+ }
892
+ return {
893
+ ...program,
894
+ kind: "program",
895
+ scope: z3.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
896
+ slug,
897
+ name,
898
+ source,
899
+ intent,
900
+ description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
901
+ rules
902
+ };
903
+ }
904
+ function defineLocalEvaluator(evaluator) {
905
+ var _a;
906
+ return { ...evaluator, scope: z3.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
907
+ }
908
+ var EVAL_DEFINITION = /* @__PURE__ */ Symbol.for("raindrop.evalDefinition");
909
+ function defineEvalSuite(definition, binding) {
910
+ if (binding) return createEvalSuite({ ...EvalSuiteManifestSchema.parse(definition), ...binding });
911
+ if (!("run" in definition) && !("agent" in definition)) {
912
+ throw new Error("Bind the eval manifest to your local run function or agent");
913
+ }
914
+ return createEvalSuite(definition);
915
+ }
916
+ function createEvalSuite(definition) {
917
+ const name = definition.name.trim();
918
+ if (name.length === 0) throw new Error("Raindrop eval suite names cannot be empty");
919
+ const dataset = typeof definition.dataset === "string" ? definition.dataset.trim() : defineDataset(definition.dataset);
920
+ if (typeof dataset === "string" && dataset.length === 0) {
921
+ throw new Error("Raindrop eval suite datasets cannot be empty");
922
+ }
923
+ if (definition.datasetVersionId !== void 0) {
924
+ z3.string().uuid().parse(definition.datasetVersionId);
925
+ if (typeof dataset !== "string")
926
+ throw new Error("A dataset version pin requires a published dataset slug");
927
+ }
928
+ if (definition.agent !== void 0 && definition.run !== void 0) {
929
+ throw new Error("An eval suite accepts either agent or run, not both");
930
+ }
931
+ if (definition.agent === void 0 && typeof definition.run !== "function") {
932
+ throw new Error("Raindrop eval suite run must be a function");
933
+ }
934
+ if (!Array.isArray(definition.evaluators) || definition.evaluators.length === 0) {
935
+ throw new Error("Raindrop eval suites require at least one evaluator");
936
+ }
937
+ validateOptionalNumber("concurrency", definition.concurrency, true, false);
938
+ validateOptionalNumber("traceWaitMs", definition.traceWaitMs, false, true);
939
+ validateOptionalNumber("evalWaitMs", definition.evalWaitMs, false, true);
940
+ validateOptionalNumber("evalPollIntervalMs", definition.evalPollIntervalMs, false, false);
941
+ const evaluators = definition.evaluators.map((evaluator) => validateEvaluator(evaluator));
942
+ const evaluatorNames = evaluators.map((entry) => evaluatorName(entry));
943
+ if (new Set(evaluatorNames).size !== evaluatorNames.length) {
944
+ throw new Error("Raindrop eval suite evaluators must be unique");
945
+ }
946
+ const branded = {
947
+ ...definition,
948
+ [EVAL_DEFINITION]: true,
949
+ name,
950
+ dataset,
951
+ evaluators
952
+ };
953
+ Object.defineProperty(branded, EVAL_DEFINITION, { enumerable: false });
954
+ return branded;
955
+ }
956
+ function isEvalSuiteDefinition(value) {
957
+ return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
958
+ }
959
+ function isPublishedEvalSuite(definition) {
960
+ return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
961
+ ({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
962
+ );
963
+ }
964
+ function evaluatorName(evaluator) {
965
+ return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
966
+ }
967
+ function evaluatorOutput(evaluator) {
968
+ return typeof evaluator.evaluator === "string" ? void 0 : evaluator.evaluator.output;
969
+ }
970
+ function validateEvaluatorOutput(evaluator, rawOutput) {
971
+ const name = evaluatorName(evaluator);
972
+ const output = EvalOutputSchema.parse(rawOutput);
973
+ const localOutput = evaluatorOutput(evaluator);
974
+ if (localOutput !== void 0 && localOutput !== output) {
975
+ throw new Error(`Raindrop local evaluator ${name} expected ${localOutput}, received ${output}`);
976
+ }
977
+ switch (output) {
978
+ case "boolean":
979
+ parseBooleanThreshold(evaluator.threshold);
980
+ break;
981
+ case "score":
982
+ parseNumericThreshold(name, "score", evaluator.threshold);
983
+ break;
984
+ case "number":
985
+ parseNumericThreshold(name, "number", evaluator.threshold);
986
+ break;
987
+ }
988
+ }
989
+ function validateEvaluator(evaluator) {
990
+ if (typeof evaluator !== "object" || evaluator === null || !("evaluator" in evaluator)) {
991
+ throw new Error("Raindrop eval evaluator entries must name an evaluator");
992
+ }
993
+ if ("output" in evaluator) {
994
+ throw new Error(
995
+ "Raindrop eval evaluator entries do not accept output; the evaluator definition owns it"
996
+ );
997
+ }
998
+ if (typeof evaluator.evaluator !== "string" && (typeof evaluator.evaluator !== "object" || evaluator.evaluator === null || typeof evaluator.evaluator.slug !== "string" || typeof evaluator.evaluator.name !== "string")) {
999
+ throw new Error("Raindrop local eval evaluators require a slug, name, output, and judge");
1000
+ }
1001
+ const name = evaluatorName(evaluator);
1002
+ if (name.length === 0) throw new Error("Raindrop eval evaluator names cannot be empty");
1003
+ if (typeof evaluator.evaluator === "string") {
1004
+ const threshold = evaluator.threshold;
1005
+ if (threshold !== void 0) {
1006
+ const boolean = BooleanThresholdSchema.safeParse(threshold);
1007
+ const numeric = NumericThresholdSchema.safeParse(threshold);
1008
+ if (!boolean.success && !numeric.success) {
1009
+ throw new Error(`Raindrop hosted evaluator ${name} has an invalid threshold`);
1010
+ }
1011
+ }
1012
+ return { ...evaluator, evaluator: name };
1013
+ }
1014
+ if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "portable") {
1015
+ validateEvaluatorOutput(evaluator, evaluator.evaluator.output);
1016
+ return evaluator;
1017
+ }
1018
+ if ("kind" in evaluator.evaluator && evaluator.evaluator.kind === "program") {
1019
+ const program = defineEvaluatorProgram(evaluator.evaluator);
1020
+ validateEvaluatorOutput({ ...evaluator, evaluator: program }, program.output);
1021
+ return { ...evaluator, evaluator: program };
1022
+ }
1023
+ if (typeof evaluator.evaluator.judge !== "function") {
1024
+ throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
1025
+ }
1026
+ if (evaluator.evaluator.scope !== "row") {
1027
+ throw new Error(`Raindrop local evaluator ${name} scope must be row`);
1028
+ }
1029
+ const output = EvalOutputSchema.parse(evaluator.evaluator.output);
1030
+ validateEvaluatorOutput(evaluator, output);
1031
+ return evaluator;
1032
+ }
1033
+ function validateOptionalNumber(name, value, integer, allowZero) {
1034
+ if (value === void 0) return;
1035
+ const valid = Number.isFinite(value) && (!integer || Number.isInteger(value)) && (allowZero ? value >= 0 : value > 0);
1036
+ if (!valid) {
1037
+ const kind = integer ? "a positive integer" : allowZero ? "non-negative" : "positive";
1038
+ throw new Error(`Raindrop eval ${name} must be ${kind}`);
1039
+ }
1040
+ }
1041
+ function parseBooleanThreshold(threshold) {
1042
+ return threshold === void 0 ? void 0 : BooleanThresholdSchema.parse(threshold);
1043
+ }
1044
+ function parseNumericThreshold(name, output, threshold) {
1045
+ const parsed = threshold === void 0 ? void 0 : NumericThresholdSchema.parse(threshold);
1046
+ if (output === "score" && parsed) {
1047
+ const lower = "gte" in parsed ? parsed.gte : void 0;
1048
+ const inRange = (lower === void 0 || lower >= 1 && lower <= 5) && (parsed.lte === void 0 || parsed.lte >= 1 && parsed.lte <= 5);
1049
+ if (!inRange)
1050
+ throw new Error(`Raindrop score evaluator ${name} thresholds must be from 1 to 5`);
1051
+ }
1052
+ return parsed;
1053
+ }
1054
+
1055
+ // src/evals/portable.ts
1056
+ import { z as z4 } from "zod";
1057
+ var PortableEvalArtifactSchema = z4.object({
1058
+ format: z4.literal("raindrop.eval-program"),
1059
+ formatVersion: z4.literal(1),
1060
+ runtimeAbi: z4.literal("raindrop-eval-v2"),
1061
+ evaluator: z4.object({
1062
+ id: z4.string().uuid(),
1063
+ slug: z4.string().min(1),
1064
+ programVersion: z4.number().int().positive(),
1065
+ executionMode: z4.enum(["deterministic", "judge"]),
1066
+ outputType: z4.enum(["boolean", "score", "number"]),
1067
+ scope: z4.literal("batch")
1068
1068
  }).strict(),
1069
- traceProjectionVersion: z3.literal(1),
1070
- traceProjectionSha256: z3.string().regex(/^[a-f0-9]{64}$/),
1071
- source: z3.string().min(1),
1072
- sha256: z3.string().regex(/^[a-f0-9]{64}$/)
1069
+ traceProjectionVersion: z4.literal(1),
1070
+ traceProjectionSha256: z4.string().regex(/^[a-f0-9]{64}$/),
1071
+ source: z4.string().min(1),
1072
+ sha256: z4.string().regex(/^[a-f0-9]{64}$/)
1073
1073
  }).strict();
1074
- var PortableEvalResultSchema = z3.object({
1075
- outcomes: z3.array(
1076
- z3.union([
1077
- z3.object({
1078
- state: z3.literal("graded"),
1079
- traceId: z3.string(),
1080
- pass: z3.boolean(),
1081
- note: z3.string().optional()
1074
+ var PortableEvalResultSchema = z4.object({
1075
+ outcomes: z4.array(
1076
+ z4.union([
1077
+ z4.object({
1078
+ state: z4.literal("graded"),
1079
+ traceId: z4.string(),
1080
+ pass: z4.boolean(),
1081
+ note: z4.string().optional()
1082
1082
  }),
1083
- z3.object({
1084
- state: z3.literal("graded"),
1085
- traceId: z3.string(),
1083
+ z4.object({
1084
+ state: z4.literal("graded"),
1085
+ traceId: z4.string(),
1086
1086
  score: EvalScoreSchema,
1087
- note: z3.string().optional()
1087
+ note: z4.string().optional()
1088
1088
  }),
1089
- z3.object({
1090
- state: z3.literal("graded"),
1091
- traceId: z3.string(),
1092
- value: z3.number().finite(),
1093
- note: z3.string().optional()
1089
+ z4.object({
1090
+ state: z4.literal("graded"),
1091
+ traceId: z4.string(),
1092
+ value: z4.number().finite(),
1093
+ note: z4.string().optional()
1094
1094
  }),
1095
- z3.object({ state: z3.literal("errored"), traceId: z3.string(), message: z3.string() }),
1096
- z3.object({
1097
- state: z3.literal("ungraded"),
1098
- traceId: z3.string(),
1099
- reason: z3.enum(["not sampled", "trace unavailable", "content unreadable"])
1095
+ z4.object({ state: z4.literal("errored"), traceId: z4.string(), message: z4.string() }),
1096
+ z4.object({
1097
+ state: z4.literal("ungraded"),
1098
+ traceId: z4.string(),
1099
+ reason: z4.enum(["not sampled", "trace unavailable", "content unreadable"])
1100
1100
  })
1101
1101
  ])
1102
1102
  ),
1103
- failures: z3.array(z3.string()),
1104
- stats: z3.object({
1105
- traceCount: z3.number().int().nonnegative(),
1106
- hydratedCount: z3.number().int().nonnegative(),
1107
- judgeCalls: z3.number().int().nonnegative(),
1108
- durationMs: z3.number().nonnegative(),
1109
- skippedRows: z3.number().int().nonnegative(),
1110
- skippedReason: z3.literal("no trace").nullable()
1103
+ failures: z4.array(z4.string()),
1104
+ stats: z4.object({
1105
+ traceCount: z4.number().int().nonnegative(),
1106
+ hydratedCount: z4.number().int().nonnegative(),
1107
+ judgeCalls: z4.number().int().nonnegative(),
1108
+ durationMs: z4.number().nonnegative(),
1109
+ skippedRows: z4.number().int().nonnegative(),
1110
+ skippedReason: z4.literal("no trace").nullable()
1111
1111
  })
1112
1112
  });
1113
1113
  function parsePortableEvalResult(value) {
@@ -1138,7 +1138,7 @@ async function evaluatePortableEvaluator(evaluator, input) {
1138
1138
  const artifact = await verifyPortableEvalArtifact(evaluator.artifact, {
1139
1139
  expectedSha256: evaluator.expectedSha256
1140
1140
  });
1141
- const { runPortableEvaluator } = await import("./portable-runtime-6X2HHB4Y.mjs");
1141
+ const { runPortableEvaluator } = await import("./portable-runtime-F7W4STGH.mjs");
1142
1142
  const cases = ((_a = input.rows) != null ? _a : []).map(({ rowId, ...pair }) => ({ ...pair, caseId: rowId }));
1143
1143
  if (cases.length > 0 && (cases.length !== input.traces.length || cases.some(
1144
1144
  (entry, index) => {
@@ -1218,11 +1218,21 @@ function canonicalJson(value) {
1218
1218
  }
1219
1219
 
1220
1220
  export {
1221
+ BooleanThresholdSchema,
1222
+ NumericThresholdSchema,
1223
+ EvalSuiteManifestSchema,
1224
+ JsonValueSchema,
1225
+ EvalDatasetCaseIdSchema,
1226
+ EvalCaseExpectationSchema,
1227
+ TraceSchema,
1228
+ CapturedTraceSnapshotSchema,
1229
+ EvalCaseTracePairSchema,
1230
+ projectWorkshopTrace,
1231
+ traceOutput,
1232
+ traceToolCalls,
1221
1233
  EvalScoreSchema,
1222
1234
  defineDataset,
1223
1235
  isEvalDataset,
1224
- BooleanThresholdSchema,
1225
- NumericThresholdSchema,
1226
1236
  defineEvaluatorProgram,
1227
1237
  defineLocalEvaluator,
1228
1238
  defineEvalSuite,
@@ -1231,14 +1241,6 @@ export {
1231
1241
  evaluatorName,
1232
1242
  evaluatorOutput,
1233
1243
  validateEvaluatorOutput,
1234
- TraceSchema,
1235
- EvalDatasetCaseIdSchema,
1236
- CapturedTraceSnapshotSchema,
1237
- EvalCaseTracePairSchema,
1238
- projectWorkshopTrace,
1239
- projectCapturedEvent,
1240
- traceOutput,
1241
- traceToolCalls,
1242
1244
  parsePortableEvalResult,
1243
1245
  importPortableEvaluator,
1244
1246
  evaluatePortableEvaluator,