raindrop-ai 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5767,17 +5767,38 @@ var require_dist = __commonJS({
5767
5767
 
5768
5768
  // src/evals/definition.ts
5769
5769
  function defineDataset(dataset) {
5770
+ var _a, _b;
5770
5771
  const id = dataset.id.trim();
5771
5772
  const name = dataset.name.trim();
5772
- const version = dataset.version.trim();
5773
+ const rows = dataset.rows.map((row) => {
5774
+ var _a2;
5775
+ const parsed = DatasetRowInputSchema.parse(row);
5776
+ return {
5777
+ id: parsed.id,
5778
+ name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
5779
+ input: parsed.input,
5780
+ output: parsed.output,
5781
+ properties: parsed.properties
5782
+ };
5783
+ });
5784
+ const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : (0, import_node_crypto2.createHash)("sha256").update(
5785
+ JSON.stringify(
5786
+ rows.map((row) => ({
5787
+ ...row,
5788
+ properties: Object.fromEntries(
5789
+ Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
5790
+ )
5791
+ }))
5792
+ )
5793
+ ).digest("hex");
5773
5794
  if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
5774
5795
  if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
5775
5796
  if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
5776
- if (dataset.cases.length === 0)
5777
- throw new Error("Raindrop eval datasets require at least one case");
5778
- const ids = dataset.cases.map((entry) => entry.id.trim());
5779
- if (ids.some((caseId) => !caseId)) throw new Error("Raindrop eval case ids cannot be empty");
5780
- if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval case ids must be unique");
5797
+ if (dataset.rows.length === 0)
5798
+ throw new Error("Raindrop eval datasets require at least one row");
5799
+ const ids = rows.map((entry) => entry.id);
5800
+ if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
5801
+ if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
5781
5802
  if (dataset.remote) {
5782
5803
  if (!dataset.remote.reference.trim())
5783
5804
  throw new Error("Raindrop remote dataset reference cannot be empty");
@@ -5790,7 +5811,7 @@ function defineDataset(dataset) {
5790
5811
  id,
5791
5812
  name,
5792
5813
  version,
5793
- cases: dataset.cases.map((entry) => ({ ...entry, id: entry.id.trim() })),
5814
+ rows,
5794
5815
  [EVAL_DATASET]: true
5795
5816
  };
5796
5817
  Object.defineProperty(defined2, EVAL_DATASET, { enumerable: false });
@@ -5800,7 +5821,7 @@ function isEvalDataset(value) {
5800
5821
  return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
5801
5822
  }
5802
5823
  function defineEvaluatorProgram(program) {
5803
- var _a, _b, _c;
5824
+ var _a, _b, _c, _d;
5804
5825
  const slug = program.slug.trim();
5805
5826
  const name = program.name.trim();
5806
5827
  const source = program.source.trim();
@@ -5832,16 +5853,18 @@ function defineEvaluatorProgram(program) {
5832
5853
  return {
5833
5854
  ...program,
5834
5855
  kind: "program",
5856
+ scope: import_zod2.z.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
5835
5857
  slug,
5836
5858
  name,
5837
5859
  source,
5838
5860
  intent,
5839
- description: ((_c = program.description) == null ? void 0 : _c.trim()) || null,
5861
+ description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
5840
5862
  rules
5841
5863
  };
5842
5864
  }
5843
5865
  function defineLocalEvaluator(evaluator) {
5844
- return evaluator;
5866
+ var _a;
5867
+ return { ...evaluator, scope: import_zod2.z.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
5845
5868
  }
5846
5869
  function defineEvalSuite(definition) {
5847
5870
  const name = definition.name.trim();
@@ -5850,6 +5873,11 @@ function defineEvalSuite(definition) {
5850
5873
  if (typeof dataset === "string" && dataset.length === 0) {
5851
5874
  throw new Error("Raindrop eval suite datasets cannot be empty");
5852
5875
  }
5876
+ if (definition.datasetVersionId !== void 0) {
5877
+ import_zod2.z.string().uuid().parse(definition.datasetVersionId);
5878
+ if (typeof dataset !== "string")
5879
+ throw new Error("A dataset version pin requires a published dataset slug");
5880
+ }
5853
5881
  if (typeof definition.run !== "function") {
5854
5882
  throw new Error("Raindrop eval suite run must be a function");
5855
5883
  }
@@ -5878,6 +5906,11 @@ function defineEvalSuite(definition) {
5878
5906
  function isEvalSuiteDefinition(value) {
5879
5907
  return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
5880
5908
  }
5909
+ function isPublishedEvalSuite(definition) {
5910
+ return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
5911
+ ({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
5912
+ );
5913
+ }
5881
5914
  function evaluatorName(evaluator) {
5882
5915
  return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
5883
5916
  }
@@ -5940,8 +5973,8 @@ function validateEvaluator(evaluator) {
5940
5973
  if (typeof evaluator.evaluator.judge !== "function") {
5941
5974
  throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
5942
5975
  }
5943
- if (evaluator.evaluator.scope !== "case") {
5944
- throw new Error(`Raindrop local evaluator ${name} scope must be case`);
5976
+ if (evaluator.evaluator.scope !== "row") {
5977
+ throw new Error(`Raindrop local evaluator ${name} scope must be row`);
5945
5978
  }
5946
5979
  const output = EvalOutputSchema.parse(evaluator.evaluator.output);
5947
5980
  validateEvaluatorOutput(evaluator, output);
@@ -5968,10 +6001,11 @@ function parseNumericThreshold(name, output, threshold) {
5968
6001
  }
5969
6002
  return parsed;
5970
6003
  }
5971
- var import_zod2, EvalScoreSchema, EVAL_DATASET, BooleanThresholdSchema, FiniteThresholdSchema, EvalOutputSchema, NumericThresholdSchema, EVAL_DEFINITION;
6004
+ var import_node_crypto2, import_zod2, EvalScoreSchema, EVAL_DATASET, DatasetRowInputSchema, BooleanThresholdSchema, FiniteThresholdSchema, EvalOutputSchema, NumericThresholdSchema, EVAL_DEFINITION;
5972
6005
  var init_definition = __esm({
5973
6006
  "src/evals/definition.ts"() {
5974
6007
  "use strict";
6008
+ import_node_crypto2 = require("crypto");
5975
6009
  import_zod2 = require("zod");
5976
6010
  EvalScoreSchema = import_zod2.z.union([
5977
6011
  import_zod2.z.literal(1),
@@ -5981,6 +6015,13 @@ var init_definition = __esm({
5981
6015
  import_zod2.z.literal(5)
5982
6016
  ]);
5983
6017
  EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
6018
+ DatasetRowInputSchema = import_zod2.z.object({
6019
+ id: import_zod2.z.string().trim().min(1),
6020
+ name: import_zod2.z.string().optional(),
6021
+ input: import_zod2.z.string().nullable(),
6022
+ output: import_zod2.z.string().nullable().default(null),
6023
+ properties: import_zod2.z.record(import_zod2.z.string()).default({})
6024
+ });
5984
6025
  BooleanThresholdSchema = import_zod2.z.object({ equals: import_zod2.z.boolean() }).strict();
5985
6026
  FiniteThresholdSchema = import_zod2.z.number().finite();
5986
6027
  EvalOutputSchema = import_zod2.z.enum(["boolean", "score", "number"]);
@@ -6929,7 +6970,7 @@ async function resolveJudge(evaluator, rawRequest, remainingMs, tracesById, case
6929
6970
  var _a2;
6930
6971
  return (_a2 = evaluator.judge) == null ? void 0 : _a2.call(evaluator, {
6931
6972
  trace: trace8,
6932
- tracePair: casesByTraceId.get(request.traceId),
6973
+ tracePair: rowTracePair(casesByTraceId.get(request.traceId)),
6933
6974
  rubric: request.rubric.slice(0, 4e4),
6934
6975
  signal: controller.signal
6935
6976
  });
@@ -6957,13 +6998,18 @@ async function resolveJudge(evaluator, rawRequest, remainingMs, tracesById, case
6957
6998
  function byteLength(value) {
6958
6999
  return new TextEncoder().encode(value).byteLength;
6959
7000
  }
6960
- var import_quickjs_singlefile_cjs_release_sync, import_quickjs_emscripten_core, import_zod6, DEFAULT_MEMORY_BYTES, DEFAULT_STACK_BYTES, DEFAULT_TIMEOUT_MS, DEFAULT_JUDGE_TIMEOUT_MS, DEFAULT_JUDGE_REQUEST_TIMEOUT_MS, DEFAULT_JUDGE_CONCURRENCY, DEFAULT_MAX_INPUT_BYTES, DEFAULT_MAX_OUTPUT_BYTES, JudgeRequestSchema;
7001
+ function rowTracePair(pair) {
7002
+ if (!pair) return void 0;
7003
+ const { caseId, ...evidence } = pair;
7004
+ return { ...evidence, rowId: caseId };
7005
+ }
7006
+ var import_quickjs_singlefile_cjs_release_sync, import_quickjs_emscripten_core, import_zod7, DEFAULT_MEMORY_BYTES, DEFAULT_STACK_BYTES, DEFAULT_TIMEOUT_MS, DEFAULT_JUDGE_TIMEOUT_MS, DEFAULT_JUDGE_REQUEST_TIMEOUT_MS, DEFAULT_JUDGE_CONCURRENCY, DEFAULT_MAX_INPUT_BYTES, DEFAULT_MAX_OUTPUT_BYTES, JudgeRequestSchema;
6961
7007
  var init_portable_runtime = __esm({
6962
7008
  "src/evals/portable-runtime.ts"() {
6963
7009
  "use strict";
6964
7010
  import_quickjs_singlefile_cjs_release_sync = __toESM(require("@jitl/quickjs-singlefile-cjs-release-sync"));
6965
7011
  import_quickjs_emscripten_core = require("quickjs-emscripten-core");
6966
- import_zod6 = require("zod");
7012
+ import_zod7 = require("zod");
6967
7013
  init_portable();
6968
7014
  DEFAULT_MEMORY_BYTES = 32 * 1024 * 1024;
6969
7015
  DEFAULT_STACK_BYTES = 64 * 1024;
@@ -6973,11 +7019,11 @@ var init_portable_runtime = __esm({
6973
7019
  DEFAULT_JUDGE_CONCURRENCY = 4;
6974
7020
  DEFAULT_MAX_INPUT_BYTES = 4 * 1024 * 1024;
6975
7021
  DEFAULT_MAX_OUTPUT_BYTES = 4 * 1024 * 1024;
6976
- JudgeRequestSchema = import_zod6.z.object({
6977
- type: import_zod6.z.literal("judge_request"),
6978
- requestId: import_zod6.z.string(),
6979
- traceId: import_zod6.z.string(),
6980
- rubric: import_zod6.z.string()
7022
+ JudgeRequestSchema = import_zod7.z.object({
7023
+ type: import_zod7.z.literal("judge_request"),
7024
+ requestId: import_zod7.z.string(),
7025
+ traceId: import_zod7.z.string(),
7026
+ rubric: import_zod7.z.string()
6981
7027
  });
6982
7028
  }
6983
7029
  });
@@ -7012,16 +7058,17 @@ async function evaluatePortableEvaluator(evaluator, input) {
7012
7058
  expectedSha256: evaluator.expectedSha256
7013
7059
  });
7014
7060
  const { runPortableEvaluator: runPortableEvaluator2 } = await Promise.resolve().then(() => (init_portable_runtime(), portable_runtime_exports));
7015
- const cases = (_a = input.cases) != null ? _a : [];
7061
+ const cases = ((_a = input.rows) != null ? _a : []).map(({ rowId, ...pair }) => ({ ...pair, caseId: rowId }));
7016
7062
  if (cases.length > 0 && (cases.length !== input.traces.length || cases.some(
7017
7063
  (entry, index) => {
7018
7064
  var _a2;
7019
7065
  return entry.candidate.trace.event.id !== ((_a2 = input.traces[index]) == null ? void 0 : _a2.event.id);
7020
7066
  }
7021
7067
  ))) {
7022
- throw new Error("Raindrop portable evaluator cases must match traces exactly, in slice order");
7068
+ throw new Error("Raindrop portable evaluator rows must match traces exactly, in slice order");
7023
7069
  }
7024
- const result = await runPortableEvaluator2(evaluator, { ...input, cases });
7070
+ const { rows: _rows, ...runtimeInput } = input;
7071
+ const result = await runPortableEvaluator2(evaluator, { ...runtimeInput, cases });
7025
7072
  validatePortableResult(artifact, input.traces, result);
7026
7073
  return result;
7027
7074
  }
@@ -7088,67 +7135,67 @@ function canonicalJson2(value) {
7088
7135
  if (Array.isArray(value)) return `[${value.map(canonicalJson2).join(",")}]`;
7089
7136
  return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${canonicalJson2(entry)}`).join(",")}}`;
7090
7137
  }
7091
- var import_zod7, PortableEvalArtifactSchema, PortableEvalResultSchema;
7138
+ var import_zod8, PortableEvalArtifactSchema, PortableEvalResultSchema;
7092
7139
  var init_portable = __esm({
7093
7140
  "src/evals/portable.ts"() {
7094
7141
  "use strict";
7095
- import_zod7 = require("zod");
7142
+ import_zod8 = require("zod");
7096
7143
  init_definition();
7097
7144
  init_trace();
7098
- PortableEvalArtifactSchema = import_zod7.z.object({
7099
- format: import_zod7.z.literal("raindrop.eval-program"),
7100
- formatVersion: import_zod7.z.literal(1),
7101
- runtimeAbi: import_zod7.z.literal("raindrop-eval-v2"),
7102
- evaluator: import_zod7.z.object({
7103
- id: import_zod7.z.string().uuid(),
7104
- slug: import_zod7.z.string().min(1),
7105
- programVersion: import_zod7.z.number().int().positive(),
7106
- executionMode: import_zod7.z.enum(["deterministic", "judge"]),
7107
- outputType: import_zod7.z.enum(["boolean", "score", "number"]),
7108
- scope: import_zod7.z.literal("batch")
7145
+ PortableEvalArtifactSchema = import_zod8.z.object({
7146
+ format: import_zod8.z.literal("raindrop.eval-program"),
7147
+ formatVersion: import_zod8.z.literal(1),
7148
+ runtimeAbi: import_zod8.z.literal("raindrop-eval-v2"),
7149
+ evaluator: import_zod8.z.object({
7150
+ id: import_zod8.z.string().uuid(),
7151
+ slug: import_zod8.z.string().min(1),
7152
+ programVersion: import_zod8.z.number().int().positive(),
7153
+ executionMode: import_zod8.z.enum(["deterministic", "judge"]),
7154
+ outputType: import_zod8.z.enum(["boolean", "score", "number"]),
7155
+ scope: import_zod8.z.literal("batch")
7109
7156
  }).strict(),
7110
- traceProjectionVersion: import_zod7.z.literal(1),
7111
- traceProjectionSha256: import_zod7.z.string().regex(/^[a-f0-9]{64}$/),
7112
- source: import_zod7.z.string().min(1),
7113
- sha256: import_zod7.z.string().regex(/^[a-f0-9]{64}$/)
7157
+ traceProjectionVersion: import_zod8.z.literal(1),
7158
+ traceProjectionSha256: import_zod8.z.string().regex(/^[a-f0-9]{64}$/),
7159
+ source: import_zod8.z.string().min(1),
7160
+ sha256: import_zod8.z.string().regex(/^[a-f0-9]{64}$/)
7114
7161
  }).strict();
7115
- PortableEvalResultSchema = import_zod7.z.object({
7116
- outcomes: import_zod7.z.array(
7117
- import_zod7.z.union([
7118
- import_zod7.z.object({
7119
- state: import_zod7.z.literal("graded"),
7120
- traceId: import_zod7.z.string(),
7121
- pass: import_zod7.z.boolean(),
7122
- note: import_zod7.z.string().optional()
7162
+ PortableEvalResultSchema = import_zod8.z.object({
7163
+ outcomes: import_zod8.z.array(
7164
+ import_zod8.z.union([
7165
+ import_zod8.z.object({
7166
+ state: import_zod8.z.literal("graded"),
7167
+ traceId: import_zod8.z.string(),
7168
+ pass: import_zod8.z.boolean(),
7169
+ note: import_zod8.z.string().optional()
7123
7170
  }),
7124
- import_zod7.z.object({
7125
- state: import_zod7.z.literal("graded"),
7126
- traceId: import_zod7.z.string(),
7171
+ import_zod8.z.object({
7172
+ state: import_zod8.z.literal("graded"),
7173
+ traceId: import_zod8.z.string(),
7127
7174
  score: EvalScoreSchema,
7128
- note: import_zod7.z.string().optional()
7175
+ note: import_zod8.z.string().optional()
7129
7176
  }),
7130
- import_zod7.z.object({
7131
- state: import_zod7.z.literal("graded"),
7132
- traceId: import_zod7.z.string(),
7133
- value: import_zod7.z.number().finite(),
7134
- note: import_zod7.z.string().optional()
7177
+ import_zod8.z.object({
7178
+ state: import_zod8.z.literal("graded"),
7179
+ traceId: import_zod8.z.string(),
7180
+ value: import_zod8.z.number().finite(),
7181
+ note: import_zod8.z.string().optional()
7135
7182
  }),
7136
- import_zod7.z.object({ state: import_zod7.z.literal("errored"), traceId: import_zod7.z.string(), message: import_zod7.z.string() }),
7137
- import_zod7.z.object({
7138
- state: import_zod7.z.literal("ungraded"),
7139
- traceId: import_zod7.z.string(),
7140
- reason: import_zod7.z.enum(["not sampled", "trace unavailable", "content unreadable"])
7183
+ import_zod8.z.object({ state: import_zod8.z.literal("errored"), traceId: import_zod8.z.string(), message: import_zod8.z.string() }),
7184
+ import_zod8.z.object({
7185
+ state: import_zod8.z.literal("ungraded"),
7186
+ traceId: import_zod8.z.string(),
7187
+ reason: import_zod8.z.enum(["not sampled", "trace unavailable", "content unreadable"])
7141
7188
  })
7142
7189
  ])
7143
7190
  ),
7144
- failures: import_zod7.z.array(import_zod7.z.string()),
7145
- stats: import_zod7.z.object({
7146
- traceCount: import_zod7.z.number().int().nonnegative(),
7147
- hydratedCount: import_zod7.z.number().int().nonnegative(),
7148
- judgeCalls: import_zod7.z.number().int().nonnegative(),
7149
- durationMs: import_zod7.z.number().nonnegative(),
7150
- skippedRows: import_zod7.z.number().int().nonnegative(),
7151
- skippedReason: import_zod7.z.literal("no trace").nullable()
7191
+ failures: import_zod8.z.array(import_zod8.z.string()),
7192
+ stats: import_zod8.z.object({
7193
+ traceCount: import_zod8.z.number().int().nonnegative(),
7194
+ hydratedCount: import_zod8.z.number().int().nonnegative(),
7195
+ judgeCalls: import_zod8.z.number().int().nonnegative(),
7196
+ durationMs: import_zod8.z.number().nonnegative(),
7197
+ skippedRows: import_zod8.z.number().int().nonnegative(),
7198
+ skippedReason: import_zod8.z.literal("no trace").nullable()
7152
7199
  })
7153
7200
  });
7154
7201
  }
@@ -7163,8 +7210,9 @@ __export(index_exports, {
7163
7210
  DEFAULT_QUERY_URL: () => DEFAULT_QUERY_URL,
7164
7211
  DEFAULT_TRACE_WAIT_MS: () => DEFAULT_TRACE_WAIT_MS,
7165
7212
  EVAL_CORRELATION_ID_ATTRIBUTE: () => EVAL_CORRELATION_ID_ATTRIBUTE,
7166
- EvalDatasetCaseInputSchema: () => EvalDatasetCaseInputSchema,
7167
7213
  EvalDatasetPublishConflictError: () => EvalDatasetPublishConflictError,
7214
+ EvalDatasetRowInputSchema: () => EvalDatasetRowInputSchema,
7215
+ EvalPublishError: () => EvalPublishError,
7168
7216
  LocalBooleanVerdictSchema: () => LocalBooleanVerdictSchema,
7169
7217
  LocalNumberVerdictSchema: () => LocalNumberVerdictSchema,
7170
7218
  LocalScoreVerdictSchema: () => LocalScoreVerdictSchema,
@@ -7181,6 +7229,7 @@ __export(index_exports, {
7181
7229
  ReplayVerdictSchema: () => ReplayVerdictSchema,
7182
7230
  TraceSchema: () => TraceSchema,
7183
7231
  claimReplay: () => claimReplay,
7232
+ compareEvalRuns: () => compareEvalRuns,
7184
7233
  createEvalSuiteRun: () => createEvalSuiteRun,
7185
7234
  createReplay: () => createReplay,
7186
7235
  currentEvalScope: () => currentEvalScope,
@@ -7189,17 +7238,22 @@ __export(index_exports, {
7189
7238
  defineEvalSuite: () => defineEvalSuite,
7190
7239
  defineEvaluatorProgram: () => defineEvaluatorProgram,
7191
7240
  defineLocalEvaluator: () => defineLocalEvaluator,
7241
+ evaluateEvalRun: () => evaluateEvalRun,
7192
7242
  evaluatePortableEvaluator: () => evaluatePortableEvaluator,
7193
7243
  evaluateReplay: () => evaluateReplay,
7194
7244
  importPortableEvaluator: () => importPortableEvaluator,
7195
7245
  isEvalDataset: () => isEvalDataset,
7196
7246
  isEvalSuiteDefinition: () => isEvalSuiteDefinition,
7247
+ isPublishedEvalSuite: () => isPublishedEvalSuite,
7197
7248
  loadEvalSnapshot: () => loadEvalSnapshot,
7198
7249
  projectWorkshopTrace: () => projectWorkshopTrace,
7199
7250
  publishEvalDataset: () => publishEvalDataset,
7251
+ publishEvalSuite: () => publishEvalSuite,
7200
7252
  pullEval: () => pullEval,
7201
- readEvalDataset: () => readEvalDataset,
7253
+ readEvalDataset: () => readEvalDataset2,
7202
7254
  readEvalManifest: () => readEvalManifest,
7255
+ readEvalRun: () => readEvalRun,
7256
+ readEvaluator: () => readEvaluator,
7203
7257
  readReplay: () => readReplay,
7204
7258
  replay: () => replay,
7205
7259
  resolveDisableBatching: () => resolveDisableBatching,
@@ -7207,6 +7261,7 @@ __export(index_exports, {
7207
7261
  runReplay: () => runReplay,
7208
7262
  traceOutput: () => traceOutput,
7209
7263
  traceToolCalls: () => traceToolCalls,
7264
+ traceTools: () => traceTools,
7210
7265
  verifyPortableEvalArtifact: () => verifyPortableEvalArtifact,
7211
7266
  withEvalScope: () => withEvalScope
7212
7267
  });
@@ -12558,10 +12613,13 @@ var SignalEventSchema = external_exports.object({
12558
12613
  // package.json
12559
12614
  var package_default = {
12560
12615
  name: "raindrop-ai",
12561
- version: "0.6.0",
12616
+ version: "0.7.0",
12562
12617
  main: "dist/index.js",
12563
12618
  module: "dist/index.mjs",
12564
12619
  types: "dist/index.d.ts",
12620
+ bin: {
12621
+ "raindrop-evals": "dist/evals/cli.js"
12622
+ },
12565
12623
  license: "MIT",
12566
12624
  homepage: "https://www.raindrop.ai/docs/sdk/typescript/",
12567
12625
  bugs: {
@@ -12658,6 +12716,7 @@ var package_default = {
12658
12716
  tsup: {
12659
12717
  entry: [
12660
12718
  "src/index.ts",
12719
+ "src/evals/cli.ts",
12661
12720
  "src/tracing/index.ts",
12662
12721
  "src/otel/index.ts"
12663
12722
  ],
@@ -16449,7 +16508,7 @@ var EvalDatasetCaseInputsSchema = import_zod4.z.array(EvalDatasetCaseInputSchema
16449
16508
  if (seen.has(entry.id)) {
16450
16509
  context9.addIssue({
16451
16510
  code: import_zod4.z.ZodIssueCode.custom,
16452
- message: `Duplicate case id ${entry.id}`,
16511
+ message: `Duplicate row id ${entry.id}`,
16453
16512
  path: [index, "id"]
16454
16513
  });
16455
16514
  }
@@ -16829,6 +16888,34 @@ var WireReplayDetailSchema = import_zod5.z.object({
16829
16888
  replay: WireReplaySchema,
16830
16889
  rows: import_zod5.z.array(WireRowSchema)
16831
16890
  });
16891
+ var EvalRunDetailSchema = WireReplayDetailSchema.extend({
16892
+ replay: WireReplaySchema.extend({
16893
+ dataset_id: import_zod5.z.string().uuid(),
16894
+ dataset_version_id: import_zod5.z.string().uuid().nullable()
16895
+ }),
16896
+ eval_runs: import_zod5.z.array(
16897
+ import_zod5.z.object({
16898
+ id: import_zod5.z.string().uuid(),
16899
+ eval_id: import_zod5.z.string().uuid(),
16900
+ program_version: import_zod5.z.number().int().positive(),
16901
+ output_type: import_zod5.z.enum(["boolean", "score", "number"]),
16902
+ executed_by: import_zod5.z.enum(["hosted", "local"]),
16903
+ status: import_zod5.z.enum(["running", "completed", "failed"])
16904
+ })
16905
+ )
16906
+ });
16907
+ async function readReplayDetails(client, runId, options = {}) {
16908
+ var _a;
16909
+ const detail = await queryApi({
16910
+ client,
16911
+ queryUrl: (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL,
16912
+ path: `/v1/replays/${encodeURIComponent(runId)}`,
16913
+ method: "GET",
16914
+ schema: EvalRunDetailSchema
16915
+ });
16916
+ if (detail.replay.id !== runId) throw new Error("Raindrop returned a different run");
16917
+ return detail;
16918
+ }
16832
16919
  var WireReportSchema = import_zod5.z.object({ status: import_zod5.z.string() });
16833
16920
  var WireAttemptStartedSchema = import_zod5.z.object({
16834
16921
  status: import_zod5.z.literal("pending"),
@@ -17007,7 +17094,7 @@ function selectEvalDatasetRows(manifest, requestedRowIds) {
17007
17094
  if (ordered.length !== selected.size) {
17008
17095
  const available = new Set(manifest.cases.map((entry) => entry.id));
17009
17096
  const unknown = rowIds.find((rowId) => !available.has(rowId));
17010
- throw new Error(`Raindrop replay rowIds contains unknown case ${unknown != null ? unknown : "unknown"}`);
17097
+ throw new Error(`Raindrop replay rowIds contains unknown row ${unknown != null ? unknown : "unknown"}`);
17011
17098
  }
17012
17099
  return ordered;
17013
17100
  }
@@ -17075,7 +17162,7 @@ async function readEvalManifest(client, dataset, options) {
17075
17162
  });
17076
17163
  const caseIds = manifest.rows.map((row) => row.id);
17077
17164
  if (new Set(caseIds).size !== caseIds.length) {
17078
- throw new Error(`Raindrop eval manifest for ${normalized} contains duplicate case ids`);
17165
+ throw new Error(`Raindrop eval manifest for ${normalized} contains duplicate row ids`);
17079
17166
  }
17080
17167
  return {
17081
17168
  datasetId: manifest.dataset_id,
@@ -17115,7 +17202,7 @@ function validateEvalDatasetManifest(value) {
17115
17202
  }
17116
17203
  const caseIds = manifest.cases.map((entry) => entry.id);
17117
17204
  if (new Set(caseIds).size !== caseIds.length) {
17118
- throw new Error(`Raindrop eval dataset ${manifest.dataset.slug} contains duplicate case ids`);
17205
+ throw new Error(`Raindrop eval dataset ${manifest.dataset.slug} contains duplicate row ids`);
17119
17206
  }
17120
17207
  return manifest;
17121
17208
  }
@@ -17180,14 +17267,17 @@ async function runReplayRows(client, handle, options) {
17180
17267
  var _a3;
17181
17268
  const parent = (_a3 = import_api7.trace.getSpan(import_api7.context.active())) == null ? void 0 : _a3.spanContext();
17182
17269
  if (!parent) return options.run(visible);
17183
- return withReplayTraceDestination({
17184
- ...replayDestination,
17185
- parentSpanContext: {
17186
- traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
17187
- spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
17188
- eventId: parent.traceId
17189
- }
17190
- }, () => options.run(visible));
17270
+ return withReplayTraceDestination(
17271
+ {
17272
+ ...replayDestination,
17273
+ parentSpanContext: {
17274
+ traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
17275
+ spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
17276
+ eventId: parent.traceId
17277
+ }
17278
+ },
17279
+ () => options.run(visible)
17280
+ );
17191
17281
  }
17192
17282
  )
17193
17283
  )
@@ -17282,7 +17372,11 @@ async function readEvalCaseTracePairs(input) {
17282
17372
  while (reading.state === "pending" && Date.now() < deadlineAt) {
17283
17373
  await sleep(Math.min(TRACE_POLL_INTERVAL_MS, Math.max(0, deadlineAt - Date.now())));
17284
17374
  if (Date.now() >= deadlineAt) break;
17285
- reading = await readEvalCaseTracePair({ ...input, deadlineAt }, evidence, (_b = input.attempt) != null ? _b : 0);
17375
+ reading = await readEvalCaseTracePair(
17376
+ { ...input, deadlineAt },
17377
+ evidence,
17378
+ (_b = input.attempt) != null ? _b : 0
17379
+ );
17286
17380
  }
17287
17381
  if (reading.state !== "ready") {
17288
17382
  const detail = reading.state === "pending" ? `did not become complete within ${input.waitMs}ms` : reading.state === "expired" ? `expired at ${reading.expiredAt}` : reading.reason;
@@ -17333,7 +17427,13 @@ async function createLocalEvalReplaySession(client, options) {
17333
17427
  throw new Error("Raindrop local eval replay requires at least one evaluator");
17334
17428
  }
17335
17429
  const queryUrl = (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL;
17336
- await validateReferenceRequirements(client, options.dataset, options.rowIds, options.evaluators, queryUrl);
17430
+ await validateReferenceRequirements(
17431
+ client,
17432
+ options.dataset,
17433
+ options.rowIds,
17434
+ options.evaluators,
17435
+ queryUrl
17436
+ );
17337
17437
  const handle = await createReplay(client, {
17338
17438
  dataset: options.dataset,
17339
17439
  rowIds: options.rowIds,
@@ -17426,14 +17526,17 @@ var LocalEvalReplaySessionImpl = class {
17426
17526
  var _a2;
17427
17527
  const parent = (_a2 = import_api7.trace.getSpan(import_api7.context.active())) == null ? void 0 : _a2.spanContext();
17428
17528
  if (!parent) return input.run(row);
17429
- return withReplayTraceDestination({
17430
- ...replayDestination,
17431
- parentSpanContext: {
17432
- traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
17433
- spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
17434
- eventId: parent.traceId
17435
- }
17436
- }, () => input.run(row));
17529
+ return withReplayTraceDestination(
17530
+ {
17531
+ ...replayDestination,
17532
+ parentSpanContext: {
17533
+ traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
17534
+ spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
17535
+ eventId: parent.traceId
17536
+ }
17537
+ },
17538
+ () => input.run(row)
17539
+ );
17437
17540
  }
17438
17541
  )
17439
17542
  )
@@ -17659,7 +17762,9 @@ function reportReplay(client, queryUrl, replayId, body) {
17659
17762
  }
17660
17763
  async function validateReferenceRequirements(client, dataset, rowIds, evaluators, queryUrl) {
17661
17764
  const selected = new Set(selectEvalDatasetRows(dataset, rowIds));
17662
- const missing = dataset.cases.filter((entry) => selected.has(entry.id) && entry.referenceTraceSha256 === void 0);
17765
+ const missing = dataset.cases.filter(
17766
+ (entry) => selected.has(entry.id) && entry.referenceTraceSha256 === void 0
17767
+ );
17663
17768
  if (missing.length === 0) return;
17664
17769
  for (const evaluator of evaluators) {
17665
17770
  const requiresReference = typeof evaluator === "string" ? (await queryApi({
@@ -17671,7 +17776,9 @@ async function validateReferenceRequirements(client, dataset, rowIds, evaluators
17671
17776
  })).data.requiresReference : evaluator.requiresReference === true;
17672
17777
  if (requiresReference) {
17673
17778
  const name = typeof evaluator === "string" ? evaluator : evaluator.slug;
17674
- throw new Error(`Raindrop evaluator ${name} requires a reference trace. Missing references for cases: ${missing.map((entry) => entry.id).join(", ")}`);
17779
+ throw new Error(
17780
+ `Raindrop evaluator ${name} requires a reference trace. Missing references for rows: ${missing.map((entry) => entry.id).join(", ")}`
17781
+ );
17675
17782
  }
17676
17783
  }
17677
17784
  }
@@ -17680,7 +17787,8 @@ async function replay(client, options) {
17680
17787
  const evaluators = normalizeEvaluators(options.evaluators);
17681
17788
  const queryUrl = (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL;
17682
17789
  const dataset = typeof options.dataset === "string" ? await readEvalDataset(client, options.dataset, { queryUrl }) : options.dataset;
17683
- await validateReferenceRequirements(client, dataset, options.rowIds, evaluators, queryUrl);
17790
+ const rowIds = options.runId === void 0 ? options.rowIds : await queuedSuiteRows(client, options.runId, dataset, options.rowIds, queryUrl);
17791
+ await validateReferenceRequirements(client, dataset, rowIds, evaluators, queryUrl);
17684
17792
  const local = evaluators.flatMap(
17685
17793
  (evaluator) => typeof evaluator === "string" ? [] : [{ evaluator, verdicts: [], durationMs: 0 }]
17686
17794
  );
@@ -17692,7 +17800,7 @@ async function replay(client, options) {
17692
17800
  }
17693
17801
  )
17694
17802
  );
17695
- const handle = await createReplay(client, { ...options, dataset });
17803
+ const handle = options.runId === void 0 ? await createReplay(client, { ...options, dataset, rowIds }) : await claimSuiteRun(client, options.runId, dataset, rowIds != null ? rowIds : [], queryUrl);
17696
17804
  const execution = await runReplayRows(client, handle, options);
17697
17805
  const { reading } = execution;
17698
17806
  const doneRowIds = new Set(
@@ -17765,6 +17873,33 @@ async function replay(client, options) {
17765
17873
  evaluators: evaluatorResults.map(({ outcomes: _outcomes, ...evaluator }) => evaluator)
17766
17874
  };
17767
17875
  }
17876
+ async function queuedSuiteRows(client, runId, dataset, requestedRows, queryUrl) {
17877
+ const detail = await readReplayDetails(client, runId, { queryUrl });
17878
+ if (detail.replay.dataset_id !== dataset.dataset.id || detail.replay.dataset_version_id !== dataset.version.id) {
17879
+ throw new Error("Queued run dataset version does not match the suite");
17880
+ }
17881
+ if (detail.replay.status !== "queued") throw new Error("Only queued runs can be claimed");
17882
+ if (detail.replay.traces_expired) throw new Error("Queued run has expired");
17883
+ const rowIds = selectEvalDatasetRows(
17884
+ dataset,
17885
+ detail.rows.map((row) => row.row_id)
17886
+ );
17887
+ const requested = requestedRows && selectEvalDatasetRows(dataset, requestedRows);
17888
+ if (requested && (requested.length !== rowIds.length || requested.some((id, index) => id !== rowIds[index]))) {
17889
+ throw new Error("Queued run rows do not match the requested selection");
17890
+ }
17891
+ return rowIds;
17892
+ }
17893
+ async function claimSuiteRun(client, runId, dataset, rowIds, queryUrl) {
17894
+ const handle = await claimReplay(client, runId, { queryUrl });
17895
+ if (handle.replayId !== runId) throw new Error("Raindrop claimed a different run");
17896
+ if (!handle.ingestUrl) throw new Error("Run has already been claimed by another runner");
17897
+ assertSelectedRows(handle, rowIds);
17898
+ return {
17899
+ ...handle,
17900
+ datasetVersion: { id: dataset.version.id, fingerprint: dataset.version.fingerprint }
17901
+ };
17902
+ }
17768
17903
  async function runLocalEvaluators(evaluators, evalCase, row, result) {
17769
17904
  await Promise.all(
17770
17905
  evaluators.map(async (state2) => {
@@ -17882,11 +18017,11 @@ async function runReplayEvaluators(client, replayId, options) {
17882
18017
  const deadlineAt = Date.now() + waitMs;
17883
18018
  return Promise.all(
17884
18019
  startedRuns.map(async (started) => {
17885
- let run = await readEvalRun(client, queryUrl, started.runId, deadlineAt);
18020
+ let run = await readReplayEvaluatorRun(client, queryUrl, started.runId, deadlineAt);
17886
18021
  while (run.status === "running" && Date.now() < deadlineAt) {
17887
18022
  await sleep(Math.min(pollIntervalMs, Math.max(0, deadlineAt - Date.now())));
17888
18023
  if (Date.now() >= deadlineAt) break;
17889
- run = await readEvalRun(client, queryUrl, started.runId, deadlineAt);
18024
+ run = await readReplayEvaluatorRun(client, queryUrl, started.runId, deadlineAt);
17890
18025
  }
17891
18026
  if (run.status === "running") {
17892
18027
  throw new Error(
@@ -17986,7 +18121,7 @@ function replayResultRows(handle, reading, evaluators) {
17986
18121
  return replayRow;
17987
18122
  });
17988
18123
  }
17989
- async function readEvalRun(client, queryUrl, runId, deadlineAt) {
18124
+ async function readReplayEvaluatorRun(client, queryUrl, runId, deadlineAt) {
17990
18125
  const response = await queryApi({
17991
18126
  client,
17992
18127
  queryUrl,
@@ -18164,6 +18299,18 @@ function describeError(cause) {
18164
18299
  // src/evals/index.ts
18165
18300
  init_definition();
18166
18301
 
18302
+ // src/evals/rows.ts
18303
+ var EvalDatasetRowInputSchema = EvalDatasetCaseInputSchema;
18304
+ function datasetFromWire({ cases, ...dataset }) {
18305
+ return { ...dataset, rows: cases };
18306
+ }
18307
+ function datasetToWire({ rows, ...dataset }) {
18308
+ return { ...dataset, cases: rows };
18309
+ }
18310
+ async function readEvalDataset2(client, dataset, options) {
18311
+ return datasetFromWire(await readEvalDataset(client, dataset, options));
18312
+ }
18313
+
18167
18314
  // src/evals/dataset.ts
18168
18315
  var MAX_PUBLISH_RETRIES = 2;
18169
18316
  var RETRY_BASE_DELAY_MS = 100;
@@ -18177,21 +18324,14 @@ var EvalDatasetPublishConflictError = class extends Error {
18177
18324
  }
18178
18325
  };
18179
18326
  async function publishEvalDataset(client, options) {
18180
- var _a, _b, _c;
18327
+ var _a, _b;
18181
18328
  if (!((_a = client.connection.apiKey) == null ? void 0 : _a.trim())) {
18182
18329
  throw new Error("Raindrop eval dataset publication requires a Query SDK API key (apiKey)");
18183
18330
  }
18184
18331
  const queryUrl = credentialEndpoint((_b = options.queryUrl) != null ? _b : DEFAULT_QUERY_URL).href;
18185
18332
  const slug = EvalDatasetSlugSchema.parse(options.slug.trim());
18186
- const fingerprint = await evalDatasetVersionFingerprint(options.cases);
18187
- const body = PublishEvalDatasetVersionInputSchema.parse({
18188
- name: options.name,
18189
- requestKey: randomUUID(),
18190
- expectedCurrentVersionId: (_c = options.expectedCurrentVersionId) != null ? _c : null,
18191
- fingerprint,
18192
- cases: options.cases
18193
- });
18194
- validateUploadSizes(body);
18333
+ const body = await prepareEvalDatasetUpload(options);
18334
+ const fingerprint = body.fingerprint;
18195
18335
  const path = `/v1/eval-datasets/${encodeURIComponent(slug)}`;
18196
18336
  const response = await publishRequest({
18197
18337
  client,
@@ -18204,7 +18344,20 @@ async function publishEvalDataset(client, options) {
18204
18344
  if (manifest.dataset.slug !== slug || manifest.dataset.id !== manifest.version.datasetId || manifest.version.fingerprint !== fingerprint) {
18205
18345
  throw new Error(`Raindrop eval dataset ${slug} returned mismatched publication identity`);
18206
18346
  }
18207
- return manifest;
18347
+ return datasetFromWire(manifest);
18348
+ }
18349
+ async function prepareEvalDatasetUpload(options) {
18350
+ var _a;
18351
+ const fingerprint = await evalDatasetVersionFingerprint(options.rows);
18352
+ const body = PublishEvalDatasetVersionInputSchema.parse({
18353
+ name: options.name,
18354
+ requestKey: randomUUID(),
18355
+ expectedCurrentVersionId: (_a = options.expectedCurrentVersionId) != null ? _a : null,
18356
+ fingerprint,
18357
+ cases: options.rows
18358
+ });
18359
+ validateUploadSizes(body);
18360
+ return body;
18208
18361
  }
18209
18362
  function validateUploadSizes(body) {
18210
18363
  for (const [index, entry] of body.cases.entries()) {
@@ -18212,7 +18365,7 @@ function validateUploadSizes(body) {
18212
18365
  const bytes2 = encodedBytes(entry.referenceTrace);
18213
18366
  if (bytes2 > EVAL_DATASET_MAX_REFERENCE_BYTES) {
18214
18367
  throw new Error(
18215
- `Raindrop eval dataset case ${entry.id} at index ${index} exceeds ${EVAL_DATASET_MAX_REFERENCE_BYTES} reference bytes`
18368
+ `Raindrop eval dataset row ${entry.id} at index ${index} exceeds ${EVAL_DATASET_MAX_REFERENCE_BYTES} reference bytes`
18216
18369
  );
18217
18370
  }
18218
18371
  }
@@ -18295,39 +18448,278 @@ function sleep2(ms) {
18295
18448
  return new Promise((resolve) => setTimeout(resolve, ms));
18296
18449
  }
18297
18450
 
18451
+ // src/evals/publish.ts
18452
+ var import_zod6 = require("zod");
18453
+ init_definition();
18454
+
18455
+ // src/evals/runs.ts
18456
+ async function readEvalRun(client, runId, options = {}) {
18457
+ const detail = await readReplayDetails(client, runId, options);
18458
+ if (detail.eval_runs.length >= 200) {
18459
+ throw new Error("Run evaluation history exceeds the read limit");
18460
+ }
18461
+ const evaluators = await Promise.all(
18462
+ detail.eval_runs.map(async (evaluation) => {
18463
+ var _a;
18464
+ const result = await readReplayEvaluatorRun(
18465
+ client,
18466
+ (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL,
18467
+ evaluation.id
18468
+ );
18469
+ if (result.id !== evaluation.id) throw new Error("Raindrop returned a different evaluation");
18470
+ return {
18471
+ ...result,
18472
+ evaluator: result.evalSlug,
18473
+ evaluatorId: evaluation.eval_id,
18474
+ programVersion: evaluation.program_version,
18475
+ executedBy: evaluation.executed_by
18476
+ };
18477
+ })
18478
+ );
18479
+ return {
18480
+ runId: detail.replay.id,
18481
+ datasetId: detail.replay.dataset_id,
18482
+ datasetVersionId: detail.replay.dataset_version_id,
18483
+ status: detail.replay.status,
18484
+ counts: detail.replay.counts,
18485
+ tracesExpireAt: detail.replay.traces_expire_at,
18486
+ tracesExpired: detail.replay.traces_expired,
18487
+ rows: detail.rows.map((row) => ({
18488
+ rowId: row.row_id,
18489
+ attempt: row.attempt,
18490
+ status: row.status,
18491
+ traceId: row.trace_id,
18492
+ error: row.error
18493
+ })),
18494
+ evaluators
18495
+ };
18496
+ }
18497
+ async function evaluateEvalRun(client, runId, options) {
18498
+ return { runId, evaluators: await evaluateReplay(client, runId, options) };
18499
+ }
18500
+
18501
+ // src/evals/publish.ts
18502
+ var EvaluatorIdentitySchema = import_zod6.z.object({
18503
+ id: import_zod6.z.string().uuid(),
18504
+ slug: import_zod6.z.string(),
18505
+ programVersion: import_zod6.z.number().int().positive(),
18506
+ outputType: import_zod6.z.enum(["boolean", "score", "number"])
18507
+ });
18508
+ var EvaluatorSourceSchema = EvaluatorIdentitySchema.extend({
18509
+ name: import_zod6.z.string(),
18510
+ description: import_zod6.z.string().nullable(),
18511
+ intent: import_zod6.z.string(),
18512
+ kind: import_zod6.z.enum(["judge", "deterministic"]),
18513
+ programSource: import_zod6.z.string(),
18514
+ rules: import_zod6.z.array(import_zod6.z.string())
18515
+ });
18516
+ var EvalPublishError = class extends Error {
18517
+ constructor(status, message) {
18518
+ super(message);
18519
+ this.status = status;
18520
+ this.name = "EvalPublishError";
18521
+ }
18522
+ };
18523
+ var ComparisonRunSchema = import_zod6.z.object({
18524
+ id: import_zod6.z.string().uuid(),
18525
+ programVersion: import_zod6.z.number().int().positive()
18526
+ });
18527
+ var ComparisonLinkSchema = import_zod6.z.object({
18528
+ url: import_zod6.z.string().url(),
18529
+ first: ComparisonRunSchema,
18530
+ second: ComparisonRunSchema
18531
+ });
18532
+ async function compareEvalRuns(client, options) {
18533
+ const [first, second] = await Promise.all([
18534
+ readEvalRun(client, options.beforeRunId, options),
18535
+ readEvalRun(client, options.afterRunId, options)
18536
+ ]);
18537
+ const candidates = first.evaluators.filter(
18538
+ (entry) => (!options.evaluator || entry.evaluator === options.evaluator) && second.evaluators.some((other) => other.evaluatorId === entry.evaluatorId)
18539
+ );
18540
+ const selected = candidates[0];
18541
+ if (!selected || candidates.length !== 1) {
18542
+ throw new Error(
18543
+ "Comparison requires one shared evaluator; specify evaluator when comparing a suite"
18544
+ );
18545
+ }
18546
+ const matches = second.evaluators.filter((entry) => entry.evaluatorId === selected.evaluatorId);
18547
+ const counterpart = matches[0];
18548
+ if (!counterpart || matches.length !== 1) {
18549
+ throw new Error("Comparison is ambiguous because this evaluator graded the run more than once");
18550
+ }
18551
+ if (selected.status === "running" || counterpart.status === "running") {
18552
+ throw new Error("Wait for both evaluations to finish before comparing");
18553
+ }
18554
+ const before = selected.id;
18555
+ const after = counterpart.id;
18556
+ const { data } = import_zod6.z.object({ data: ComparisonLinkSchema }).parse(await evalRequest(client, `/v1/eval-runs/${before}/compare/${after}`, options));
18557
+ if (data.first.id !== before || data.second.id !== after) {
18558
+ throw new Error("Raindrop returned a different comparison pair");
18559
+ }
18560
+ return {
18561
+ ...data,
18562
+ beforeRunId: first.runId,
18563
+ afterRunId: second.runId,
18564
+ evaluator: selected.evaluator,
18565
+ sameEvaluatorVersion: data.first.programVersion === data.second.programVersion
18566
+ };
18567
+ }
18568
+ async function readEvaluator(client, slug, options = {}) {
18569
+ const { data } = import_zod6.z.object({ data: EvaluatorSourceSchema }).parse(await evalRequest(client, `/v1/evals/${encodeURIComponent(slug)}`, options));
18570
+ if (data.slug !== slug) throw new Error("Raindrop returned a different evaluator slug");
18571
+ return defineEvaluatorProgram({
18572
+ slug: data.slug,
18573
+ name: data.name,
18574
+ output: data.outputType,
18575
+ scope: "batch",
18576
+ executionMode: data.kind,
18577
+ source: data.programSource,
18578
+ description: data.description,
18579
+ intent: data.intent,
18580
+ rules: data.rules,
18581
+ expected: { evalId: data.id, programVersion: data.programVersion }
18582
+ });
18583
+ }
18584
+ async function publishEvalSuite(client, definition, options = {}) {
18585
+ const suite = defineEvalSuite(definition);
18586
+ const localDataset = isEvalDataset(suite.dataset) ? suite.dataset : void 0;
18587
+ const slug = EvalDatasetSlugSchema.parse(localDataset ? localDataset.id : suite.dataset);
18588
+ const rows = localDataset == null ? void 0 : localDataset.rows.map((row) => {
18589
+ if (row.output !== null) {
18590
+ throw new Error(
18591
+ "Publish reference evidence with publishEvalDataset first; a local row output is not a reference trace"
18592
+ );
18593
+ }
18594
+ return EvalDatasetCaseInputSchema.parse({
18595
+ id: row.id,
18596
+ name: row.name,
18597
+ input: row.input,
18598
+ properties: row.properties
18599
+ });
18600
+ });
18601
+ if (localDataset && rows) {
18602
+ await prepareEvalDatasetUpload({
18603
+ slug,
18604
+ name: localDataset.name,
18605
+ rows,
18606
+ expectedCurrentVersionId: options.expectedCurrentDatasetVersionId
18607
+ });
18608
+ }
18609
+ const entries = suite.evaluators.map((entry) => {
18610
+ if (typeof entry.evaluator === "string") return entry;
18611
+ if ("kind" in entry.evaluator && entry.evaluator.kind === "program") {
18612
+ if (!entry.evaluator.expected) {
18613
+ throw new Error(
18614
+ "Evaluator source publication is not supported. Create the evaluator in Raindrop and reference its slug."
18615
+ );
18616
+ }
18617
+ return { ...entry, evaluator: defineEvaluatorProgram(entry.evaluator) };
18618
+ }
18619
+ return entry;
18620
+ });
18621
+ const evaluators = [];
18622
+ for (const entry of entries) {
18623
+ if (typeof entry.evaluator === "string") {
18624
+ evaluators.push({
18625
+ ...entry,
18626
+ evaluator: await readEvaluator(client, entry.evaluator, options)
18627
+ });
18628
+ } else {
18629
+ evaluators.push(entry);
18630
+ }
18631
+ }
18632
+ const dataset = localDataset && rows ? await publishEvalDataset(client, {
18633
+ slug,
18634
+ name: localDataset.name,
18635
+ rows,
18636
+ queryUrl: options.queryUrl,
18637
+ expectedCurrentVersionId: options.expectedCurrentDatasetVersionId
18638
+ }) : await readEvalDataset2(client, slug, {
18639
+ queryUrl: options.queryUrl,
18640
+ versionId: suite.datasetVersionId
18641
+ });
18642
+ return defineEvalSuite({
18643
+ ...suite,
18644
+ dataset: slug,
18645
+ datasetVersionId: dataset.version.id,
18646
+ evaluators
18647
+ });
18648
+ }
18649
+ async function evalRequest(client, path, options) {
18650
+ var _a, _b;
18651
+ const apiKey = (_a = client.connection.apiKey) == null ? void 0 : _a.trim();
18652
+ if (!apiKey) throw new Error("Raindrop evals require a Query SDK API key (apiKey)");
18653
+ const base = credentialEndpoint((_b = options.queryUrl) != null ? _b : DEFAULT_QUERY_URL).href.replace(/\/+$/, "");
18654
+ const headers = {
18655
+ Authorization: `Bearer ${apiKey}`,
18656
+ "Content-Type": "application/json"
18657
+ };
18658
+ if (client.connection.projectId !== void 0)
18659
+ headers["X-Raindrop-Project-Id"] = client.connection.projectId;
18660
+ const controller = new AbortController();
18661
+ const timeout = setTimeout(() => controller.abort(), 3e4);
18662
+ try {
18663
+ const response = await runWithTracingSuppressed(
18664
+ () => fetch(`${base}${path}`, {
18665
+ method: "GET",
18666
+ headers,
18667
+ redirect: "error",
18668
+ signal: controller.signal
18669
+ })
18670
+ );
18671
+ const body = await response.text();
18672
+ if (!response.ok)
18673
+ throw new EvalPublishError(
18674
+ response.status,
18675
+ `Raindrop GET ${path}: ${response.status} ${body}`
18676
+ );
18677
+ return JSON.parse(body);
18678
+ } finally {
18679
+ clearTimeout(timeout);
18680
+ }
18681
+ }
18682
+
18683
+ // src/evals/trace-tools.ts
18684
+ init_trace();
18685
+ var traceTools = Object.freeze({
18686
+ getOutput: traceOutput,
18687
+ getToolCalls: traceToolCalls
18688
+ });
18689
+
18298
18690
  // src/evals/index.ts
18299
18691
  init_trace();
18300
18692
  init_portable();
18301
18693
 
18302
18694
  // src/evals/pull.ts
18303
- var import_zod8 = require("zod");
18695
+ var import_zod9 = require("zod");
18304
18696
  init_definition();
18305
18697
  init_portable();
18306
18698
  var SNAPSHOT_VERSION = 1;
18307
- var PulledEvaluatorSchema = import_zod8.z.object({
18308
- slug: import_zod8.z.string().min(1),
18309
- name: import_zod8.z.string().min(1),
18310
- expectedSha256: import_zod8.z.string().regex(/^[a-f0-9]{64}$/),
18311
- artifact: import_zod8.z.unknown()
18699
+ var PulledEvaluatorSchema = import_zod9.z.object({
18700
+ slug: import_zod9.z.string().min(1),
18701
+ name: import_zod9.z.string().min(1),
18702
+ expectedSha256: import_zod9.z.string().regex(/^[a-f0-9]{64}$/),
18703
+ artifact: import_zod9.z.unknown()
18312
18704
  });
18313
- var EvalSnapshotSchema = import_zod8.z.object({
18314
- schemaVersion: import_zod8.z.literal(SNAPSHOT_VERSION),
18315
- pulledAt: import_zod8.z.string().datetime(),
18316
- dataset: import_zod8.z.object({
18317
- reference: import_zod8.z.string().min(1),
18318
- datasetId: import_zod8.z.string().uuid(),
18319
- fingerprint: import_zod8.z.string().regex(/^[a-f0-9]{64}$/),
18320
- rows: import_zod8.z.array(
18321
- import_zod8.z.object({
18322
- id: import_zod8.z.string(),
18323
- name: import_zod8.z.string(),
18324
- input: import_zod8.z.string().nullable(),
18325
- output: import_zod8.z.string().nullable(),
18326
- properties: import_zod8.z.record(import_zod8.z.string())
18705
+ var EvalSnapshotSchema = import_zod9.z.object({
18706
+ schemaVersion: import_zod9.z.literal(SNAPSHOT_VERSION),
18707
+ pulledAt: import_zod9.z.string().datetime(),
18708
+ dataset: import_zod9.z.object({
18709
+ reference: import_zod9.z.string().min(1),
18710
+ datasetId: import_zod9.z.string().uuid(),
18711
+ fingerprint: import_zod9.z.string().regex(/^[a-f0-9]{64}$/),
18712
+ rows: import_zod9.z.array(
18713
+ import_zod9.z.object({
18714
+ id: import_zod9.z.string(),
18715
+ name: import_zod9.z.string(),
18716
+ input: import_zod9.z.string().nullable(),
18717
+ output: import_zod9.z.string().nullable(),
18718
+ properties: import_zod9.z.record(import_zod9.z.string())
18327
18719
  })
18328
18720
  )
18329
18721
  }),
18330
- evaluators: import_zod8.z.array(PulledEvaluatorSchema)
18722
+ evaluators: import_zod9.z.array(PulledEvaluatorSchema)
18331
18723
  });
18332
18724
  async function pullEval(client, options) {
18333
18725
  var _a;
@@ -18392,11 +18784,11 @@ async function pullEvaluator(client, queryUrl, rawSlug) {
18392
18784
  `Raindrop evaluator ${rawSlug} pull failed with ${response.status}: ${await response.text()}`
18393
18785
  );
18394
18786
  }
18395
- const body = import_zod8.z.object({
18396
- data: import_zod8.z.object({
18397
- slug: import_zod8.z.string(),
18398
- name: import_zod8.z.string(),
18399
- artifact: import_zod8.z.unknown().nullable()
18787
+ const body = import_zod9.z.object({
18788
+ data: import_zod9.z.object({
18789
+ slug: import_zod9.z.string(),
18790
+ name: import_zod9.z.string(),
18791
+ artifact: import_zod9.z.unknown().nullable()
18400
18792
  })
18401
18793
  }).parse(await response.json());
18402
18794
  if (!body.data.artifact)
@@ -18425,7 +18817,7 @@ async function materializeSnapshot(snapshotPath, raw, judges) {
18425
18817
  id: snapshot.dataset.datasetId,
18426
18818
  name: snapshot.dataset.reference,
18427
18819
  version: snapshot.dataset.fingerprint,
18428
- cases: snapshot.dataset.rows,
18820
+ rows: snapshot.dataset.rows,
18429
18821
  remote: {
18430
18822
  reference: snapshot.dataset.reference,
18431
18823
  datasetId: snapshot.dataset.datasetId,
@@ -18533,46 +18925,46 @@ function outputMismatch(name, output) {
18533
18925
  init_portable();
18534
18926
 
18535
18927
  // src/evals/workshop.ts
18536
- var import_zod9 = require("zod");
18537
- var WorkshopRunSchema = import_zod9.z.object({
18538
- id: import_zod9.z.string(),
18539
- event_id: import_zod9.z.string().nullable(),
18540
- name: import_zod9.z.string().nullable(),
18541
- event_name: import_zod9.z.string().nullable(),
18542
- user_id: import_zod9.z.string().nullable(),
18543
- convo_id: import_zod9.z.string().nullable(),
18544
- started_at: import_zod9.z.number().nullable(),
18545
- last_updated_at: import_zod9.z.number().nullable(),
18546
- metadata: import_zod9.z.string().nullable(),
18547
- model: import_zod9.z.string().nullable(),
18548
- finished: import_zod9.z.number().nullable()
18928
+ var import_zod10 = require("zod");
18929
+ var WorkshopRunSchema = import_zod10.z.object({
18930
+ id: import_zod10.z.string(),
18931
+ event_id: import_zod10.z.string().nullable(),
18932
+ name: import_zod10.z.string().nullable(),
18933
+ event_name: import_zod10.z.string().nullable(),
18934
+ user_id: import_zod10.z.string().nullable(),
18935
+ convo_id: import_zod10.z.string().nullable(),
18936
+ started_at: import_zod10.z.number().nullable(),
18937
+ last_updated_at: import_zod10.z.number().nullable(),
18938
+ metadata: import_zod10.z.string().nullable(),
18939
+ model: import_zod10.z.string().nullable(),
18940
+ finished: import_zod10.z.number().nullable()
18549
18941
  });
18550
- var WorkshopSpanSchema = import_zod9.z.object({
18551
- id: import_zod9.z.string(),
18552
- run_id: import_zod9.z.string(),
18553
- parent_span_id: import_zod9.z.string().nullable(),
18554
- name: import_zod9.z.string(),
18555
- span_type: import_zod9.z.string().nullable(),
18556
- status: import_zod9.z.string().nullable(),
18557
- input_payload: import_zod9.z.string().nullable(),
18558
- output_payload: import_zod9.z.string().nullable(),
18559
- start_time_ms: import_zod9.z.number().nullable(),
18560
- end_time_ms: import_zod9.z.number().nullable(),
18561
- duration_ms: import_zod9.z.number().nullable(),
18562
- model: import_zod9.z.string().nullable(),
18563
- provider: import_zod9.z.string().nullable(),
18564
- input_tokens: import_zod9.z.number().nullable(),
18565
- output_tokens: import_zod9.z.number().nullable(),
18566
- attributes: import_zod9.z.string().nullable()
18942
+ var WorkshopSpanSchema = import_zod10.z.object({
18943
+ id: import_zod10.z.string(),
18944
+ run_id: import_zod10.z.string(),
18945
+ parent_span_id: import_zod10.z.string().nullable(),
18946
+ name: import_zod10.z.string(),
18947
+ span_type: import_zod10.z.string().nullable(),
18948
+ status: import_zod10.z.string().nullable(),
18949
+ input_payload: import_zod10.z.string().nullable(),
18950
+ output_payload: import_zod10.z.string().nullable(),
18951
+ start_time_ms: import_zod10.z.number().nullable(),
18952
+ end_time_ms: import_zod10.z.number().nullable(),
18953
+ duration_ms: import_zod10.z.number().nullable(),
18954
+ model: import_zod10.z.string().nullable(),
18955
+ provider: import_zod10.z.string().nullable(),
18956
+ input_tokens: import_zod10.z.number().nullable(),
18957
+ output_tokens: import_zod10.z.number().nullable(),
18958
+ attributes: import_zod10.z.string().nullable()
18567
18959
  });
18568
- var WorkshopTraceEnvelopeSchema = import_zod9.z.object({
18569
- status: import_zod9.z.literal("complete"),
18570
- correlationId: import_zod9.z.string(),
18571
- traceId: import_zod9.z.string(),
18572
- rootSpanId: import_zod9.z.string(),
18573
- spanCount: import_zod9.z.number().int().positive(),
18960
+ var WorkshopTraceEnvelopeSchema = import_zod10.z.object({
18961
+ status: import_zod10.z.literal("complete"),
18962
+ correlationId: import_zod10.z.string(),
18963
+ traceId: import_zod10.z.string(),
18964
+ rootSpanId: import_zod10.z.string(),
18965
+ spanCount: import_zod10.z.number().int().positive(),
18574
18966
  run: WorkshopRunSchema,
18575
- spans: import_zod9.z.array(WorkshopSpanSchema).min(1)
18967
+ spans: import_zod10.z.array(WorkshopSpanSchema).min(1)
18576
18968
  });
18577
18969
  async function readWorkshopTrace(input) {
18578
18970
  if (!Number.isFinite(input.waitMs) || input.waitMs < 0)
@@ -18658,9 +19050,12 @@ function serializableOutput(value) {
18658
19050
  var DEFAULT_WORKSHOP_TRACE_WAIT_MS = 3e4;
18659
19051
  function createEvalSuiteRun(client, definition, options) {
18660
19052
  const validated = validateDefinition(definition);
19053
+ if (validated.datasetVersionId && "selection" in options && options.selection.dataset.version.id !== validated.datasetVersionId) {
19054
+ throw new Error("Selected dataset version does not match the published suite");
19055
+ }
18661
19056
  if (options.destination.kind === "raindrop") {
18662
19057
  if (!("selection" in options)) {
18663
- throw new Error("Raindrop remote eval sessions require selected cases");
19058
+ throw new Error("Raindrop remote eval sessions require selected rows");
18664
19059
  }
18665
19060
  if (typeof validated.dataset !== "string") {
18666
19061
  throw new Error("Raindrop remote eval sessions require a dataset slug");
@@ -18675,16 +19070,16 @@ function createEvalSuiteRun(client, definition, options) {
18675
19070
  return typeof evaluator === "string" ? [] : [evaluator];
18676
19071
  });
18677
19072
  if (localEvaluators.length !== validated.evaluators.length) {
18678
- throw new Error("Raindrop case-scoped eval sessions support local evaluators only");
19073
+ throw new Error("Raindrop row-scoped eval sessions support local evaluators only");
18679
19074
  }
18680
19075
  const selection = validateRemoteSelection(validated.dataset, options.selection);
18681
- if (!selection) throw new Error("Raindrop remote eval sessions require selected cases");
18682
- return new RaindropCaseEvalRun(
19076
+ if (!selection) throw new Error("Raindrop remote eval sessions require selected rows");
19077
+ return new RaindropRowEvalRun(
18683
19078
  client,
18684
19079
  validated,
18685
19080
  options.destination,
18686
19081
  options.selection.dataset,
18687
- selection.caseIds,
19082
+ selection.rowIds,
18688
19083
  localEvaluators
18689
19084
  );
18690
19085
  }
@@ -18709,34 +19104,56 @@ function createEvalSuiteRun(client, definition, options) {
18709
19104
  options.destination
18710
19105
  );
18711
19106
  }
18712
- async function runEvalSuite(client, definition, options) {
18713
- var _a, _b, _c, _d;
18714
- const validated = validateDefinition(definition);
18715
- if (options === void 0) throw new Error("Raindrop runEvalSuite requires an explicit destination");
18716
- if (options.destination.kind === "workshop") {
18717
- const session = createEvalSuiteRun(client, validated, { destination: options.destination });
19107
+ async function runEvalSuite(client, definition, options = {}) {
19108
+ var _a, _b, _c, _d, _e, _f, _g, _h;
19109
+ let validated = validateDefinition(definition);
19110
+ const destination = (_a = options.destination) != null ? _a : { kind: "raindrop" };
19111
+ if (validated.datasetVersionId && ((_b = options.selection) == null ? void 0 : _b.dataset) && options.selection.dataset.version.id !== validated.datasetVersionId) {
19112
+ throw new Error("Selected dataset version does not match the published suite");
19113
+ }
19114
+ if (destination.kind === "workshop") {
19115
+ if (options.runId !== void 0) throw new Error("Queued runs are only supported in Raindrop");
19116
+ const session = createEvalSuiteRun(client, validated, { destination });
18718
19117
  await session.run({ selection: options.selection, attempt: options.attempt });
18719
19118
  return session.finish();
18720
19119
  }
18721
- if (isCaseScopedLocalEval(validated)) {
19120
+ const alreadyPublished = isPublishedEvalSuite(validated);
19121
+ if (typeof validated.dataset === "string" && ((_c = options.selection) == null ? void 0 : _c.dataset)) {
19122
+ validateRemoteSelection(validated.dataset, options.selection);
19123
+ validated = defineEvalSuite({
19124
+ ...validated,
19125
+ datasetVersionId: options.selection.dataset.version.id
19126
+ });
19127
+ }
19128
+ if (!alreadyPublished) {
19129
+ validated = await publishEvalSuite(client, validated, {
19130
+ queryUrl: destination.queryUrl,
19131
+ expectedCurrentDatasetVersionId: options.expectedCurrentDatasetVersionId
19132
+ });
19133
+ }
19134
+ if (((_d = options.selection) == null ? void 0 : _d.dataset) && options.selection.dataset.version.id !== validated.datasetVersionId) {
19135
+ throw new Error("Selected dataset version does not match the published suite");
19136
+ }
19137
+ if (isRowScopedLocalEval(validated) && options.runId === void 0) {
18722
19138
  if (typeof validated.dataset !== "string") {
18723
19139
  throw new Error("Raindrop remote execution requires an eval dataset slug");
18724
19140
  }
18725
- const dataset = (_b = (_a = options.selection) == null ? void 0 : _a.dataset) != null ? _b : await readEvalDataset(client, validated.dataset, {
18726
- queryUrl: options.destination.queryUrl
19141
+ const dataset = (_f = (_e = options.selection) == null ? void 0 : _e.dataset) != null ? _f : await readEvalDataset2(client, validated.dataset, {
19142
+ queryUrl: destination.queryUrl,
19143
+ versionId: validated.datasetVersionId
18727
19144
  });
18728
19145
  const selection = validateRemoteSelection(validated.dataset, {
18729
19146
  dataset,
18730
- caseIds: (_d = (_c = options.selection) == null ? void 0 : _c.caseIds) != null ? _d : dataset.cases.map((entry) => entry.id)
19147
+ rowIds: (_h = (_g = options.selection) == null ? void 0 : _g.rowIds) != null ? _h : dataset.rows.map((entry) => entry.id)
18731
19148
  });
18732
- if (!selection) throw new Error("Raindrop remote eval requires selected cases");
19149
+ if (!selection) throw new Error("Raindrop remote eval requires selected rows");
18733
19150
  const session = createEvalSuiteRun(client, validated, {
18734
- destination: options.destination,
18735
- selection: { dataset, caseIds: selection.caseIds }
19151
+ destination,
19152
+ selection: { dataset, rowIds: selection.rowIds }
18736
19153
  });
18737
19154
  const attempts = await Promise.allSettled(
18738
- selection.caseIds.map(
18739
- (caseId) => session.run({ selection: { dataset, caseIds: [caseId] }, attempt: 0 })
19155
+ selection.rowIds.map(
19156
+ (rowId) => session.run({ selection: { dataset, rowIds: [rowId] }, attempt: 0 })
18740
19157
  )
18741
19158
  );
18742
19159
  const failures = attempts.filter(
@@ -18749,7 +19166,7 @@ async function runEvalSuite(client, definition, options) {
18749
19166
  if (failures.length > 0) {
18750
19167
  throw new AggregateError(
18751
19168
  [...failures.map((failure) => failure.reason), cause],
18752
- "Raindrop eval case execution and run finalization failed"
19169
+ "Raindrop eval row execution and run finalization failed"
18753
19170
  );
18754
19171
  }
18755
19172
  throw cause;
@@ -18759,33 +19176,33 @@ async function runEvalSuite(client, definition, options) {
18759
19176
  if (failures.length > 1) {
18760
19177
  throw new AggregateError(
18761
19178
  failures.map((failure) => failure.reason),
18762
- "Multiple Raindrop eval cases failed"
19179
+ "Multiple Raindrop eval rows failed"
18763
19180
  );
18764
19181
  }
18765
19182
  return finished;
18766
19183
  }
18767
- return runRaindropEval(client, validated, options.destination, options);
19184
+ return runRaindropEval(client, validated, destination, options);
18768
19185
  }
18769
- function isCaseScopedLocalEval(definition) {
19186
+ function isRowScopedLocalEval(definition) {
18770
19187
  return definition.evaluators.every(
18771
- (entry) => typeof entry.evaluator !== "string" && entry.evaluator.scope === "case" && !("kind" in entry.evaluator)
19188
+ (entry) => typeof entry.evaluator !== "string" && entry.evaluator.scope === "row" && !("kind" in entry.evaluator)
18772
19189
  );
18773
19190
  }
18774
- var RaindropCaseEvalRun = class {
18775
- constructor(client, definition, destination, dataset, caseIds, evaluators) {
19191
+ var RaindropRowEvalRun = class {
19192
+ constructor(client, definition, destination, dataset, rowIds, evaluators) {
18776
19193
  this.client = client;
18777
19194
  this.definition = definition;
18778
19195
  this.destination = destination;
18779
19196
  this.dataset = dataset;
18780
- this.caseIds = caseIds;
19197
+ this.rowIds = rowIds;
18781
19198
  this.evaluators = evaluators;
18782
19199
  this.id = randomId();
18783
19200
  this.active = /* @__PURE__ */ new Set();
18784
19201
  this.completed = [];
18785
19202
  this.attempts = /* @__PURE__ */ new Set();
18786
19203
  this.state = "open";
18787
- this.runningCases = 0;
18788
- this.caseWaiters = [];
19204
+ this.runningRows = 0;
19205
+ this.rowWaiters = [];
18789
19206
  }
18790
19207
  async run(options = {}) {
18791
19208
  if (this.state !== "open") throw new Error(`Raindrop eval run ${this.id} is already finishing`);
@@ -18805,24 +19222,24 @@ var RaindropCaseEvalRun = class {
18805
19222
  }
18806
19223
  async runOnce(options) {
18807
19224
  var _a, _b;
18808
- const selected = (_a = options.selection) == null ? void 0 : _a.caseIds;
19225
+ const selected = (_a = options.selection) == null ? void 0 : _a.rowIds;
18809
19226
  if (!selected || selected.length !== 1) {
18810
- throw new Error("Raindrop case-scoped eval session run requires exactly one case");
19227
+ throw new Error("Raindrop row-scoped eval session run requires exactly one row");
18811
19228
  }
18812
- const caseId = selected[0];
18813
- if (caseId === void 0 || !this.caseIds.includes(caseId)) {
18814
- throw new Error(`Raindrop eval ${this.definition.name} did not select a planned case`);
19229
+ const rowId = selected[0];
19230
+ if (rowId === void 0 || !this.rowIds.includes(rowId)) {
19231
+ throw new Error(`Raindrop eval ${this.definition.name} did not select a planned row`);
18815
19232
  }
18816
19233
  const attempt = (_b = options.attempt) != null ? _b : 0;
18817
- const attemptKey = `${caseId}:${attempt}`;
19234
+ const attemptKey = `${rowId}:${attempt}`;
18818
19235
  if (this.attempts.has(attemptKey)) {
18819
- throw new Error(`Raindrop eval case ${caseId} attempt ${attempt} already ran`);
19236
+ throw new Error(`Raindrop eval row ${rowId} attempt ${attempt} already ran`);
18820
19237
  }
18821
19238
  this.attempts.add(attemptKey);
18822
- return this.withCasePermit(async () => {
19239
+ return this.withRowPermit(async () => {
18823
19240
  const replay2 = await this.getReplay();
18824
19241
  const attemptResult = await replay2.run({
18825
- rowId: caseId,
19242
+ rowId,
18826
19243
  attempt,
18827
19244
  run: this.definition.run,
18828
19245
  traceWaitMs: this.definition.traceWaitMs
@@ -18846,7 +19263,7 @@ var RaindropCaseEvalRun = class {
18846
19263
  };
18847
19264
  this.completed.push(result);
18848
19265
  return {
18849
- replayId: replay2.replayId,
19266
+ runId: replay2.replayId,
18850
19267
  name: this.definition.name,
18851
19268
  status: "running",
18852
19269
  counts: {
@@ -18858,7 +19275,7 @@ var RaindropCaseEvalRun = class {
18858
19275
  tracesExpireAt: null,
18859
19276
  tracesExpired: false,
18860
19277
  evaluators: replayEvaluatorsFromVerdicts(attemptResult.verdicts),
18861
- cases: [result]
19278
+ rows: [result]
18862
19279
  };
18863
19280
  });
18864
19281
  }
@@ -18866,20 +19283,21 @@ var RaindropCaseEvalRun = class {
18866
19283
  await Promise.allSettled([...this.active]);
18867
19284
  const replay2 = await this.getReplay();
18868
19285
  const finished = await replay2.finish();
18869
- const { rows: _rows, ...reading } = finished.reading;
19286
+ const { rows: _rows, replayId, ...reading } = finished.reading;
18870
19287
  this.state = "finished";
18871
19288
  return {
18872
19289
  ...reading,
19290
+ runId: replayId,
18873
19291
  name: this.definition.name,
18874
19292
  evaluators: finished.evaluators,
18875
- cases: this.sortedResults()
19293
+ rows: this.sortedResults()
18876
19294
  };
18877
19295
  }
18878
19296
  getReplay() {
18879
19297
  if (!this.replayPromise) {
18880
19298
  this.replayPromise = createLocalEvalReplaySession(this.client, {
18881
- dataset: this.dataset,
18882
- rowIds: this.caseIds,
19299
+ dataset: datasetToWire(this.dataset),
19300
+ rowIds: this.rowIds,
18883
19301
  evaluators: this.evaluators,
18884
19302
  queryUrl: this.destination.queryUrl,
18885
19303
  replayIngestUrl: this.destination.replayIngestUrl
@@ -18887,28 +19305,28 @@ var RaindropCaseEvalRun = class {
18887
19305
  }
18888
19306
  return this.replayPromise;
18889
19307
  }
18890
- async withCasePermit(run) {
19308
+ async withRowPermit(run) {
18891
19309
  var _a;
18892
19310
  const limit = (_a = this.definition.concurrency) != null ? _a : 4;
18893
- if (this.runningCases >= limit) {
18894
- await new Promise((resolve) => this.caseWaiters.push(resolve));
19311
+ if (this.runningRows >= limit) {
19312
+ await new Promise((resolve) => this.rowWaiters.push(resolve));
18895
19313
  } else {
18896
- this.runningCases += 1;
19314
+ this.runningRows += 1;
18897
19315
  }
18898
19316
  try {
18899
19317
  return await run();
18900
19318
  } finally {
18901
- const next = this.caseWaiters.shift();
19319
+ const next = this.rowWaiters.shift();
18902
19320
  if (next) next();
18903
- else this.runningCases -= 1;
19321
+ else this.runningRows -= 1;
18904
19322
  }
18905
19323
  }
18906
19324
  sortedResults() {
18907
- const order = new Map(this.caseIds.map((caseId, index) => [caseId, index]));
19325
+ const order = new Map(this.rowIds.map((rowId, index) => [rowId, index]));
18908
19326
  return [...this.completed].sort((left, right) => {
18909
19327
  var _a, _b, _c, _d;
18910
- const caseOrder = ((_a = order.get(left.id)) != null ? _a : Number.MAX_SAFE_INTEGER) - ((_b = order.get(right.id)) != null ? _b : Number.MAX_SAFE_INTEGER);
18911
- return caseOrder || ((_c = left.attempt) != null ? _c : 0) - ((_d = right.attempt) != null ? _d : 0);
19328
+ const rowOrder = ((_a = order.get(left.id)) != null ? _a : Number.MAX_SAFE_INTEGER) - ((_b = order.get(right.id)) != null ? _b : Number.MAX_SAFE_INTEGER);
19329
+ return rowOrder || ((_c = left.attempt) != null ? _c : 0) - ((_d = right.attempt) != null ? _d : 0);
18912
19330
  });
18913
19331
  }
18914
19332
  };
@@ -18937,8 +19355,8 @@ var WorkshopEvalRun = class {
18937
19355
  this.completed = [];
18938
19356
  this.active = /* @__PURE__ */ new Set();
18939
19357
  this.state = "open";
18940
- this.runningCases = 0;
18941
- this.caseWaiters = [];
19358
+ this.runningRows = 0;
19359
+ this.rowWaiters = [];
18942
19360
  }
18943
19361
  async run(options = {}) {
18944
19362
  if (this.state !== "open") throw new Error(`Raindrop eval run ${this.id} is already finishing`);
@@ -18952,11 +19370,11 @@ var WorkshopEvalRun = class {
18952
19370
  }
18953
19371
  async runOnce(options) {
18954
19372
  var _a;
18955
- const cases = selectLocalCases(this.definition, options.selection);
19373
+ const rows = selectLocalRows(this.definition, options.selection);
18956
19374
  const completed = await Promise.all(
18957
- cases.map((row) => this.withCasePermit(() => {
19375
+ rows.map((row) => this.withRowPermit(() => {
18958
19376
  var _a2;
18959
- return this.executeCase(row, (_a2 = options.attempt) != null ? _a2 : 0);
19377
+ return this.executeRow(row, (_a2 = options.attempt) != null ? _a2 : 0);
18960
19378
  }))
18961
19379
  );
18962
19380
  const graded = await this.applyPortableEvaluators(completed, (_a = options.attempt) != null ? _a : 0);
@@ -19017,7 +19435,7 @@ var WorkshopEvalRun = class {
19017
19435
  name: entry.evaluator.name,
19018
19436
  output: entry.evaluator.output,
19019
19437
  execution: "local",
19020
- scope: entry.evaluator.scope,
19438
+ scope: entry.evaluator.scope === "row" ? "case" : entry.evaluator.scope,
19021
19439
  threshold: entry.threshold
19022
19440
  };
19023
19441
  })
@@ -19029,7 +19447,7 @@ var WorkshopEvalRun = class {
19029
19447
  this.state = "finished";
19030
19448
  return result;
19031
19449
  }
19032
- async executeCase(row, attempt) {
19450
+ async executeRow(row, attempt) {
19033
19451
  var _a, _b, _c;
19034
19452
  const correlationId = randomId();
19035
19453
  const exportedSpanIds = /* @__PURE__ */ new Set();
@@ -19171,20 +19589,20 @@ var WorkshopEvalRun = class {
19171
19589
  };
19172
19590
  return { result, packet, trace: trace8 };
19173
19591
  }
19174
- async withCasePermit(run) {
19592
+ async withRowPermit(run) {
19175
19593
  var _a;
19176
19594
  const limit = (_a = this.definition.concurrency) != null ? _a : 4;
19177
- if (this.runningCases >= limit) {
19178
- await new Promise((resolve) => this.caseWaiters.push(resolve));
19595
+ if (this.runningRows >= limit) {
19596
+ await new Promise((resolve) => this.rowWaiters.push(resolve));
19179
19597
  } else {
19180
- this.runningCases += 1;
19598
+ this.runningRows += 1;
19181
19599
  }
19182
19600
  try {
19183
19601
  return await run();
19184
19602
  } finally {
19185
- const next = this.caseWaiters.shift();
19603
+ const next = this.rowWaiters.shift();
19186
19604
  if (next) next();
19187
- else this.runningCases -= 1;
19605
+ else this.runningRows -= 1;
19188
19606
  }
19189
19607
  }
19190
19608
  async applyPortableEvaluators(completed, attempt) {
@@ -19249,14 +19667,17 @@ var WorkshopEvalRun = class {
19249
19667
  }
19250
19668
  };
19251
19669
  async function runRaindropEval(client, definition, destination, options) {
19252
- var _a, _b, _c;
19670
+ var _a, _b, _c, _d;
19253
19671
  if (typeof definition.dataset !== "string") {
19254
19672
  throw new Error(
19255
19673
  "Raindrop remote execution requires an eval dataset slug; legacy pulled snapshots do not carry an immutable dataset version id"
19256
19674
  );
19257
19675
  }
19258
19676
  const datasetReference = definition.dataset;
19259
- const pinnedDataset = (_a = options.selection) == null ? void 0 : _a.dataset;
19677
+ const pinnedDataset = (_b = (_a = options.selection) == null ? void 0 : _a.dataset) != null ? _b : definition.datasetVersionId ? await readEvalDataset2(client, datasetReference, {
19678
+ queryUrl: destination.queryUrl,
19679
+ versionId: definition.datasetVersionId
19680
+ }) : void 0;
19260
19681
  if (pinnedDataset && pinnedDataset.dataset.slug !== datasetReference) {
19261
19682
  throw new Error(
19262
19683
  `Raindrop eval ${definition.name} pinned dataset ${pinnedDataset.dataset.slug} does not match ${datasetReference}`
@@ -19268,7 +19689,8 @@ async function runRaindropEval(client, definition, destination, options) {
19268
19689
  definition.evaluators.map((evaluator) => [evaluatorName(evaluator), evaluator])
19269
19690
  );
19270
19691
  const replayOptions = {
19271
- dataset: pinnedDataset != null ? pinnedDataset : datasetReference,
19692
+ runId: options.runId,
19693
+ dataset: pinnedDataset ? datasetToWire(pinnedDataset) : datasetReference,
19272
19694
  evaluators: definition.evaluators.map((entry) => replayEvaluator(entry)),
19273
19695
  evaluatorPins: hostedEvaluatorPins(definition.evaluators),
19274
19696
  run: definition.run,
@@ -19283,7 +19705,7 @@ async function runRaindropEval(client, definition, destination, options) {
19283
19705
  client,
19284
19706
  selection ? {
19285
19707
  ...replayOptions,
19286
- rowIds: selection.caseIds
19708
+ rowIds: selection.rowIds
19287
19709
  } : replayOptions
19288
19710
  );
19289
19711
  const outputByEvaluator = /* @__PURE__ */ new Map();
@@ -19296,7 +19718,7 @@ async function runRaindropEval(client, definition, destination, options) {
19296
19718
  if (evaluatorResult.status === "failed") {
19297
19719
  continue;
19298
19720
  }
19299
- const output = (_c = (_b = evaluatorResult.summary) == null ? void 0 : _b.outputType) != null ? _c : evaluatorOutput(evaluator);
19721
+ const output = (_d = (_c = evaluatorResult.summary) == null ? void 0 : _c.outputType) != null ? _d : evaluatorOutput(evaluator);
19300
19722
  if (output === void 0)
19301
19723
  throw new Error(
19302
19724
  `Raindrop evaluator ${evaluatorResult.evaluator} completed without an output type`
@@ -19304,11 +19726,12 @@ async function runRaindropEval(client, definition, destination, options) {
19304
19726
  validateEvaluatorOutput(evaluator, output);
19305
19727
  outputByEvaluator.set(evaluatorResult.evaluator, output);
19306
19728
  }
19307
- const { rows, ...summary } = result;
19729
+ const { rows, replayId, ...summary } = result;
19308
19730
  return {
19309
19731
  ...summary,
19732
+ runId: replayId,
19310
19733
  name: definition.name,
19311
- cases: rows.map((row) => ({
19734
+ rows: rows.map((row) => ({
19312
19735
  ...row,
19313
19736
  verdicts: row.verdicts.map((verdict) => {
19314
19737
  const evaluator = configured.get(verdict.evaluator);
@@ -19344,32 +19767,38 @@ function replayEvaluator(entry) {
19344
19767
  if (typeof entry.evaluator === "string") return entry.evaluator;
19345
19768
  if (isPortableEvaluator(entry.evaluator)) return entry.evaluator.slug;
19346
19769
  if ("kind" in entry.evaluator && entry.evaluator.kind === "program") {
19770
+ if (!entry.evaluator.expected) {
19771
+ throw new Error(
19772
+ `Create evaluator ${entry.evaluator.slug} in Raindrop and reference its slug; uploading evaluator source is not supported`
19773
+ );
19774
+ }
19347
19775
  return entry.evaluator.slug;
19348
19776
  }
19349
- switch (entry.evaluator.output) {
19777
+ const evaluator = entry.evaluator;
19778
+ switch (evaluator.output) {
19350
19779
  case "boolean":
19351
19780
  return {
19352
- slug: entry.evaluator.slug,
19353
- name: entry.evaluator.name,
19354
- output: entry.evaluator.output,
19355
- judge: entry.evaluator.judge,
19356
- requiresReference: entry.evaluator.requiresReference
19781
+ slug: evaluator.slug,
19782
+ name: evaluator.name,
19783
+ output: evaluator.output,
19784
+ judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
19785
+ requiresReference: evaluator.requiresReference
19357
19786
  };
19358
19787
  case "score":
19359
19788
  return {
19360
- slug: entry.evaluator.slug,
19361
- name: entry.evaluator.name,
19362
- output: entry.evaluator.output,
19363
- judge: entry.evaluator.judge,
19364
- requiresReference: entry.evaluator.requiresReference
19789
+ slug: evaluator.slug,
19790
+ name: evaluator.name,
19791
+ output: evaluator.output,
19792
+ judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
19793
+ requiresReference: evaluator.requiresReference
19365
19794
  };
19366
19795
  case "number":
19367
19796
  return {
19368
- slug: entry.evaluator.slug,
19369
- name: entry.evaluator.name,
19370
- output: entry.evaluator.output,
19371
- judge: entry.evaluator.judge,
19372
- requiresReference: entry.evaluator.requiresReference
19797
+ slug: evaluator.slug,
19798
+ name: evaluator.name,
19799
+ output: evaluator.output,
19800
+ judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
19801
+ requiresReference: evaluator.requiresReference
19373
19802
  };
19374
19803
  }
19375
19804
  }
@@ -19378,34 +19807,34 @@ function validateDefinition(definition) {
19378
19807
  throw new Error("Raindrop runEvalSuite requires a definition created by defineEvalSuite");
19379
19808
  return definition;
19380
19809
  }
19381
- function selectLocalCases(definition, selection) {
19382
- if (!selection) return definition.dataset.cases;
19383
- if (selection.caseIds.length === 0)
19810
+ function selectLocalRows(definition, selection) {
19811
+ if (!selection) return definition.dataset.rows;
19812
+ if (selection.rowIds.length === 0)
19384
19813
  throw new Error(`Raindrop eval ${definition.name} selection cannot be empty`);
19385
- if (new Set(selection.caseIds).size !== selection.caseIds.length)
19386
- throw new Error(`Raindrop eval ${definition.name} selection case ids must be unique`);
19387
- const byId = new Map(definition.dataset.cases.map((row) => [row.id, row]));
19388
- return selection.caseIds.map((id) => {
19814
+ if (new Set(selection.rowIds).size !== selection.rowIds.length)
19815
+ throw new Error(`Raindrop eval ${definition.name} selection row ids must be unique`);
19816
+ const byId = new Map(definition.dataset.rows.map((row) => [row.id, row]));
19817
+ return selection.rowIds.map((id) => {
19389
19818
  const row = byId.get(id);
19390
19819
  if (!row)
19391
- throw new Error(`Raindrop eval ${definition.name} selection contains unknown case ${id}`);
19820
+ throw new Error(`Raindrop eval ${definition.name} selection contains unknown row ${id}`);
19392
19821
  return row;
19393
19822
  });
19394
19823
  }
19395
19824
  function validateRemoteSelection(dataset, selection) {
19396
- var _a, _b, _c;
19825
+ var _a;
19397
19826
  if (!selection) return void 0;
19398
- if (selection.caseIds.length === 0)
19827
+ if (selection.rowIds.length === 0)
19399
19828
  throw new Error(`Raindrop eval ${dataset} selection cannot be empty`);
19400
- if (new Set(selection.caseIds).size !== selection.caseIds.length)
19401
- throw new Error(`Raindrop eval ${dataset} selection case ids must be unique`);
19402
- const cases = (_c = (_a = selection.dataset) == null ? void 0 : _a.cases) != null ? _c : (_b = selection.manifest) == null ? void 0 : _b.cases;
19403
- if (!cases) throw new Error(`Raindrop eval ${dataset} selection requires its dataset manifest`);
19404
- const available = new Set(cases.map((row) => row.id));
19405
- const unknown = selection.caseIds.find((caseId) => !available.has(caseId));
19829
+ if (new Set(selection.rowIds).size !== selection.rowIds.length)
19830
+ throw new Error(`Raindrop eval ${dataset} selection row ids must be unique`);
19831
+ const rows = (_a = selection.dataset) == null ? void 0 : _a.rows;
19832
+ if (!rows) throw new Error(`Raindrop eval ${dataset} selection requires its dataset manifest`);
19833
+ const available = new Set(rows.map((row) => row.id));
19834
+ const unknown = selection.rowIds.find((rowId) => !available.has(rowId));
19406
19835
  if (unknown)
19407
- throw new Error(`Raindrop eval ${dataset} selection contains unknown case ${unknown}`);
19408
- return { caseIds: selection.caseIds };
19836
+ throw new Error(`Raindrop eval ${dataset} selection contains unknown row ${unknown}`);
19837
+ return { rowIds: selection.rowIds };
19409
19838
  }
19410
19839
  function canonicalWorkshopTrace(envelope) {
19411
19840
  var _a, _b, _c, _d, _e, _f;
@@ -19531,18 +19960,18 @@ function portableVerdict(evaluator, runId, workshopUrl, outcome) {
19531
19960
  ...outcome.note ? { note: outcome.note } : {}
19532
19961
  };
19533
19962
  }
19534
- function workshopResult(id, name, cases) {
19535
- const done = cases.filter((entry) => entry.status === "done").length;
19536
- const missing = cases.filter((entry) => entry.status === "missing").length;
19963
+ function workshopResult(id, name, rows) {
19964
+ const done = rows.filter((entry) => entry.status === "done").length;
19965
+ const missing = rows.filter((entry) => entry.status === "missing").length;
19537
19966
  return {
19538
- replayId: id,
19967
+ runId: id,
19539
19968
  name,
19540
19969
  status: missing > 0 ? "failed" : "complete",
19541
- counts: { total: cases.length, pending: 0, done, missing },
19970
+ counts: { total: rows.length, pending: 0, done, missing },
19542
19971
  tracesExpireAt: null,
19543
19972
  tracesExpired: false,
19544
19973
  evaluators: [],
19545
- cases
19974
+ rows
19546
19975
  };
19547
19976
  }
19548
19977
  function workshopEvalUrl(baseUrl, id) {
@@ -20859,8 +21288,9 @@ var index_default = Raindrop;
20859
21288
  DEFAULT_QUERY_URL,
20860
21289
  DEFAULT_TRACE_WAIT_MS,
20861
21290
  EVAL_CORRELATION_ID_ATTRIBUTE,
20862
- EvalDatasetCaseInputSchema,
20863
21291
  EvalDatasetPublishConflictError,
21292
+ EvalDatasetRowInputSchema,
21293
+ EvalPublishError,
20864
21294
  LocalBooleanVerdictSchema,
20865
21295
  LocalNumberVerdictSchema,
20866
21296
  LocalScoreVerdictSchema,
@@ -20877,6 +21307,7 @@ var index_default = Raindrop;
20877
21307
  ReplayVerdictSchema,
20878
21308
  TraceSchema,
20879
21309
  claimReplay,
21310
+ compareEvalRuns,
20880
21311
  createEvalSuiteRun,
20881
21312
  createReplay,
20882
21313
  currentEvalScope,
@@ -20884,17 +21315,22 @@ var index_default = Raindrop;
20884
21315
  defineEvalSuite,
20885
21316
  defineEvaluatorProgram,
20886
21317
  defineLocalEvaluator,
21318
+ evaluateEvalRun,
20887
21319
  evaluatePortableEvaluator,
20888
21320
  evaluateReplay,
20889
21321
  importPortableEvaluator,
20890
21322
  isEvalDataset,
20891
21323
  isEvalSuiteDefinition,
21324
+ isPublishedEvalSuite,
20892
21325
  loadEvalSnapshot,
20893
21326
  projectWorkshopTrace,
20894
21327
  publishEvalDataset,
21328
+ publishEvalSuite,
20895
21329
  pullEval,
20896
21330
  readEvalDataset,
20897
21331
  readEvalManifest,
21332
+ readEvalRun,
21333
+ readEvaluator,
20898
21334
  readReplay,
20899
21335
  replay,
20900
21336
  resolveDisableBatching,
@@ -20902,6 +21338,7 @@ var index_default = Raindrop;
20902
21338
  runReplay,
20903
21339
  traceOutput,
20904
21340
  traceToolCalls,
21341
+ traceTools,
20905
21342
  verifyPortableEvalArtifact,
20906
21343
  withEvalScope
20907
21344
  });