raindrop-ai 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -8
- package/dist/chunk-4Z7UHC7X.mjs +14850 -0
- package/dist/{chunk-ZVHQBBST.mjs → chunk-OX6X4ZWQ.mjs} +5 -1
- package/dist/{chunk-CFGEE5ID.mjs → chunk-RMP6BZSY.mjs} +59 -16
- package/dist/evals/cli.d.mts +1 -0
- package/dist/evals/cli.d.ts +1 -0
- package/dist/evals/cli.js +21135 -0
- package/dist/evals/cli.mjs +125 -0
- package/dist/{index-APN1jN-i.d.mts → index-n8AIAiRR.d.mts} +18838 -18626
- package/dist/{index-APN1jN-i.d.ts → index-n8AIAiRR.d.ts} +18838 -18626
- package/dist/index.d.mts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +731 -294
- package/dist/index.mjs +57 -14425
- package/dist/{portable-runtime-QZKU2VXS.mjs → portable-runtime-MFVOVFPB.mjs} +7 -2
- package/dist/tracing/index.d.mts +1 -1
- package/dist/tracing/index.d.ts +1 -1
- package/dist/tracing/index.js +5 -1
- package/dist/tracing/index.mjs +1 -1
- package/package.json +5 -1
package/dist/index.js
CHANGED
|
@@ -5767,17 +5767,38 @@ var require_dist = __commonJS({
|
|
|
5767
5767
|
|
|
5768
5768
|
// src/evals/definition.ts
|
|
5769
5769
|
function defineDataset(dataset) {
|
|
5770
|
+
var _a, _b;
|
|
5770
5771
|
const id = dataset.id.trim();
|
|
5771
5772
|
const name = dataset.name.trim();
|
|
5772
|
-
const
|
|
5773
|
+
const rows = dataset.rows.map((row) => {
|
|
5774
|
+
var _a2;
|
|
5775
|
+
const parsed = DatasetRowInputSchema.parse(row);
|
|
5776
|
+
return {
|
|
5777
|
+
id: parsed.id,
|
|
5778
|
+
name: (_a2 = parsed.name) != null ? _a2 : parsed.id,
|
|
5779
|
+
input: parsed.input,
|
|
5780
|
+
output: parsed.output,
|
|
5781
|
+
properties: parsed.properties
|
|
5782
|
+
};
|
|
5783
|
+
});
|
|
5784
|
+
const version = (_b = (_a = dataset.version) == null ? void 0 : _a.trim()) != null ? _b : (0, import_node_crypto2.createHash)("sha256").update(
|
|
5785
|
+
JSON.stringify(
|
|
5786
|
+
rows.map((row) => ({
|
|
5787
|
+
...row,
|
|
5788
|
+
properties: Object.fromEntries(
|
|
5789
|
+
Object.keys(row.properties).sort().map((key) => [key, row.properties[key]])
|
|
5790
|
+
)
|
|
5791
|
+
}))
|
|
5792
|
+
)
|
|
5793
|
+
).digest("hex");
|
|
5773
5794
|
if (!id) throw new Error("Raindrop eval dataset ids cannot be empty");
|
|
5774
5795
|
if (!name) throw new Error("Raindrop eval dataset names cannot be empty");
|
|
5775
5796
|
if (!version) throw new Error("Raindrop eval dataset versions cannot be empty");
|
|
5776
|
-
if (dataset.
|
|
5777
|
-
throw new Error("Raindrop eval datasets require at least one
|
|
5778
|
-
const ids =
|
|
5779
|
-
if (ids.some((
|
|
5780
|
-
if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval
|
|
5797
|
+
if (dataset.rows.length === 0)
|
|
5798
|
+
throw new Error("Raindrop eval datasets require at least one row");
|
|
5799
|
+
const ids = rows.map((entry) => entry.id);
|
|
5800
|
+
if (ids.some((rowId) => !rowId)) throw new Error("Raindrop eval row ids cannot be empty");
|
|
5801
|
+
if (new Set(ids).size !== ids.length) throw new Error("Raindrop eval row ids must be unique");
|
|
5781
5802
|
if (dataset.remote) {
|
|
5782
5803
|
if (!dataset.remote.reference.trim())
|
|
5783
5804
|
throw new Error("Raindrop remote dataset reference cannot be empty");
|
|
@@ -5790,7 +5811,7 @@ function defineDataset(dataset) {
|
|
|
5790
5811
|
id,
|
|
5791
5812
|
name,
|
|
5792
5813
|
version,
|
|
5793
|
-
|
|
5814
|
+
rows,
|
|
5794
5815
|
[EVAL_DATASET]: true
|
|
5795
5816
|
};
|
|
5796
5817
|
Object.defineProperty(defined2, EVAL_DATASET, { enumerable: false });
|
|
@@ -5800,7 +5821,7 @@ function isEvalDataset(value) {
|
|
|
5800
5821
|
return typeof value === "object" && value !== null && EVAL_DATASET in value && value[EVAL_DATASET] === true;
|
|
5801
5822
|
}
|
|
5802
5823
|
function defineEvaluatorProgram(program) {
|
|
5803
|
-
var _a, _b, _c;
|
|
5824
|
+
var _a, _b, _c, _d;
|
|
5804
5825
|
const slug = program.slug.trim();
|
|
5805
5826
|
const name = program.name.trim();
|
|
5806
5827
|
const source = program.source.trim();
|
|
@@ -5832,16 +5853,18 @@ function defineEvaluatorProgram(program) {
|
|
|
5832
5853
|
return {
|
|
5833
5854
|
...program,
|
|
5834
5855
|
kind: "program",
|
|
5856
|
+
scope: import_zod2.z.literal("batch").parse((_c = program.scope) != null ? _c : "batch"),
|
|
5835
5857
|
slug,
|
|
5836
5858
|
name,
|
|
5837
5859
|
source,
|
|
5838
5860
|
intent,
|
|
5839
|
-
description: ((
|
|
5861
|
+
description: ((_d = program.description) == null ? void 0 : _d.trim()) || null,
|
|
5840
5862
|
rules
|
|
5841
5863
|
};
|
|
5842
5864
|
}
|
|
5843
5865
|
function defineLocalEvaluator(evaluator) {
|
|
5844
|
-
|
|
5866
|
+
var _a;
|
|
5867
|
+
return { ...evaluator, scope: import_zod2.z.literal("row").parse((_a = evaluator.scope) != null ? _a : "row") };
|
|
5845
5868
|
}
|
|
5846
5869
|
function defineEvalSuite(definition) {
|
|
5847
5870
|
const name = definition.name.trim();
|
|
@@ -5850,6 +5873,11 @@ function defineEvalSuite(definition) {
|
|
|
5850
5873
|
if (typeof dataset === "string" && dataset.length === 0) {
|
|
5851
5874
|
throw new Error("Raindrop eval suite datasets cannot be empty");
|
|
5852
5875
|
}
|
|
5876
|
+
if (definition.datasetVersionId !== void 0) {
|
|
5877
|
+
import_zod2.z.string().uuid().parse(definition.datasetVersionId);
|
|
5878
|
+
if (typeof dataset !== "string")
|
|
5879
|
+
throw new Error("A dataset version pin requires a published dataset slug");
|
|
5880
|
+
}
|
|
5853
5881
|
if (typeof definition.run !== "function") {
|
|
5854
5882
|
throw new Error("Raindrop eval suite run must be a function");
|
|
5855
5883
|
}
|
|
@@ -5878,6 +5906,11 @@ function defineEvalSuite(definition) {
|
|
|
5878
5906
|
function isEvalSuiteDefinition(value) {
|
|
5879
5907
|
return typeof value === "object" && value !== null && EVAL_DEFINITION in value && value[EVAL_DEFINITION] === true;
|
|
5880
5908
|
}
|
|
5909
|
+
function isPublishedEvalSuite(definition) {
|
|
5910
|
+
return typeof definition.dataset === "string" && definition.datasetVersionId !== void 0 && definition.evaluators.every(
|
|
5911
|
+
({ evaluator }) => typeof evaluator !== "string" && (!("kind" in evaluator) || evaluator.kind !== "program" || evaluator.expected !== void 0)
|
|
5912
|
+
);
|
|
5913
|
+
}
|
|
5881
5914
|
function evaluatorName(evaluator) {
|
|
5882
5915
|
return typeof evaluator.evaluator === "string" ? evaluator.evaluator.trim() : evaluator.evaluator.slug.trim();
|
|
5883
5916
|
}
|
|
@@ -5940,8 +5973,8 @@ function validateEvaluator(evaluator) {
|
|
|
5940
5973
|
if (typeof evaluator.evaluator.judge !== "function") {
|
|
5941
5974
|
throw new Error(`Raindrop local evaluator ${name} judge must be a function`);
|
|
5942
5975
|
}
|
|
5943
|
-
if (evaluator.evaluator.scope !== "
|
|
5944
|
-
throw new Error(`Raindrop local evaluator ${name} scope must be
|
|
5976
|
+
if (evaluator.evaluator.scope !== "row") {
|
|
5977
|
+
throw new Error(`Raindrop local evaluator ${name} scope must be row`);
|
|
5945
5978
|
}
|
|
5946
5979
|
const output = EvalOutputSchema.parse(evaluator.evaluator.output);
|
|
5947
5980
|
validateEvaluatorOutput(evaluator, output);
|
|
@@ -5968,10 +6001,11 @@ function parseNumericThreshold(name, output, threshold) {
|
|
|
5968
6001
|
}
|
|
5969
6002
|
return parsed;
|
|
5970
6003
|
}
|
|
5971
|
-
var import_zod2, EvalScoreSchema, EVAL_DATASET, BooleanThresholdSchema, FiniteThresholdSchema, EvalOutputSchema, NumericThresholdSchema, EVAL_DEFINITION;
|
|
6004
|
+
var import_node_crypto2, import_zod2, EvalScoreSchema, EVAL_DATASET, DatasetRowInputSchema, BooleanThresholdSchema, FiniteThresholdSchema, EvalOutputSchema, NumericThresholdSchema, EVAL_DEFINITION;
|
|
5972
6005
|
var init_definition = __esm({
|
|
5973
6006
|
"src/evals/definition.ts"() {
|
|
5974
6007
|
"use strict";
|
|
6008
|
+
import_node_crypto2 = require("crypto");
|
|
5975
6009
|
import_zod2 = require("zod");
|
|
5976
6010
|
EvalScoreSchema = import_zod2.z.union([
|
|
5977
6011
|
import_zod2.z.literal(1),
|
|
@@ -5981,6 +6015,13 @@ var init_definition = __esm({
|
|
|
5981
6015
|
import_zod2.z.literal(5)
|
|
5982
6016
|
]);
|
|
5983
6017
|
EVAL_DATASET = /* @__PURE__ */ Symbol.for("raindrop.evalDataset");
|
|
6018
|
+
DatasetRowInputSchema = import_zod2.z.object({
|
|
6019
|
+
id: import_zod2.z.string().trim().min(1),
|
|
6020
|
+
name: import_zod2.z.string().optional(),
|
|
6021
|
+
input: import_zod2.z.string().nullable(),
|
|
6022
|
+
output: import_zod2.z.string().nullable().default(null),
|
|
6023
|
+
properties: import_zod2.z.record(import_zod2.z.string()).default({})
|
|
6024
|
+
});
|
|
5984
6025
|
BooleanThresholdSchema = import_zod2.z.object({ equals: import_zod2.z.boolean() }).strict();
|
|
5985
6026
|
FiniteThresholdSchema = import_zod2.z.number().finite();
|
|
5986
6027
|
EvalOutputSchema = import_zod2.z.enum(["boolean", "score", "number"]);
|
|
@@ -6929,7 +6970,7 @@ async function resolveJudge(evaluator, rawRequest, remainingMs, tracesById, case
|
|
|
6929
6970
|
var _a2;
|
|
6930
6971
|
return (_a2 = evaluator.judge) == null ? void 0 : _a2.call(evaluator, {
|
|
6931
6972
|
trace: trace8,
|
|
6932
|
-
tracePair: casesByTraceId.get(request.traceId),
|
|
6973
|
+
tracePair: rowTracePair(casesByTraceId.get(request.traceId)),
|
|
6933
6974
|
rubric: request.rubric.slice(0, 4e4),
|
|
6934
6975
|
signal: controller.signal
|
|
6935
6976
|
});
|
|
@@ -6957,13 +6998,18 @@ async function resolveJudge(evaluator, rawRequest, remainingMs, tracesById, case
|
|
|
6957
6998
|
function byteLength(value) {
|
|
6958
6999
|
return new TextEncoder().encode(value).byteLength;
|
|
6959
7000
|
}
|
|
6960
|
-
|
|
7001
|
+
function rowTracePair(pair) {
|
|
7002
|
+
if (!pair) return void 0;
|
|
7003
|
+
const { caseId, ...evidence } = pair;
|
|
7004
|
+
return { ...evidence, rowId: caseId };
|
|
7005
|
+
}
|
|
7006
|
+
var import_quickjs_singlefile_cjs_release_sync, import_quickjs_emscripten_core, import_zod7, DEFAULT_MEMORY_BYTES, DEFAULT_STACK_BYTES, DEFAULT_TIMEOUT_MS, DEFAULT_JUDGE_TIMEOUT_MS, DEFAULT_JUDGE_REQUEST_TIMEOUT_MS, DEFAULT_JUDGE_CONCURRENCY, DEFAULT_MAX_INPUT_BYTES, DEFAULT_MAX_OUTPUT_BYTES, JudgeRequestSchema;
|
|
6961
7007
|
var init_portable_runtime = __esm({
|
|
6962
7008
|
"src/evals/portable-runtime.ts"() {
|
|
6963
7009
|
"use strict";
|
|
6964
7010
|
import_quickjs_singlefile_cjs_release_sync = __toESM(require("@jitl/quickjs-singlefile-cjs-release-sync"));
|
|
6965
7011
|
import_quickjs_emscripten_core = require("quickjs-emscripten-core");
|
|
6966
|
-
|
|
7012
|
+
import_zod7 = require("zod");
|
|
6967
7013
|
init_portable();
|
|
6968
7014
|
DEFAULT_MEMORY_BYTES = 32 * 1024 * 1024;
|
|
6969
7015
|
DEFAULT_STACK_BYTES = 64 * 1024;
|
|
@@ -6973,11 +7019,11 @@ var init_portable_runtime = __esm({
|
|
|
6973
7019
|
DEFAULT_JUDGE_CONCURRENCY = 4;
|
|
6974
7020
|
DEFAULT_MAX_INPUT_BYTES = 4 * 1024 * 1024;
|
|
6975
7021
|
DEFAULT_MAX_OUTPUT_BYTES = 4 * 1024 * 1024;
|
|
6976
|
-
JudgeRequestSchema =
|
|
6977
|
-
type:
|
|
6978
|
-
requestId:
|
|
6979
|
-
traceId:
|
|
6980
|
-
rubric:
|
|
7022
|
+
JudgeRequestSchema = import_zod7.z.object({
|
|
7023
|
+
type: import_zod7.z.literal("judge_request"),
|
|
7024
|
+
requestId: import_zod7.z.string(),
|
|
7025
|
+
traceId: import_zod7.z.string(),
|
|
7026
|
+
rubric: import_zod7.z.string()
|
|
6981
7027
|
});
|
|
6982
7028
|
}
|
|
6983
7029
|
});
|
|
@@ -7012,16 +7058,17 @@ async function evaluatePortableEvaluator(evaluator, input) {
|
|
|
7012
7058
|
expectedSha256: evaluator.expectedSha256
|
|
7013
7059
|
});
|
|
7014
7060
|
const { runPortableEvaluator: runPortableEvaluator2 } = await Promise.resolve().then(() => (init_portable_runtime(), portable_runtime_exports));
|
|
7015
|
-
const cases = (_a = input.
|
|
7061
|
+
const cases = ((_a = input.rows) != null ? _a : []).map(({ rowId, ...pair }) => ({ ...pair, caseId: rowId }));
|
|
7016
7062
|
if (cases.length > 0 && (cases.length !== input.traces.length || cases.some(
|
|
7017
7063
|
(entry, index) => {
|
|
7018
7064
|
var _a2;
|
|
7019
7065
|
return entry.candidate.trace.event.id !== ((_a2 = input.traces[index]) == null ? void 0 : _a2.event.id);
|
|
7020
7066
|
}
|
|
7021
7067
|
))) {
|
|
7022
|
-
throw new Error("Raindrop portable evaluator
|
|
7068
|
+
throw new Error("Raindrop portable evaluator rows must match traces exactly, in slice order");
|
|
7023
7069
|
}
|
|
7024
|
-
const
|
|
7070
|
+
const { rows: _rows, ...runtimeInput } = input;
|
|
7071
|
+
const result = await runPortableEvaluator2(evaluator, { ...runtimeInput, cases });
|
|
7025
7072
|
validatePortableResult(artifact, input.traces, result);
|
|
7026
7073
|
return result;
|
|
7027
7074
|
}
|
|
@@ -7088,67 +7135,67 @@ function canonicalJson2(value) {
|
|
|
7088
7135
|
if (Array.isArray(value)) return `[${value.map(canonicalJson2).join(",")}]`;
|
|
7089
7136
|
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${canonicalJson2(entry)}`).join(",")}}`;
|
|
7090
7137
|
}
|
|
7091
|
-
var
|
|
7138
|
+
var import_zod8, PortableEvalArtifactSchema, PortableEvalResultSchema;
|
|
7092
7139
|
var init_portable = __esm({
|
|
7093
7140
|
"src/evals/portable.ts"() {
|
|
7094
7141
|
"use strict";
|
|
7095
|
-
|
|
7142
|
+
import_zod8 = require("zod");
|
|
7096
7143
|
init_definition();
|
|
7097
7144
|
init_trace();
|
|
7098
|
-
PortableEvalArtifactSchema =
|
|
7099
|
-
format:
|
|
7100
|
-
formatVersion:
|
|
7101
|
-
runtimeAbi:
|
|
7102
|
-
evaluator:
|
|
7103
|
-
id:
|
|
7104
|
-
slug:
|
|
7105
|
-
programVersion:
|
|
7106
|
-
executionMode:
|
|
7107
|
-
outputType:
|
|
7108
|
-
scope:
|
|
7145
|
+
PortableEvalArtifactSchema = import_zod8.z.object({
|
|
7146
|
+
format: import_zod8.z.literal("raindrop.eval-program"),
|
|
7147
|
+
formatVersion: import_zod8.z.literal(1),
|
|
7148
|
+
runtimeAbi: import_zod8.z.literal("raindrop-eval-v2"),
|
|
7149
|
+
evaluator: import_zod8.z.object({
|
|
7150
|
+
id: import_zod8.z.string().uuid(),
|
|
7151
|
+
slug: import_zod8.z.string().min(1),
|
|
7152
|
+
programVersion: import_zod8.z.number().int().positive(),
|
|
7153
|
+
executionMode: import_zod8.z.enum(["deterministic", "judge"]),
|
|
7154
|
+
outputType: import_zod8.z.enum(["boolean", "score", "number"]),
|
|
7155
|
+
scope: import_zod8.z.literal("batch")
|
|
7109
7156
|
}).strict(),
|
|
7110
|
-
traceProjectionVersion:
|
|
7111
|
-
traceProjectionSha256:
|
|
7112
|
-
source:
|
|
7113
|
-
sha256:
|
|
7157
|
+
traceProjectionVersion: import_zod8.z.literal(1),
|
|
7158
|
+
traceProjectionSha256: import_zod8.z.string().regex(/^[a-f0-9]{64}$/),
|
|
7159
|
+
source: import_zod8.z.string().min(1),
|
|
7160
|
+
sha256: import_zod8.z.string().regex(/^[a-f0-9]{64}$/)
|
|
7114
7161
|
}).strict();
|
|
7115
|
-
PortableEvalResultSchema =
|
|
7116
|
-
outcomes:
|
|
7117
|
-
|
|
7118
|
-
|
|
7119
|
-
state:
|
|
7120
|
-
traceId:
|
|
7121
|
-
pass:
|
|
7122
|
-
note:
|
|
7162
|
+
PortableEvalResultSchema = import_zod8.z.object({
|
|
7163
|
+
outcomes: import_zod8.z.array(
|
|
7164
|
+
import_zod8.z.union([
|
|
7165
|
+
import_zod8.z.object({
|
|
7166
|
+
state: import_zod8.z.literal("graded"),
|
|
7167
|
+
traceId: import_zod8.z.string(),
|
|
7168
|
+
pass: import_zod8.z.boolean(),
|
|
7169
|
+
note: import_zod8.z.string().optional()
|
|
7123
7170
|
}),
|
|
7124
|
-
|
|
7125
|
-
state:
|
|
7126
|
-
traceId:
|
|
7171
|
+
import_zod8.z.object({
|
|
7172
|
+
state: import_zod8.z.literal("graded"),
|
|
7173
|
+
traceId: import_zod8.z.string(),
|
|
7127
7174
|
score: EvalScoreSchema,
|
|
7128
|
-
note:
|
|
7175
|
+
note: import_zod8.z.string().optional()
|
|
7129
7176
|
}),
|
|
7130
|
-
|
|
7131
|
-
state:
|
|
7132
|
-
traceId:
|
|
7133
|
-
value:
|
|
7134
|
-
note:
|
|
7177
|
+
import_zod8.z.object({
|
|
7178
|
+
state: import_zod8.z.literal("graded"),
|
|
7179
|
+
traceId: import_zod8.z.string(),
|
|
7180
|
+
value: import_zod8.z.number().finite(),
|
|
7181
|
+
note: import_zod8.z.string().optional()
|
|
7135
7182
|
}),
|
|
7136
|
-
|
|
7137
|
-
|
|
7138
|
-
state:
|
|
7139
|
-
traceId:
|
|
7140
|
-
reason:
|
|
7183
|
+
import_zod8.z.object({ state: import_zod8.z.literal("errored"), traceId: import_zod8.z.string(), message: import_zod8.z.string() }),
|
|
7184
|
+
import_zod8.z.object({
|
|
7185
|
+
state: import_zod8.z.literal("ungraded"),
|
|
7186
|
+
traceId: import_zod8.z.string(),
|
|
7187
|
+
reason: import_zod8.z.enum(["not sampled", "trace unavailable", "content unreadable"])
|
|
7141
7188
|
})
|
|
7142
7189
|
])
|
|
7143
7190
|
),
|
|
7144
|
-
failures:
|
|
7145
|
-
stats:
|
|
7146
|
-
traceCount:
|
|
7147
|
-
hydratedCount:
|
|
7148
|
-
judgeCalls:
|
|
7149
|
-
durationMs:
|
|
7150
|
-
skippedRows:
|
|
7151
|
-
skippedReason:
|
|
7191
|
+
failures: import_zod8.z.array(import_zod8.z.string()),
|
|
7192
|
+
stats: import_zod8.z.object({
|
|
7193
|
+
traceCount: import_zod8.z.number().int().nonnegative(),
|
|
7194
|
+
hydratedCount: import_zod8.z.number().int().nonnegative(),
|
|
7195
|
+
judgeCalls: import_zod8.z.number().int().nonnegative(),
|
|
7196
|
+
durationMs: import_zod8.z.number().nonnegative(),
|
|
7197
|
+
skippedRows: import_zod8.z.number().int().nonnegative(),
|
|
7198
|
+
skippedReason: import_zod8.z.literal("no trace").nullable()
|
|
7152
7199
|
})
|
|
7153
7200
|
});
|
|
7154
7201
|
}
|
|
@@ -7163,8 +7210,9 @@ __export(index_exports, {
|
|
|
7163
7210
|
DEFAULT_QUERY_URL: () => DEFAULT_QUERY_URL,
|
|
7164
7211
|
DEFAULT_TRACE_WAIT_MS: () => DEFAULT_TRACE_WAIT_MS,
|
|
7165
7212
|
EVAL_CORRELATION_ID_ATTRIBUTE: () => EVAL_CORRELATION_ID_ATTRIBUTE,
|
|
7166
|
-
EvalDatasetCaseInputSchema: () => EvalDatasetCaseInputSchema,
|
|
7167
7213
|
EvalDatasetPublishConflictError: () => EvalDatasetPublishConflictError,
|
|
7214
|
+
EvalDatasetRowInputSchema: () => EvalDatasetRowInputSchema,
|
|
7215
|
+
EvalPublishError: () => EvalPublishError,
|
|
7168
7216
|
LocalBooleanVerdictSchema: () => LocalBooleanVerdictSchema,
|
|
7169
7217
|
LocalNumberVerdictSchema: () => LocalNumberVerdictSchema,
|
|
7170
7218
|
LocalScoreVerdictSchema: () => LocalScoreVerdictSchema,
|
|
@@ -7181,6 +7229,7 @@ __export(index_exports, {
|
|
|
7181
7229
|
ReplayVerdictSchema: () => ReplayVerdictSchema,
|
|
7182
7230
|
TraceSchema: () => TraceSchema,
|
|
7183
7231
|
claimReplay: () => claimReplay,
|
|
7232
|
+
compareEvalRuns: () => compareEvalRuns,
|
|
7184
7233
|
createEvalSuiteRun: () => createEvalSuiteRun,
|
|
7185
7234
|
createReplay: () => createReplay,
|
|
7186
7235
|
currentEvalScope: () => currentEvalScope,
|
|
@@ -7189,17 +7238,22 @@ __export(index_exports, {
|
|
|
7189
7238
|
defineEvalSuite: () => defineEvalSuite,
|
|
7190
7239
|
defineEvaluatorProgram: () => defineEvaluatorProgram,
|
|
7191
7240
|
defineLocalEvaluator: () => defineLocalEvaluator,
|
|
7241
|
+
evaluateEvalRun: () => evaluateEvalRun,
|
|
7192
7242
|
evaluatePortableEvaluator: () => evaluatePortableEvaluator,
|
|
7193
7243
|
evaluateReplay: () => evaluateReplay,
|
|
7194
7244
|
importPortableEvaluator: () => importPortableEvaluator,
|
|
7195
7245
|
isEvalDataset: () => isEvalDataset,
|
|
7196
7246
|
isEvalSuiteDefinition: () => isEvalSuiteDefinition,
|
|
7247
|
+
isPublishedEvalSuite: () => isPublishedEvalSuite,
|
|
7197
7248
|
loadEvalSnapshot: () => loadEvalSnapshot,
|
|
7198
7249
|
projectWorkshopTrace: () => projectWorkshopTrace,
|
|
7199
7250
|
publishEvalDataset: () => publishEvalDataset,
|
|
7251
|
+
publishEvalSuite: () => publishEvalSuite,
|
|
7200
7252
|
pullEval: () => pullEval,
|
|
7201
|
-
readEvalDataset: () =>
|
|
7253
|
+
readEvalDataset: () => readEvalDataset2,
|
|
7202
7254
|
readEvalManifest: () => readEvalManifest,
|
|
7255
|
+
readEvalRun: () => readEvalRun,
|
|
7256
|
+
readEvaluator: () => readEvaluator,
|
|
7203
7257
|
readReplay: () => readReplay,
|
|
7204
7258
|
replay: () => replay,
|
|
7205
7259
|
resolveDisableBatching: () => resolveDisableBatching,
|
|
@@ -7207,6 +7261,7 @@ __export(index_exports, {
|
|
|
7207
7261
|
runReplay: () => runReplay,
|
|
7208
7262
|
traceOutput: () => traceOutput,
|
|
7209
7263
|
traceToolCalls: () => traceToolCalls,
|
|
7264
|
+
traceTools: () => traceTools,
|
|
7210
7265
|
verifyPortableEvalArtifact: () => verifyPortableEvalArtifact,
|
|
7211
7266
|
withEvalScope: () => withEvalScope
|
|
7212
7267
|
});
|
|
@@ -12558,10 +12613,13 @@ var SignalEventSchema = external_exports.object({
|
|
|
12558
12613
|
// package.json
|
|
12559
12614
|
var package_default = {
|
|
12560
12615
|
name: "raindrop-ai",
|
|
12561
|
-
version: "0.
|
|
12616
|
+
version: "0.7.0",
|
|
12562
12617
|
main: "dist/index.js",
|
|
12563
12618
|
module: "dist/index.mjs",
|
|
12564
12619
|
types: "dist/index.d.ts",
|
|
12620
|
+
bin: {
|
|
12621
|
+
"raindrop-evals": "dist/evals/cli.js"
|
|
12622
|
+
},
|
|
12565
12623
|
license: "MIT",
|
|
12566
12624
|
homepage: "https://www.raindrop.ai/docs/sdk/typescript/",
|
|
12567
12625
|
bugs: {
|
|
@@ -12658,6 +12716,7 @@ var package_default = {
|
|
|
12658
12716
|
tsup: {
|
|
12659
12717
|
entry: [
|
|
12660
12718
|
"src/index.ts",
|
|
12719
|
+
"src/evals/cli.ts",
|
|
12661
12720
|
"src/tracing/index.ts",
|
|
12662
12721
|
"src/otel/index.ts"
|
|
12663
12722
|
],
|
|
@@ -16449,7 +16508,7 @@ var EvalDatasetCaseInputsSchema = import_zod4.z.array(EvalDatasetCaseInputSchema
|
|
|
16449
16508
|
if (seen.has(entry.id)) {
|
|
16450
16509
|
context9.addIssue({
|
|
16451
16510
|
code: import_zod4.z.ZodIssueCode.custom,
|
|
16452
|
-
message: `Duplicate
|
|
16511
|
+
message: `Duplicate row id ${entry.id}`,
|
|
16453
16512
|
path: [index, "id"]
|
|
16454
16513
|
});
|
|
16455
16514
|
}
|
|
@@ -16829,6 +16888,34 @@ var WireReplayDetailSchema = import_zod5.z.object({
|
|
|
16829
16888
|
replay: WireReplaySchema,
|
|
16830
16889
|
rows: import_zod5.z.array(WireRowSchema)
|
|
16831
16890
|
});
|
|
16891
|
+
var EvalRunDetailSchema = WireReplayDetailSchema.extend({
|
|
16892
|
+
replay: WireReplaySchema.extend({
|
|
16893
|
+
dataset_id: import_zod5.z.string().uuid(),
|
|
16894
|
+
dataset_version_id: import_zod5.z.string().uuid().nullable()
|
|
16895
|
+
}),
|
|
16896
|
+
eval_runs: import_zod5.z.array(
|
|
16897
|
+
import_zod5.z.object({
|
|
16898
|
+
id: import_zod5.z.string().uuid(),
|
|
16899
|
+
eval_id: import_zod5.z.string().uuid(),
|
|
16900
|
+
program_version: import_zod5.z.number().int().positive(),
|
|
16901
|
+
output_type: import_zod5.z.enum(["boolean", "score", "number"]),
|
|
16902
|
+
executed_by: import_zod5.z.enum(["hosted", "local"]),
|
|
16903
|
+
status: import_zod5.z.enum(["running", "completed", "failed"])
|
|
16904
|
+
})
|
|
16905
|
+
)
|
|
16906
|
+
});
|
|
16907
|
+
async function readReplayDetails(client, runId, options = {}) {
|
|
16908
|
+
var _a;
|
|
16909
|
+
const detail = await queryApi({
|
|
16910
|
+
client,
|
|
16911
|
+
queryUrl: (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL,
|
|
16912
|
+
path: `/v1/replays/${encodeURIComponent(runId)}`,
|
|
16913
|
+
method: "GET",
|
|
16914
|
+
schema: EvalRunDetailSchema
|
|
16915
|
+
});
|
|
16916
|
+
if (detail.replay.id !== runId) throw new Error("Raindrop returned a different run");
|
|
16917
|
+
return detail;
|
|
16918
|
+
}
|
|
16832
16919
|
var WireReportSchema = import_zod5.z.object({ status: import_zod5.z.string() });
|
|
16833
16920
|
var WireAttemptStartedSchema = import_zod5.z.object({
|
|
16834
16921
|
status: import_zod5.z.literal("pending"),
|
|
@@ -17007,7 +17094,7 @@ function selectEvalDatasetRows(manifest, requestedRowIds) {
|
|
|
17007
17094
|
if (ordered.length !== selected.size) {
|
|
17008
17095
|
const available = new Set(manifest.cases.map((entry) => entry.id));
|
|
17009
17096
|
const unknown = rowIds.find((rowId) => !available.has(rowId));
|
|
17010
|
-
throw new Error(`Raindrop replay rowIds contains unknown
|
|
17097
|
+
throw new Error(`Raindrop replay rowIds contains unknown row ${unknown != null ? unknown : "unknown"}`);
|
|
17011
17098
|
}
|
|
17012
17099
|
return ordered;
|
|
17013
17100
|
}
|
|
@@ -17075,7 +17162,7 @@ async function readEvalManifest(client, dataset, options) {
|
|
|
17075
17162
|
});
|
|
17076
17163
|
const caseIds = manifest.rows.map((row) => row.id);
|
|
17077
17164
|
if (new Set(caseIds).size !== caseIds.length) {
|
|
17078
|
-
throw new Error(`Raindrop eval manifest for ${normalized} contains duplicate
|
|
17165
|
+
throw new Error(`Raindrop eval manifest for ${normalized} contains duplicate row ids`);
|
|
17079
17166
|
}
|
|
17080
17167
|
return {
|
|
17081
17168
|
datasetId: manifest.dataset_id,
|
|
@@ -17115,7 +17202,7 @@ function validateEvalDatasetManifest(value) {
|
|
|
17115
17202
|
}
|
|
17116
17203
|
const caseIds = manifest.cases.map((entry) => entry.id);
|
|
17117
17204
|
if (new Set(caseIds).size !== caseIds.length) {
|
|
17118
|
-
throw new Error(`Raindrop eval dataset ${manifest.dataset.slug} contains duplicate
|
|
17205
|
+
throw new Error(`Raindrop eval dataset ${manifest.dataset.slug} contains duplicate row ids`);
|
|
17119
17206
|
}
|
|
17120
17207
|
return manifest;
|
|
17121
17208
|
}
|
|
@@ -17180,14 +17267,17 @@ async function runReplayRows(client, handle, options) {
|
|
|
17180
17267
|
var _a3;
|
|
17181
17268
|
const parent = (_a3 = import_api7.trace.getSpan(import_api7.context.active())) == null ? void 0 : _a3.spanContext();
|
|
17182
17269
|
if (!parent) return options.run(visible);
|
|
17183
|
-
return withReplayTraceDestination(
|
|
17184
|
-
|
|
17185
|
-
|
|
17186
|
-
|
|
17187
|
-
|
|
17188
|
-
|
|
17189
|
-
|
|
17190
|
-
|
|
17270
|
+
return withReplayTraceDestination(
|
|
17271
|
+
{
|
|
17272
|
+
...replayDestination,
|
|
17273
|
+
parentSpanContext: {
|
|
17274
|
+
traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
|
|
17275
|
+
spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
|
|
17276
|
+
eventId: parent.traceId
|
|
17277
|
+
}
|
|
17278
|
+
},
|
|
17279
|
+
() => options.run(visible)
|
|
17280
|
+
);
|
|
17191
17281
|
}
|
|
17192
17282
|
)
|
|
17193
17283
|
)
|
|
@@ -17282,7 +17372,11 @@ async function readEvalCaseTracePairs(input) {
|
|
|
17282
17372
|
while (reading.state === "pending" && Date.now() < deadlineAt) {
|
|
17283
17373
|
await sleep(Math.min(TRACE_POLL_INTERVAL_MS, Math.max(0, deadlineAt - Date.now())));
|
|
17284
17374
|
if (Date.now() >= deadlineAt) break;
|
|
17285
|
-
reading = await readEvalCaseTracePair(
|
|
17375
|
+
reading = await readEvalCaseTracePair(
|
|
17376
|
+
{ ...input, deadlineAt },
|
|
17377
|
+
evidence,
|
|
17378
|
+
(_b = input.attempt) != null ? _b : 0
|
|
17379
|
+
);
|
|
17286
17380
|
}
|
|
17287
17381
|
if (reading.state !== "ready") {
|
|
17288
17382
|
const detail = reading.state === "pending" ? `did not become complete within ${input.waitMs}ms` : reading.state === "expired" ? `expired at ${reading.expiredAt}` : reading.reason;
|
|
@@ -17333,7 +17427,13 @@ async function createLocalEvalReplaySession(client, options) {
|
|
|
17333
17427
|
throw new Error("Raindrop local eval replay requires at least one evaluator");
|
|
17334
17428
|
}
|
|
17335
17429
|
const queryUrl = (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL;
|
|
17336
|
-
await validateReferenceRequirements(
|
|
17430
|
+
await validateReferenceRequirements(
|
|
17431
|
+
client,
|
|
17432
|
+
options.dataset,
|
|
17433
|
+
options.rowIds,
|
|
17434
|
+
options.evaluators,
|
|
17435
|
+
queryUrl
|
|
17436
|
+
);
|
|
17337
17437
|
const handle = await createReplay(client, {
|
|
17338
17438
|
dataset: options.dataset,
|
|
17339
17439
|
rowIds: options.rowIds,
|
|
@@ -17426,14 +17526,17 @@ var LocalEvalReplaySessionImpl = class {
|
|
|
17426
17526
|
var _a2;
|
|
17427
17527
|
const parent = (_a2 = import_api7.trace.getSpan(import_api7.context.active())) == null ? void 0 : _a2.spanContext();
|
|
17428
17528
|
if (!parent) return input.run(row);
|
|
17429
|
-
return withReplayTraceDestination(
|
|
17430
|
-
|
|
17431
|
-
|
|
17432
|
-
|
|
17433
|
-
|
|
17434
|
-
|
|
17435
|
-
|
|
17436
|
-
|
|
17529
|
+
return withReplayTraceDestination(
|
|
17530
|
+
{
|
|
17531
|
+
...replayDestination,
|
|
17532
|
+
parentSpanContext: {
|
|
17533
|
+
traceIdB64: Buffer.from(parent.traceId, "hex").toString("base64"),
|
|
17534
|
+
spanIdB64: Buffer.from(parent.spanId, "hex").toString("base64"),
|
|
17535
|
+
eventId: parent.traceId
|
|
17536
|
+
}
|
|
17537
|
+
},
|
|
17538
|
+
() => input.run(row)
|
|
17539
|
+
);
|
|
17437
17540
|
}
|
|
17438
17541
|
)
|
|
17439
17542
|
)
|
|
@@ -17659,7 +17762,9 @@ function reportReplay(client, queryUrl, replayId, body) {
|
|
|
17659
17762
|
}
|
|
17660
17763
|
async function validateReferenceRequirements(client, dataset, rowIds, evaluators, queryUrl) {
|
|
17661
17764
|
const selected = new Set(selectEvalDatasetRows(dataset, rowIds));
|
|
17662
|
-
const missing = dataset.cases.filter(
|
|
17765
|
+
const missing = dataset.cases.filter(
|
|
17766
|
+
(entry) => selected.has(entry.id) && entry.referenceTraceSha256 === void 0
|
|
17767
|
+
);
|
|
17663
17768
|
if (missing.length === 0) return;
|
|
17664
17769
|
for (const evaluator of evaluators) {
|
|
17665
17770
|
const requiresReference = typeof evaluator === "string" ? (await queryApi({
|
|
@@ -17671,7 +17776,9 @@ async function validateReferenceRequirements(client, dataset, rowIds, evaluators
|
|
|
17671
17776
|
})).data.requiresReference : evaluator.requiresReference === true;
|
|
17672
17777
|
if (requiresReference) {
|
|
17673
17778
|
const name = typeof evaluator === "string" ? evaluator : evaluator.slug;
|
|
17674
|
-
throw new Error(
|
|
17779
|
+
throw new Error(
|
|
17780
|
+
`Raindrop evaluator ${name} requires a reference trace. Missing references for rows: ${missing.map((entry) => entry.id).join(", ")}`
|
|
17781
|
+
);
|
|
17675
17782
|
}
|
|
17676
17783
|
}
|
|
17677
17784
|
}
|
|
@@ -17680,7 +17787,8 @@ async function replay(client, options) {
|
|
|
17680
17787
|
const evaluators = normalizeEvaluators(options.evaluators);
|
|
17681
17788
|
const queryUrl = (_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL;
|
|
17682
17789
|
const dataset = typeof options.dataset === "string" ? await readEvalDataset(client, options.dataset, { queryUrl }) : options.dataset;
|
|
17683
|
-
await
|
|
17790
|
+
const rowIds = options.runId === void 0 ? options.rowIds : await queuedSuiteRows(client, options.runId, dataset, options.rowIds, queryUrl);
|
|
17791
|
+
await validateReferenceRequirements(client, dataset, rowIds, evaluators, queryUrl);
|
|
17684
17792
|
const local = evaluators.flatMap(
|
|
17685
17793
|
(evaluator) => typeof evaluator === "string" ? [] : [{ evaluator, verdicts: [], durationMs: 0 }]
|
|
17686
17794
|
);
|
|
@@ -17692,7 +17800,7 @@ async function replay(client, options) {
|
|
|
17692
17800
|
}
|
|
17693
17801
|
)
|
|
17694
17802
|
);
|
|
17695
|
-
const handle = await createReplay(client, { ...options, dataset });
|
|
17803
|
+
const handle = options.runId === void 0 ? await createReplay(client, { ...options, dataset, rowIds }) : await claimSuiteRun(client, options.runId, dataset, rowIds != null ? rowIds : [], queryUrl);
|
|
17696
17804
|
const execution = await runReplayRows(client, handle, options);
|
|
17697
17805
|
const { reading } = execution;
|
|
17698
17806
|
const doneRowIds = new Set(
|
|
@@ -17765,6 +17873,33 @@ async function replay(client, options) {
|
|
|
17765
17873
|
evaluators: evaluatorResults.map(({ outcomes: _outcomes, ...evaluator }) => evaluator)
|
|
17766
17874
|
};
|
|
17767
17875
|
}
|
|
17876
|
+
async function queuedSuiteRows(client, runId, dataset, requestedRows, queryUrl) {
|
|
17877
|
+
const detail = await readReplayDetails(client, runId, { queryUrl });
|
|
17878
|
+
if (detail.replay.dataset_id !== dataset.dataset.id || detail.replay.dataset_version_id !== dataset.version.id) {
|
|
17879
|
+
throw new Error("Queued run dataset version does not match the suite");
|
|
17880
|
+
}
|
|
17881
|
+
if (detail.replay.status !== "queued") throw new Error("Only queued runs can be claimed");
|
|
17882
|
+
if (detail.replay.traces_expired) throw new Error("Queued run has expired");
|
|
17883
|
+
const rowIds = selectEvalDatasetRows(
|
|
17884
|
+
dataset,
|
|
17885
|
+
detail.rows.map((row) => row.row_id)
|
|
17886
|
+
);
|
|
17887
|
+
const requested = requestedRows && selectEvalDatasetRows(dataset, requestedRows);
|
|
17888
|
+
if (requested && (requested.length !== rowIds.length || requested.some((id, index) => id !== rowIds[index]))) {
|
|
17889
|
+
throw new Error("Queued run rows do not match the requested selection");
|
|
17890
|
+
}
|
|
17891
|
+
return rowIds;
|
|
17892
|
+
}
|
|
17893
|
+
async function claimSuiteRun(client, runId, dataset, rowIds, queryUrl) {
|
|
17894
|
+
const handle = await claimReplay(client, runId, { queryUrl });
|
|
17895
|
+
if (handle.replayId !== runId) throw new Error("Raindrop claimed a different run");
|
|
17896
|
+
if (!handle.ingestUrl) throw new Error("Run has already been claimed by another runner");
|
|
17897
|
+
assertSelectedRows(handle, rowIds);
|
|
17898
|
+
return {
|
|
17899
|
+
...handle,
|
|
17900
|
+
datasetVersion: { id: dataset.version.id, fingerprint: dataset.version.fingerprint }
|
|
17901
|
+
};
|
|
17902
|
+
}
|
|
17768
17903
|
async function runLocalEvaluators(evaluators, evalCase, row, result) {
|
|
17769
17904
|
await Promise.all(
|
|
17770
17905
|
evaluators.map(async (state2) => {
|
|
@@ -17882,11 +18017,11 @@ async function runReplayEvaluators(client, replayId, options) {
|
|
|
17882
18017
|
const deadlineAt = Date.now() + waitMs;
|
|
17883
18018
|
return Promise.all(
|
|
17884
18019
|
startedRuns.map(async (started) => {
|
|
17885
|
-
let run = await
|
|
18020
|
+
let run = await readReplayEvaluatorRun(client, queryUrl, started.runId, deadlineAt);
|
|
17886
18021
|
while (run.status === "running" && Date.now() < deadlineAt) {
|
|
17887
18022
|
await sleep(Math.min(pollIntervalMs, Math.max(0, deadlineAt - Date.now())));
|
|
17888
18023
|
if (Date.now() >= deadlineAt) break;
|
|
17889
|
-
run = await
|
|
18024
|
+
run = await readReplayEvaluatorRun(client, queryUrl, started.runId, deadlineAt);
|
|
17890
18025
|
}
|
|
17891
18026
|
if (run.status === "running") {
|
|
17892
18027
|
throw new Error(
|
|
@@ -17986,7 +18121,7 @@ function replayResultRows(handle, reading, evaluators) {
|
|
|
17986
18121
|
return replayRow;
|
|
17987
18122
|
});
|
|
17988
18123
|
}
|
|
17989
|
-
async function
|
|
18124
|
+
async function readReplayEvaluatorRun(client, queryUrl, runId, deadlineAt) {
|
|
17990
18125
|
const response = await queryApi({
|
|
17991
18126
|
client,
|
|
17992
18127
|
queryUrl,
|
|
@@ -18164,6 +18299,18 @@ function describeError(cause) {
|
|
|
18164
18299
|
// src/evals/index.ts
|
|
18165
18300
|
init_definition();
|
|
18166
18301
|
|
|
18302
|
+
// src/evals/rows.ts
|
|
18303
|
+
var EvalDatasetRowInputSchema = EvalDatasetCaseInputSchema;
|
|
18304
|
+
function datasetFromWire({ cases, ...dataset }) {
|
|
18305
|
+
return { ...dataset, rows: cases };
|
|
18306
|
+
}
|
|
18307
|
+
function datasetToWire({ rows, ...dataset }) {
|
|
18308
|
+
return { ...dataset, cases: rows };
|
|
18309
|
+
}
|
|
18310
|
+
async function readEvalDataset2(client, dataset, options) {
|
|
18311
|
+
return datasetFromWire(await readEvalDataset(client, dataset, options));
|
|
18312
|
+
}
|
|
18313
|
+
|
|
18167
18314
|
// src/evals/dataset.ts
|
|
18168
18315
|
var MAX_PUBLISH_RETRIES = 2;
|
|
18169
18316
|
var RETRY_BASE_DELAY_MS = 100;
|
|
@@ -18177,21 +18324,14 @@ var EvalDatasetPublishConflictError = class extends Error {
|
|
|
18177
18324
|
}
|
|
18178
18325
|
};
|
|
18179
18326
|
async function publishEvalDataset(client, options) {
|
|
18180
|
-
var _a, _b
|
|
18327
|
+
var _a, _b;
|
|
18181
18328
|
if (!((_a = client.connection.apiKey) == null ? void 0 : _a.trim())) {
|
|
18182
18329
|
throw new Error("Raindrop eval dataset publication requires a Query SDK API key (apiKey)");
|
|
18183
18330
|
}
|
|
18184
18331
|
const queryUrl = credentialEndpoint((_b = options.queryUrl) != null ? _b : DEFAULT_QUERY_URL).href;
|
|
18185
18332
|
const slug = EvalDatasetSlugSchema.parse(options.slug.trim());
|
|
18186
|
-
const
|
|
18187
|
-
const
|
|
18188
|
-
name: options.name,
|
|
18189
|
-
requestKey: randomUUID(),
|
|
18190
|
-
expectedCurrentVersionId: (_c = options.expectedCurrentVersionId) != null ? _c : null,
|
|
18191
|
-
fingerprint,
|
|
18192
|
-
cases: options.cases
|
|
18193
|
-
});
|
|
18194
|
-
validateUploadSizes(body);
|
|
18333
|
+
const body = await prepareEvalDatasetUpload(options);
|
|
18334
|
+
const fingerprint = body.fingerprint;
|
|
18195
18335
|
const path = `/v1/eval-datasets/${encodeURIComponent(slug)}`;
|
|
18196
18336
|
const response = await publishRequest({
|
|
18197
18337
|
client,
|
|
@@ -18204,7 +18344,20 @@ async function publishEvalDataset(client, options) {
|
|
|
18204
18344
|
if (manifest.dataset.slug !== slug || manifest.dataset.id !== manifest.version.datasetId || manifest.version.fingerprint !== fingerprint) {
|
|
18205
18345
|
throw new Error(`Raindrop eval dataset ${slug} returned mismatched publication identity`);
|
|
18206
18346
|
}
|
|
18207
|
-
return manifest;
|
|
18347
|
+
return datasetFromWire(manifest);
|
|
18348
|
+
}
|
|
18349
|
+
async function prepareEvalDatasetUpload(options) {
|
|
18350
|
+
var _a;
|
|
18351
|
+
const fingerprint = await evalDatasetVersionFingerprint(options.rows);
|
|
18352
|
+
const body = PublishEvalDatasetVersionInputSchema.parse({
|
|
18353
|
+
name: options.name,
|
|
18354
|
+
requestKey: randomUUID(),
|
|
18355
|
+
expectedCurrentVersionId: (_a = options.expectedCurrentVersionId) != null ? _a : null,
|
|
18356
|
+
fingerprint,
|
|
18357
|
+
cases: options.rows
|
|
18358
|
+
});
|
|
18359
|
+
validateUploadSizes(body);
|
|
18360
|
+
return body;
|
|
18208
18361
|
}
|
|
18209
18362
|
function validateUploadSizes(body) {
|
|
18210
18363
|
for (const [index, entry] of body.cases.entries()) {
|
|
@@ -18212,7 +18365,7 @@ function validateUploadSizes(body) {
|
|
|
18212
18365
|
const bytes2 = encodedBytes(entry.referenceTrace);
|
|
18213
18366
|
if (bytes2 > EVAL_DATASET_MAX_REFERENCE_BYTES) {
|
|
18214
18367
|
throw new Error(
|
|
18215
|
-
`Raindrop eval dataset
|
|
18368
|
+
`Raindrop eval dataset row ${entry.id} at index ${index} exceeds ${EVAL_DATASET_MAX_REFERENCE_BYTES} reference bytes`
|
|
18216
18369
|
);
|
|
18217
18370
|
}
|
|
18218
18371
|
}
|
|
@@ -18295,39 +18448,278 @@ function sleep2(ms) {
|
|
|
18295
18448
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
18296
18449
|
}
|
|
18297
18450
|
|
|
18451
|
+
// src/evals/publish.ts
|
|
18452
|
+
var import_zod6 = require("zod");
|
|
18453
|
+
init_definition();
|
|
18454
|
+
|
|
18455
|
+
// src/evals/runs.ts
|
|
18456
|
+
async function readEvalRun(client, runId, options = {}) {
|
|
18457
|
+
const detail = await readReplayDetails(client, runId, options);
|
|
18458
|
+
if (detail.eval_runs.length >= 200) {
|
|
18459
|
+
throw new Error("Run evaluation history exceeds the read limit");
|
|
18460
|
+
}
|
|
18461
|
+
const evaluators = await Promise.all(
|
|
18462
|
+
detail.eval_runs.map(async (evaluation) => {
|
|
18463
|
+
var _a;
|
|
18464
|
+
const result = await readReplayEvaluatorRun(
|
|
18465
|
+
client,
|
|
18466
|
+
(_a = options.queryUrl) != null ? _a : DEFAULT_QUERY_URL,
|
|
18467
|
+
evaluation.id
|
|
18468
|
+
);
|
|
18469
|
+
if (result.id !== evaluation.id) throw new Error("Raindrop returned a different evaluation");
|
|
18470
|
+
return {
|
|
18471
|
+
...result,
|
|
18472
|
+
evaluator: result.evalSlug,
|
|
18473
|
+
evaluatorId: evaluation.eval_id,
|
|
18474
|
+
programVersion: evaluation.program_version,
|
|
18475
|
+
executedBy: evaluation.executed_by
|
|
18476
|
+
};
|
|
18477
|
+
})
|
|
18478
|
+
);
|
|
18479
|
+
return {
|
|
18480
|
+
runId: detail.replay.id,
|
|
18481
|
+
datasetId: detail.replay.dataset_id,
|
|
18482
|
+
datasetVersionId: detail.replay.dataset_version_id,
|
|
18483
|
+
status: detail.replay.status,
|
|
18484
|
+
counts: detail.replay.counts,
|
|
18485
|
+
tracesExpireAt: detail.replay.traces_expire_at,
|
|
18486
|
+
tracesExpired: detail.replay.traces_expired,
|
|
18487
|
+
rows: detail.rows.map((row) => ({
|
|
18488
|
+
rowId: row.row_id,
|
|
18489
|
+
attempt: row.attempt,
|
|
18490
|
+
status: row.status,
|
|
18491
|
+
traceId: row.trace_id,
|
|
18492
|
+
error: row.error
|
|
18493
|
+
})),
|
|
18494
|
+
evaluators
|
|
18495
|
+
};
|
|
18496
|
+
}
|
|
18497
|
+
async function evaluateEvalRun(client, runId, options) {
|
|
18498
|
+
return { runId, evaluators: await evaluateReplay(client, runId, options) };
|
|
18499
|
+
}
|
|
18500
|
+
|
|
18501
|
+
// src/evals/publish.ts
|
|
18502
|
+
var EvaluatorIdentitySchema = import_zod6.z.object({
|
|
18503
|
+
id: import_zod6.z.string().uuid(),
|
|
18504
|
+
slug: import_zod6.z.string(),
|
|
18505
|
+
programVersion: import_zod6.z.number().int().positive(),
|
|
18506
|
+
outputType: import_zod6.z.enum(["boolean", "score", "number"])
|
|
18507
|
+
});
|
|
18508
|
+
var EvaluatorSourceSchema = EvaluatorIdentitySchema.extend({
|
|
18509
|
+
name: import_zod6.z.string(),
|
|
18510
|
+
description: import_zod6.z.string().nullable(),
|
|
18511
|
+
intent: import_zod6.z.string(),
|
|
18512
|
+
kind: import_zod6.z.enum(["judge", "deterministic"]),
|
|
18513
|
+
programSource: import_zod6.z.string(),
|
|
18514
|
+
rules: import_zod6.z.array(import_zod6.z.string())
|
|
18515
|
+
});
|
|
18516
|
+
var EvalPublishError = class extends Error {
|
|
18517
|
+
constructor(status, message) {
|
|
18518
|
+
super(message);
|
|
18519
|
+
this.status = status;
|
|
18520
|
+
this.name = "EvalPublishError";
|
|
18521
|
+
}
|
|
18522
|
+
};
|
|
18523
|
+
var ComparisonRunSchema = import_zod6.z.object({
|
|
18524
|
+
id: import_zod6.z.string().uuid(),
|
|
18525
|
+
programVersion: import_zod6.z.number().int().positive()
|
|
18526
|
+
});
|
|
18527
|
+
var ComparisonLinkSchema = import_zod6.z.object({
|
|
18528
|
+
url: import_zod6.z.string().url(),
|
|
18529
|
+
first: ComparisonRunSchema,
|
|
18530
|
+
second: ComparisonRunSchema
|
|
18531
|
+
});
|
|
18532
|
+
async function compareEvalRuns(client, options) {
|
|
18533
|
+
const [first, second] = await Promise.all([
|
|
18534
|
+
readEvalRun(client, options.beforeRunId, options),
|
|
18535
|
+
readEvalRun(client, options.afterRunId, options)
|
|
18536
|
+
]);
|
|
18537
|
+
const candidates = first.evaluators.filter(
|
|
18538
|
+
(entry) => (!options.evaluator || entry.evaluator === options.evaluator) && second.evaluators.some((other) => other.evaluatorId === entry.evaluatorId)
|
|
18539
|
+
);
|
|
18540
|
+
const selected = candidates[0];
|
|
18541
|
+
if (!selected || candidates.length !== 1) {
|
|
18542
|
+
throw new Error(
|
|
18543
|
+
"Comparison requires one shared evaluator; specify evaluator when comparing a suite"
|
|
18544
|
+
);
|
|
18545
|
+
}
|
|
18546
|
+
const matches = second.evaluators.filter((entry) => entry.evaluatorId === selected.evaluatorId);
|
|
18547
|
+
const counterpart = matches[0];
|
|
18548
|
+
if (!counterpart || matches.length !== 1) {
|
|
18549
|
+
throw new Error("Comparison is ambiguous because this evaluator graded the run more than once");
|
|
18550
|
+
}
|
|
18551
|
+
if (selected.status === "running" || counterpart.status === "running") {
|
|
18552
|
+
throw new Error("Wait for both evaluations to finish before comparing");
|
|
18553
|
+
}
|
|
18554
|
+
const before = selected.id;
|
|
18555
|
+
const after = counterpart.id;
|
|
18556
|
+
const { data } = import_zod6.z.object({ data: ComparisonLinkSchema }).parse(await evalRequest(client, `/v1/eval-runs/${before}/compare/${after}`, options));
|
|
18557
|
+
if (data.first.id !== before || data.second.id !== after) {
|
|
18558
|
+
throw new Error("Raindrop returned a different comparison pair");
|
|
18559
|
+
}
|
|
18560
|
+
return {
|
|
18561
|
+
...data,
|
|
18562
|
+
beforeRunId: first.runId,
|
|
18563
|
+
afterRunId: second.runId,
|
|
18564
|
+
evaluator: selected.evaluator,
|
|
18565
|
+
sameEvaluatorVersion: data.first.programVersion === data.second.programVersion
|
|
18566
|
+
};
|
|
18567
|
+
}
|
|
18568
|
+
async function readEvaluator(client, slug, options = {}) {
|
|
18569
|
+
const { data } = import_zod6.z.object({ data: EvaluatorSourceSchema }).parse(await evalRequest(client, `/v1/evals/${encodeURIComponent(slug)}`, options));
|
|
18570
|
+
if (data.slug !== slug) throw new Error("Raindrop returned a different evaluator slug");
|
|
18571
|
+
return defineEvaluatorProgram({
|
|
18572
|
+
slug: data.slug,
|
|
18573
|
+
name: data.name,
|
|
18574
|
+
output: data.outputType,
|
|
18575
|
+
scope: "batch",
|
|
18576
|
+
executionMode: data.kind,
|
|
18577
|
+
source: data.programSource,
|
|
18578
|
+
description: data.description,
|
|
18579
|
+
intent: data.intent,
|
|
18580
|
+
rules: data.rules,
|
|
18581
|
+
expected: { evalId: data.id, programVersion: data.programVersion }
|
|
18582
|
+
});
|
|
18583
|
+
}
|
|
18584
|
+
async function publishEvalSuite(client, definition, options = {}) {
|
|
18585
|
+
const suite = defineEvalSuite(definition);
|
|
18586
|
+
const localDataset = isEvalDataset(suite.dataset) ? suite.dataset : void 0;
|
|
18587
|
+
const slug = EvalDatasetSlugSchema.parse(localDataset ? localDataset.id : suite.dataset);
|
|
18588
|
+
const rows = localDataset == null ? void 0 : localDataset.rows.map((row) => {
|
|
18589
|
+
if (row.output !== null) {
|
|
18590
|
+
throw new Error(
|
|
18591
|
+
"Publish reference evidence with publishEvalDataset first; a local row output is not a reference trace"
|
|
18592
|
+
);
|
|
18593
|
+
}
|
|
18594
|
+
return EvalDatasetCaseInputSchema.parse({
|
|
18595
|
+
id: row.id,
|
|
18596
|
+
name: row.name,
|
|
18597
|
+
input: row.input,
|
|
18598
|
+
properties: row.properties
|
|
18599
|
+
});
|
|
18600
|
+
});
|
|
18601
|
+
if (localDataset && rows) {
|
|
18602
|
+
await prepareEvalDatasetUpload({
|
|
18603
|
+
slug,
|
|
18604
|
+
name: localDataset.name,
|
|
18605
|
+
rows,
|
|
18606
|
+
expectedCurrentVersionId: options.expectedCurrentDatasetVersionId
|
|
18607
|
+
});
|
|
18608
|
+
}
|
|
18609
|
+
const entries = suite.evaluators.map((entry) => {
|
|
18610
|
+
if (typeof entry.evaluator === "string") return entry;
|
|
18611
|
+
if ("kind" in entry.evaluator && entry.evaluator.kind === "program") {
|
|
18612
|
+
if (!entry.evaluator.expected) {
|
|
18613
|
+
throw new Error(
|
|
18614
|
+
"Evaluator source publication is not supported. Create the evaluator in Raindrop and reference its slug."
|
|
18615
|
+
);
|
|
18616
|
+
}
|
|
18617
|
+
return { ...entry, evaluator: defineEvaluatorProgram(entry.evaluator) };
|
|
18618
|
+
}
|
|
18619
|
+
return entry;
|
|
18620
|
+
});
|
|
18621
|
+
const evaluators = [];
|
|
18622
|
+
for (const entry of entries) {
|
|
18623
|
+
if (typeof entry.evaluator === "string") {
|
|
18624
|
+
evaluators.push({
|
|
18625
|
+
...entry,
|
|
18626
|
+
evaluator: await readEvaluator(client, entry.evaluator, options)
|
|
18627
|
+
});
|
|
18628
|
+
} else {
|
|
18629
|
+
evaluators.push(entry);
|
|
18630
|
+
}
|
|
18631
|
+
}
|
|
18632
|
+
const dataset = localDataset && rows ? await publishEvalDataset(client, {
|
|
18633
|
+
slug,
|
|
18634
|
+
name: localDataset.name,
|
|
18635
|
+
rows,
|
|
18636
|
+
queryUrl: options.queryUrl,
|
|
18637
|
+
expectedCurrentVersionId: options.expectedCurrentDatasetVersionId
|
|
18638
|
+
}) : await readEvalDataset2(client, slug, {
|
|
18639
|
+
queryUrl: options.queryUrl,
|
|
18640
|
+
versionId: suite.datasetVersionId
|
|
18641
|
+
});
|
|
18642
|
+
return defineEvalSuite({
|
|
18643
|
+
...suite,
|
|
18644
|
+
dataset: slug,
|
|
18645
|
+
datasetVersionId: dataset.version.id,
|
|
18646
|
+
evaluators
|
|
18647
|
+
});
|
|
18648
|
+
}
|
|
18649
|
+
async function evalRequest(client, path, options) {
|
|
18650
|
+
var _a, _b;
|
|
18651
|
+
const apiKey = (_a = client.connection.apiKey) == null ? void 0 : _a.trim();
|
|
18652
|
+
if (!apiKey) throw new Error("Raindrop evals require a Query SDK API key (apiKey)");
|
|
18653
|
+
const base = credentialEndpoint((_b = options.queryUrl) != null ? _b : DEFAULT_QUERY_URL).href.replace(/\/+$/, "");
|
|
18654
|
+
const headers = {
|
|
18655
|
+
Authorization: `Bearer ${apiKey}`,
|
|
18656
|
+
"Content-Type": "application/json"
|
|
18657
|
+
};
|
|
18658
|
+
if (client.connection.projectId !== void 0)
|
|
18659
|
+
headers["X-Raindrop-Project-Id"] = client.connection.projectId;
|
|
18660
|
+
const controller = new AbortController();
|
|
18661
|
+
const timeout = setTimeout(() => controller.abort(), 3e4);
|
|
18662
|
+
try {
|
|
18663
|
+
const response = await runWithTracingSuppressed(
|
|
18664
|
+
() => fetch(`${base}${path}`, {
|
|
18665
|
+
method: "GET",
|
|
18666
|
+
headers,
|
|
18667
|
+
redirect: "error",
|
|
18668
|
+
signal: controller.signal
|
|
18669
|
+
})
|
|
18670
|
+
);
|
|
18671
|
+
const body = await response.text();
|
|
18672
|
+
if (!response.ok)
|
|
18673
|
+
throw new EvalPublishError(
|
|
18674
|
+
response.status,
|
|
18675
|
+
`Raindrop GET ${path}: ${response.status} ${body}`
|
|
18676
|
+
);
|
|
18677
|
+
return JSON.parse(body);
|
|
18678
|
+
} finally {
|
|
18679
|
+
clearTimeout(timeout);
|
|
18680
|
+
}
|
|
18681
|
+
}
|
|
18682
|
+
|
|
18683
|
+
// src/evals/trace-tools.ts
|
|
18684
|
+
init_trace();
|
|
18685
|
+
var traceTools = Object.freeze({
|
|
18686
|
+
getOutput: traceOutput,
|
|
18687
|
+
getToolCalls: traceToolCalls
|
|
18688
|
+
});
|
|
18689
|
+
|
|
18298
18690
|
// src/evals/index.ts
|
|
18299
18691
|
init_trace();
|
|
18300
18692
|
init_portable();
|
|
18301
18693
|
|
|
18302
18694
|
// src/evals/pull.ts
|
|
18303
|
-
var
|
|
18695
|
+
var import_zod9 = require("zod");
|
|
18304
18696
|
init_definition();
|
|
18305
18697
|
init_portable();
|
|
18306
18698
|
var SNAPSHOT_VERSION = 1;
|
|
18307
|
-
var PulledEvaluatorSchema =
|
|
18308
|
-
slug:
|
|
18309
|
-
name:
|
|
18310
|
-
expectedSha256:
|
|
18311
|
-
artifact:
|
|
18699
|
+
var PulledEvaluatorSchema = import_zod9.z.object({
|
|
18700
|
+
slug: import_zod9.z.string().min(1),
|
|
18701
|
+
name: import_zod9.z.string().min(1),
|
|
18702
|
+
expectedSha256: import_zod9.z.string().regex(/^[a-f0-9]{64}$/),
|
|
18703
|
+
artifact: import_zod9.z.unknown()
|
|
18312
18704
|
});
|
|
18313
|
-
var EvalSnapshotSchema =
|
|
18314
|
-
schemaVersion:
|
|
18315
|
-
pulledAt:
|
|
18316
|
-
dataset:
|
|
18317
|
-
reference:
|
|
18318
|
-
datasetId:
|
|
18319
|
-
fingerprint:
|
|
18320
|
-
rows:
|
|
18321
|
-
|
|
18322
|
-
id:
|
|
18323
|
-
name:
|
|
18324
|
-
input:
|
|
18325
|
-
output:
|
|
18326
|
-
properties:
|
|
18705
|
+
var EvalSnapshotSchema = import_zod9.z.object({
|
|
18706
|
+
schemaVersion: import_zod9.z.literal(SNAPSHOT_VERSION),
|
|
18707
|
+
pulledAt: import_zod9.z.string().datetime(),
|
|
18708
|
+
dataset: import_zod9.z.object({
|
|
18709
|
+
reference: import_zod9.z.string().min(1),
|
|
18710
|
+
datasetId: import_zod9.z.string().uuid(),
|
|
18711
|
+
fingerprint: import_zod9.z.string().regex(/^[a-f0-9]{64}$/),
|
|
18712
|
+
rows: import_zod9.z.array(
|
|
18713
|
+
import_zod9.z.object({
|
|
18714
|
+
id: import_zod9.z.string(),
|
|
18715
|
+
name: import_zod9.z.string(),
|
|
18716
|
+
input: import_zod9.z.string().nullable(),
|
|
18717
|
+
output: import_zod9.z.string().nullable(),
|
|
18718
|
+
properties: import_zod9.z.record(import_zod9.z.string())
|
|
18327
18719
|
})
|
|
18328
18720
|
)
|
|
18329
18721
|
}),
|
|
18330
|
-
evaluators:
|
|
18722
|
+
evaluators: import_zod9.z.array(PulledEvaluatorSchema)
|
|
18331
18723
|
});
|
|
18332
18724
|
async function pullEval(client, options) {
|
|
18333
18725
|
var _a;
|
|
@@ -18392,11 +18784,11 @@ async function pullEvaluator(client, queryUrl, rawSlug) {
|
|
|
18392
18784
|
`Raindrop evaluator ${rawSlug} pull failed with ${response.status}: ${await response.text()}`
|
|
18393
18785
|
);
|
|
18394
18786
|
}
|
|
18395
|
-
const body =
|
|
18396
|
-
data:
|
|
18397
|
-
slug:
|
|
18398
|
-
name:
|
|
18399
|
-
artifact:
|
|
18787
|
+
const body = import_zod9.z.object({
|
|
18788
|
+
data: import_zod9.z.object({
|
|
18789
|
+
slug: import_zod9.z.string(),
|
|
18790
|
+
name: import_zod9.z.string(),
|
|
18791
|
+
artifact: import_zod9.z.unknown().nullable()
|
|
18400
18792
|
})
|
|
18401
18793
|
}).parse(await response.json());
|
|
18402
18794
|
if (!body.data.artifact)
|
|
@@ -18425,7 +18817,7 @@ async function materializeSnapshot(snapshotPath, raw, judges) {
|
|
|
18425
18817
|
id: snapshot.dataset.datasetId,
|
|
18426
18818
|
name: snapshot.dataset.reference,
|
|
18427
18819
|
version: snapshot.dataset.fingerprint,
|
|
18428
|
-
|
|
18820
|
+
rows: snapshot.dataset.rows,
|
|
18429
18821
|
remote: {
|
|
18430
18822
|
reference: snapshot.dataset.reference,
|
|
18431
18823
|
datasetId: snapshot.dataset.datasetId,
|
|
@@ -18533,46 +18925,46 @@ function outputMismatch(name, output) {
|
|
|
18533
18925
|
init_portable();
|
|
18534
18926
|
|
|
18535
18927
|
// src/evals/workshop.ts
|
|
18536
|
-
var
|
|
18537
|
-
var WorkshopRunSchema =
|
|
18538
|
-
id:
|
|
18539
|
-
event_id:
|
|
18540
|
-
name:
|
|
18541
|
-
event_name:
|
|
18542
|
-
user_id:
|
|
18543
|
-
convo_id:
|
|
18544
|
-
started_at:
|
|
18545
|
-
last_updated_at:
|
|
18546
|
-
metadata:
|
|
18547
|
-
model:
|
|
18548
|
-
finished:
|
|
18928
|
+
var import_zod10 = require("zod");
|
|
18929
|
+
var WorkshopRunSchema = import_zod10.z.object({
|
|
18930
|
+
id: import_zod10.z.string(),
|
|
18931
|
+
event_id: import_zod10.z.string().nullable(),
|
|
18932
|
+
name: import_zod10.z.string().nullable(),
|
|
18933
|
+
event_name: import_zod10.z.string().nullable(),
|
|
18934
|
+
user_id: import_zod10.z.string().nullable(),
|
|
18935
|
+
convo_id: import_zod10.z.string().nullable(),
|
|
18936
|
+
started_at: import_zod10.z.number().nullable(),
|
|
18937
|
+
last_updated_at: import_zod10.z.number().nullable(),
|
|
18938
|
+
metadata: import_zod10.z.string().nullable(),
|
|
18939
|
+
model: import_zod10.z.string().nullable(),
|
|
18940
|
+
finished: import_zod10.z.number().nullable()
|
|
18549
18941
|
});
|
|
18550
|
-
var WorkshopSpanSchema =
|
|
18551
|
-
id:
|
|
18552
|
-
run_id:
|
|
18553
|
-
parent_span_id:
|
|
18554
|
-
name:
|
|
18555
|
-
span_type:
|
|
18556
|
-
status:
|
|
18557
|
-
input_payload:
|
|
18558
|
-
output_payload:
|
|
18559
|
-
start_time_ms:
|
|
18560
|
-
end_time_ms:
|
|
18561
|
-
duration_ms:
|
|
18562
|
-
model:
|
|
18563
|
-
provider:
|
|
18564
|
-
input_tokens:
|
|
18565
|
-
output_tokens:
|
|
18566
|
-
attributes:
|
|
18942
|
+
var WorkshopSpanSchema = import_zod10.z.object({
|
|
18943
|
+
id: import_zod10.z.string(),
|
|
18944
|
+
run_id: import_zod10.z.string(),
|
|
18945
|
+
parent_span_id: import_zod10.z.string().nullable(),
|
|
18946
|
+
name: import_zod10.z.string(),
|
|
18947
|
+
span_type: import_zod10.z.string().nullable(),
|
|
18948
|
+
status: import_zod10.z.string().nullable(),
|
|
18949
|
+
input_payload: import_zod10.z.string().nullable(),
|
|
18950
|
+
output_payload: import_zod10.z.string().nullable(),
|
|
18951
|
+
start_time_ms: import_zod10.z.number().nullable(),
|
|
18952
|
+
end_time_ms: import_zod10.z.number().nullable(),
|
|
18953
|
+
duration_ms: import_zod10.z.number().nullable(),
|
|
18954
|
+
model: import_zod10.z.string().nullable(),
|
|
18955
|
+
provider: import_zod10.z.string().nullable(),
|
|
18956
|
+
input_tokens: import_zod10.z.number().nullable(),
|
|
18957
|
+
output_tokens: import_zod10.z.number().nullable(),
|
|
18958
|
+
attributes: import_zod10.z.string().nullable()
|
|
18567
18959
|
});
|
|
18568
|
-
var WorkshopTraceEnvelopeSchema =
|
|
18569
|
-
status:
|
|
18570
|
-
correlationId:
|
|
18571
|
-
traceId:
|
|
18572
|
-
rootSpanId:
|
|
18573
|
-
spanCount:
|
|
18960
|
+
var WorkshopTraceEnvelopeSchema = import_zod10.z.object({
|
|
18961
|
+
status: import_zod10.z.literal("complete"),
|
|
18962
|
+
correlationId: import_zod10.z.string(),
|
|
18963
|
+
traceId: import_zod10.z.string(),
|
|
18964
|
+
rootSpanId: import_zod10.z.string(),
|
|
18965
|
+
spanCount: import_zod10.z.number().int().positive(),
|
|
18574
18966
|
run: WorkshopRunSchema,
|
|
18575
|
-
spans:
|
|
18967
|
+
spans: import_zod10.z.array(WorkshopSpanSchema).min(1)
|
|
18576
18968
|
});
|
|
18577
18969
|
async function readWorkshopTrace(input) {
|
|
18578
18970
|
if (!Number.isFinite(input.waitMs) || input.waitMs < 0)
|
|
@@ -18658,9 +19050,12 @@ function serializableOutput(value) {
|
|
|
18658
19050
|
var DEFAULT_WORKSHOP_TRACE_WAIT_MS = 3e4;
|
|
18659
19051
|
function createEvalSuiteRun(client, definition, options) {
|
|
18660
19052
|
const validated = validateDefinition(definition);
|
|
19053
|
+
if (validated.datasetVersionId && "selection" in options && options.selection.dataset.version.id !== validated.datasetVersionId) {
|
|
19054
|
+
throw new Error("Selected dataset version does not match the published suite");
|
|
19055
|
+
}
|
|
18661
19056
|
if (options.destination.kind === "raindrop") {
|
|
18662
19057
|
if (!("selection" in options)) {
|
|
18663
|
-
throw new Error("Raindrop remote eval sessions require selected
|
|
19058
|
+
throw new Error("Raindrop remote eval sessions require selected rows");
|
|
18664
19059
|
}
|
|
18665
19060
|
if (typeof validated.dataset !== "string") {
|
|
18666
19061
|
throw new Error("Raindrop remote eval sessions require a dataset slug");
|
|
@@ -18675,16 +19070,16 @@ function createEvalSuiteRun(client, definition, options) {
|
|
|
18675
19070
|
return typeof evaluator === "string" ? [] : [evaluator];
|
|
18676
19071
|
});
|
|
18677
19072
|
if (localEvaluators.length !== validated.evaluators.length) {
|
|
18678
|
-
throw new Error("Raindrop
|
|
19073
|
+
throw new Error("Raindrop row-scoped eval sessions support local evaluators only");
|
|
18679
19074
|
}
|
|
18680
19075
|
const selection = validateRemoteSelection(validated.dataset, options.selection);
|
|
18681
|
-
if (!selection) throw new Error("Raindrop remote eval sessions require selected
|
|
18682
|
-
return new
|
|
19076
|
+
if (!selection) throw new Error("Raindrop remote eval sessions require selected rows");
|
|
19077
|
+
return new RaindropRowEvalRun(
|
|
18683
19078
|
client,
|
|
18684
19079
|
validated,
|
|
18685
19080
|
options.destination,
|
|
18686
19081
|
options.selection.dataset,
|
|
18687
|
-
selection.
|
|
19082
|
+
selection.rowIds,
|
|
18688
19083
|
localEvaluators
|
|
18689
19084
|
);
|
|
18690
19085
|
}
|
|
@@ -18709,34 +19104,56 @@ function createEvalSuiteRun(client, definition, options) {
|
|
|
18709
19104
|
options.destination
|
|
18710
19105
|
);
|
|
18711
19106
|
}
|
|
18712
|
-
async function runEvalSuite(client, definition, options) {
|
|
18713
|
-
var _a, _b, _c, _d;
|
|
18714
|
-
|
|
18715
|
-
|
|
18716
|
-
if (options.
|
|
18717
|
-
|
|
19107
|
+
async function runEvalSuite(client, definition, options = {}) {
|
|
19108
|
+
var _a, _b, _c, _d, _e, _f, _g, _h;
|
|
19109
|
+
let validated = validateDefinition(definition);
|
|
19110
|
+
const destination = (_a = options.destination) != null ? _a : { kind: "raindrop" };
|
|
19111
|
+
if (validated.datasetVersionId && ((_b = options.selection) == null ? void 0 : _b.dataset) && options.selection.dataset.version.id !== validated.datasetVersionId) {
|
|
19112
|
+
throw new Error("Selected dataset version does not match the published suite");
|
|
19113
|
+
}
|
|
19114
|
+
if (destination.kind === "workshop") {
|
|
19115
|
+
if (options.runId !== void 0) throw new Error("Queued runs are only supported in Raindrop");
|
|
19116
|
+
const session = createEvalSuiteRun(client, validated, { destination });
|
|
18718
19117
|
await session.run({ selection: options.selection, attempt: options.attempt });
|
|
18719
19118
|
return session.finish();
|
|
18720
19119
|
}
|
|
18721
|
-
|
|
19120
|
+
const alreadyPublished = isPublishedEvalSuite(validated);
|
|
19121
|
+
if (typeof validated.dataset === "string" && ((_c = options.selection) == null ? void 0 : _c.dataset)) {
|
|
19122
|
+
validateRemoteSelection(validated.dataset, options.selection);
|
|
19123
|
+
validated = defineEvalSuite({
|
|
19124
|
+
...validated,
|
|
19125
|
+
datasetVersionId: options.selection.dataset.version.id
|
|
19126
|
+
});
|
|
19127
|
+
}
|
|
19128
|
+
if (!alreadyPublished) {
|
|
19129
|
+
validated = await publishEvalSuite(client, validated, {
|
|
19130
|
+
queryUrl: destination.queryUrl,
|
|
19131
|
+
expectedCurrentDatasetVersionId: options.expectedCurrentDatasetVersionId
|
|
19132
|
+
});
|
|
19133
|
+
}
|
|
19134
|
+
if (((_d = options.selection) == null ? void 0 : _d.dataset) && options.selection.dataset.version.id !== validated.datasetVersionId) {
|
|
19135
|
+
throw new Error("Selected dataset version does not match the published suite");
|
|
19136
|
+
}
|
|
19137
|
+
if (isRowScopedLocalEval(validated) && options.runId === void 0) {
|
|
18722
19138
|
if (typeof validated.dataset !== "string") {
|
|
18723
19139
|
throw new Error("Raindrop remote execution requires an eval dataset slug");
|
|
18724
19140
|
}
|
|
18725
|
-
const dataset = (
|
|
18726
|
-
queryUrl:
|
|
19141
|
+
const dataset = (_f = (_e = options.selection) == null ? void 0 : _e.dataset) != null ? _f : await readEvalDataset2(client, validated.dataset, {
|
|
19142
|
+
queryUrl: destination.queryUrl,
|
|
19143
|
+
versionId: validated.datasetVersionId
|
|
18727
19144
|
});
|
|
18728
19145
|
const selection = validateRemoteSelection(validated.dataset, {
|
|
18729
19146
|
dataset,
|
|
18730
|
-
|
|
19147
|
+
rowIds: (_h = (_g = options.selection) == null ? void 0 : _g.rowIds) != null ? _h : dataset.rows.map((entry) => entry.id)
|
|
18731
19148
|
});
|
|
18732
|
-
if (!selection) throw new Error("Raindrop remote eval requires selected
|
|
19149
|
+
if (!selection) throw new Error("Raindrop remote eval requires selected rows");
|
|
18733
19150
|
const session = createEvalSuiteRun(client, validated, {
|
|
18734
|
-
destination
|
|
18735
|
-
selection: { dataset,
|
|
19151
|
+
destination,
|
|
19152
|
+
selection: { dataset, rowIds: selection.rowIds }
|
|
18736
19153
|
});
|
|
18737
19154
|
const attempts = await Promise.allSettled(
|
|
18738
|
-
selection.
|
|
18739
|
-
(
|
|
19155
|
+
selection.rowIds.map(
|
|
19156
|
+
(rowId) => session.run({ selection: { dataset, rowIds: [rowId] }, attempt: 0 })
|
|
18740
19157
|
)
|
|
18741
19158
|
);
|
|
18742
19159
|
const failures = attempts.filter(
|
|
@@ -18749,7 +19166,7 @@ async function runEvalSuite(client, definition, options) {
|
|
|
18749
19166
|
if (failures.length > 0) {
|
|
18750
19167
|
throw new AggregateError(
|
|
18751
19168
|
[...failures.map((failure) => failure.reason), cause],
|
|
18752
|
-
"Raindrop eval
|
|
19169
|
+
"Raindrop eval row execution and run finalization failed"
|
|
18753
19170
|
);
|
|
18754
19171
|
}
|
|
18755
19172
|
throw cause;
|
|
@@ -18759,33 +19176,33 @@ async function runEvalSuite(client, definition, options) {
|
|
|
18759
19176
|
if (failures.length > 1) {
|
|
18760
19177
|
throw new AggregateError(
|
|
18761
19178
|
failures.map((failure) => failure.reason),
|
|
18762
|
-
"Multiple Raindrop eval
|
|
19179
|
+
"Multiple Raindrop eval rows failed"
|
|
18763
19180
|
);
|
|
18764
19181
|
}
|
|
18765
19182
|
return finished;
|
|
18766
19183
|
}
|
|
18767
|
-
return runRaindropEval(client, validated,
|
|
19184
|
+
return runRaindropEval(client, validated, destination, options);
|
|
18768
19185
|
}
|
|
18769
|
-
function
|
|
19186
|
+
function isRowScopedLocalEval(definition) {
|
|
18770
19187
|
return definition.evaluators.every(
|
|
18771
|
-
(entry) => typeof entry.evaluator !== "string" && entry.evaluator.scope === "
|
|
19188
|
+
(entry) => typeof entry.evaluator !== "string" && entry.evaluator.scope === "row" && !("kind" in entry.evaluator)
|
|
18772
19189
|
);
|
|
18773
19190
|
}
|
|
18774
|
-
var
|
|
18775
|
-
constructor(client, definition, destination, dataset,
|
|
19191
|
+
var RaindropRowEvalRun = class {
|
|
19192
|
+
constructor(client, definition, destination, dataset, rowIds, evaluators) {
|
|
18776
19193
|
this.client = client;
|
|
18777
19194
|
this.definition = definition;
|
|
18778
19195
|
this.destination = destination;
|
|
18779
19196
|
this.dataset = dataset;
|
|
18780
|
-
this.
|
|
19197
|
+
this.rowIds = rowIds;
|
|
18781
19198
|
this.evaluators = evaluators;
|
|
18782
19199
|
this.id = randomId();
|
|
18783
19200
|
this.active = /* @__PURE__ */ new Set();
|
|
18784
19201
|
this.completed = [];
|
|
18785
19202
|
this.attempts = /* @__PURE__ */ new Set();
|
|
18786
19203
|
this.state = "open";
|
|
18787
|
-
this.
|
|
18788
|
-
this.
|
|
19204
|
+
this.runningRows = 0;
|
|
19205
|
+
this.rowWaiters = [];
|
|
18789
19206
|
}
|
|
18790
19207
|
async run(options = {}) {
|
|
18791
19208
|
if (this.state !== "open") throw new Error(`Raindrop eval run ${this.id} is already finishing`);
|
|
@@ -18805,24 +19222,24 @@ var RaindropCaseEvalRun = class {
|
|
|
18805
19222
|
}
|
|
18806
19223
|
async runOnce(options) {
|
|
18807
19224
|
var _a, _b;
|
|
18808
|
-
const selected = (_a = options.selection) == null ? void 0 : _a.
|
|
19225
|
+
const selected = (_a = options.selection) == null ? void 0 : _a.rowIds;
|
|
18809
19226
|
if (!selected || selected.length !== 1) {
|
|
18810
|
-
throw new Error("Raindrop
|
|
19227
|
+
throw new Error("Raindrop row-scoped eval session run requires exactly one row");
|
|
18811
19228
|
}
|
|
18812
|
-
const
|
|
18813
|
-
if (
|
|
18814
|
-
throw new Error(`Raindrop eval ${this.definition.name} did not select a planned
|
|
19229
|
+
const rowId = selected[0];
|
|
19230
|
+
if (rowId === void 0 || !this.rowIds.includes(rowId)) {
|
|
19231
|
+
throw new Error(`Raindrop eval ${this.definition.name} did not select a planned row`);
|
|
18815
19232
|
}
|
|
18816
19233
|
const attempt = (_b = options.attempt) != null ? _b : 0;
|
|
18817
|
-
const attemptKey = `${
|
|
19234
|
+
const attemptKey = `${rowId}:${attempt}`;
|
|
18818
19235
|
if (this.attempts.has(attemptKey)) {
|
|
18819
|
-
throw new Error(`Raindrop eval
|
|
19236
|
+
throw new Error(`Raindrop eval row ${rowId} attempt ${attempt} already ran`);
|
|
18820
19237
|
}
|
|
18821
19238
|
this.attempts.add(attemptKey);
|
|
18822
|
-
return this.
|
|
19239
|
+
return this.withRowPermit(async () => {
|
|
18823
19240
|
const replay2 = await this.getReplay();
|
|
18824
19241
|
const attemptResult = await replay2.run({
|
|
18825
|
-
rowId
|
|
19242
|
+
rowId,
|
|
18826
19243
|
attempt,
|
|
18827
19244
|
run: this.definition.run,
|
|
18828
19245
|
traceWaitMs: this.definition.traceWaitMs
|
|
@@ -18846,7 +19263,7 @@ var RaindropCaseEvalRun = class {
|
|
|
18846
19263
|
};
|
|
18847
19264
|
this.completed.push(result);
|
|
18848
19265
|
return {
|
|
18849
|
-
|
|
19266
|
+
runId: replay2.replayId,
|
|
18850
19267
|
name: this.definition.name,
|
|
18851
19268
|
status: "running",
|
|
18852
19269
|
counts: {
|
|
@@ -18858,7 +19275,7 @@ var RaindropCaseEvalRun = class {
|
|
|
18858
19275
|
tracesExpireAt: null,
|
|
18859
19276
|
tracesExpired: false,
|
|
18860
19277
|
evaluators: replayEvaluatorsFromVerdicts(attemptResult.verdicts),
|
|
18861
|
-
|
|
19278
|
+
rows: [result]
|
|
18862
19279
|
};
|
|
18863
19280
|
});
|
|
18864
19281
|
}
|
|
@@ -18866,20 +19283,21 @@ var RaindropCaseEvalRun = class {
|
|
|
18866
19283
|
await Promise.allSettled([...this.active]);
|
|
18867
19284
|
const replay2 = await this.getReplay();
|
|
18868
19285
|
const finished = await replay2.finish();
|
|
18869
|
-
const { rows: _rows, ...reading } = finished.reading;
|
|
19286
|
+
const { rows: _rows, replayId, ...reading } = finished.reading;
|
|
18870
19287
|
this.state = "finished";
|
|
18871
19288
|
return {
|
|
18872
19289
|
...reading,
|
|
19290
|
+
runId: replayId,
|
|
18873
19291
|
name: this.definition.name,
|
|
18874
19292
|
evaluators: finished.evaluators,
|
|
18875
|
-
|
|
19293
|
+
rows: this.sortedResults()
|
|
18876
19294
|
};
|
|
18877
19295
|
}
|
|
18878
19296
|
getReplay() {
|
|
18879
19297
|
if (!this.replayPromise) {
|
|
18880
19298
|
this.replayPromise = createLocalEvalReplaySession(this.client, {
|
|
18881
|
-
dataset: this.dataset,
|
|
18882
|
-
rowIds: this.
|
|
19299
|
+
dataset: datasetToWire(this.dataset),
|
|
19300
|
+
rowIds: this.rowIds,
|
|
18883
19301
|
evaluators: this.evaluators,
|
|
18884
19302
|
queryUrl: this.destination.queryUrl,
|
|
18885
19303
|
replayIngestUrl: this.destination.replayIngestUrl
|
|
@@ -18887,28 +19305,28 @@ var RaindropCaseEvalRun = class {
|
|
|
18887
19305
|
}
|
|
18888
19306
|
return this.replayPromise;
|
|
18889
19307
|
}
|
|
18890
|
-
async
|
|
19308
|
+
async withRowPermit(run) {
|
|
18891
19309
|
var _a;
|
|
18892
19310
|
const limit = (_a = this.definition.concurrency) != null ? _a : 4;
|
|
18893
|
-
if (this.
|
|
18894
|
-
await new Promise((resolve) => this.
|
|
19311
|
+
if (this.runningRows >= limit) {
|
|
19312
|
+
await new Promise((resolve) => this.rowWaiters.push(resolve));
|
|
18895
19313
|
} else {
|
|
18896
|
-
this.
|
|
19314
|
+
this.runningRows += 1;
|
|
18897
19315
|
}
|
|
18898
19316
|
try {
|
|
18899
19317
|
return await run();
|
|
18900
19318
|
} finally {
|
|
18901
|
-
const next = this.
|
|
19319
|
+
const next = this.rowWaiters.shift();
|
|
18902
19320
|
if (next) next();
|
|
18903
|
-
else this.
|
|
19321
|
+
else this.runningRows -= 1;
|
|
18904
19322
|
}
|
|
18905
19323
|
}
|
|
18906
19324
|
sortedResults() {
|
|
18907
|
-
const order = new Map(this.
|
|
19325
|
+
const order = new Map(this.rowIds.map((rowId, index) => [rowId, index]));
|
|
18908
19326
|
return [...this.completed].sort((left, right) => {
|
|
18909
19327
|
var _a, _b, _c, _d;
|
|
18910
|
-
const
|
|
18911
|
-
return
|
|
19328
|
+
const rowOrder = ((_a = order.get(left.id)) != null ? _a : Number.MAX_SAFE_INTEGER) - ((_b = order.get(right.id)) != null ? _b : Number.MAX_SAFE_INTEGER);
|
|
19329
|
+
return rowOrder || ((_c = left.attempt) != null ? _c : 0) - ((_d = right.attempt) != null ? _d : 0);
|
|
18912
19330
|
});
|
|
18913
19331
|
}
|
|
18914
19332
|
};
|
|
@@ -18937,8 +19355,8 @@ var WorkshopEvalRun = class {
|
|
|
18937
19355
|
this.completed = [];
|
|
18938
19356
|
this.active = /* @__PURE__ */ new Set();
|
|
18939
19357
|
this.state = "open";
|
|
18940
|
-
this.
|
|
18941
|
-
this.
|
|
19358
|
+
this.runningRows = 0;
|
|
19359
|
+
this.rowWaiters = [];
|
|
18942
19360
|
}
|
|
18943
19361
|
async run(options = {}) {
|
|
18944
19362
|
if (this.state !== "open") throw new Error(`Raindrop eval run ${this.id} is already finishing`);
|
|
@@ -18952,11 +19370,11 @@ var WorkshopEvalRun = class {
|
|
|
18952
19370
|
}
|
|
18953
19371
|
async runOnce(options) {
|
|
18954
19372
|
var _a;
|
|
18955
|
-
const
|
|
19373
|
+
const rows = selectLocalRows(this.definition, options.selection);
|
|
18956
19374
|
const completed = await Promise.all(
|
|
18957
|
-
|
|
19375
|
+
rows.map((row) => this.withRowPermit(() => {
|
|
18958
19376
|
var _a2;
|
|
18959
|
-
return this.
|
|
19377
|
+
return this.executeRow(row, (_a2 = options.attempt) != null ? _a2 : 0);
|
|
18960
19378
|
}))
|
|
18961
19379
|
);
|
|
18962
19380
|
const graded = await this.applyPortableEvaluators(completed, (_a = options.attempt) != null ? _a : 0);
|
|
@@ -19017,7 +19435,7 @@ var WorkshopEvalRun = class {
|
|
|
19017
19435
|
name: entry.evaluator.name,
|
|
19018
19436
|
output: entry.evaluator.output,
|
|
19019
19437
|
execution: "local",
|
|
19020
|
-
scope: entry.evaluator.scope,
|
|
19438
|
+
scope: entry.evaluator.scope === "row" ? "case" : entry.evaluator.scope,
|
|
19021
19439
|
threshold: entry.threshold
|
|
19022
19440
|
};
|
|
19023
19441
|
})
|
|
@@ -19029,7 +19447,7 @@ var WorkshopEvalRun = class {
|
|
|
19029
19447
|
this.state = "finished";
|
|
19030
19448
|
return result;
|
|
19031
19449
|
}
|
|
19032
|
-
async
|
|
19450
|
+
async executeRow(row, attempt) {
|
|
19033
19451
|
var _a, _b, _c;
|
|
19034
19452
|
const correlationId = randomId();
|
|
19035
19453
|
const exportedSpanIds = /* @__PURE__ */ new Set();
|
|
@@ -19171,20 +19589,20 @@ var WorkshopEvalRun = class {
|
|
|
19171
19589
|
};
|
|
19172
19590
|
return { result, packet, trace: trace8 };
|
|
19173
19591
|
}
|
|
19174
|
-
async
|
|
19592
|
+
async withRowPermit(run) {
|
|
19175
19593
|
var _a;
|
|
19176
19594
|
const limit = (_a = this.definition.concurrency) != null ? _a : 4;
|
|
19177
|
-
if (this.
|
|
19178
|
-
await new Promise((resolve) => this.
|
|
19595
|
+
if (this.runningRows >= limit) {
|
|
19596
|
+
await new Promise((resolve) => this.rowWaiters.push(resolve));
|
|
19179
19597
|
} else {
|
|
19180
|
-
this.
|
|
19598
|
+
this.runningRows += 1;
|
|
19181
19599
|
}
|
|
19182
19600
|
try {
|
|
19183
19601
|
return await run();
|
|
19184
19602
|
} finally {
|
|
19185
|
-
const next = this.
|
|
19603
|
+
const next = this.rowWaiters.shift();
|
|
19186
19604
|
if (next) next();
|
|
19187
|
-
else this.
|
|
19605
|
+
else this.runningRows -= 1;
|
|
19188
19606
|
}
|
|
19189
19607
|
}
|
|
19190
19608
|
async applyPortableEvaluators(completed, attempt) {
|
|
@@ -19249,14 +19667,17 @@ var WorkshopEvalRun = class {
|
|
|
19249
19667
|
}
|
|
19250
19668
|
};
|
|
19251
19669
|
async function runRaindropEval(client, definition, destination, options) {
|
|
19252
|
-
var _a, _b, _c;
|
|
19670
|
+
var _a, _b, _c, _d;
|
|
19253
19671
|
if (typeof definition.dataset !== "string") {
|
|
19254
19672
|
throw new Error(
|
|
19255
19673
|
"Raindrop remote execution requires an eval dataset slug; legacy pulled snapshots do not carry an immutable dataset version id"
|
|
19256
19674
|
);
|
|
19257
19675
|
}
|
|
19258
19676
|
const datasetReference = definition.dataset;
|
|
19259
|
-
const pinnedDataset = (_a = options.selection) == null ? void 0 : _a.dataset
|
|
19677
|
+
const pinnedDataset = (_b = (_a = options.selection) == null ? void 0 : _a.dataset) != null ? _b : definition.datasetVersionId ? await readEvalDataset2(client, datasetReference, {
|
|
19678
|
+
queryUrl: destination.queryUrl,
|
|
19679
|
+
versionId: definition.datasetVersionId
|
|
19680
|
+
}) : void 0;
|
|
19260
19681
|
if (pinnedDataset && pinnedDataset.dataset.slug !== datasetReference) {
|
|
19261
19682
|
throw new Error(
|
|
19262
19683
|
`Raindrop eval ${definition.name} pinned dataset ${pinnedDataset.dataset.slug} does not match ${datasetReference}`
|
|
@@ -19268,7 +19689,8 @@ async function runRaindropEval(client, definition, destination, options) {
|
|
|
19268
19689
|
definition.evaluators.map((evaluator) => [evaluatorName(evaluator), evaluator])
|
|
19269
19690
|
);
|
|
19270
19691
|
const replayOptions = {
|
|
19271
|
-
|
|
19692
|
+
runId: options.runId,
|
|
19693
|
+
dataset: pinnedDataset ? datasetToWire(pinnedDataset) : datasetReference,
|
|
19272
19694
|
evaluators: definition.evaluators.map((entry) => replayEvaluator(entry)),
|
|
19273
19695
|
evaluatorPins: hostedEvaluatorPins(definition.evaluators),
|
|
19274
19696
|
run: definition.run,
|
|
@@ -19283,7 +19705,7 @@ async function runRaindropEval(client, definition, destination, options) {
|
|
|
19283
19705
|
client,
|
|
19284
19706
|
selection ? {
|
|
19285
19707
|
...replayOptions,
|
|
19286
|
-
rowIds: selection.
|
|
19708
|
+
rowIds: selection.rowIds
|
|
19287
19709
|
} : replayOptions
|
|
19288
19710
|
);
|
|
19289
19711
|
const outputByEvaluator = /* @__PURE__ */ new Map();
|
|
@@ -19296,7 +19718,7 @@ async function runRaindropEval(client, definition, destination, options) {
|
|
|
19296
19718
|
if (evaluatorResult.status === "failed") {
|
|
19297
19719
|
continue;
|
|
19298
19720
|
}
|
|
19299
|
-
const output = (
|
|
19721
|
+
const output = (_d = (_c = evaluatorResult.summary) == null ? void 0 : _c.outputType) != null ? _d : evaluatorOutput(evaluator);
|
|
19300
19722
|
if (output === void 0)
|
|
19301
19723
|
throw new Error(
|
|
19302
19724
|
`Raindrop evaluator ${evaluatorResult.evaluator} completed without an output type`
|
|
@@ -19304,11 +19726,12 @@ async function runRaindropEval(client, definition, destination, options) {
|
|
|
19304
19726
|
validateEvaluatorOutput(evaluator, output);
|
|
19305
19727
|
outputByEvaluator.set(evaluatorResult.evaluator, output);
|
|
19306
19728
|
}
|
|
19307
|
-
const { rows, ...summary } = result;
|
|
19729
|
+
const { rows, replayId, ...summary } = result;
|
|
19308
19730
|
return {
|
|
19309
19731
|
...summary,
|
|
19732
|
+
runId: replayId,
|
|
19310
19733
|
name: definition.name,
|
|
19311
|
-
|
|
19734
|
+
rows: rows.map((row) => ({
|
|
19312
19735
|
...row,
|
|
19313
19736
|
verdicts: row.verdicts.map((verdict) => {
|
|
19314
19737
|
const evaluator = configured.get(verdict.evaluator);
|
|
@@ -19344,32 +19767,38 @@ function replayEvaluator(entry) {
|
|
|
19344
19767
|
if (typeof entry.evaluator === "string") return entry.evaluator;
|
|
19345
19768
|
if (isPortableEvaluator(entry.evaluator)) return entry.evaluator.slug;
|
|
19346
19769
|
if ("kind" in entry.evaluator && entry.evaluator.kind === "program") {
|
|
19770
|
+
if (!entry.evaluator.expected) {
|
|
19771
|
+
throw new Error(
|
|
19772
|
+
`Create evaluator ${entry.evaluator.slug} in Raindrop and reference its slug; uploading evaluator source is not supported`
|
|
19773
|
+
);
|
|
19774
|
+
}
|
|
19347
19775
|
return entry.evaluator.slug;
|
|
19348
19776
|
}
|
|
19349
|
-
|
|
19777
|
+
const evaluator = entry.evaluator;
|
|
19778
|
+
switch (evaluator.output) {
|
|
19350
19779
|
case "boolean":
|
|
19351
19780
|
return {
|
|
19352
|
-
slug:
|
|
19353
|
-
name:
|
|
19354
|
-
output:
|
|
19355
|
-
judge:
|
|
19356
|
-
requiresReference:
|
|
19781
|
+
slug: evaluator.slug,
|
|
19782
|
+
name: evaluator.name,
|
|
19783
|
+
output: evaluator.output,
|
|
19784
|
+
judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
|
|
19785
|
+
requiresReference: evaluator.requiresReference
|
|
19357
19786
|
};
|
|
19358
19787
|
case "score":
|
|
19359
19788
|
return {
|
|
19360
|
-
slug:
|
|
19361
|
-
name:
|
|
19362
|
-
output:
|
|
19363
|
-
judge:
|
|
19364
|
-
requiresReference:
|
|
19789
|
+
slug: evaluator.slug,
|
|
19790
|
+
name: evaluator.name,
|
|
19791
|
+
output: evaluator.output,
|
|
19792
|
+
judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
|
|
19793
|
+
requiresReference: evaluator.requiresReference
|
|
19365
19794
|
};
|
|
19366
19795
|
case "number":
|
|
19367
19796
|
return {
|
|
19368
|
-
slug:
|
|
19369
|
-
name:
|
|
19370
|
-
output:
|
|
19371
|
-
judge:
|
|
19372
|
-
requiresReference:
|
|
19797
|
+
slug: evaluator.slug,
|
|
19798
|
+
name: evaluator.name,
|
|
19799
|
+
output: evaluator.output,
|
|
19800
|
+
judge: ({ trace: trace8, row, result, case: pair }) => evaluator.judge({ trace: trace8, row, result, reference: pair == null ? void 0 : pair.reference }),
|
|
19801
|
+
requiresReference: evaluator.requiresReference
|
|
19373
19802
|
};
|
|
19374
19803
|
}
|
|
19375
19804
|
}
|
|
@@ -19378,34 +19807,34 @@ function validateDefinition(definition) {
|
|
|
19378
19807
|
throw new Error("Raindrop runEvalSuite requires a definition created by defineEvalSuite");
|
|
19379
19808
|
return definition;
|
|
19380
19809
|
}
|
|
19381
|
-
function
|
|
19382
|
-
if (!selection) return definition.dataset.
|
|
19383
|
-
if (selection.
|
|
19810
|
+
function selectLocalRows(definition, selection) {
|
|
19811
|
+
if (!selection) return definition.dataset.rows;
|
|
19812
|
+
if (selection.rowIds.length === 0)
|
|
19384
19813
|
throw new Error(`Raindrop eval ${definition.name} selection cannot be empty`);
|
|
19385
|
-
if (new Set(selection.
|
|
19386
|
-
throw new Error(`Raindrop eval ${definition.name} selection
|
|
19387
|
-
const byId = new Map(definition.dataset.
|
|
19388
|
-
return selection.
|
|
19814
|
+
if (new Set(selection.rowIds).size !== selection.rowIds.length)
|
|
19815
|
+
throw new Error(`Raindrop eval ${definition.name} selection row ids must be unique`);
|
|
19816
|
+
const byId = new Map(definition.dataset.rows.map((row) => [row.id, row]));
|
|
19817
|
+
return selection.rowIds.map((id) => {
|
|
19389
19818
|
const row = byId.get(id);
|
|
19390
19819
|
if (!row)
|
|
19391
|
-
throw new Error(`Raindrop eval ${definition.name} selection contains unknown
|
|
19820
|
+
throw new Error(`Raindrop eval ${definition.name} selection contains unknown row ${id}`);
|
|
19392
19821
|
return row;
|
|
19393
19822
|
});
|
|
19394
19823
|
}
|
|
19395
19824
|
function validateRemoteSelection(dataset, selection) {
|
|
19396
|
-
var _a
|
|
19825
|
+
var _a;
|
|
19397
19826
|
if (!selection) return void 0;
|
|
19398
|
-
if (selection.
|
|
19827
|
+
if (selection.rowIds.length === 0)
|
|
19399
19828
|
throw new Error(`Raindrop eval ${dataset} selection cannot be empty`);
|
|
19400
|
-
if (new Set(selection.
|
|
19401
|
-
throw new Error(`Raindrop eval ${dataset} selection
|
|
19402
|
-
const
|
|
19403
|
-
if (!
|
|
19404
|
-
const available = new Set(
|
|
19405
|
-
const unknown = selection.
|
|
19829
|
+
if (new Set(selection.rowIds).size !== selection.rowIds.length)
|
|
19830
|
+
throw new Error(`Raindrop eval ${dataset} selection row ids must be unique`);
|
|
19831
|
+
const rows = (_a = selection.dataset) == null ? void 0 : _a.rows;
|
|
19832
|
+
if (!rows) throw new Error(`Raindrop eval ${dataset} selection requires its dataset manifest`);
|
|
19833
|
+
const available = new Set(rows.map((row) => row.id));
|
|
19834
|
+
const unknown = selection.rowIds.find((rowId) => !available.has(rowId));
|
|
19406
19835
|
if (unknown)
|
|
19407
|
-
throw new Error(`Raindrop eval ${dataset} selection contains unknown
|
|
19408
|
-
return {
|
|
19836
|
+
throw new Error(`Raindrop eval ${dataset} selection contains unknown row ${unknown}`);
|
|
19837
|
+
return { rowIds: selection.rowIds };
|
|
19409
19838
|
}
|
|
19410
19839
|
function canonicalWorkshopTrace(envelope) {
|
|
19411
19840
|
var _a, _b, _c, _d, _e, _f;
|
|
@@ -19531,18 +19960,18 @@ function portableVerdict(evaluator, runId, workshopUrl, outcome) {
|
|
|
19531
19960
|
...outcome.note ? { note: outcome.note } : {}
|
|
19532
19961
|
};
|
|
19533
19962
|
}
|
|
19534
|
-
function workshopResult(id, name,
|
|
19535
|
-
const done =
|
|
19536
|
-
const missing =
|
|
19963
|
+
function workshopResult(id, name, rows) {
|
|
19964
|
+
const done = rows.filter((entry) => entry.status === "done").length;
|
|
19965
|
+
const missing = rows.filter((entry) => entry.status === "missing").length;
|
|
19537
19966
|
return {
|
|
19538
|
-
|
|
19967
|
+
runId: id,
|
|
19539
19968
|
name,
|
|
19540
19969
|
status: missing > 0 ? "failed" : "complete",
|
|
19541
|
-
counts: { total:
|
|
19970
|
+
counts: { total: rows.length, pending: 0, done, missing },
|
|
19542
19971
|
tracesExpireAt: null,
|
|
19543
19972
|
tracesExpired: false,
|
|
19544
19973
|
evaluators: [],
|
|
19545
|
-
|
|
19974
|
+
rows
|
|
19546
19975
|
};
|
|
19547
19976
|
}
|
|
19548
19977
|
function workshopEvalUrl(baseUrl, id) {
|
|
@@ -20859,8 +21288,9 @@ var index_default = Raindrop;
|
|
|
20859
21288
|
DEFAULT_QUERY_URL,
|
|
20860
21289
|
DEFAULT_TRACE_WAIT_MS,
|
|
20861
21290
|
EVAL_CORRELATION_ID_ATTRIBUTE,
|
|
20862
|
-
EvalDatasetCaseInputSchema,
|
|
20863
21291
|
EvalDatasetPublishConflictError,
|
|
21292
|
+
EvalDatasetRowInputSchema,
|
|
21293
|
+
EvalPublishError,
|
|
20864
21294
|
LocalBooleanVerdictSchema,
|
|
20865
21295
|
LocalNumberVerdictSchema,
|
|
20866
21296
|
LocalScoreVerdictSchema,
|
|
@@ -20877,6 +21307,7 @@ var index_default = Raindrop;
|
|
|
20877
21307
|
ReplayVerdictSchema,
|
|
20878
21308
|
TraceSchema,
|
|
20879
21309
|
claimReplay,
|
|
21310
|
+
compareEvalRuns,
|
|
20880
21311
|
createEvalSuiteRun,
|
|
20881
21312
|
createReplay,
|
|
20882
21313
|
currentEvalScope,
|
|
@@ -20884,17 +21315,22 @@ var index_default = Raindrop;
|
|
|
20884
21315
|
defineEvalSuite,
|
|
20885
21316
|
defineEvaluatorProgram,
|
|
20886
21317
|
defineLocalEvaluator,
|
|
21318
|
+
evaluateEvalRun,
|
|
20887
21319
|
evaluatePortableEvaluator,
|
|
20888
21320
|
evaluateReplay,
|
|
20889
21321
|
importPortableEvaluator,
|
|
20890
21322
|
isEvalDataset,
|
|
20891
21323
|
isEvalSuiteDefinition,
|
|
21324
|
+
isPublishedEvalSuite,
|
|
20892
21325
|
loadEvalSnapshot,
|
|
20893
21326
|
projectWorkshopTrace,
|
|
20894
21327
|
publishEvalDataset,
|
|
21328
|
+
publishEvalSuite,
|
|
20895
21329
|
pullEval,
|
|
20896
21330
|
readEvalDataset,
|
|
20897
21331
|
readEvalManifest,
|
|
21332
|
+
readEvalRun,
|
|
21333
|
+
readEvaluator,
|
|
20898
21334
|
readReplay,
|
|
20899
21335
|
replay,
|
|
20900
21336
|
resolveDisableBatching,
|
|
@@ -20902,6 +21338,7 @@ var index_default = Raindrop;
|
|
|
20902
21338
|
runReplay,
|
|
20903
21339
|
traceOutput,
|
|
20904
21340
|
traceToolCalls,
|
|
21341
|
+
traceTools,
|
|
20905
21342
|
verifyPortableEvalArtifact,
|
|
20906
21343
|
withEvalScope
|
|
20907
21344
|
});
|