raindrop-ai 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,125 @@
1
+ #!/usr/bin/env node
2
+ import {
3
+ compareEvalRuns,
4
+ index_default,
5
+ publishEvalSuite,
6
+ readEvaluator,
7
+ runEvalSuite
8
+ } from "../chunk-4Z7UHC7X.mjs";
9
+ import "../chunk-OX6X4ZWQ.mjs";
10
+ import {
11
+ isEvalSuiteDefinition
12
+ } from "../chunk-RMP6BZSY.mjs";
13
+ import "../chunk-UJCSKKID.mjs";
14
+
15
+ // src/evals/cli.ts
16
+ import { resolve } from "path";
17
+ import { pathToFileURL } from "url";
18
+ import { parseArgs } from "util";
19
+ import { z } from "zod";
20
+ var SuiteModuleSchema = z.object({
21
+ default: z.custom(isEvalSuiteDefinition)
22
+ });
23
+ async function main() {
24
+ var _a, _b;
25
+ const { positionals, values } = parseArgs({
26
+ allowPositionals: true,
27
+ options: {
28
+ project: { type: "string" },
29
+ "query-url": { type: "string" },
30
+ "replay-ingest-url": { type: "string" },
31
+ "expected-dataset-version": { type: "string" },
32
+ "run-id": { type: "string" },
33
+ evaluator: { type: "string" },
34
+ json: { type: "boolean" },
35
+ help: { type: "boolean" }
36
+ }
37
+ });
38
+ const [command, file, afterRunId] = positionals;
39
+ if (values.help || !command) {
40
+ process.stdout.write(
41
+ [
42
+ "raindrop-evals publish <suite.mjs> [--project slug] [--json]",
43
+ "raindrop-evals run <suite.mjs> [--run-id queued-run-id] [--project slug] [--json]",
44
+ "raindrop-evals read <evaluator-slug> [--json]",
45
+ "raindrop-evals compare <before-run-id> <after-run-id> [--evaluator slug] [--json]",
46
+ "",
47
+ "Publish uploads dataset rows and pins existing evaluators without running them.",
48
+ "The module must default-export defineEvalSuite(...). Run uploads dataset rows first.",
49
+ "Create hosted evaluators in Raindrop and reference their slugs. Evaluator source is not uploaded.",
50
+ "Hosted judges run on Raindrop; agent callbacks and local evaluators run locally.",
51
+ "Requires RAINDROP_QUERY_API_KEY. Project defaults to RAINDROP_PROJECT_ID or default.",
52
+ "Use --query-url and --replay-ingest-url for a full-local Raindrop stack.",
53
+ "TypeScript modules need your runtime's TypeScript loader, such as node --import tsx.",
54
+ "Output is JSON. Run exits 1 for failed/incomplete evals; setup errors exit 2."
55
+ ].join("\n") + "\n"
56
+ );
57
+ return;
58
+ }
59
+ if (!file || !["publish", "run", "read", "compare"].includes(command))
60
+ throw new Error("Invalid command. Use --help.");
61
+ const apiKey = process.env.RAINDROP_QUERY_API_KEY;
62
+ if (!apiKey) throw new Error("Set RAINDROP_QUERY_API_KEY to a Query SDK organization key");
63
+ const client = new index_default({
64
+ apiKey,
65
+ projectId: (_b = (_a = values.project) != null ? _a : process.env.RAINDROP_PROJECT_ID) != null ? _b : "default",
66
+ localWorkshopUrl: false
67
+ });
68
+ const options = { queryUrl: values["query-url"] };
69
+ try {
70
+ if (command === "compare") {
71
+ if (!afterRunId) throw new Error("Compare requires before and after run IDs");
72
+ print(await compareEvalRuns(client, { ...options, beforeRunId: file, afterRunId, evaluator: values.evaluator }));
73
+ } else if (command === "read") {
74
+ print(await readEvaluator(client, file, options));
75
+ } else {
76
+ const module = SuiteModuleSchema.parse(await import(pathToFileURL(resolve(file)).href));
77
+ if (command === "publish") {
78
+ const suite = await publishEvalSuite(client, module.default, {
79
+ ...options,
80
+ expectedCurrentDatasetVersionId: values["expected-dataset-version"]
81
+ });
82
+ print({
83
+ name: suite.name,
84
+ dataset: suite.dataset,
85
+ datasetVersionId: suite.datasetVersionId,
86
+ evaluators: suite.evaluators.map(
87
+ ({ evaluator }) => typeof evaluator === "string" ? { slug: evaluator } : {
88
+ slug: evaluator.slug,
89
+ output: evaluator.output,
90
+ expected: "kind" in evaluator && evaluator.kind === "program" ? evaluator.expected : void 0
91
+ }
92
+ )
93
+ });
94
+ } else {
95
+ const result = await runEvalSuite(client, module.default, {
96
+ runId: values["run-id"],
97
+ expectedCurrentDatasetVersionId: values["expected-dataset-version"],
98
+ destination: {
99
+ kind: "raindrop",
100
+ ...options,
101
+ replayIngestUrl: values["replay-ingest-url"]
102
+ }
103
+ });
104
+ print(result);
105
+ if (result.status !== "complete" || result.evaluators.some((evaluator) => evaluator.status !== "completed") || result.rows.some(
106
+ (entry) => entry.status !== "done" || entry.verdicts.some(
107
+ (verdict) => verdict.state !== "passed" && verdict.state !== "measurement"
108
+ )
109
+ ))
110
+ process.exitCode = 1;
111
+ }
112
+ }
113
+ } finally {
114
+ await client.close();
115
+ }
116
+ }
117
+ function print(value) {
118
+ process.stdout.write(JSON.stringify(value, null, 2) + "\n");
119
+ }
120
+ void main().catch((error) => {
121
+ process.stderr.write(
122
+ JSON.stringify({ error: error instanceof Error ? error.message : String(error) }) + "\n"
123
+ );
124
+ process.exitCode = 2;
125
+ });