evals-lab 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/run.js ADDED
@@ -0,0 +1,531 @@
1
+ // `evals-lab run <bundle>`: an Export for CI bundle (#254, docs/pipeline-yaml.md)
2
+ // run on this machine, with no lab (#255).
3
+ //
4
+ // The bundle is read into a run document by the core's readBundle -- the
5
+ // import and the resolve the page does at submit -- and handed to the worker
6
+ // a lab's queue runs, lab/run-evals.js --run, so a run here and the same run
7
+ // in a lab are one code path. What this adds is what a queue would have:
8
+ //
9
+ // THE ITEMS. --items, else the bundle's items/. Every file there runs, as
10
+ // every file of a Source does; an item a case names that is not there runs
11
+ // too, as Missing, so the run is incomplete (exit 2) rather than quietly
12
+ // grading fewer cases.
13
+ // THE PLUGINS. The bundle's plugins/, registered here before it is read and
14
+ // stamped on the run so the worker loads the same ones.
15
+ // THE KEYS. Nothing: the worker reads each from $EVALSLAB_API_KEY_<SLUG>.
16
+ //
17
+ // The exit code is the worker's: 0 pass, 1 an eval failed, 2 incomplete, 3
18
+ // broken or refused. The table goes to stderr, so stdout is the JSON report's
19
+ // alone when --json is -. --summary appends the same as Markdown, for a CI
20
+ // page that shows it: GitHub's $GITHUB_STEP_SUMMARY is a file to append to.
21
+ //
22
+ // `evals-lab run --lab <url> --pipeline <name>` (#256) is the same command
23
+ // against a running lab instead: the pipeline is the lab's, resolved by the
24
+ // core as the page resolves it at submit, and queued there, so the run is in
25
+ // that lab's History. The keys stay in the lab -- the run asks the lab's own
26
+ // profiles, which the lab serves without their keys -- and with --wait the
27
+ // exit code is the verdict the lab kept on the run's row (#252).
28
+ const { spawn } = require("child_process");
29
+ const fs = require("fs");
30
+ const os = require("os");
31
+ const path = require("path");
32
+ const { pathToFileURL } = require("url");
33
+
34
+ const EXIT = { broken: 3 };
35
+
36
+ const USAGE = `Usage: evals-lab run <bundle-dir | pipeline.yaml> [options]
37
+ evals-lab run --lab <url> --pipeline <name> [--wait] [options]
38
+
39
+ --items <dir> the items the pipeline's Source reads. Default: the
40
+ bundle's items/.
41
+ --json <file|-> write the JSON report. "-" means stdout.
42
+ --junit <file> write JUnit XML: a testsuite per Target and eval, a
43
+ testcase per item.
44
+ --summary <file> add the results to <file> as Markdown: each Target and
45
+ eval's pass rate, then the items that failed. In GitHub
46
+ Actions, --summary "$GITHUB_STEP_SUMMARY".
47
+ --min-pass <0..1> an eval passes when at least this share of its items
48
+ pass. It relaxes each eval's own rule, never tightens it.
49
+ --progress a line on stderr as each item finishes.
50
+ --lab <url> run on a running lab instead: its pipeline, its profiles
51
+ and their keys, and the run in its History. LAB_PASSWORD
52
+ is sent when it is set.
53
+ --pipeline <name> the lab's pipeline to run, by its name or its id.
54
+ --wait wait for the lab's run to finish, and exit with its
55
+ verdict. Without it the run's id is printed once it is
56
+ queued. --json, --junit, --summary, --min-pass and
57
+ --progress need it.
58
+ --help this.
59
+
60
+ A bundle's Target profile keys come from $EVALSLAB_API_KEY_<SLUG>. Exit
61
+ codes: 0 pass (or queued, without --wait), 1 an eval failed, 2 incomplete (an
62
+ item missing, one that did not run, or the run cancelled), 3 broken or
63
+ refused.`;
64
+
65
+ // A run refused before it starts; [usage] when it was the command line.
66
+ class Refused extends Error {
67
+ constructor(message, usage = false) { super(message); this.usage = usage; }
68
+ }
69
+
70
+ // What the command line asks for, or a Refused saying what is wrong with it.
71
+ function runOptions(argv) {
72
+ const o = { bundle: null, items: null, json: null, junit: null, summary: null, minPass: null, progress: false,
73
+ lab: null, pipeline: null, wait: false };
74
+ const value = { "--items": "items", "--json": "json", "--junit": "junit", "--summary": "summary", "--min-pass": "minPass",
75
+ "--lab": "lab", "--pipeline": "pipeline" };
76
+ for (let i = 0; i < argv.length; i++) {
77
+ const a = argv[i];
78
+ if (a === "--help" || a === "-h") o.help = true;
79
+ else if (a === "--progress") o.progress = true;
80
+ else if (a === "--wait") o.wait = true;
81
+ else if (value[a]) {
82
+ if (argv[i + 1] === undefined) throw new Refused(`${a} needs a value`, true);
83
+ o[value[a]] = argv[++i];
84
+ } else if (a.startsWith("-") && a !== "-") throw new Refused(`unknown option ${a}`, true);
85
+ else if (o.bundle != null) throw new Refused(`one bundle at a time: ${o.bundle} and ${a}`, true);
86
+ else o.bundle = a;
87
+ }
88
+ if (o.help) return o;
89
+ if (o.lab != null || o.pipeline != null) {
90
+ if (o.bundle != null) throw new Refused(`a bundle or a lab, not both: ${o.bundle} and --lab`, true);
91
+ if (o.lab == null) throw new Refused("--pipeline names a lab's pipeline: name the lab with --lab", true);
92
+ if (o.pipeline == null) throw new Refused("name the lab's pipeline to run with --pipeline", true);
93
+ if (!/^https?:\/\/[^/]/i.test(o.lab)) throw new Refused(`--lab ${o.lab} is not an http or https address`, true);
94
+ if (o.items != null) throw new Refused("--items is a bundle's: a lab's run reads its own Source", true);
95
+ const waits = ["json", "junit", "summary", "minPass", "progress"].filter(k => o[k] != null && o[k] !== false);
96
+ if (waits.length && !o.wait) {
97
+ throw new Refused(`--${waits[0].replace("minPass", "min-pass")} reads the finished run: add --wait`, true);
98
+ }
99
+ } else if (o.wait) {
100
+ throw new Refused("--wait is for a lab's run: a bundle's always waits", true);
101
+ } else if (o.bundle == null) {
102
+ throw new Refused("name the bundle: its directory, or its pipeline.yaml", true);
103
+ }
104
+ if (o.minPass != null) {
105
+ const n = Number(o.minPass);
106
+ if (o.minPass.trim() === "" || !(n >= 0 && n <= 1)) throw new Refused(`--min-pass ${o.minPass} is not a share from 0 to 1`, true);
107
+ }
108
+ return o;
109
+ }
110
+
111
+ // The bundle's text files by path inside it -- what readBundle reads -- and
112
+ // the directory they sit in. [where] is the directory or its pipeline.yaml.
113
+ function bundleFiles(where) {
114
+ let stat;
115
+ try { stat = fs.statSync(where); } catch { throw new Refused(`${where} is not there`); }
116
+ const dir = stat.isDirectory() ? where : path.dirname(where);
117
+ const pipeline = stat.isDirectory() ? path.join(dir, "pipeline.yaml") : where;
118
+ if (!fs.existsSync(pipeline)) throw new Refused(`${dir} holds no pipeline.yaml: name an Export for CI bundle`);
119
+ const files = { "pipeline.yaml": fs.readFileSync(pipeline, "utf8") };
120
+ const profiles = path.join(dir, "profiles.yaml");
121
+ if (fs.existsSync(profiles)) files["profiles.yaml"] = fs.readFileSync(profiles, "utf8");
122
+ const datasets = path.join(dir, "datasets");
123
+ if (fs.existsSync(datasets)) {
124
+ for (const d of fs.readdirSync(datasets, { withFileTypes: true })) {
125
+ if (d.isFile()) files[`datasets/${d.name}`] = fs.readFileSync(path.join(datasets, d.name), "utf8");
126
+ }
127
+ }
128
+ return { dir, files };
129
+ }
130
+
131
+ // The bundle's plugins/<id>/<version>/, each registered into [core] as the
132
+ // worker registers it, and the stamps that have the worker load the same.
133
+ async function loadPlugins(core, dir) {
134
+ const root = path.join(dir, "plugins");
135
+ if (!fs.existsSync(root)) return [];
136
+ const stamps = [];
137
+ for (const id of fs.readdirSync(root, { withFileTypes: true }).filter(d => d.isDirectory())) {
138
+ for (const version of fs.readdirSync(path.join(root, id.name), { withFileTypes: true }).filter(d => d.isDirectory())) {
139
+ const at = path.join(root, id.name, version.name);
140
+ try {
141
+ const manifest = JSON.parse(fs.readFileSync(path.join(at, "manifest.json"), "utf8"));
142
+ const mod = await import(pathToFileURL(path.join(at, manifest.entry)).href);
143
+ await (mod.default ?? mod.register)(core.pluginHost(manifest.id));
144
+ } catch (e) {
145
+ throw new Refused(`the plugin in ${at} did not load: ${e.message}`);
146
+ }
147
+ stamps.push({ id: id.name, version: version.name, sha256: "" });
148
+ }
149
+ }
150
+ return stamps;
151
+ }
152
+
153
+ // The files of an items directory, as a Source names its own: by name, no
154
+ // directories, nothing hidden.
155
+ function itemsIn(dir) {
156
+ return fs.readdirSync(dir, { withFileTypes: true })
157
+ .filter(d => d.isFile() && !d.name.startsWith(".")).map(d => d.name).sort();
158
+ }
159
+
160
+ const word = v => v.charAt(0).toUpperCase() + v.slice(1);
161
+
162
+ /** Why an item's score failed, in a line: its failing metrics, else what it missed. */
163
+ function reasonOf(score) {
164
+ const off = (score.metrics || []).filter(m => !m.pass).map(m => `${m.label}: ${m.reason}`);
165
+ if (off.length) return off.join(" · ");
166
+ const missed = (score.missed || []).map(r => (Array.isArray(r) ? r.join(" or ") : r));
167
+ return missed.length ? `missed ${missed.join(", ")}` : "failed";
168
+ }
169
+
170
+ // The report as a person reads it: each Target's evals and their verdicts,
171
+ // the items each failed, and the items that never ran.
172
+ function table(core, run, report, missing) {
173
+ const lines = [`${run.name || "the pipeline"}`];
174
+ const items = report.run.items || [];
175
+ const MAX = 10;
176
+ (report.run.verdicts || []).forEach((evals, i) => {
177
+ for (const [id, e] of Object.entries(evals)) {
178
+ const n = e.ran != null ? `${e.passed} of ${e.ran} passed` : e.detail || "";
179
+ lines.push(` ${word(e.verdict).padEnd(11)} ${core.targetLabel(run, i)} › ${e.name}${n ? ` ${n}` : ""}`);
180
+ if (e.verdict !== "fail" || e.ran == null) continue;
181
+ const failed = items.filter(it => {
182
+ const s = it?.scenarios?.[i]?.scores?.[id];
183
+ return s && !core.isSkipped(s) && !s.pass;
184
+ });
185
+ for (const it of failed.slice(0, MAX)) {
186
+ lines.push(` ${"".padEnd(11)} ${it.name ?? "the text"}: ${reasonOf(it.scenarios[i].scores[id])}`);
187
+ }
188
+ if (failed.length > MAX) lines.push(` ${"".padEnd(11)} and ${failed.length - MAX} more, in the JSON report`);
189
+ }
190
+ });
191
+ const gone = new Set(missing);
192
+ for (const name of missing) lines.push(` ${"Missing".padEnd(11)} ${name}`);
193
+ for (const it of items) {
194
+ if (it?.unrun && !gone.has(it.name)) lines.push(` ${"Unrun".padEnd(11)} ${it.name ?? "the text"}: ${it.unrun}`);
195
+ }
196
+ const ran = items.filter(it => it && !it.unrun).length;
197
+ lines.push(`${word(report.verdict)}${report.cancelled ? " (cancelled)" : ""} · ${ran} of ${items.length} items ran`
198
+ + (report.minPass != null ? ` · min pass ${report.minPass}` : ""));
199
+ return lines.join("\n");
200
+ }
201
+
202
+ // A table cell's text: a pipe or a line break would end the cell early.
203
+ const cell = v => String(v).replace(/\|/g, "\\|").replace(/\s*\n\s*/g, " ");
204
+
205
+ // The report as a CI page shows it, in Markdown: the run's verdict, a row per
206
+ // Target and eval with its pass rate, then the items each failed and those
207
+ // that never ran -- the table's content, laid out for a page.
208
+ function markdown(core, run, report, missing) {
209
+ const items = report.run.items || [];
210
+ const ran = items.filter(it => it && !it.unrun).length;
211
+ const lines = [`### ${cell(run.name || "the pipeline")}: ${word(report.verdict)}${report.cancelled ? " (cancelled)" : ""}`, "",
212
+ `${ran} of ${items.length} items ran` + (report.minPass != null ? ` · min pass ${report.minPass}` : ""), "",
213
+ "| Target | Eval | Verdict | Passed |", "|---|---|---|---|"];
214
+ const failures = [];
215
+ (report.run.verdicts || []).forEach((evals, i) => {
216
+ for (const [id, e] of Object.entries(evals)) {
217
+ const rate = e.ran ? `${e.passed} of ${e.ran} (${Math.round(100 * e.passed / e.ran)}%)` : cell(e.detail || "");
218
+ lines.push(`| ${cell(core.targetLabel(run, i))} | ${cell(e.name)} | ${word(e.verdict)} | ${rate} |`);
219
+ if (e.verdict !== "fail") continue;
220
+ const failed = e.ran == null ? [] : items.filter(it => {
221
+ const s = it?.scenarios?.[i]?.scores?.[id];
222
+ return s && !core.isSkipped(s) && !s.pass;
223
+ });
224
+ failures.push({ title: `${core.targetLabel(run, i)} › ${e.name}`, failed, i, id });
225
+ }
226
+ });
227
+ const MAX = 25;
228
+ for (const f of failures) {
229
+ if (!f.failed.length) continue;
230
+ lines.push("", `#### Failed: ${cell(f.title)}`, "");
231
+ for (const it of f.failed.slice(0, MAX)) {
232
+ lines.push(`- \`${cell(it.name ?? "the text")}\`: ${cell(reasonOf(it.scenarios[f.i].scores[f.id]))}`);
233
+ }
234
+ if (f.failed.length > MAX) lines.push(`- and ${f.failed.length - MAX} more, in the JSON report`);
235
+ }
236
+ const gone = new Set(missing);
237
+ const unrun = [...missing.map(name => `- \`${cell(name)}\`: Missing`),
238
+ ...items.filter(it => it?.unrun && !gone.has(it.name)).map(it => `- \`${cell(it.name ?? "the text")}\`: ${cell(it.unrun)}`)];
239
+ if (unrun.length) lines.push("", "#### Not run", "", ...unrun);
240
+ return lines.join("\n") + "\n\n";
241
+ }
242
+
243
+ // Each line the worker adds to [file] as an item finishes, said on stderr.
244
+ function follow(file, total, say) {
245
+ let at = 0, rest = "";
246
+ const read = () => {
247
+ let text;
248
+ try {
249
+ const fd = fs.openSync(file, "r");
250
+ const buf = Buffer.alloc(Math.max(0, fs.fstatSync(fd).size - at));
251
+ fs.readSync(fd, buf, 0, buf.length, at);
252
+ fs.closeSync(fd);
253
+ at += buf.length;
254
+ text = rest + buf.toString("utf8");
255
+ } catch { return; }
256
+ const lines = text.split("\n");
257
+ rest = lines.pop();
258
+ for (const line of lines) {
259
+ let row;
260
+ try { row = JSON.parse(line); } catch { continue; }
261
+ if (row.event) continue;
262
+ say(` ${row.item + 1} of ${total} ${row.name ?? "the text"}${row.unrun ? ` ${row.unrun}` : ""}`);
263
+ }
264
+ };
265
+ const timer = setInterval(read, 250);
266
+ return () => { clearInterval(timer); read(); };
267
+ }
268
+
269
+ // ---- a lab's run (#256) -------------------------------------------------------
270
+
271
+ const sleep = ms => new Promise(r => setTimeout(r, ms));
272
+ // How long between reads of a queued run, and how many reads in a row may
273
+ // go unanswered -- a lab restarting under a long run -- before it is given up.
274
+ const POLL_MS = 500;
275
+ const POLL_MISSES = 20;
276
+
277
+ // The lab's API at [base], as [env]'s LAB_PASSWORD opens it: HTTP Basic, the
278
+ // password alone, as server.py takes it. A refusal says what the lab said.
279
+ function labApi(base, env) {
280
+ const auth = env.LAB_PASSWORD
281
+ ? { Authorization: `Basic ${Buffer.from(`:${env.LAB_PASSWORD}`).toString("base64")}` } : {};
282
+ return async (method, route, body, { missing = false } = {}) => {
283
+ let res;
284
+ try {
285
+ res = await fetch(base + route, {
286
+ method, redirect: "manual",
287
+ headers: { ...auth, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
288
+ ...(body !== undefined ? { body: JSON.stringify(body) } : {}),
289
+ });
290
+ } catch (e) {
291
+ throw Object.assign(new Refused(`${base} did not answer: ${e.cause?.message ?? e.message}`), { unanswered: true });
292
+ }
293
+ const text = await res.text();
294
+ if (res.status === 401) {
295
+ throw new Refused(env.LAB_PASSWORD ? `${base} did not take LAB_PASSWORD` : `${base} asks for a password: set LAB_PASSWORD`);
296
+ }
297
+ if (res.status === 404 && missing) return null;
298
+ let json = null;
299
+ try { json = JSON.parse(text); } catch { /* a plain body, said as it is */ }
300
+ if (!res.ok) throw new Refused(`${base} refused ${method} ${route}: ${json?.error ?? `${res.status} ${text.trim().slice(0, 200)}`}`);
301
+ if (json === null) throw new Refused(`${base} answered ${method} ${route} with no JSON: is it a lab?`);
302
+ return json;
303
+ };
304
+ }
305
+
306
+ // The lab's saved pipeline [name] names -- by name, else by id -- or a
307
+ // Refused saying which ones it has.
308
+ function pipelineNamed(saved, name, base) {
309
+ const byName = saved.filter(w => w?.name === name);
310
+ const found = byName.length ? byName : saved.filter(w => w?.id === name);
311
+ if (!found.length) {
312
+ throw new Refused(`${base} has no pipeline named ${JSON.stringify(name)}`
313
+ + (saved.length ? `: it has ${saved.map(w => JSON.stringify(w?.name)).join(", ")}` : ""));
314
+ }
315
+ if (found.length > 1) {
316
+ throw new Refused(`${base} has ${found.length} pipelines named ${JSON.stringify(name)}: `
317
+ + `name one by its id (${found.map(w => w.id).join(", ")})`);
318
+ }
319
+ return found[0];
320
+ }
321
+
322
+ // The row's verdicts with --min-pass relaxing each eval read item by item,
323
+ // as the worker's --min-pass does: it only ever turns a fail into a pass.
324
+ const relaxed = (verdicts, minPass) => minPass == null ? verdicts : verdicts.map(evals =>
325
+ Object.fromEntries(Object.entries(evals).map(([id, e]) =>
326
+ [id, e.verdict === "fail" && e.ran && e.passed / e.ran >= minPass ? { ...e, verdict: "pass" } : e])));
327
+
328
+ // The JUnit suites of a lab's run, from its row: what the worker's runSuites
329
+ // writes from its own outcomes, read here from the verdicts and scores the
330
+ // row kept.
331
+ function rowSuites(core, run, verdicts, items) {
332
+ return verdicts.flatMap((evals, i) => Object.entries(evals).map(([id, e]) => {
333
+ const name = `${core.targetLabel(run, i)} › ${e.name}`;
334
+ if (e.ran == null) {
335
+ // Settled over the run rather than item by item: one case, the run.
336
+ const c = { name: e.name };
337
+ if (e.verdict === "skipped") c.skipped = "an earlier eval failed";
338
+ else if (e.verdict === "none") c.skipped = "nothing to read";
339
+ else if (e.verdict === "incomplete") c.error = "the run did not finish";
340
+ else if (e.verdict === "fail") c.failure = e.detail || "failed";
341
+ return { name, cases: [c] };
342
+ }
343
+ return { name, cases: items.map((it, x) => {
344
+ const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
345
+ const read = it?.scenarios?.[i]?.scores?.[id];
346
+ if (!it) c.error = "not run";
347
+ else if (it.unrun) c.error = it.unrun;
348
+ else if (!read) c.skipped = "not graded";
349
+ else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
350
+ else if (!read.pass) c.failure = reasonOf(read);
351
+ return c;
352
+ }) };
353
+ }));
354
+ }
355
+
356
+ /** [o]'s pipeline, queued on its lab; with --wait, followed to its verdict. */
357
+ async function runOnLab(core, o, { env, stdout, say }) {
358
+ const base = o.lab.replace(/\/+$/, "");
359
+ const call = labApi(base, env);
360
+
361
+ // What the page reads before it submits: the lab's pipelines and Target
362
+ // profiles, the Source the pipeline reads and the eval groups it may link.
363
+ const docs = (await call("GET", "/api/state")).docs ?? {};
364
+ const setup = docs["promptlab.profiles"]?.body ?? {};
365
+ const profiles = Array.isArray(setup.list) ? setup.list : [];
366
+ const saved = docs["promptlab.workflows"]?.body?.list;
367
+ const found = pipelineNamed(Array.isArray(saved) ? saved : [], o.pipeline, base);
368
+ const types = new Map(profiles.map(p => [p.id, p.type]));
369
+ const doc = core.upgradePipeline(found.work, { profileType: id => types.get(id) });
370
+ const g = profiles.find(p => p.id === setup.grader);
371
+ const grader = g ? { id: g.id, name: g.name } : null;
372
+ const content = core.contentOf(doc);
373
+ let source = null;
374
+ if (content?.type === "source" && content.ref?.id) {
375
+ source = await call("GET", `/api/sources/${encodeURIComponent(content.ref.id)}`, undefined, { missing: true });
376
+ }
377
+ const files = source ? source.files.map(f => f.name) : undefined;
378
+ const { datasets } = await call("GET", "/api/datasets");
379
+ const problems = core.validatePipeline(doc, {
380
+ profiles: id => profiles.find(p => p.id === id) ?? null,
381
+ sources: id => (source && id === content.ref.id ? { name: source.name, type: source.type, files } : null),
382
+ grader, groups: datasets ?? [],
383
+ });
384
+ if (problems.length) throw new Refused(`${found.name} cannot run: ${problems.join("; ")}`);
385
+ // As RunsTab's submit resolves it: slugged, so the run carries the slug
386
+ // its keys are spelt from (#253), and keyless -- the lab fills each key in
387
+ // from its own store.
388
+ const slugged = core.withSlugs(profiles);
389
+ const run = core.resolvePipeline(doc, {
390
+ profiles: id => slugged.find(p => p.id === id) ?? null, grader, ...(files ? { files } : {}),
391
+ });
392
+
393
+ const queued = (await call("POST", "/api/queue", run)).run;
394
+ say(`evals-lab run: ${found.name} is queued on ${base} as run ${queued.id}`);
395
+ if (!o.wait) {
396
+ stdout.write(`${queued.id}\n`);
397
+ return 0;
398
+ }
399
+
400
+ let row = queued, misses = 0, said = null;
401
+ while (row.status === "queued" || row.status === "running") {
402
+ await sleep(POLL_MS);
403
+ try {
404
+ row = await call("GET", `/api/queue/${encodeURIComponent(queued.id)}`);
405
+ misses = 0;
406
+ } catch (e) {
407
+ if (!e.unanswered || ++misses >= POLL_MISSES) throw e;
408
+ continue;
409
+ }
410
+ const p = row.progress ?? {};
411
+ if (o.progress && row.status === "running" && p.n !== said) {
412
+ said = p.n;
413
+ say(` ${p.n} of ${p.total}${p.current ? ` ${p.current}` : ""}`);
414
+ }
415
+ }
416
+
417
+ // A run the lab could not finish has no verdict to exit with.
418
+ if (row.status === "failed") throw new Refused(`run ${row.id} failed on ${base}: ${row.error || "the lab said no more"}`);
419
+ const finished = row.status === "done";
420
+ if (finished && !row.verdict) {
421
+ throw new Refused(`${base} keeps no verdict on its runs: it is older than evals-lab run --lab`);
422
+ }
423
+ const minPass = o.minPass == null ? null : Number(o.minPass);
424
+ const verdicts = row.verdicts ? relaxed(row.verdicts, minPass) : [];
425
+ // The lab's own verdict; --min-pass only turns a fail into a pass, and
426
+ // only once no eval fails under it.
427
+ const failing = verdicts.some(v => Object.values(v).some(e => e.verdict === "fail"));
428
+ const verdict = !finished ? "incomplete" : row.verdict === "fail" && !failing ? "pass" : row.verdict;
429
+ const items = core.upgradeResults(Array.isArray(row.results) ? row.results : []);
430
+ const report = {
431
+ runner: "evals-lab run --lab", lab: base, id: row.id, status: row.status, verdict,
432
+ ...(row.status === "cancelled" ? { cancelled: true } : {}),
433
+ ...(minPass != null ? { minPass } : {}),
434
+ run: { name: row.snapshot?.name ?? null, evals: row.snapshot?.evals ?? [], verdicts, items },
435
+ };
436
+ say(table(core, row.snapshot ?? run, report, []));
437
+ if (o.summary) fs.appendFileSync(o.summary, markdown(core, row.snapshot ?? run, report, []));
438
+ const text = JSON.stringify(report, null, 2) + "\n";
439
+ if (o.json === "-") stdout.write(text);
440
+ else if (o.json) fs.writeFileSync(o.json, text);
441
+ if (o.junit) {
442
+ fs.writeFileSync(o.junit, core.junitXml(rowSuites(core, row.snapshot ?? run, verdicts, items), report.runner));
443
+ }
444
+ return { pass: 0, fail: 1 }[verdict] ?? 2;
445
+ }
446
+
447
+ /**
448
+ * Runs [argv] -- what follows `evals-lab run` -- against the staged lab in
449
+ * [lab], and resolves to the exit code.
450
+ */
451
+ async function run(argv, { lab, env = process.env, stdout = process.stdout, stderr = process.stderr } = {}) {
452
+ const say = text => stderr.write(text + "\n");
453
+ let tmp = null;
454
+ try {
455
+ const o = runOptions(argv);
456
+ if (o.help) { stdout.write(USAGE + "\n"); return 0; }
457
+ const corePath = path.join(lab, "evals-core.mjs");
458
+ if (!fs.existsSync(corePath)) {
459
+ throw new Refused(`no ${corePath} -- in a checkout the package is staged by \`node npm/build.js\``);
460
+ }
461
+ const core = require(corePath);
462
+ if (o.lab != null) return await runOnLab(core, o, { env, stdout, say });
463
+ const { dir, files } = bundleFiles(o.bundle);
464
+ const plugins = await loadPlugins(core, dir);
465
+
466
+ // Read once without items, for the items its cases name; then with the
467
+ // ones there and those, so a case's item that is not there runs as missing.
468
+ let read = core.readBundle(files);
469
+ if ("error" in read) throw new Refused(`${o.bundle}: ${read.error}`);
470
+ let items = null, missing = [];
471
+ if (core.contentOf(read.run)?.type === "source") {
472
+ items = o.items ?? path.join(dir, "items");
473
+ if (!fs.existsSync(items) || !fs.statSync(items).isDirectory()) {
474
+ throw new Refused(o.items ? `--items ${o.items} is not a directory`
475
+ : `the pipeline reads a Source's items, and ${dir} holds no items/: name them with --items`);
476
+ }
477
+ const there = itemsIn(items);
478
+ const named = core.contentOf(read.run).files || [];
479
+ missing = named.filter(n => !there.includes(n));
480
+ read = core.readBundle(files, { items: [...new Set([...there, ...named])] });
481
+ if ("error" in read) throw new Refused(`${o.bundle}: ${read.error}`);
482
+ }
483
+ const doc = plugins.length ? { ...read.run, plugins } : read.run;
484
+
485
+ tmp = fs.mkdtempSync(path.join(os.tmpdir(), "evals-lab-run-"));
486
+ fs.writeFileSync(path.join(tmp, "run.json"), JSON.stringify(doc));
487
+ const args = [path.join(lab, "run-evals.js"), "--run", path.join(tmp, "run.json"),
488
+ "--json", path.join(tmp, "report.json")];
489
+ if (items) args.push("--source", items);
490
+ const ref = core.evalsDataset(read.run);
491
+ if (ref && read.datasets[ref.id]) {
492
+ fs.writeFileSync(path.join(tmp, "dataset.json"), JSON.stringify(read.datasets[ref.id]));
493
+ args.push("--dataset", path.join(tmp, "dataset.json"));
494
+ }
495
+ if (plugins.length) args.push("--plugins", path.join(dir, "plugins"));
496
+ if (o.minPass != null) args.push("--min-pass", o.minPass);
497
+ if (o.junit) args.push("--junit", path.resolve(o.junit));
498
+ if (o.progress) args.push("--progress-file", path.join(tmp, "progress.jsonl"));
499
+
500
+ const content = core.contentOf(read.run);
501
+ const total = content?.type === "source" ? (content.files || []).length : 1;
502
+ const stop = o.progress ? follow(path.join(tmp, "progress.jsonl"), total, say) : () => {};
503
+ // The worker's own lines are the table's raw form, so they are not
504
+ // shown; what it says on stderr -- a refusal -- is.
505
+ const code = await new Promise(resolve => {
506
+ const child = spawn(process.execPath, args, { stdio: ["ignore", "ignore", "inherit"], env, windowsHide: true });
507
+ child.on("error", e => { say(`evals-lab run: the worker did not start: ${e.message}`); resolve(EXIT.broken); });
508
+ child.on("exit", c => resolve(c ?? EXIT.broken));
509
+ });
510
+ stop();
511
+
512
+ const reportFile = path.join(tmp, "report.json");
513
+ if (!fs.existsSync(reportFile)) return code || EXIT.broken;
514
+ const text = fs.readFileSync(reportFile, "utf8");
515
+ const report = JSON.parse(text);
516
+ say(table(core, read.run, report, missing));
517
+ if (o.summary) fs.appendFileSync(o.summary, markdown(core, read.run, report, missing));
518
+ if (o.json === "-") stdout.write(text);
519
+ else if (o.json) fs.writeFileSync(o.json, text);
520
+ return code;
521
+ } catch (e) {
522
+ if (!(e instanceof Refused)) throw e;
523
+ say(`evals-lab run: ${e.message}`);
524
+ if (e.usage) say(`\n${USAGE}`);
525
+ return EXIT.broken;
526
+ } finally {
527
+ if (tmp) fs.rmSync(tmp, { recursive: true, force: true });
528
+ }
529
+ }
530
+
531
+ module.exports = { run, runOptions, USAGE };
package/lab/VERSION CHANGED
@@ -1 +1 @@
1
- 0.4.0 (2026.10.02-361)
1
+ 0.5.0 (2026.10.04-397)
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": 12,
2
+ "version": 13,
3
3
  "id": "demo-1",
4
4
  "name": "Demo 1",
5
5
  "jobs": [
@@ -94,17 +94,16 @@
94
94
  {
95
95
  "id": "t1",
96
96
  "name": "",
97
+ "type": "group",
97
98
  "continueOnFailure": true,
98
- "type": "metrics",
99
- "mode": "all",
100
- "threshold": null,
101
- "grader": null,
102
- "dataset": {
99
+ "group": {
103
100
  "id": "demo-1",
104
101
  "name": "Demo 1"
105
102
  },
106
- "over": "item",
107
- "metrics": []
103
+ "pin": null
108
104
  }
109
- ]
105
+ ],
106
+ "pass": {
107
+ "mode": "all"
108
+ }
110
109
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": 12,
2
+ "version": 13,
3
3
  "id": "demo-2",
4
4
  "name": "Demo 2",
5
5
  "jobs": [
@@ -94,17 +94,16 @@
94
94
  {
95
95
  "id": "t1",
96
96
  "name": "",
97
+ "type": "group",
97
98
  "continueOnFailure": true,
98
- "type": "metrics",
99
- "mode": "all",
100
- "threshold": null,
101
- "grader": null,
102
- "dataset": {
99
+ "group": {
103
100
  "id": "demo-2",
104
101
  "name": "Demo 2"
105
102
  },
106
- "over": "item",
107
- "metrics": []
103
+ "pin": null
108
104
  }
109
- ]
105
+ ],
106
+ "pass": {
107
+ "mode": "all"
108
+ }
110
109
  }