evals-lab 0.4.1 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -6,15 +6,19 @@
6
6
  // what it did twice: a short human report, and a JSON one with every verdict
7
7
  // in it. The exit code is the third statement and the coarsest:
8
8
  //
9
- // 0 every graded case passed
10
- // 1 the set ran and at least one case failed
11
- // 2 the set did not run in full -- nothing is graded, or an image or a
12
- // recorded reply was missing. NOT a pass. An eval set that scores well
13
- // because it never ran is the failure this whole ticket is about.
9
+ // 0 every eval passed
10
+ // 1 the run ran and at least one eval failed
11
+ // 2 the run did not run in full -- nothing is graded, an image or a
12
+ // recorded reply was missing, or it was cancelled. NOT a pass. An eval
13
+ // set that scores well because it never ran is the failure this whole
14
+ // ticket is about.
14
15
  // 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
15
16
  //
16
- // The gating policy -- which of those a pull request may merge on -- is #250's
17
- // to decide, and it has the JSON to decide it from. This only reports.
17
+ // The same in every mode, --run and --rescore included, so CI can gate on it
18
+ // (#251): each eval is held to its own rule -- every item it reads passes, or
19
+ // a whole-run eval's verdict -- and --min-pass relaxes the per-item rule to a
20
+ // share. The report's `verdict` (pass, fail, incomplete) says the same as the
21
+ // code, and --junit writes it for a CI test panel.
18
22
  //
19
23
  // Ollama is firewalled to a handful of hosts, so a live run has to happen on
20
24
  // one of them. That is a fact about the network and not something a flag here
@@ -64,7 +68,10 @@ const { execFileSync } = require("child_process");
64
68
  const core = require("./evals-core.mjs");
65
69
 
66
70
  const HERE = __dirname;
67
- const ROOT = path.join(HERE, "..", "..");
71
+ const ROOT = HERE;
72
+
73
+ // What a report names as its runner: this file, from the lab's root.
74
+ const RUNNER = "run-evals.js";
68
75
 
69
76
  // Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
70
77
  // into 448px tiles and the tile COUNT is what costs. The same budget the tab
@@ -82,11 +89,12 @@ const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
82
89
  // single-flight queue for good; `--timeout` can shorten it, never lengthen.
83
90
  const REQUEST_CAP = 600;
84
91
 
85
- const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
92
+ const USAGE = `Usage: node run-evals.js [options]
86
93
 
87
94
  --model <id> the model to grade. Required for a live run.
88
95
  --url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
89
- Ollama on this machine. A key comes from $EVAL_API_KEY, never argv.
96
+ Ollama on this machine. A key comes from
97
+ $EVALSLAB_API_KEY, never argv.
90
98
  --prompt <text> the prompt to grade, as a template: its tokens resolve
91
99
  under --tokens. Required without --pipeline: a dataset
92
100
  holds no prompt (the lab's Prompt library does).
@@ -103,17 +111,22 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
103
111
  --pipeline <file> a run document (docs/pipeline-model.md) with one
104
112
  scenario and a graded eval, graded case by case against
105
113
  the set. Instead of --prompt, --model and --url. Each
106
- profile's key comes from $EVAL_API_KEY_<ID>, never the
114
+ profile's key comes from $EVALSLAB_API_KEY_<SLUG> -- its
115
+ id's, in a file whose profile has no slug -- never the
107
116
  file; its content.files, when not empty, is the file
108
117
  list a case has to be in, as --files says.
109
118
  --samples <dir> where the images are. Default: the lab's samples/.
110
- --dataset <file> the dataset a graded run is scored against: its body
111
- ({"cases"}), or the file the lab's Export writes.
119
+ --dataset <file> the one eval group a graded run is scored against: its
120
+ body ({"cases"}), or the file the lab's Export writes.
112
121
  Versions 1 to 3 are read too (a prompt they hold is
113
122
  not used: --prompt names it); a version-2
114
123
  body's rules are what an older run's jobs, and a run
115
124
  without --pipeline, read replies under. Its cases grade.
116
- Required for anything graded.
125
+ Required for anything graded (or --groups, for a --run).
126
+ --groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
127
+ grades against each one its evals link (§17), so a run
128
+ linking several groups is handed them here instead of the
129
+ one --dataset. For --run and --rescore; not with --dataset.
117
130
  --source <dir> the same, named the way a run names it: the directory a
118
131
  Source is stored at. --samples and --source are one
119
132
  flag by two names; both together are refused.
@@ -153,6 +166,12 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
153
166
  alone. No model, no files, no network.
154
167
  --json <path|-> write the machine-readable report. "-" means stdout, and
155
168
  sends the human report to stderr.
169
+ --min-pass <0..1> an eval whose items are read one by one passes when at
170
+ least this share of them pass. It relaxes each eval's
171
+ own rule -- every item passes -- and never tightens it;
172
+ a whole-run eval keeps its own verdict.
173
+ --junit <file> write JUnit XML: one testsuite per target and eval, one
174
+ testcase per item, a failure carrying its reason.
156
175
  `;
157
176
 
158
177
  // NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
@@ -189,10 +208,10 @@ function broken(msg) {
189
208
 
190
209
  function parseArgs(argv) {
191
210
  const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
192
- replies: "", json: "", source: "", files: "", item: "", resume: "",
211
+ groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
193
212
  timeout: null, progress: false, cancelFile: "",
194
213
  run: "", progressFile: "", resultsFile: "", from: null, only: null,
195
- rescore: false, tokens: "", plugins: "" };
214
+ rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
196
215
  for (let i = 0; i < argv.length; i++) {
197
216
  const a = argv[i];
198
217
  const value = () => {
@@ -210,6 +229,7 @@ function parseArgs(argv) {
210
229
  case "--pipeline": o.pipeline = value(); break;
211
230
  case "--samples": o.samples = value(); break;
212
231
  case "--dataset": o.dataset = value(); break;
232
+ case "--groups": o.groups = value(); break;
213
233
  case "--replies": o.replies = value(); break;
214
234
  case "--source": o.source = value(); break;
215
235
  case "--files": o.files = value(); break;
@@ -225,6 +245,13 @@ function parseArgs(argv) {
225
245
  case "--only": o.only = value(); break;
226
246
  case "--rescore": o.rescore = true; break;
227
247
  case "--json": o.json = value(); break;
248
+ case "--min-pass": {
249
+ const v = value(), n = Number(v);
250
+ if (v.trim() === "" || !(n >= 0 && n <= 1)) broken(`--min-pass is a share from 0 to 1, and ${v} is not one`);
251
+ o.minPass = n;
252
+ break;
253
+ }
254
+ case "--junit": o.junit = value(); break;
228
255
  case "-h": case "--help":
229
256
  process.stdout.write(USAGE);
230
257
  throw new Stop("", EXIT.passed);
@@ -259,10 +286,10 @@ const inRepo = p => {
259
286
  const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
260
287
  const isText = v => typeof v === "string";
261
288
 
262
- // [dataset] is the body --dataset handed over, or null: the one dataset a
263
- // graded run can name, whatever id it names it by, since the queue hands the
264
- // worker the body of exactly the dataset the run was submitted against.
265
- function readRun(file, dataset) {
289
+ // [gctx] is the grading context (gradingContext): the eval group bodies a run
290
+ // grades against -- one, as --dataset, or a body per group kept with the run,
291
+ // as --groups (§17). The run reads each link's body by its reference.
292
+ function readRun(file, gctx) {
266
293
  let doc;
267
294
  try {
268
295
  // A run queued before the current version is read as one of today's:
@@ -270,12 +297,20 @@ function readRun(file, dataset) {
270
297
  // submitted with. A v2 run's list jobs read their replies under the
271
298
  // rules of the dataset it was graded against -- the body handed over.
272
299
  doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
273
- { rulesFor: () => rulesOf(dataset) });
300
+ { rulesFor: () => rulesOf(gctx.primary()) });
274
301
  } catch (e) {
275
302
  broken(`${file}: ${e.message}`);
276
303
  }
277
- const named = isObject(doc) ? core.evalsDataset(doc) : null;
278
- const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
304
+ // Every Library group the evals name -- a link's group, a private group's
305
+ // Cases from -- each as validatePipeline reads one. Pins were resolved at
306
+ // submit (the reference carries `n`), so they are not checked again here.
307
+ const groups = [], seen = new Set();
308
+ for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
309
+ if (seen.has(ref.id)) continue;
310
+ seen.add(ref.id);
311
+ groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
312
+ }
313
+ const bad = core.validatePipeline(doc, { groups });
279
314
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
280
315
  return doc;
281
316
  }
@@ -283,9 +318,25 @@ function readRun(file, dataset) {
283
318
  // The rules a version-2 body held, by the body readDataset made of it: what
284
319
  // an older run's jobs read replies under. A body of today's holds none.
285
320
  const heldRules = new WeakMap();
286
- const rulesOf = (dataset) => (dataset && heldRules.get(dataset)) ?? null;
321
+ const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
322
+
323
+ // An eval group's body as one of today's, with its version-2 rules taken
324
+ // first: a bare body, or the body the lab's Export wraps, unwrapped by its
325
+ // caller. [where] names it in a refusal.
326
+ function bodyFrom(doc, where) {
327
+ if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
328
+ broken(`${where}: rules has to be null or { "rules": [...] }`);
329
+ }
330
+ const rules = core.datasetRules(doc);
331
+ const body = core.upgradeDatasetBody(doc);
332
+ if (!isObject(body) || !Array.isArray(body.cases)) {
333
+ broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
334
+ }
335
+ if (rules) heldRules.set(body, rules);
336
+ return body;
337
+ }
287
338
 
288
- // The dataset --dataset names: a dataset's body, or the file the lab's
339
+ // The dataset --dataset names: an eval group's body, or the file the lab's
289
340
  // Export writes for one, as one of today's.
290
341
  function readDataset(file) {
291
342
  if (!file) return null;
@@ -300,26 +351,57 @@ function readDataset(file) {
300
351
  broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
301
352
  }
302
353
  doc = isObject(doc.dataset) ? doc.dataset.body : null;
354
+ } else if (isObject(doc) && doc.format === "evals-lab/eval-group") {
355
+ // The lab's Export of an eval group (docs/pipeline-model.md §17), which
356
+ // began at version 7.
357
+ if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
358
+ doc = isObject(doc.group) ? doc.group.body : null;
303
359
  }
304
- if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
305
- broken(`${file}: rules has to be null or { "rules": [...] }`);
360
+ return bodyFrom(doc, file);
361
+ }
362
+
363
+ // The bodies a run kept, as --groups hands them (§17): a JSON object of
364
+ // `<id>@<n>` to body -- the group's id and the version it graded with, its id
365
+ // alone for a run from before the lab numbered them -- each read as today's.
366
+ function readGroups(file) {
367
+ let raw;
368
+ try {
369
+ raw = JSON.parse(fs.readFileSync(file, "utf8"));
370
+ } catch (e) {
371
+ broken(`${file}: ${e.message}`);
306
372
  }
307
- // A body from an earlier version -- a run queued then, resumed now -- reads
308
- // as one of today's, its rules taken first.
309
- const rules = core.datasetRules(doc);
310
- doc = core.upgradeDatasetBody(doc);
311
- if (!isObject(doc) || !Array.isArray(doc.cases)) {
312
- broken(`${file} is not a dataset: its body has a cases list, or it is the lab's Export of one`);
373
+ if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
374
+ const bodies = new Map();
375
+ for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
376
+ return bodies;
377
+ }
378
+
379
+ // The key a run keeps a group's body under, mirroring the server's group_key:
380
+ // `<id>@<n>`, or the id alone for a reference from before versions.
381
+ const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
382
+
383
+ // The eval groups a --run grades against: one body, as --dataset, or a body
384
+ // per group, as --groups -- never both. `resolve` gives a link's body by its
385
+ // reference; `primary` the one a single-group run names, for a v2 run's rules.
386
+ function gradingContext(o) {
387
+ if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
388
+ if (o.groups) {
389
+ const bodies = readGroups(o.groups);
390
+ const first = bodies.values().next().value ?? null;
391
+ return { single: null, bodies, primary: () => first,
392
+ resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
313
393
  }
314
- if (rules) heldRules.set(doc, rules);
315
- return doc;
394
+ const single = readDataset(o.dataset);
395
+ return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
316
396
  }
317
397
 
318
398
  // The cases a graded run is scored against, by the item each names: exactly,
319
- // as the Source names it.
320
- function gradedBy(dataset) {
399
+ // as the Source names it. Built from the one group a single-group run names,
400
+ // for an eval type of its own that reads a case (a plugin's); a `group` eval
401
+ // reads each group's own case, by the item's name, as it grades.
402
+ function gradedBy(body) {
321
403
  const out = new Map();
322
- for (const c of core.gradedSetFrom(dataset || {})) {
404
+ for (const c of core.gradedSetFrom(body || {})) {
323
405
  if (!c.todo) out.set(c.item, c);
324
406
  }
325
407
  return out;
@@ -505,7 +587,7 @@ function graderFor(run) {
505
587
  const c = run.profiles?.[ref.id];
506
588
  if (!c) return undefined;
507
589
  if (!made.has(ref.id)) {
508
- const link = reach(c, core.keyVar(ref.id));
590
+ const link = reach(c, core.keyVar(ref.id, c.slug));
509
591
  made.set(ref.id, async prompt => {
510
592
  const r = await ask(link, { id: ref.id, ...c }, prompt, null);
511
593
  if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
@@ -613,17 +695,138 @@ function readUtf8(file) {
613
695
  return fs.readFileSync(file).toString("utf8");
614
696
  }
615
697
 
616
- // Every eval's reading of each scenario, keyed by the eval's id, exactly as
617
- // the page reads it: a whole-run eval's verdict, settled once every item is
618
- // in, and a per-item eval's counts. [settled] is false for a run that
619
- // stopped short, whose whole-run evals have not settled.
620
- const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
621
- Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
622
- name: o.label, skipped: o.skipped,
623
- ...(o.whole
624
- ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
625
- : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
626
- }])));
698
+ // Every eval's reading of each scenario, exactly as the page reads it.
699
+ // [settled] is false for a run that stopped short, whose whole-run evals
700
+ // have not settled.
701
+ const outcomesOf = (run, items, settled) =>
702
+ core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
703
+
704
+ // A link's Whole run verdict, per Target, keyed by the eval's id: a Library
705
+ // group's `run` metrics read over the replies so far (readGroupRun). A private
706
+ // group's Whole run is scenarioEvals' own (it holds the body); a link's body
707
+ // is held apart, in --groups, so reading it beside the link's items is the
708
+ // runner's (§17). [resolve] gives a link's body by its reference.
709
+ function wholeRunsOf(run, items, resolve) {
710
+ const kind = core.lastKind(run);
711
+ return core.targetsOf(run).map((_, i) => {
712
+ const out = {};
713
+ const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
714
+ for (const t of core.evalsOf(run)) {
715
+ // A private whole-run group is already scenarioEvals' verdict; here is
716
+ // the run beside a link's (or private group's) own items.
717
+ if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
718
+ const body = core.groupOf(t, { group: resolve });
719
+ if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
720
+ }
721
+ return out;
722
+ });
723
+ }
724
+
725
+ /**
726
+ * One eval's verdict on one target, for the build: its own rule -- every
727
+ * item it read passed, or a whole-run eval's verdict -- with [minPass] the
728
+ * share of items that is enough instead. It only relaxes: an eval whose
729
+ * every item passed passes whatever the share. One an earlier failure
730
+ * stopped is skipped -- that failure is the run's verdict already -- and one
731
+ * that had nothing to read, a group grading none of the run's items, says
732
+ * none: the run ran in full, which is what the server's status reads from
733
+ * the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
734
+ * run verdict, where its group reads one beside its items (§17): the group
735
+ * passes when both its items and its Whole run do, and either failing fails it.
736
+ */
737
+ function evalVerdict(o, settled, minPass, wholeRun) {
738
+ if (o.skipped) return "skipped";
739
+ if (!settled) return "incomplete";
740
+ if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
741
+ const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
742
+ : o.passed === o.ran ? "pass"
743
+ : minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
744
+ if (!wholeRun) return item;
745
+ const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
746
+ if (item === "fail" || whole === "fail") return "fail";
747
+ if (item === "pass" || whole === "pass") return "pass";
748
+ return item === "skipped" || whole === "skipped" ? "skipped" : "none";
749
+ }
750
+
751
+ // Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
752
+ // per-item eval's counts -- and, where a link reads a Whole run beside its
753
+ // items, that verdict too -- with the build's verdict on each. [wholes] is
754
+ // wholeRunsOf, one map per target.
755
+ const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
756
+ Object.fromEntries(evals.map(o => {
757
+ const wr = wholes[i] && wholes[i][o.id];
758
+ return [o.id, {
759
+ name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
760
+ ...(o.whole
761
+ ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
762
+ : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
763
+ ...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
764
+ }];
765
+ })));
766
+
767
+ /** Each Target's verdict under the overall pass rule (§17), from its groups'
768
+ verdicts: a group that passed counts, one that failed counts against, and
769
+ one that read nothing or was skipped is neither. */
770
+ const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
771
+ Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
772
+
773
+ /** The run's verdict from its evals' and the overall rule: a run that did not
774
+ finish is incomplete, whatever its evals read so far; then it passes when
775
+ every Target does under the rule. */
776
+ function runVerdict(verdicts, complete, pass) {
777
+ if (!complete) return "incomplete";
778
+ return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
779
+ }
780
+
781
+ const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
782
+
783
+ // ---- JUnit -----------------------------------------------------------------
784
+ // What a CI test panel reads, written by the core's junitXml.
785
+
786
+ /** Why a score failed, in a line: its failing metrics, else what it missed. */
787
+ function reasonOf(score) {
788
+ const off = (score.metrics || []).filter(m => !m.pass).map(m => `${m.label}: ${m.reason}`);
789
+ if (off.length) return off.join(" · ");
790
+ const missed = (score.missed || []).map(r => (Array.isArray(r) ? r.join(" or ") : r));
791
+ return missed.length ? `missed ${missed.join(", ")}` : "failed";
792
+ }
793
+
794
+ /** A --run's or a --rescore's suites: each target's evals over [items], with a
795
+ Whole run case where a link reads one beside its items ([wholes]). */
796
+ function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
797
+ return outcomes.flatMap((evals, i) => evals.map(o => {
798
+ const name = `${core.targetLabel(run, i)} › ${o.label}`;
799
+ const wr = wholes[i] && wholes[i][o.id];
800
+ const v = evalVerdict(o, settled, minPass, wr);
801
+ if (o.whole) {
802
+ // Settled over the run rather than item by item: one case, the run.
803
+ const c = { name: o.label };
804
+ if (v === "skipped") c.skipped = "an earlier eval failed";
805
+ else if (v === "none") c.skipped = "nothing to read";
806
+ else if (v === "incomplete") c.error = "the run did not finish";
807
+ else if (v === "fail") c.failure = o.verdict.detail || "failed";
808
+ return { name, cases: [c] };
809
+ }
810
+ const cases = items.map((it, x) => {
811
+ const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
812
+ const read = o.items[x];
813
+ if (!it) c.error = "not run";
814
+ else if (it.unrun) c.error = it.unrun;
815
+ else if (!read) c.skipped = "not graded";
816
+ else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
817
+ else if (!read.pass) c.failure = reasonOf(read);
818
+ return c;
819
+ });
820
+ if (wr) {
821
+ const c = { name: "Whole run" };
822
+ if (!settled) c.error = "the run did not finish";
823
+ else if (!wr.ran) c.skipped = "nothing to read";
824
+ else if (!wr.pass) c.failure = wr.detail || "failed";
825
+ cases.push(c);
826
+ }
827
+ return { name, cases };
828
+ }));
829
+ }
627
830
 
628
831
  // The run as the report restates it: each scenario's stages with the
629
832
  // connection each one asked.
@@ -639,8 +842,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
639
842
  });
640
843
 
641
844
  async function runSnapshot(o) {
642
- const dataset = readDataset(o.dataset);
643
- const run = readRun(o.run, dataset);
845
+ const gctx = gradingContext(o);
846
+ const run = readRun(o.run, gctx);
644
847
  // One item a pipeline holds itself (Text, or Prompt only's bare one), or
645
848
  // the Source's files.
646
849
  const content = core.contentOf(run);
@@ -652,8 +855,9 @@ async function runSnapshot(o) {
652
855
  broken(`${o.run} names no files and no text, so there is nothing to run`);
653
856
  }
654
857
 
655
- // The dataset the run was submitted against, for scoring its items.
656
- const graded = gradedBy(dataset);
858
+ // The one group a single-group run names, for an eval type of its own that
859
+ // reads a case; a `group` eval reads each group's case by the item's name.
860
+ const graded = gradedBy(gctx.primary());
657
861
 
658
862
  const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
659
863
  : text.trim() ? [{ name: null, kind: "text", text }]
@@ -678,7 +882,7 @@ async function runSnapshot(o) {
678
882
  // profile's id.
679
883
  const plans = core.targetsOf(run).map((_, i) => {
680
884
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
681
- const links = connections.map(c => reach(c, core.keyVar(c.id)));
885
+ const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
682
886
  return { stages, tokens, connections, links, calls, cells };
683
887
  });
684
888
 
@@ -728,6 +932,12 @@ async function runSnapshot(o) {
728
932
  break;
729
933
  }
730
934
  }
935
+ // A text file the Source does not hold is unrun, as an image is: the
936
+ // run is incomplete rather than broken.
937
+ if (item.kind === "text" && !item.bare && item.text == null && !fs.existsSync(path.join(o.source, item.name))) {
938
+ failed = `the file ${item.name} is not in the source`;
939
+ break;
940
+ }
731
941
  const textOf = item.kind === "text" && !item.bare
732
942
  ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
733
943
  const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
@@ -736,7 +946,8 @@ async function runSnapshot(o) {
736
946
  // text item a dataset grades is graded like an image.
737
947
  const kase = item.name ? graded.get(item.name) ?? null : null;
738
948
  production = core.productionOf(run, record);
739
- sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run) }) });
949
+ sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
950
+ { production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
740
951
  }
741
952
  // Production's reply to the item, where its record holds one: what a
742
953
  // metric compares with, kept so a re-score reads it again.
@@ -749,16 +960,27 @@ async function runSnapshot(o) {
749
960
  }
750
961
  if (progressFd != null) fs.closeSync(progressFd);
751
962
 
963
+ // Every item in, none of them unrun: only then has the run settled. A
964
+ // re-run of one item reads the rest from the results it was handed.
965
+ const ranItems = results.slice(0, total);
966
+ const complete = !cancelled && !unrun.length
967
+ && ranItems.length === total && ranItems.every(it => it && !it.unrun);
968
+ const outcomes = outcomesOf(run, ranItems, complete);
969
+ const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
970
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
752
971
  const report = {
753
- runner: "tools/prompt-lab/run-evals.js",
972
+ runner: RUNNER,
754
973
  ranAt: new Date().toISOString(),
755
- verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
974
+ verdict: runVerdict(verdicts, complete, run.pass),
975
+ ...(cancelled ? { cancelled: true } : {}),
976
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
756
977
  run: {
757
978
  name: run.name ?? null,
758
979
  evals: run.evals,
759
- verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
980
+ verdicts,
981
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
760
982
  scenarios: scenariosOf(run),
761
- items: results.slice(0, total),
983
+ items: ranItems,
762
984
  },
763
985
  unrun,
764
986
  };
@@ -767,14 +989,37 @@ async function runSnapshot(o) {
767
989
  say(`${report.runner} — run ${report.run.name ?? ""}`);
768
990
  for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
769
991
  if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
770
- say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled" : report.verdict}`);
992
+ sayVerdicts(say, run, verdicts, report.run.pass);
993
+ say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
994
+
995
+ writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
996
+ process.exitCode = exitOf(report.verdict);
997
+ }
998
+
999
+ /** Each target's groups and their verdicts, a line each, and the overall rule
1000
+ beneath them where [pass] gives a figure per target. */
1001
+ function sayVerdicts(say, run, verdicts, pass) {
1002
+ verdicts.forEach((evals, i) => {
1003
+ for (const e of Object.values(evals)) {
1004
+ const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
1005
+ const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
1006
+ say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
1007
+ }
1008
+ if (pass) {
1009
+ const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
1010
+ say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
1011
+ }
1012
+ });
1013
+ }
771
1014
 
1015
+ /** The JSON report, and the JUnit one from [suites], where each was asked for. */
1016
+ function writeReports(o, report, suites) {
772
1017
  if (o.json) {
773
- const text2 = JSON.stringify(report, null, 2);
774
- if (o.json === "-") process.stdout.write(text2 + "\n");
775
- else fs.writeFileSync(o.json, text2 + "\n");
1018
+ const text = JSON.stringify(report, null, 2);
1019
+ if (o.json === "-") process.stdout.write(text + "\n");
1020
+ else fs.writeFileSync(o.json, text + "\n");
776
1021
  }
777
- process.exitCode = cancelled ? EXIT.passed : (unrun.length ? EXIT.incomplete : EXIT.passed);
1022
+ if (o.junit) fs.writeFileSync(o.junit, core.junitXml(suites(), RUNNER));
778
1023
  }
779
1024
 
780
1025
  /** Re-score a run's stored results against the dataset --dataset hands over,
@@ -783,8 +1028,8 @@ async function runSnapshot(o) {
783
1028
  * a --run's, with the stored results as its items, and a whole-run eval's
784
1029
  * verdict settled the same way. */
785
1030
  async function runRescore(o){
786
- const dataset = readDataset(o.dataset);
787
- const run = readRun(o.run, dataset);
1031
+ const gctx = gradingContext(o);
1032
+ const run = readRun(o.run, gctx);
788
1033
  let results;
789
1034
  try {
790
1035
  results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
@@ -796,50 +1041,53 @@ async function runRescore(o){
796
1041
  }
797
1042
  results = core.upgradeResults(results);
798
1043
 
799
- // The cases the run's items are scored against. A file the set does
800
- // not grade keeps its reply and has no score -- honestly: nothing else would
801
- // say what the set covers.
802
- const graded = gradedBy(dataset);
1044
+ // The one group a single-group run names, for an eval type of its own that
1045
+ // reads a case. A file no group grades keeps its reply and has no score --
1046
+ // honestly: nothing else would say what the groups cover.
1047
+ const graded = gradedBy(gctx.primary());
803
1048
 
804
1049
  const only = o.only != null ? parseInt(o.only, 10) : null;
805
- // An eval that reads every item (the Metrics) re-reads one with no case
1050
+ // An eval that reads every item (a group) re-reads one with no case
806
1051
  // too, against the production reply the item kept.
807
1052
  const items = await Promise.all(results.map(async (it, i) => {
808
1053
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
809
1054
  const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
810
- const more = { production: it.production ?? null, grader: graderFor(run) };
1055
+ const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
811
1056
  return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
812
1057
  (isObject(side) && side.res)
813
1058
  ? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
814
1059
  }));
815
1060
 
1061
+ // A stored run that stopped short re-scores as what it is: incomplete.
1062
+ const complete = items.every(it => isObject(it) && !it.unrun);
1063
+ const outcomes = outcomesOf(run, items, complete);
1064
+ const wholes = wholeRunsOf(run, items, gctx.resolve);
1065
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
816
1066
  const report = {
817
- runner: "tools/prompt-lab/run-evals.js",
1067
+ runner: RUNNER,
818
1068
  ranAt: new Date().toISOString(),
819
- verdict: "done",
1069
+ verdict: runVerdict(verdicts, complete, run.pass),
1070
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
820
1071
  rescored: true,
821
1072
  run: {
822
1073
  name: run.name ?? null,
823
1074
  evals: run.evals,
824
- verdicts: verdictsOf(run, items, true),
1075
+ verdicts,
1076
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
825
1077
  scenarios: scenariosOf(run),
826
1078
  items,
827
1079
  },
828
1080
  };
829
1081
 
830
- if (o.json) {
831
- const text = JSON.stringify(report, null, 2);
832
- if (o.json === "-") process.stdout.write(text + "\n");
833
- else fs.writeFileSync(o.json, text + "\n");
834
- }
835
- process.exitCode = EXIT.passed;
1082
+ writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
1083
+ process.exitCode = exitOf(report.verdict);
836
1084
  }
837
1085
 
838
1086
  /**
839
1087
  * The run --prompt, --model and --url describe, as a run document: one job
840
1088
  * that sees the image and answers in the default kind, one scenario,
841
1089
  * graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
842
- * connection, as it never did, and its key is $EVAL_API_KEY.
1090
+ * connection, as it never did, and its key is $EVALSLAB_API_KEY.
843
1091
  */
844
1092
  function cliRun(o, dataset) {
845
1093
  let doc = core.blankPipeline();
@@ -857,8 +1105,7 @@ function cliRun(o, dataset) {
857
1105
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
858
1106
  doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
859
1107
  steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
860
- doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
861
- mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
1108
+ doc.evals = [{ id: "cli", type: "group", name: "", continueOnFailure: true, group: { id: "cli", name: "" }, pin: null }];
862
1109
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
863
1110
  return doc;
864
1111
  }
@@ -933,19 +1180,22 @@ async function main() {
933
1180
  }
934
1181
 
935
1182
  // The dataset the run is graded against. Everything below grades, so there
936
- // is no run without one.
1183
+ // is no run without one. --groups hands a --run the bodies it kept; a
1184
+ // --pipeline or --prompt grades one scenario against one, named by --dataset.
1185
+ if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
937
1186
  if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
938
1187
  const dataset = readDataset(o.dataset);
1188
+ const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
939
1189
 
940
1190
  // The run to grade. A --pipeline document is one scenario graded against
941
1191
  // its dataset; without one it is the stage this script always ran -- the
942
- // prompt, seeing the image, on --url with $EVAL_API_KEY -- stated as
1192
+ // prompt, seeing the image, on --url with $EVALSLAB_API_KEY -- stated as
943
1193
  // the same kind of document, so both go through one reading of it.
944
1194
  if (!o.pipeline && !(o.prompt ?? "").trim()) {
945
1195
  broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
946
1196
  + "the lab's Prompt library does.");
947
1197
  }
948
- const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
1198
+ const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
949
1199
  if (o.pipeline) {
950
1200
  if (core.targetsOf(run).length !== 1) {
951
1201
  broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
@@ -1038,7 +1288,7 @@ async function main() {
1038
1288
  if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
1039
1289
  links = connections.map((c, i) => {
1040
1290
  if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
1041
- const to = reach(c, o.pipeline ? core.keyVar(c.id) : "EVAL_API_KEY");
1291
+ const to = reach(c, o.pipeline ? core.keyVar(c.id, c.slug) : "EVALSLAB_API_KEY");
1042
1292
  // Once here rather than once per image: a llama.cpp stage on a
1043
1293
  // hosted model is a request that cannot be built at all.
1044
1294
  try {
@@ -1186,10 +1436,12 @@ async function main() {
1186
1436
  // passed, and 49 items absent with the fiftieth green is exactly how
1187
1437
  // that happens.
1188
1438
  const complete = graded.length > 0 && unrun.length === 0;
1189
- const verdict = !complete ? "incomplete" : tally.passed === tally.ran ? "pass" : "fail";
1439
+ const enough = tally.passed === tally.ran
1440
+ || (o.minPass != null && tally.passed / tally.ran >= o.minPass);
1441
+ const verdict = !complete ? "incomplete" : enough ? "pass" : "fail";
1190
1442
 
1191
1443
  const report = {
1192
- runner: "tools/prompt-lab/run-evals.js",
1444
+ runner: RUNNER,
1193
1445
  ranAt: new Date().toISOString(),
1194
1446
  dataset: inRepo(o.dataset),
1195
1447
  prompt,
@@ -1217,6 +1469,7 @@ async function main() {
1217
1469
  filesListed: snapshotSet.size } : {}),
1218
1470
  ...(o.source ? { source: inRepo(o.source) } : {}),
1219
1471
  verdict,
1472
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
1220
1473
  graded: (itemsOnly ?? graded).length,
1221
1474
  ungraded: set.length - graded.length,
1222
1475
  ran: tally.ran,
@@ -1269,14 +1522,16 @@ async function main() {
1269
1522
  + `so this is a partial run and not a result.`);
1270
1523
  }
1271
1524
 
1272
- if (o.json) {
1273
- const text = JSON.stringify(report, null, 2);
1274
- if (toStdout) process.stdout.write(text + "\n");
1275
- else fs.writeFileSync(o.json, text + "\n");
1276
- }
1277
-
1278
- process.exitCode = !complete ? EXIT.incomplete
1279
- : report.failed ? EXIT.failed : EXIT.passed;
1525
+ // The one eval this mode grades by, as a suite of its cases.
1526
+ const suites = () => {
1527
+ const j = Math.max(0, core.evalsOf(run).findIndex(t => core.evalsDataset({ evals: [t] })));
1528
+ return [{ name: `${core.targetLabel(run, 0)} › ${core.evalLabel(run, j)}`, cases: [
1529
+ ...rows.map(r => ({ name: r.id, ...(r.pass ? {} : { failure: r.reasons.join(" · ") || "failed" }) })),
1530
+ ...unrun.map(u => { const at = u.indexOf(": "); return { name: u.slice(0, at), error: u.slice(at + 2) }; }),
1531
+ ] }];
1532
+ };
1533
+ writeReports(o, report, suites);
1534
+ process.exitCode = exitOf(verdict);
1280
1535
  }
1281
1536
 
1282
1537
  main().catch(e => {