evals-lab 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -116,13 +116,17 @@ const USAGE = `Usage: node run-evals.js [options]
116
116
  file; its content.files, when not empty, is the file
117
117
  list a case has to be in, as --files says.
118
118
  --samples <dir> where the images are. Default: the lab's samples/.
119
- --dataset <file> the dataset a graded run is scored against: its body
120
- ({"cases"}), or the file the lab's Export writes.
119
+ --dataset <file> the one eval group a graded run is scored against: its
120
+ body ({"cases"}), or the file the lab's Export writes.
121
121
  Versions 1 to 3 are read too (a prompt they hold is
122
122
  not used: --prompt names it); a version-2
123
123
  body's rules are what an older run's jobs, and a run
124
124
  without --pipeline, read replies under. Its cases grade.
125
- Required for anything graded.
125
+ Required for anything graded (or --groups, for a --run).
126
+ --groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
127
+ grades against each one its evals link (§17), so a run
128
+ linking several groups is handed them here instead of the
129
+ one --dataset. For --run and --rescore; not with --dataset.
126
130
  --source <dir> the same, named the way a run names it: the directory a
127
131
  Source is stored at. --samples and --source are one
128
132
  flag by two names; both together are refused.
@@ -204,7 +208,7 @@ function broken(msg) {
204
208
 
205
209
  function parseArgs(argv) {
206
210
  const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
207
- replies: "", json: "", source: "", files: "", item: "", resume: "",
211
+ groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
208
212
  timeout: null, progress: false, cancelFile: "",
209
213
  run: "", progressFile: "", resultsFile: "", from: null, only: null,
210
214
  rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
@@ -225,6 +229,7 @@ function parseArgs(argv) {
225
229
  case "--pipeline": o.pipeline = value(); break;
226
230
  case "--samples": o.samples = value(); break;
227
231
  case "--dataset": o.dataset = value(); break;
232
+ case "--groups": o.groups = value(); break;
228
233
  case "--replies": o.replies = value(); break;
229
234
  case "--source": o.source = value(); break;
230
235
  case "--files": o.files = value(); break;
@@ -281,10 +286,10 @@ const inRepo = p => {
281
286
  const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
282
287
  const isText = v => typeof v === "string";
283
288
 
284
- // [dataset] is the body --dataset handed over, or null: the one dataset a
285
- // graded run can name, whatever id it names it by, since the queue hands the
286
- // worker the body of exactly the dataset the run was submitted against.
287
- function readRun(file, dataset) {
289
+ // [gctx] is the grading context (gradingContext): the eval group bodies a run
290
+ // grades against -- one, as --dataset, or a body per group kept with the run,
291
+ // as --groups (§17). The run reads each link's body by its reference.
292
+ function readRun(file, gctx) {
288
293
  let doc;
289
294
  try {
290
295
  // A run queued before the current version is read as one of today's:
@@ -292,12 +297,20 @@ function readRun(file, dataset) {
292
297
  // submitted with. A v2 run's list jobs read their replies under the
293
298
  // rules of the dataset it was graded against -- the body handed over.
294
299
  doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
295
- { rulesFor: () => rulesOf(dataset) });
300
+ { rulesFor: () => rulesOf(gctx.primary()) });
296
301
  } catch (e) {
297
302
  broken(`${file}: ${e.message}`);
298
303
  }
299
- const named = isObject(doc) ? core.evalsDataset(doc) : null;
300
- const bad = core.validatePipeline(doc, { groups: dataset && named ? [named] : [] });
304
+ // Every Library group the evals name -- a link's group, a private group's
305
+ // Cases from -- each as validatePipeline reads one. Pins were resolved at
306
+ // submit (the reference carries `n`), so they are not checked again here.
307
+ const groups = [], seen = new Set();
308
+ for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
309
+ if (seen.has(ref.id)) continue;
310
+ seen.add(ref.id);
311
+ groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
312
+ }
313
+ const bad = core.validatePipeline(doc, { groups });
301
314
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
302
315
  return doc;
303
316
  }
@@ -305,9 +318,25 @@ function readRun(file, dataset) {
305
318
  // The rules a version-2 body held, by the body readDataset made of it: what
306
319
  // an older run's jobs read replies under. A body of today's holds none.
307
320
  const heldRules = new WeakMap();
308
- const rulesOf = (dataset) => (dataset && heldRules.get(dataset)) ?? null;
321
+ const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
322
+
323
+ // An eval group's body as one of today's, with its version-2 rules taken
324
+ // first: a bare body, or the body the lab's Export wraps, unwrapped by its
325
+ // caller. [where] names it in a refusal.
326
+ function bodyFrom(doc, where) {
327
+ if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
328
+ broken(`${where}: rules has to be null or { "rules": [...] }`);
329
+ }
330
+ const rules = core.datasetRules(doc);
331
+ const body = core.upgradeDatasetBody(doc);
332
+ if (!isObject(body) || !Array.isArray(body.cases)) {
333
+ broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
334
+ }
335
+ if (rules) heldRules.set(body, rules);
336
+ return body;
337
+ }
309
338
 
310
- // The dataset --dataset names: a dataset's body, or the file the lab's
339
+ // The dataset --dataset names: an eval group's body, or the file the lab's
311
340
  // Export writes for one, as one of today's.
312
341
  function readDataset(file) {
313
342
  if (!file) return null;
@@ -328,29 +357,51 @@ function readDataset(file) {
328
357
  if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
329
358
  doc = isObject(doc.group) ? doc.group.body : null;
330
359
  }
331
- if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
332
- broken(`${file}: rules has to be null or { "rules": [...] }`);
333
- }
334
- // A body from an earlier version -- a run queued then, resumed now -- reads
335
- // as one of today's, its rules taken first.
336
- const rules = core.datasetRules(doc);
337
- doc = core.upgradeDatasetBody(doc);
338
- if (!isObject(doc) || !Array.isArray(doc.cases)) {
339
- broken(`${file} is not a dataset: its body has a cases list, or it is the lab's Export of one`);
360
+ return bodyFrom(doc, file);
361
+ }
362
+
363
+ // The bodies a run kept, as --groups hands them (§17): a JSON object of
364
+ // `<id>@<n>` to body -- the group's id and the version it graded with, its id
365
+ // alone for a run from before the lab numbered them -- each read as today's.
366
+ function readGroups(file) {
367
+ let raw;
368
+ try {
369
+ raw = JSON.parse(fs.readFileSync(file, "utf8"));
370
+ } catch (e) {
371
+ broken(`${file}: ${e.message}`);
340
372
  }
341
- if (rules) heldRules.set(doc, rules);
342
- return doc;
373
+ if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
374
+ const bodies = new Map();
375
+ for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
376
+ return bodies;
343
377
  }
344
378
 
345
- // A link's eval group, by reference: the body --dataset handed over, the one
346
- // Library group a run reads, whatever id it names it by.
347
- const groupOf = (dataset) => () => dataset ?? undefined;
379
+ // The key a run keeps a group's body under, mirroring the server's group_key:
380
+ // `<id>@<n>`, or the id alone for a reference from before versions.
381
+ const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
382
+
383
+ // The eval groups a --run grades against: one body, as --dataset, or a body
384
+ // per group, as --groups -- never both. `resolve` gives a link's body by its
385
+ // reference; `primary` the one a single-group run names, for a v2 run's rules.
386
+ function gradingContext(o) {
387
+ if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
388
+ if (o.groups) {
389
+ const bodies = readGroups(o.groups);
390
+ const first = bodies.values().next().value ?? null;
391
+ return { single: null, bodies, primary: () => first,
392
+ resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
393
+ }
394
+ const single = readDataset(o.dataset);
395
+ return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
396
+ }
348
397
 
349
398
  // The cases a graded run is scored against, by the item each names: exactly,
350
- // as the Source names it.
351
- function gradedBy(dataset) {
399
+ // as the Source names it. Built from the one group a single-group run names,
400
+ // for an eval type of its own that reads a case (a plugin's); a `group` eval
401
+ // reads each group's own case, by the item's name, as it grades.
402
+ function gradedBy(body) {
352
403
  const out = new Map();
353
- for (const c of core.gradedSetFrom(dataset || {})) {
404
+ for (const c of core.gradedSetFrom(body || {})) {
354
405
  if (!c.todo) out.set(c.item, c);
355
406
  }
356
407
  return out;
@@ -499,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
499
550
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
500
551
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
501
552
  }
553
+ // The Recorded target replays each call's recorded reply, read as the flow
554
+ // reads one (like send, but sending nothing); it does not go through the
555
+ // model Read-as wrapper below -- the recorded body is read directly.
556
+ if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
557
+ return () => {
558
+ const r = record ?? textRecord(text);
559
+ const result = r && r.result;
560
+ if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
561
+ try {
562
+ const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
563
+ { scope: r.scope || {}, now: r.at ?? null });
564
+ return { raw: read.raw, said: read.said, ms: 0, conn: null };
565
+ } catch (e) {
566
+ return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
567
+ }
568
+ };
569
+ }
502
570
  const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
503
- ? async sent => core.localAnswer(c, text, sent)
571
+ ? async sent => core.localAnswer(c, text, sent, record)
504
572
  : (sent, url) => ask(links[k], c, sent, url);
505
573
  if (!step?.readAs || !step?.step) return call;
506
574
  return async (sent, url) => {
@@ -650,40 +718,81 @@ function readUtf8(file) {
650
718
  const outcomesOf = (run, items, settled) =>
651
719
  core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
652
720
 
721
+ // A link's Whole run verdict, per Target, keyed by the eval's id: a Library
722
+ // group's `run` metrics read over the replies so far (readGroupRun). A private
723
+ // group's Whole run is scenarioEvals' own (it holds the body); a link's body
724
+ // is held apart, in --groups, so reading it beside the link's items is the
725
+ // runner's (§17). [resolve] gives a link's body by its reference.
726
+ function wholeRunsOf(run, items, resolve) {
727
+ const kind = core.lastKind(run);
728
+ return core.targetsOf(run).map((_, i) => {
729
+ const out = {};
730
+ const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
731
+ for (const t of core.evalsOf(run)) {
732
+ // A private whole-run group is already scenarioEvals' verdict; here is
733
+ // the run beside a link's (or private group's) own items.
734
+ if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
735
+ const body = core.groupOf(t, { group: resolve });
736
+ if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
737
+ }
738
+ return out;
739
+ });
740
+ }
741
+
653
742
  /**
654
743
  * One eval's verdict on one target, for the build: its own rule -- every
655
744
  * item it read passed, or a whole-run eval's verdict -- with [minPass] the
656
745
  * share of items that is enough instead. It only relaxes: an eval whose
657
746
  * every item passed passes whatever the share. One an earlier failure
658
747
  * stopped is skipped -- that failure is the run's verdict already -- and one
659
- * that had nothing to read, a dataset grading none of the run's items, says
748
+ * that had nothing to read, a group grading none of the run's items, says
660
749
  * none: the run ran in full, which is what the server's status reads from
661
- * the exit code, so it is not incomplete either.
750
+ * the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
751
+ * run verdict, where its group reads one beside its items (§17): the group
752
+ * passes when both its items and its Whole run do, and either failing fails it.
662
753
  */
663
- function evalVerdict(o, settled, minPass) {
754
+ function evalVerdict(o, settled, minPass, wholeRun) {
664
755
  if (o.skipped) return "skipped";
665
756
  if (!settled) return "incomplete";
666
757
  if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
667
- if (!o.ran) return o.skippedItems ? "skipped" : "none";
668
- if (o.passed === o.ran) return "pass";
669
- return minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
758
+ const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
759
+ : o.passed === o.ran ? "pass"
760
+ : minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
761
+ if (!wholeRun) return item;
762
+ const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
763
+ if (item === "fail" || whole === "fail") return "fail";
764
+ if (item === "pass" || whole === "pass") return "pass";
765
+ return item === "skipped" || whole === "skipped" ? "skipped" : "none";
670
766
  }
671
767
 
672
- // Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
673
- // a per-item eval's counts, and the build's verdict on each.
674
- const verdictsOf = (outcomes, settled, minPass) => outcomes.map(evals =>
675
- Object.fromEntries(evals.map(o => [o.id, {
676
- name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass),
677
- ...(o.whole
678
- ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
679
- : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
680
- }])));
681
-
682
- /** The run's verdict from its evals': a run that did not finish is
683
- incomplete, whatever its evals read so far; then any eval failing fails it. */
684
- function runVerdict(verdicts, complete) {
768
+ // Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
769
+ // per-item eval's counts -- and, where a link reads a Whole run beside its
770
+ // items, that verdict too -- with the build's verdict on each. [wholes] is
771
+ // wholeRunsOf, one map per target.
772
+ const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
773
+ Object.fromEntries(evals.map(o => {
774
+ const wr = wholes[i] && wholes[i][o.id];
775
+ return [o.id, {
776
+ name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
777
+ ...(o.whole
778
+ ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
779
+ : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
780
+ ...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
781
+ }];
782
+ })));
783
+
784
+ /** Each Target's verdict under the overall pass rule (§17), from its groups'
785
+ verdicts: a group that passed counts, one that failed counts against, and
786
+ one that read nothing or was skipped is neither. */
787
+ const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
788
+ Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
789
+
790
+ /** The run's verdict from its evals' and the overall rule: a run that did not
791
+ finish is incomplete, whatever its evals read so far; then it passes when
792
+ every Target does under the rule. */
793
+ function runVerdict(verdicts, complete, pass) {
685
794
  if (!complete) return "incomplete";
686
- return verdicts.some(v => Object.values(v).some(e => e.verdict === "fail")) ? "fail" : "pass";
795
+ return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
687
796
  }
688
797
 
689
798
  const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
@@ -699,11 +808,13 @@ function reasonOf(score) {
699
808
  return missed.length ? `missed ${missed.join(", ")}` : "failed";
700
809
  }
701
810
 
702
- /** A --run's or a --rescore's suites: each target's evals over [items]. */
703
- function runSuites(run, outcomes, items, settled, minPass) {
811
+ /** A --run's or a --rescore's suites: each target's evals over [items], with a
812
+ Whole run case where a link reads one beside its items ([wholes]). */
813
+ function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
704
814
  return outcomes.flatMap((evals, i) => evals.map(o => {
705
815
  const name = `${core.targetLabel(run, i)} › ${o.label}`;
706
- const v = evalVerdict(o, settled, minPass);
816
+ const wr = wholes[i] && wholes[i][o.id];
817
+ const v = evalVerdict(o, settled, minPass, wr);
707
818
  if (o.whole) {
708
819
  // Settled over the run rather than item by item: one case, the run.
709
820
  const c = { name: o.label };
@@ -713,7 +824,7 @@ function runSuites(run, outcomes, items, settled, minPass) {
713
824
  else if (v === "fail") c.failure = o.verdict.detail || "failed";
714
825
  return { name, cases: [c] };
715
826
  }
716
- return { name, cases: items.map((it, x) => {
827
+ const cases = items.map((it, x) => {
717
828
  const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
718
829
  const read = o.items[x];
719
830
  if (!it) c.error = "not run";
@@ -722,7 +833,15 @@ function runSuites(run, outcomes, items, settled, minPass) {
722
833
  else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
723
834
  else if (!read.pass) c.failure = reasonOf(read);
724
835
  return c;
725
- }) };
836
+ });
837
+ if (wr) {
838
+ const c = { name: "Whole run" };
839
+ if (!settled) c.error = "the run did not finish";
840
+ else if (!wr.ran) c.skipped = "nothing to read";
841
+ else if (!wr.pass) c.failure = wr.detail || "failed";
842
+ cases.push(c);
843
+ }
844
+ return { name, cases };
726
845
  }));
727
846
  }
728
847
 
@@ -740,8 +859,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
740
859
  });
741
860
 
742
861
  async function runSnapshot(o) {
743
- const dataset = readDataset(o.dataset);
744
- const run = readRun(o.run, dataset);
862
+ const gctx = gradingContext(o);
863
+ const run = readRun(o.run, gctx);
745
864
  // One item a pipeline holds itself (Text, or Prompt only's bare one), or
746
865
  // the Source's files.
747
866
  const content = core.contentOf(run);
@@ -753,8 +872,9 @@ async function runSnapshot(o) {
753
872
  broken(`${o.run} names no files and no text, so there is nothing to run`);
754
873
  }
755
874
 
756
- // The dataset the run was submitted against, for scoring its items.
757
- const graded = gradedBy(dataset);
875
+ // The one group a single-group run names, for an eval type of its own that
876
+ // reads a case; a `group` eval reads each group's case by the item's name.
877
+ const graded = gradedBy(gctx.primary());
758
878
 
759
879
  const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
760
880
  : text.trim() ? [{ name: null, kind: "text", text }]
@@ -780,7 +900,10 @@ async function runSnapshot(o) {
780
900
  const plans = core.targetsOf(run).map((_, i) => {
781
901
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
782
902
  const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
783
- return { stages, tokens, connections, links, calls, cells };
903
+ // The Recorded target replays the recorded reply, so a metric that compares
904
+ // with the recorded reply has nothing to say of it (n/a).
905
+ const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
906
+ return { stages, tokens, connections, links, calls, cells, recordedTarget };
784
907
  });
785
908
 
786
909
  let results = [];
@@ -842,8 +965,9 @@ async function runSnapshot(o) {
842
965
  // A case is found by its file's name, whatever kind of file it is: a
843
966
  // text item a dataset grades is graded like an image.
844
967
  const kase = item.name ? graded.get(item.name) ?? null : null;
845
- production = core.productionOf(run, record);
846
- sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run), group: groupOf(dataset) }) });
968
+ production = core.recordedReplyOf(run, record);
969
+ sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
970
+ { production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
847
971
  }
848
972
  // Production's reply to the item, where its record holds one: what a
849
973
  // metric compares with, kept so a re-score reads it again.
@@ -862,17 +986,19 @@ async function runSnapshot(o) {
862
986
  const complete = !cancelled && !unrun.length
863
987
  && ranItems.length === total && ranItems.every(it => it && !it.unrun);
864
988
  const outcomes = outcomesOf(run, ranItems, complete);
865
- const verdicts = verdictsOf(outcomes, complete, o.minPass);
989
+ const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
990
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
866
991
  const report = {
867
992
  runner: RUNNER,
868
993
  ranAt: new Date().toISOString(),
869
- verdict: runVerdict(verdicts, complete),
994
+ verdict: runVerdict(verdicts, complete, run.pass),
870
995
  ...(cancelled ? { cancelled: true } : {}),
871
996
  ...(o.minPass != null ? { minPass: o.minPass } : {}),
872
997
  run: {
873
998
  name: run.name ?? null,
874
999
  evals: run.evals,
875
1000
  verdicts,
1001
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
876
1002
  scenarios: scenariosOf(run),
877
1003
  items: ranItems,
878
1004
  },
@@ -883,19 +1009,25 @@ async function runSnapshot(o) {
883
1009
  say(`${report.runner} — run ${report.run.name ?? ""}`);
884
1010
  for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
885
1011
  if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
886
- sayVerdicts(say, run, verdicts);
1012
+ sayVerdicts(say, run, verdicts, report.run.pass);
887
1013
  say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
888
1014
 
889
- writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
1015
+ writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
890
1016
  process.exitCode = exitOf(report.verdict);
891
1017
  }
892
1018
 
893
- /** Each target's evals and their verdicts, a line each. */
894
- function sayVerdicts(say, run, verdicts) {
1019
+ /** Each target's groups and their verdicts, a line each, and the overall rule
1020
+ beneath them where [pass] gives a figure per target. */
1021
+ function sayVerdicts(say, run, verdicts, pass) {
895
1022
  verdicts.forEach((evals, i) => {
896
1023
  for (const e of Object.values(evals)) {
897
1024
  const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
898
- say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}`);
1025
+ const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
1026
+ say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
1027
+ }
1028
+ if (pass) {
1029
+ const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
1030
+ say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
899
1031
  }
900
1032
  });
901
1033
  }
@@ -916,8 +1048,8 @@ function writeReports(o, report, suites) {
916
1048
  * a --run's, with the stored results as its items, and a whole-run eval's
917
1049
  * verdict settled the same way. */
918
1050
  async function runRescore(o){
919
- const dataset = readDataset(o.dataset);
920
- const run = readRun(o.run, dataset);
1051
+ const gctx = gradingContext(o);
1052
+ const run = readRun(o.run, gctx);
921
1053
  let results;
922
1054
  try {
923
1055
  results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
@@ -929,18 +1061,18 @@ async function runRescore(o){
929
1061
  }
930
1062
  results = core.upgradeResults(results);
931
1063
 
932
- // The cases the run's items are scored against. A file the set does
933
- // not grade keeps its reply and has no score -- honestly: nothing else would
934
- // say what the set covers.
935
- const graded = gradedBy(dataset);
1064
+ // The one group a single-group run names, for an eval type of its own that
1065
+ // reads a case. A file no group grades keeps its reply and has no score --
1066
+ // honestly: nothing else would say what the groups cover.
1067
+ const graded = gradedBy(gctx.primary());
936
1068
 
937
1069
  const only = o.only != null ? parseInt(o.only, 10) : null;
938
- // An eval that reads every item (the Metrics) re-reads one with no case
1070
+ // An eval that reads every item (a group) re-reads one with no case
939
1071
  // too, against the production reply the item kept.
940
1072
  const items = await Promise.all(results.map(async (it, i) => {
941
1073
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
942
1074
  const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
943
- const more = { production: it.production ?? null, grader: graderFor(run), group: groupOf(dataset) };
1075
+ const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
944
1076
  return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
945
1077
  (isObject(side) && side.res)
946
1078
  ? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
@@ -949,23 +1081,25 @@ async function runRescore(o){
949
1081
  // A stored run that stopped short re-scores as what it is: incomplete.
950
1082
  const complete = items.every(it => isObject(it) && !it.unrun);
951
1083
  const outcomes = outcomesOf(run, items, complete);
952
- const verdicts = verdictsOf(outcomes, complete, o.minPass);
1084
+ const wholes = wholeRunsOf(run, items, gctx.resolve);
1085
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
953
1086
  const report = {
954
1087
  runner: RUNNER,
955
1088
  ranAt: new Date().toISOString(),
956
- verdict: runVerdict(verdicts, complete),
1089
+ verdict: runVerdict(verdicts, complete, run.pass),
957
1090
  ...(o.minPass != null ? { minPass: o.minPass } : {}),
958
1091
  rescored: true,
959
1092
  run: {
960
1093
  name: run.name ?? null,
961
1094
  evals: run.evals,
962
1095
  verdicts,
1096
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
963
1097
  scenarios: scenariosOf(run),
964
1098
  items,
965
1099
  },
966
1100
  };
967
1101
 
968
- writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
1102
+ writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
969
1103
  process.exitCode = exitOf(report.verdict);
970
1104
  }
971
1105
 
@@ -1066,9 +1200,12 @@ async function main() {
1066
1200
  }
1067
1201
 
1068
1202
  // The dataset the run is graded against. Everything below grades, so there
1069
- // is no run without one.
1203
+ // is no run without one. --groups hands a --run the bodies it kept; a
1204
+ // --pipeline or --prompt grades one scenario against one, named by --dataset.
1205
+ if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
1070
1206
  if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
1071
1207
  const dataset = readDataset(o.dataset);
1208
+ const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
1072
1209
 
1073
1210
  // The run to grade. A --pipeline document is one scenario graded against
1074
1211
  // its dataset; without one it is the stage this script always ran -- the
@@ -1078,7 +1215,7 @@ async function main() {
1078
1215
  broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
1079
1216
  + "the lab's Prompt library does.");
1080
1217
  }
1081
- const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
1218
+ const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
1082
1219
  if (o.pipeline) {
1083
1220
  if (core.targetsOf(run).length !== 1) {
1084
1221
  broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `