evals-lab 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -116,13 +116,17 @@ const USAGE = `Usage: node run-evals.js [options]
116
116
  file; its content.files, when not empty, is the file
117
117
  list a case has to be in, as --files says.
118
118
  --samples <dir> where the images are. Default: the lab's samples/.
119
- --dataset <file> the dataset a graded run is scored against: its body
120
- ({"cases"}), or the file the lab's Export writes.
119
+ --dataset <file> the one eval group a graded run is scored against: its
120
+ body ({"cases"}), or the file the lab's Export writes.
121
121
  Versions 1 to 3 are read too (a prompt they hold is
122
122
  not used: --prompt names it); a version-2
123
123
  body's rules are what an older run's jobs, and a run
124
124
  without --pipeline, read replies under. Its cases grade.
125
- Required for anything graded.
125
+ Required for anything graded (or --groups, for a --run).
126
+ --groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
127
+ grades against each one its evals link (§17), so a run
128
+ linking several groups is handed them here instead of the
129
+ one --dataset. For --run and --rescore; not with --dataset.
126
130
  --source <dir> the same, named the way a run names it: the directory a
127
131
  Source is stored at. --samples and --source are one
128
132
  flag by two names; both together are refused.
@@ -204,7 +208,7 @@ function broken(msg) {
204
208
 
205
209
  function parseArgs(argv) {
206
210
  const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
207
- replies: "", json: "", source: "", files: "", item: "", resume: "",
211
+ groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
208
212
  timeout: null, progress: false, cancelFile: "",
209
213
  run: "", progressFile: "", resultsFile: "", from: null, only: null,
210
214
  rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
@@ -225,6 +229,7 @@ function parseArgs(argv) {
225
229
  case "--pipeline": o.pipeline = value(); break;
226
230
  case "--samples": o.samples = value(); break;
227
231
  case "--dataset": o.dataset = value(); break;
232
+ case "--groups": o.groups = value(); break;
228
233
  case "--replies": o.replies = value(); break;
229
234
  case "--source": o.source = value(); break;
230
235
  case "--files": o.files = value(); break;
@@ -281,10 +286,10 @@ const inRepo = p => {
281
286
  const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
282
287
  const isText = v => typeof v === "string";
283
288
 
284
- // [dataset] is the body --dataset handed over, or null: the one dataset a
285
- // graded run can name, whatever id it names it by, since the queue hands the
286
- // worker the body of exactly the dataset the run was submitted against.
287
- function readRun(file, dataset) {
289
+ // [gctx] is the grading context (gradingContext): the eval group bodies a run
290
+ // grades against -- one, as --dataset, or a body per group kept with the run,
291
+ // as --groups (§17). The run reads each link's body by its reference.
292
+ function readRun(file, gctx) {
288
293
  let doc;
289
294
  try {
290
295
  // A run queued before the current version is read as one of today's:
@@ -292,12 +297,20 @@ function readRun(file, dataset) {
292
297
  // submitted with. A v2 run's list jobs read their replies under the
293
298
  // rules of the dataset it was graded against -- the body handed over.
294
299
  doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
295
- { rulesFor: () => rulesOf(dataset) });
300
+ { rulesFor: () => rulesOf(gctx.primary()) });
296
301
  } catch (e) {
297
302
  broken(`${file}: ${e.message}`);
298
303
  }
299
- const named = isObject(doc) ? core.evalsDataset(doc) : null;
300
- const bad = core.validatePipeline(doc, { groups: dataset && named ? [named] : [] });
304
+ // Every Library group the evals name -- a link's group, a private group's
305
+ // Cases from -- each as validatePipeline reads one. Pins were resolved at
306
+ // submit (the reference carries `n`), so they are not checked again here.
307
+ const groups = [], seen = new Set();
308
+ for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
309
+ if (seen.has(ref.id)) continue;
310
+ seen.add(ref.id);
311
+ groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
312
+ }
313
+ const bad = core.validatePipeline(doc, { groups });
301
314
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
302
315
  return doc;
303
316
  }
@@ -305,9 +318,25 @@ function readRun(file, dataset) {
305
318
  // The rules a version-2 body held, by the body readDataset made of it: what
306
319
  // an older run's jobs read replies under. A body of today's holds none.
307
320
  const heldRules = new WeakMap();
308
- const rulesOf = (dataset) => (dataset && heldRules.get(dataset)) ?? null;
321
+ const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
322
+
323
+ // An eval group's body as one of today's, with its version-2 rules taken
324
+ // first: a bare body, or the body the lab's Export wraps, unwrapped by its
325
+ // caller. [where] names it in a refusal.
326
+ function bodyFrom(doc, where) {
327
+ if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
328
+ broken(`${where}: rules has to be null or { "rules": [...] }`);
329
+ }
330
+ const rules = core.datasetRules(doc);
331
+ const body = core.upgradeDatasetBody(doc);
332
+ if (!isObject(body) || !Array.isArray(body.cases)) {
333
+ broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
334
+ }
335
+ if (rules) heldRules.set(body, rules);
336
+ return body;
337
+ }
309
338
 
310
- // The dataset --dataset names: a dataset's body, or the file the lab's
339
+ // The dataset --dataset names: an eval group's body, or the file the lab's
311
340
  // Export writes for one, as one of today's.
312
341
  function readDataset(file) {
313
342
  if (!file) return null;
@@ -328,29 +357,51 @@ function readDataset(file) {
328
357
  if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
329
358
  doc = isObject(doc.group) ? doc.group.body : null;
330
359
  }
331
- if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
332
- broken(`${file}: rules has to be null or { "rules": [...] }`);
333
- }
334
- // A body from an earlier version -- a run queued then, resumed now -- reads
335
- // as one of today's, its rules taken first.
336
- const rules = core.datasetRules(doc);
337
- doc = core.upgradeDatasetBody(doc);
338
- if (!isObject(doc) || !Array.isArray(doc.cases)) {
339
- broken(`${file} is not a dataset: its body has a cases list, or it is the lab's Export of one`);
360
+ return bodyFrom(doc, file);
361
+ }
362
+
363
+ // The bodies a run kept, as --groups hands them (§17): a JSON object of
364
+ // `<id>@<n>` to body -- the group's id and the version it graded with, its id
365
+ // alone for a run from before the lab numbered them -- each read as today's.
366
+ function readGroups(file) {
367
+ let raw;
368
+ try {
369
+ raw = JSON.parse(fs.readFileSync(file, "utf8"));
370
+ } catch (e) {
371
+ broken(`${file}: ${e.message}`);
340
372
  }
341
- if (rules) heldRules.set(doc, rules);
342
- return doc;
373
+ if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
374
+ const bodies = new Map();
375
+ for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
376
+ return bodies;
343
377
  }
344
378
 
345
- // A link's eval group, by reference: the body --dataset handed over, the one
346
- // Library group a run reads, whatever id it names it by.
347
- const groupOf = (dataset) => () => dataset ?? undefined;
379
+ // The key a run keeps a group's body under, mirroring the server's group_key:
380
+ // `<id>@<n>`, or the id alone for a reference from before versions.
381
+ const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
382
+
383
+ // The eval groups a --run grades against: one body, as --dataset, or a body
384
+ // per group, as --groups -- never both. `resolve` gives a link's body by its
385
+ // reference; `primary` the one a single-group run names, for a v2 run's rules.
386
+ function gradingContext(o) {
387
+ if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
388
+ if (o.groups) {
389
+ const bodies = readGroups(o.groups);
390
+ const first = bodies.values().next().value ?? null;
391
+ return { single: null, bodies, primary: () => first,
392
+ resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
393
+ }
394
+ const single = readDataset(o.dataset);
395
+ return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
396
+ }
348
397
 
349
398
  // The cases a graded run is scored against, by the item each names: exactly,
350
- // as the Source names it.
351
- function gradedBy(dataset) {
399
+ // as the Source names it. Built from the one group a single-group run names,
400
+ // for an eval type of its own that reads a case (a plugin's); a `group` eval
401
+ // reads each group's own case, by the item's name, as it grades.
402
+ function gradedBy(body) {
352
403
  const out = new Map();
353
- for (const c of core.gradedSetFrom(dataset || {})) {
404
+ for (const c of core.gradedSetFrom(body || {})) {
354
405
  if (!c.todo) out.set(c.item, c);
355
406
  }
356
407
  return out;
@@ -650,40 +701,81 @@ function readUtf8(file) {
650
701
  const outcomesOf = (run, items, settled) =>
651
702
  core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
652
703
 
704
+ // A link's Whole run verdict, per Target, keyed by the eval's id: a Library
705
+ // group's `run` metrics read over the replies so far (readGroupRun). A private
706
+ // group's Whole run is scenarioEvals' own (it holds the body); a link's body
707
+ // is held apart, in --groups, so reading it beside the link's items is the
708
+ // runner's (§17). [resolve] gives a link's body by its reference.
709
+ function wholeRunsOf(run, items, resolve) {
710
+ const kind = core.lastKind(run);
711
+ return core.targetsOf(run).map((_, i) => {
712
+ const out = {};
713
+ const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
714
+ for (const t of core.evalsOf(run)) {
715
+ // A private whole-run group is already scenarioEvals' verdict; here is
716
+ // the run beside a link's (or private group's) own items.
717
+ if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
718
+ const body = core.groupOf(t, { group: resolve });
719
+ if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
720
+ }
721
+ return out;
722
+ });
723
+ }
724
+
653
725
  /**
654
726
  * One eval's verdict on one target, for the build: its own rule -- every
655
727
  * item it read passed, or a whole-run eval's verdict -- with [minPass] the
656
728
  * share of items that is enough instead. It only relaxes: an eval whose
657
729
  * every item passed passes whatever the share. One an earlier failure
658
730
  * stopped is skipped -- that failure is the run's verdict already -- and one
659
- * that had nothing to read, a dataset grading none of the run's items, says
731
+ * that had nothing to read, a group grading none of the run's items, says
660
732
  * none: the run ran in full, which is what the server's status reads from
661
- * the exit code, so it is not incomplete either.
733
+ * the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
734
+ * run verdict, where its group reads one beside its items (§17): the group
735
+ * passes when both its items and its Whole run do, and either failing fails it.
662
736
  */
663
- function evalVerdict(o, settled, minPass) {
737
+ function evalVerdict(o, settled, minPass, wholeRun) {
664
738
  if (o.skipped) return "skipped";
665
739
  if (!settled) return "incomplete";
666
740
  if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
667
- if (!o.ran) return o.skippedItems ? "skipped" : "none";
668
- if (o.passed === o.ran) return "pass";
669
- return minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
741
+ const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
742
+ : o.passed === o.ran ? "pass"
743
+ : minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
744
+ if (!wholeRun) return item;
745
+ const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
746
+ if (item === "fail" || whole === "fail") return "fail";
747
+ if (item === "pass" || whole === "pass") return "pass";
748
+ return item === "skipped" || whole === "skipped" ? "skipped" : "none";
670
749
  }
671
750
 
672
- // Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
673
- // a per-item eval's counts, and the build's verdict on each.
674
- const verdictsOf = (outcomes, settled, minPass) => outcomes.map(evals =>
675
- Object.fromEntries(evals.map(o => [o.id, {
676
- name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass),
677
- ...(o.whole
678
- ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
679
- : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
680
- }])));
681
-
682
- /** The run's verdict from its evals': a run that did not finish is
683
- incomplete, whatever its evals read so far; then any eval failing fails it. */
684
- function runVerdict(verdicts, complete) {
751
+ // Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
752
+ // per-item eval's counts -- and, where a link reads a Whole run beside its
753
+ // items, that verdict too -- with the build's verdict on each. [wholes] is
754
+ // wholeRunsOf, one map per target.
755
+ const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
756
+ Object.fromEntries(evals.map(o => {
757
+ const wr = wholes[i] && wholes[i][o.id];
758
+ return [o.id, {
759
+ name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
760
+ ...(o.whole
761
+ ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
762
+ : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
763
+ ...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
764
+ }];
765
+ })));
766
+
767
+ /** Each Target's verdict under the overall pass rule (§17), from its groups'
768
+ verdicts: a group that passed counts, one that failed counts against, and
769
+ one that read nothing or was skipped is neither. */
770
+ const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
771
+ Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
772
+
773
+ /** The run's verdict from its evals' and the overall rule: a run that did not
774
+ finish is incomplete, whatever its evals read so far; then it passes when
775
+ every Target does under the rule. */
776
+ function runVerdict(verdicts, complete, pass) {
685
777
  if (!complete) return "incomplete";
686
- return verdicts.some(v => Object.values(v).some(e => e.verdict === "fail")) ? "fail" : "pass";
778
+ return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
687
779
  }
688
780
 
689
781
  const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
@@ -699,11 +791,13 @@ function reasonOf(score) {
699
791
  return missed.length ? `missed ${missed.join(", ")}` : "failed";
700
792
  }
701
793
 
702
- /** A --run's or a --rescore's suites: each target's evals over [items]. */
703
- function runSuites(run, outcomes, items, settled, minPass) {
794
+ /** A --run's or a --rescore's suites: each target's evals over [items], with a
795
+ Whole run case where a link reads one beside its items ([wholes]). */
796
+ function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
704
797
  return outcomes.flatMap((evals, i) => evals.map(o => {
705
798
  const name = `${core.targetLabel(run, i)} › ${o.label}`;
706
- const v = evalVerdict(o, settled, minPass);
799
+ const wr = wholes[i] && wholes[i][o.id];
800
+ const v = evalVerdict(o, settled, minPass, wr);
707
801
  if (o.whole) {
708
802
  // Settled over the run rather than item by item: one case, the run.
709
803
  const c = { name: o.label };
@@ -713,7 +807,7 @@ function runSuites(run, outcomes, items, settled, minPass) {
713
807
  else if (v === "fail") c.failure = o.verdict.detail || "failed";
714
808
  return { name, cases: [c] };
715
809
  }
716
- return { name, cases: items.map((it, x) => {
810
+ const cases = items.map((it, x) => {
717
811
  const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
718
812
  const read = o.items[x];
719
813
  if (!it) c.error = "not run";
@@ -722,7 +816,15 @@ function runSuites(run, outcomes, items, settled, minPass) {
722
816
  else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
723
817
  else if (!read.pass) c.failure = reasonOf(read);
724
818
  return c;
725
- }) };
819
+ });
820
+ if (wr) {
821
+ const c = { name: "Whole run" };
822
+ if (!settled) c.error = "the run did not finish";
823
+ else if (!wr.ran) c.skipped = "nothing to read";
824
+ else if (!wr.pass) c.failure = wr.detail || "failed";
825
+ cases.push(c);
826
+ }
827
+ return { name, cases };
726
828
  }));
727
829
  }
728
830
 
@@ -740,8 +842,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
740
842
  });
741
843
 
742
844
  async function runSnapshot(o) {
743
- const dataset = readDataset(o.dataset);
744
- const run = readRun(o.run, dataset);
845
+ const gctx = gradingContext(o);
846
+ const run = readRun(o.run, gctx);
745
847
  // One item a pipeline holds itself (Text, or Prompt only's bare one), or
746
848
  // the Source's files.
747
849
  const content = core.contentOf(run);
@@ -753,8 +855,9 @@ async function runSnapshot(o) {
753
855
  broken(`${o.run} names no files and no text, so there is nothing to run`);
754
856
  }
755
857
 
756
- // The dataset the run was submitted against, for scoring its items.
757
- const graded = gradedBy(dataset);
858
+ // The one group a single-group run names, for an eval type of its own that
859
+ // reads a case; a `group` eval reads each group's case by the item's name.
860
+ const graded = gradedBy(gctx.primary());
758
861
 
759
862
  const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
760
863
  : text.trim() ? [{ name: null, kind: "text", text }]
@@ -843,7 +946,8 @@ async function runSnapshot(o) {
843
946
  // text item a dataset grades is graded like an image.
844
947
  const kase = item.name ? graded.get(item.name) ?? null : null;
845
948
  production = core.productionOf(run, record);
846
- sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run), group: groupOf(dataset) }) });
949
+ sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
950
+ { production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
847
951
  }
848
952
  // Production's reply to the item, where its record holds one: what a
849
953
  // metric compares with, kept so a re-score reads it again.
@@ -862,17 +966,19 @@ async function runSnapshot(o) {
862
966
  const complete = !cancelled && !unrun.length
863
967
  && ranItems.length === total && ranItems.every(it => it && !it.unrun);
864
968
  const outcomes = outcomesOf(run, ranItems, complete);
865
- const verdicts = verdictsOf(outcomes, complete, o.minPass);
969
+ const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
970
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
866
971
  const report = {
867
972
  runner: RUNNER,
868
973
  ranAt: new Date().toISOString(),
869
- verdict: runVerdict(verdicts, complete),
974
+ verdict: runVerdict(verdicts, complete, run.pass),
870
975
  ...(cancelled ? { cancelled: true } : {}),
871
976
  ...(o.minPass != null ? { minPass: o.minPass } : {}),
872
977
  run: {
873
978
  name: run.name ?? null,
874
979
  evals: run.evals,
875
980
  verdicts,
981
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
876
982
  scenarios: scenariosOf(run),
877
983
  items: ranItems,
878
984
  },
@@ -883,19 +989,25 @@ async function runSnapshot(o) {
883
989
  say(`${report.runner} — run ${report.run.name ?? ""}`);
884
990
  for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
885
991
  if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
886
- sayVerdicts(say, run, verdicts);
992
+ sayVerdicts(say, run, verdicts, report.run.pass);
887
993
  say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
888
994
 
889
- writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
995
+ writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
890
996
  process.exitCode = exitOf(report.verdict);
891
997
  }
892
998
 
893
- /** Each target's evals and their verdicts, a line each. */
894
- function sayVerdicts(say, run, verdicts) {
999
+ /** Each target's groups and their verdicts, a line each, and the overall rule
1000
+ beneath them where [pass] gives a figure per target. */
1001
+ function sayVerdicts(say, run, verdicts, pass) {
895
1002
  verdicts.forEach((evals, i) => {
896
1003
  for (const e of Object.values(evals)) {
897
1004
  const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
898
- say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}`);
1005
+ const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
1006
+ say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
1007
+ }
1008
+ if (pass) {
1009
+ const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
1010
+ say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
899
1011
  }
900
1012
  });
901
1013
  }
@@ -916,8 +1028,8 @@ function writeReports(o, report, suites) {
916
1028
  * a --run's, with the stored results as its items, and a whole-run eval's
917
1029
  * verdict settled the same way. */
918
1030
  async function runRescore(o){
919
- const dataset = readDataset(o.dataset);
920
- const run = readRun(o.run, dataset);
1031
+ const gctx = gradingContext(o);
1032
+ const run = readRun(o.run, gctx);
921
1033
  let results;
922
1034
  try {
923
1035
  results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
@@ -929,18 +1041,18 @@ async function runRescore(o){
929
1041
  }
930
1042
  results = core.upgradeResults(results);
931
1043
 
932
- // The cases the run's items are scored against. A file the set does
933
- // not grade keeps its reply and has no score -- honestly: nothing else would
934
- // say what the set covers.
935
- const graded = gradedBy(dataset);
1044
+ // The one group a single-group run names, for an eval type of its own that
1045
+ // reads a case. A file no group grades keeps its reply and has no score --
1046
+ // honestly: nothing else would say what the groups cover.
1047
+ const graded = gradedBy(gctx.primary());
936
1048
 
937
1049
  const only = o.only != null ? parseInt(o.only, 10) : null;
938
- // An eval that reads every item (the Metrics) re-reads one with no case
1050
+ // An eval that reads every item (a group) re-reads one with no case
939
1051
  // too, against the production reply the item kept.
940
1052
  const items = await Promise.all(results.map(async (it, i) => {
941
1053
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
942
1054
  const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
943
- const more = { production: it.production ?? null, grader: graderFor(run), group: groupOf(dataset) };
1055
+ const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
944
1056
  return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
945
1057
  (isObject(side) && side.res)
946
1058
  ? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
@@ -949,23 +1061,25 @@ async function runRescore(o){
949
1061
  // A stored run that stopped short re-scores as what it is: incomplete.
950
1062
  const complete = items.every(it => isObject(it) && !it.unrun);
951
1063
  const outcomes = outcomesOf(run, items, complete);
952
- const verdicts = verdictsOf(outcomes, complete, o.minPass);
1064
+ const wholes = wholeRunsOf(run, items, gctx.resolve);
1065
+ const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
953
1066
  const report = {
954
1067
  runner: RUNNER,
955
1068
  ranAt: new Date().toISOString(),
956
- verdict: runVerdict(verdicts, complete),
1069
+ verdict: runVerdict(verdicts, complete, run.pass),
957
1070
  ...(o.minPass != null ? { minPass: o.minPass } : {}),
958
1071
  rescored: true,
959
1072
  run: {
960
1073
  name: run.name ?? null,
961
1074
  evals: run.evals,
962
1075
  verdicts,
1076
+ pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
963
1077
  scenarios: scenariosOf(run),
964
1078
  items,
965
1079
  },
966
1080
  };
967
1081
 
968
- writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
1082
+ writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
969
1083
  process.exitCode = exitOf(report.verdict);
970
1084
  }
971
1085
 
@@ -1066,9 +1180,12 @@ async function main() {
1066
1180
  }
1067
1181
 
1068
1182
  // The dataset the run is graded against. Everything below grades, so there
1069
- // is no run without one.
1183
+ // is no run without one. --groups hands a --run the bodies it kept; a
1184
+ // --pipeline or --prompt grades one scenario against one, named by --dataset.
1185
+ if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
1070
1186
  if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
1071
1187
  const dataset = readDataset(o.dataset);
1188
+ const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
1072
1189
 
1073
1190
  // The run to grade. A --pipeline document is one scenario graded against
1074
1191
  // its dataset; without one it is the stage this script always ran -- the
@@ -1078,7 +1195,7 @@ async function main() {
1078
1195
  broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
1079
1196
  + "the lab's Prompt library does.");
1080
1197
  }
1081
- const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
1198
+ const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
1082
1199
  if (o.pipeline) {
1083
1200
  if (core.targetsOf(run).length !== 1) {
1084
1201
  broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
package/lab/server.py CHANGED
@@ -809,7 +809,7 @@ class Sources:
809
809
  out = []
810
810
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
811
811
  db.row_factory = sqlite3.Row
812
- for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
812
+ for r in db.execute("SELECT id, name, system, bytes, type, created_at FROM sources "
813
813
  "ORDER BY system DESC, name COLLATE NOCASE"):
814
814
  if r["system"]:
815
815
  files = self._sample_files()
@@ -817,10 +817,13 @@ class Sources:
817
817
  "type": r["type"],
818
818
  "files": len(files), "bytes": sum(f["bytes"] for f in files)})
819
819
  else:
820
- n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
821
- (r["id"],)).fetchone()[0]
820
+ n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
821
+ (r["id"],)).fetchone()
822
+ # When it last changed: made, or a file added -- what a
823
+ # picker orders its recent Sources by.
822
824
  out.append({"id": r["id"], "name": r["name"], "system": False,
823
- "type": r["type"], "files": n, "bytes": r["bytes"]})
825
+ "type": r["type"], "files": n, "bytes": r["bytes"],
826
+ "changed": max(filter(None, (r["created_at"], last)))})
824
827
  return out
825
828
 
826
829
  def get(self, sid) -> dict:
@@ -1754,7 +1757,7 @@ def scenario_ref(sc, i):
1754
1757
  what version 4's upgrade gives one, by position."""
1755
1758
  sid = sc.get("id") if isinstance(sc, dict) else None
1756
1759
  name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
1757
- return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
1760
+ return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
1758
1761
 
1759
1762
 
1760
1763
  class Prompts:
@@ -3882,6 +3885,27 @@ class Queue:
3882
3885
  (rundir / "dataset.json").write_text(json.dumps(body))
3883
3886
  return ["--dataset", str(rundir / "dataset.json")], None
3884
3887
 
3888
+ def _grading_args(self, run, rundir):
3889
+ """The worker's eval group bodies: --dataset for a run that grades
3890
+ against one group, which keeps its single-body path and old-row
3891
+ pinning, and --groups for a run that links several (#233) -- every body
3892
+ it kept, by `<id>@<n>`, written beside the run document. Returns
3893
+ (args, None) or (None, why)."""
3894
+ refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
3895
+ keys = {group_key(r) for r in refs if r is not None}
3896
+ if len(keys) <= 1:
3897
+ return self._dataset_args(run, rundir)
3898
+ kept = self.groups(run["id"], raw=True)
3899
+ if kept is None:
3900
+ return None, "the run kept no eval group bodies"
3901
+ missing = sorted({(r.get("name") or r.get("id")) for r in refs
3902
+ if r is not None and group_key(r) not in kept})
3903
+ if missing:
3904
+ return None, f"the run kept no body of the eval group {', '.join(missing)}"
3905
+ rundir.mkdir(parents=True, exist_ok=True)
3906
+ (rundir / "groups.json").write_text(json.dumps(kept))
3907
+ return ["--groups", str(rundir / "groups.json")], None
3908
+
3885
3909
  def _behind(self, rid):
3886
3910
  """How many submissions stand between this one and the worker, by
3887
3911
  submit time -- what a waiting form names when it says what it is
@@ -4020,7 +4044,7 @@ class Queue:
4020
4044
  return None, (403, err)
4021
4045
  if NODE is None:
4022
4046
  return None, (500, "node is not installed, so nothing can run")
4023
- dataset, err = self._dataset_args(run, rundir)
4047
+ dataset, err = self._grading_args(run, rundir)
4024
4048
  if err:
4025
4049
  return None, (409, err)
4026
4050
  plugins, err = self._plugin_args(run)
@@ -4086,7 +4110,7 @@ class Queue:
4086
4110
  return None, (500, "node is not installed, so nothing can run")
4087
4111
  rundir = self.dir / run["id"]
4088
4112
  rundir.mkdir(parents=True, exist_ok=True)
4089
- dataset, err = self._dataset_args(run, rundir)
4113
+ dataset, err = self._grading_args(run, rundir)
4090
4114
  if err:
4091
4115
  return None, (409, err)
4092
4116
  plugins, err = self._plugin_args(run)
@@ -4202,7 +4226,7 @@ class Queue:
4202
4226
  return self._finish(rid, "failed", error=err)
4203
4227
  if NODE is None:
4204
4228
  return self._finish(rid, "failed", error="node is not installed, so nothing can run")
4205
- dataset, err = self._dataset_args(run, rundir)
4229
+ dataset, err = self._grading_args(run, rundir)
4206
4230
  if err:
4207
4231
  return self._finish(rid, "failed", error=err)
4208
4232
  plugins, err = self._plugin_args(run)
@@ -5687,10 +5711,8 @@ class Handler(BaseHTTPRequestHandler):
5687
5711
  return self._json(400, {"error": f"the run cannot be queued: {body}"})
5688
5712
  ref["n"], ref["version"] = n, fingerprint(body)
5689
5713
  groups[group_key(ref)] = body
5690
- # The worker is handed one body until it reads one per group (#233).
5691
- if len(groups) > 1:
5692
- return self._json(400, {"error": "the run cannot be queued: its evals grade against "
5693
- "one version of one Library eval group at a time"})
5714
+ # The worker reads a body per group now (#233), so a run may link
5715
+ # several; each body is kept with the row under `<id>@<n>`.
5694
5716
  return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
5695
5717
 
5696
5718
  # ---- Datasets ----------------------------------------------------------