evals-lab 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -6,15 +6,19 @@
6
6
  // what it did twice: a short human report, and a JSON one with every verdict
7
7
  // in it. The exit code is the third statement and the coarsest:
8
8
  //
9
- // 0 every graded case passed
10
- // 1 the set ran and at least one case failed
11
- // 2 the set did not run in full -- nothing is graded, or an image or a
12
- // recorded reply was missing. NOT a pass. An eval set that scores well
13
- // because it never ran is the failure this whole ticket is about.
9
+ // 0 every eval passed
10
+ // 1 the run ran and at least one eval failed
11
+ // 2 the run did not run in full -- nothing is graded, an image or a
12
+ // recorded reply was missing, or it was cancelled. NOT a pass. An eval
13
+ // set that scores well because it never ran is the failure this whole
14
+ // ticket is about.
14
15
  // 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
15
16
  //
16
- // The gating policy -- which of those a pull request may merge on -- is #250's
17
- // to decide, and it has the JSON to decide it from. This only reports.
17
+ // The same in every mode, --run and --rescore included, so CI can gate on it
18
+ // (#251): each eval is held to its own rule -- every item it reads passes, or
19
+ // a whole-run eval's verdict -- and --min-pass relaxes the per-item rule to a
20
+ // share. The report's `verdict` (pass, fail, incomplete) says the same as the
21
+ // code, and --junit writes it for a CI test panel.
18
22
  //
19
23
  // Ollama is firewalled to a handful of hosts, so a live run has to happen on
20
24
  // one of them. That is a fact about the network and not something a flag here
@@ -64,7 +68,10 @@ const { execFileSync } = require("child_process");
64
68
  const core = require("./evals-core.mjs");
65
69
 
66
70
  const HERE = __dirname;
67
- const ROOT = path.join(HERE, "..", "..");
71
+ const ROOT = HERE;
72
+
73
+ // What a report names as its runner: this file, from the lab's root.
74
+ const RUNNER = "run-evals.js";
68
75
 
69
76
  // Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
70
77
  // into 448px tiles and the tile COUNT is what costs. The same budget the tab
@@ -82,11 +89,12 @@ const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
82
89
  // single-flight queue for good; `--timeout` can shorten it, never lengthen.
83
90
  const REQUEST_CAP = 600;
84
91
 
85
- const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
92
+ const USAGE = `Usage: node run-evals.js [options]
86
93
 
87
94
  --model <id> the model to grade. Required for a live run.
88
95
  --url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
89
- Ollama on this machine. A key comes from $EVAL_API_KEY, never argv.
96
+ Ollama on this machine. A key comes from
97
+ $EVALSLAB_API_KEY, never argv.
90
98
  --prompt <text> the prompt to grade, as a template: its tokens resolve
91
99
  under --tokens. Required without --pipeline: a dataset
92
100
  holds no prompt (the lab's Prompt library does).
@@ -103,7 +111,8 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
103
111
  --pipeline <file> a run document (docs/pipeline-model.md) with one
104
112
  scenario and a graded eval, graded case by case against
105
113
  the set. Instead of --prompt, --model and --url. Each
106
- profile's key comes from $EVAL_API_KEY_<ID>, never the
114
+ profile's key comes from $EVALSLAB_API_KEY_<SLUG> -- its
115
+ id's, in a file whose profile has no slug -- never the
107
116
  file; its content.files, when not empty, is the file
108
117
  list a case has to be in, as --files says.
109
118
  --samples <dir> where the images are. Default: the lab's samples/.
@@ -153,6 +162,12 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
153
162
  alone. No model, no files, no network.
154
163
  --json <path|-> write the machine-readable report. "-" means stdout, and
155
164
  sends the human report to stderr.
165
+ --min-pass <0..1> an eval whose items are read one by one passes when at
166
+ least this share of them pass. It relaxes each eval's
167
+ own rule -- every item passes -- and never tightens it;
168
+ a whole-run eval keeps its own verdict.
169
+ --junit <file> write JUnit XML: one testsuite per target and eval, one
170
+ testcase per item, a failure carrying its reason.
156
171
  `;
157
172
 
158
173
  // NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
@@ -192,7 +207,7 @@ function parseArgs(argv) {
192
207
  replies: "", json: "", source: "", files: "", item: "", resume: "",
193
208
  timeout: null, progress: false, cancelFile: "",
194
209
  run: "", progressFile: "", resultsFile: "", from: null, only: null,
195
- rescore: false, tokens: "", plugins: "" };
210
+ rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
196
211
  for (let i = 0; i < argv.length; i++) {
197
212
  const a = argv[i];
198
213
  const value = () => {
@@ -225,6 +240,13 @@ function parseArgs(argv) {
225
240
  case "--only": o.only = value(); break;
226
241
  case "--rescore": o.rescore = true; break;
227
242
  case "--json": o.json = value(); break;
243
+ case "--min-pass": {
244
+ const v = value(), n = Number(v);
245
+ if (v.trim() === "" || !(n >= 0 && n <= 1)) broken(`--min-pass is a share from 0 to 1, and ${v} is not one`);
246
+ o.minPass = n;
247
+ break;
248
+ }
249
+ case "--junit": o.junit = value(); break;
228
250
  case "-h": case "--help":
229
251
  process.stdout.write(USAGE);
230
252
  throw new Stop("", EXIT.passed);
@@ -275,7 +297,7 @@ function readRun(file, dataset) {
275
297
  broken(`${file}: ${e.message}`);
276
298
  }
277
299
  const named = isObject(doc) ? core.evalsDataset(doc) : null;
278
- const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
300
+ const bad = core.validatePipeline(doc, { groups: dataset && named ? [named] : [] });
279
301
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
280
302
  return doc;
281
303
  }
@@ -300,6 +322,11 @@ function readDataset(file) {
300
322
  broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
301
323
  }
302
324
  doc = isObject(doc.dataset) ? doc.dataset.body : null;
325
+ } else if (isObject(doc) && doc.format === "evals-lab/eval-group") {
326
+ // The lab's Export of an eval group (docs/pipeline-model.md §17), which
327
+ // began at version 7.
328
+ if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
329
+ doc = isObject(doc.group) ? doc.group.body : null;
303
330
  }
304
331
  if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
305
332
  broken(`${file}: rules has to be null or { "rules": [...] }`);
@@ -315,6 +342,10 @@ function readDataset(file) {
315
342
  return doc;
316
343
  }
317
344
 
345
+ // A link's eval group, by reference: the body --dataset handed over, the one
346
+ // Library group a run reads, whatever id it names it by.
347
+ const groupOf = (dataset) => () => dataset ?? undefined;
348
+
318
349
  // The cases a graded run is scored against, by the item each names: exactly,
319
350
  // as the Source names it.
320
351
  function gradedBy(dataset) {
@@ -505,7 +536,7 @@ function graderFor(run) {
505
536
  const c = run.profiles?.[ref.id];
506
537
  if (!c) return undefined;
507
538
  if (!made.has(ref.id)) {
508
- const link = reach(c, core.keyVar(ref.id));
539
+ const link = reach(c, core.keyVar(ref.id, c.slug));
509
540
  made.set(ref.id, async prompt => {
510
541
  const r = await ask(link, { id: ref.id, ...c }, prompt, null);
511
542
  if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
@@ -613,18 +644,88 @@ function readUtf8(file) {
613
644
  return fs.readFileSync(file).toString("utf8");
614
645
  }
615
646
 
616
- // Every eval's reading of each scenario, keyed by the eval's id, exactly as
617
- // the page reads it: a whole-run eval's verdict, settled once every item is
618
- // in, and a per-item eval's counts. [settled] is false for a run that
619
- // stopped short, whose whole-run evals have not settled.
620
- const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
621
- Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
622
- name: o.label, skipped: o.skipped,
647
+ // Every eval's reading of each scenario, exactly as the page reads it.
648
+ // [settled] is false for a run that stopped short, whose whole-run evals
649
+ // have not settled.
650
+ const outcomesOf = (run, items, settled) =>
651
+ core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
652
+
653
+ /**
654
+ * One eval's verdict on one target, for the build: its own rule -- every
655
+ * item it read passed, or a whole-run eval's verdict -- with [minPass] the
656
+ * share of items that is enough instead. It only relaxes: an eval whose
657
+ * every item passed passes whatever the share. One an earlier failure
658
+ * stopped is skipped -- that failure is the run's verdict already -- and one
659
+ * that had nothing to read, a dataset grading none of the run's items, says
660
+ * none: the run ran in full, which is what the server's status reads from
661
+ * the exit code, so it is not incomplete either.
662
+ */
663
+ function evalVerdict(o, settled, minPass) {
664
+ if (o.skipped) return "skipped";
665
+ if (!settled) return "incomplete";
666
+ if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
667
+ if (!o.ran) return o.skippedItems ? "skipped" : "none";
668
+ if (o.passed === o.ran) return "pass";
669
+ return minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
670
+ }
671
+
672
+ // Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
673
+ // a per-item eval's counts, and the build's verdict on each.
674
+ const verdictsOf = (outcomes, settled, minPass) => outcomes.map(evals =>
675
+ Object.fromEntries(evals.map(o => [o.id, {
676
+ name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass),
623
677
  ...(o.whole
624
678
  ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
625
679
  : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
626
680
  }])));
627
681
 
682
+ /** The run's verdict from its evals': a run that did not finish is
683
+ incomplete, whatever its evals read so far; then any eval failing fails it. */
684
+ function runVerdict(verdicts, complete) {
685
+ if (!complete) return "incomplete";
686
+ return verdicts.some(v => Object.values(v).some(e => e.verdict === "fail")) ? "fail" : "pass";
687
+ }
688
+
689
+ const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
690
+
691
+ // ---- JUnit -----------------------------------------------------------------
692
+ // What a CI test panel reads, written by the core's junitXml.
693
+
694
+ /** Why a score failed, in a line: its failing metrics, else what it missed. */
695
+ function reasonOf(score) {
696
+ const off = (score.metrics || []).filter(m => !m.pass).map(m => `${m.label}: ${m.reason}`);
697
+ if (off.length) return off.join(" · ");
698
+ const missed = (score.missed || []).map(r => (Array.isArray(r) ? r.join(" or ") : r));
699
+ return missed.length ? `missed ${missed.join(", ")}` : "failed";
700
+ }
701
+
702
+ /** A --run's or a --rescore's suites: each target's evals over [items]. */
703
+ function runSuites(run, outcomes, items, settled, minPass) {
704
+ return outcomes.flatMap((evals, i) => evals.map(o => {
705
+ const name = `${core.targetLabel(run, i)} › ${o.label}`;
706
+ const v = evalVerdict(o, settled, minPass);
707
+ if (o.whole) {
708
+ // Settled over the run rather than item by item: one case, the run.
709
+ const c = { name: o.label };
710
+ if (v === "skipped") c.skipped = "an earlier eval failed";
711
+ else if (v === "none") c.skipped = "nothing to read";
712
+ else if (v === "incomplete") c.error = "the run did not finish";
713
+ else if (v === "fail") c.failure = o.verdict.detail || "failed";
714
+ return { name, cases: [c] };
715
+ }
716
+ return { name, cases: items.map((it, x) => {
717
+ const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
718
+ const read = o.items[x];
719
+ if (!it) c.error = "not run";
720
+ else if (it.unrun) c.error = it.unrun;
721
+ else if (!read) c.skipped = "not graded";
722
+ else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
723
+ else if (!read.pass) c.failure = reasonOf(read);
724
+ return c;
725
+ }) };
726
+ }));
727
+ }
728
+
628
729
  // The run as the report restates it: each scenario's stages with the
629
730
  // connection each one asked.
630
731
  const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
@@ -678,7 +779,7 @@ async function runSnapshot(o) {
678
779
  // profile's id.
679
780
  const plans = core.targetsOf(run).map((_, i) => {
680
781
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
681
- const links = connections.map(c => reach(c, core.keyVar(c.id)));
782
+ const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
682
783
  return { stages, tokens, connections, links, calls, cells };
683
784
  });
684
785
 
@@ -728,6 +829,12 @@ async function runSnapshot(o) {
728
829
  break;
729
830
  }
730
831
  }
832
+ // A text file the Source does not hold is unrun, as an image is: the
833
+ // run is incomplete rather than broken.
834
+ if (item.kind === "text" && !item.bare && item.text == null && !fs.existsSync(path.join(o.source, item.name))) {
835
+ failed = `the file ${item.name} is not in the source`;
836
+ break;
837
+ }
731
838
  const textOf = item.kind === "text" && !item.bare
732
839
  ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
733
840
  const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
@@ -736,7 +843,7 @@ async function runSnapshot(o) {
736
843
  // text item a dataset grades is graded like an image.
737
844
  const kase = item.name ? graded.get(item.name) ?? null : null;
738
845
  production = core.productionOf(run, record);
739
- sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run) }) });
846
+ sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run), group: groupOf(dataset) }) });
740
847
  }
741
848
  // Production's reply to the item, where its record holds one: what a
742
849
  // metric compares with, kept so a re-score reads it again.
@@ -749,16 +856,25 @@ async function runSnapshot(o) {
749
856
  }
750
857
  if (progressFd != null) fs.closeSync(progressFd);
751
858
 
859
+ // Every item in, none of them unrun: only then has the run settled. A
860
+ // re-run of one item reads the rest from the results it was handed.
861
+ const ranItems = results.slice(0, total);
862
+ const complete = !cancelled && !unrun.length
863
+ && ranItems.length === total && ranItems.every(it => it && !it.unrun);
864
+ const outcomes = outcomesOf(run, ranItems, complete);
865
+ const verdicts = verdictsOf(outcomes, complete, o.minPass);
752
866
  const report = {
753
- runner: "tools/prompt-lab/run-evals.js",
867
+ runner: RUNNER,
754
868
  ranAt: new Date().toISOString(),
755
- verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
869
+ verdict: runVerdict(verdicts, complete),
870
+ ...(cancelled ? { cancelled: true } : {}),
871
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
756
872
  run: {
757
873
  name: run.name ?? null,
758
874
  evals: run.evals,
759
- verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
875
+ verdicts,
760
876
  scenarios: scenariosOf(run),
761
- items: results.slice(0, total),
877
+ items: ranItems,
762
878
  },
763
879
  unrun,
764
880
  };
@@ -767,14 +883,31 @@ async function runSnapshot(o) {
767
883
  say(`${report.runner} — run ${report.run.name ?? ""}`);
768
884
  for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
769
885
  if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
770
- say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled" : report.verdict}`);
886
+ sayVerdicts(say, run, verdicts);
887
+ say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
771
888
 
889
+ writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
890
+ process.exitCode = exitOf(report.verdict);
891
+ }
892
+
893
+ /** Each target's evals and their verdicts, a line each. */
894
+ function sayVerdicts(say, run, verdicts) {
895
+ verdicts.forEach((evals, i) => {
896
+ for (const e of Object.values(evals)) {
897
+ const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
898
+ say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}`);
899
+ }
900
+ });
901
+ }
902
+
903
+ /** The JSON report, and the JUnit one from [suites], where each was asked for. */
904
+ function writeReports(o, report, suites) {
772
905
  if (o.json) {
773
- const text2 = JSON.stringify(report, null, 2);
774
- if (o.json === "-") process.stdout.write(text2 + "\n");
775
- else fs.writeFileSync(o.json, text2 + "\n");
906
+ const text = JSON.stringify(report, null, 2);
907
+ if (o.json === "-") process.stdout.write(text + "\n");
908
+ else fs.writeFileSync(o.json, text + "\n");
776
909
  }
777
- process.exitCode = cancelled ? EXIT.passed : (unrun.length ? EXIT.incomplete : EXIT.passed);
910
+ if (o.junit) fs.writeFileSync(o.junit, core.junitXml(suites(), RUNNER));
778
911
  }
779
912
 
780
913
  /** Re-score a run's stored results against the dataset --dataset hands over,
@@ -807,39 +940,40 @@ async function runRescore(o){
807
940
  const items = await Promise.all(results.map(async (it, i) => {
808
941
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
809
942
  const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
810
- const more = { production: it.production ?? null, grader: graderFor(run) };
943
+ const more = { production: it.production ?? null, grader: graderFor(run), group: groupOf(dataset) };
811
944
  return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
812
945
  (isObject(side) && side.res)
813
946
  ? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
814
947
  }));
815
948
 
949
+ // A stored run that stopped short re-scores as what it is: incomplete.
950
+ const complete = items.every(it => isObject(it) && !it.unrun);
951
+ const outcomes = outcomesOf(run, items, complete);
952
+ const verdicts = verdictsOf(outcomes, complete, o.minPass);
816
953
  const report = {
817
- runner: "tools/prompt-lab/run-evals.js",
954
+ runner: RUNNER,
818
955
  ranAt: new Date().toISOString(),
819
- verdict: "done",
956
+ verdict: runVerdict(verdicts, complete),
957
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
820
958
  rescored: true,
821
959
  run: {
822
960
  name: run.name ?? null,
823
961
  evals: run.evals,
824
- verdicts: verdictsOf(run, items, true),
962
+ verdicts,
825
963
  scenarios: scenariosOf(run),
826
964
  items,
827
965
  },
828
966
  };
829
967
 
830
- if (o.json) {
831
- const text = JSON.stringify(report, null, 2);
832
- if (o.json === "-") process.stdout.write(text + "\n");
833
- else fs.writeFileSync(o.json, text + "\n");
834
- }
835
- process.exitCode = EXIT.passed;
968
+ writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
969
+ process.exitCode = exitOf(report.verdict);
836
970
  }
837
971
 
838
972
  /**
839
973
  * The run --prompt, --model and --url describe, as a run document: one job
840
974
  * that sees the image and answers in the default kind, one scenario,
841
975
  * graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
842
- * connection, as it never did, and its key is $EVAL_API_KEY.
976
+ * connection, as it never did, and its key is $EVALSLAB_API_KEY.
843
977
  */
844
978
  function cliRun(o, dataset) {
845
979
  let doc = core.blankPipeline();
@@ -857,8 +991,7 @@ function cliRun(o, dataset) {
857
991
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
858
992
  doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
859
993
  steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
860
- doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
861
- mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
994
+ doc.evals = [{ id: "cli", type: "group", name: "", continueOnFailure: true, group: { id: "cli", name: "" }, pin: null }];
862
995
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
863
996
  return doc;
864
997
  }
@@ -939,7 +1072,7 @@ async function main() {
939
1072
 
940
1073
  // The run to grade. A --pipeline document is one scenario graded against
941
1074
  // its dataset; without one it is the stage this script always ran -- the
942
- // prompt, seeing the image, on --url with $EVAL_API_KEY -- stated as
1075
+ // prompt, seeing the image, on --url with $EVALSLAB_API_KEY -- stated as
943
1076
  // the same kind of document, so both go through one reading of it.
944
1077
  if (!o.pipeline && !(o.prompt ?? "").trim()) {
945
1078
  broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
@@ -1038,7 +1171,7 @@ async function main() {
1038
1171
  if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
1039
1172
  links = connections.map((c, i) => {
1040
1173
  if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
1041
- const to = reach(c, o.pipeline ? core.keyVar(c.id) : "EVAL_API_KEY");
1174
+ const to = reach(c, o.pipeline ? core.keyVar(c.id, c.slug) : "EVALSLAB_API_KEY");
1042
1175
  // Once here rather than once per image: a llama.cpp stage on a
1043
1176
  // hosted model is a request that cannot be built at all.
1044
1177
  try {
@@ -1186,10 +1319,12 @@ async function main() {
1186
1319
  // passed, and 49 items absent with the fiftieth green is exactly how
1187
1320
  // that happens.
1188
1321
  const complete = graded.length > 0 && unrun.length === 0;
1189
- const verdict = !complete ? "incomplete" : tally.passed === tally.ran ? "pass" : "fail";
1322
+ const enough = tally.passed === tally.ran
1323
+ || (o.minPass != null && tally.passed / tally.ran >= o.minPass);
1324
+ const verdict = !complete ? "incomplete" : enough ? "pass" : "fail";
1190
1325
 
1191
1326
  const report = {
1192
- runner: "tools/prompt-lab/run-evals.js",
1327
+ runner: RUNNER,
1193
1328
  ranAt: new Date().toISOString(),
1194
1329
  dataset: inRepo(o.dataset),
1195
1330
  prompt,
@@ -1217,6 +1352,7 @@ async function main() {
1217
1352
  filesListed: snapshotSet.size } : {}),
1218
1353
  ...(o.source ? { source: inRepo(o.source) } : {}),
1219
1354
  verdict,
1355
+ ...(o.minPass != null ? { minPass: o.minPass } : {}),
1220
1356
  graded: (itemsOnly ?? graded).length,
1221
1357
  ungraded: set.length - graded.length,
1222
1358
  ran: tally.ran,
@@ -1269,14 +1405,16 @@ async function main() {
1269
1405
  + `so this is a partial run and not a result.`);
1270
1406
  }
1271
1407
 
1272
- if (o.json) {
1273
- const text = JSON.stringify(report, null, 2);
1274
- if (toStdout) process.stdout.write(text + "\n");
1275
- else fs.writeFileSync(o.json, text + "\n");
1276
- }
1277
-
1278
- process.exitCode = !complete ? EXIT.incomplete
1279
- : report.failed ? EXIT.failed : EXIT.passed;
1408
+ // The one eval this mode grades by, as a suite of its cases.
1409
+ const suites = () => {
1410
+ const j = Math.max(0, core.evalsOf(run).findIndex(t => core.evalsDataset({ evals: [t] })));
1411
+ return [{ name: `${core.targetLabel(run, 0)} › ${core.evalLabel(run, j)}`, cases: [
1412
+ ...rows.map(r => ({ name: r.id, ...(r.pass ? {} : { failure: r.reasons.join(" · ") || "failed" }) })),
1413
+ ...unrun.map(u => { const at = u.indexOf(": "); return { name: u.slice(0, at), error: u.slice(at + 2) }; }),
1414
+ ] }];
1415
+ };
1416
+ writeReports(o, report, suites);
1417
+ process.exitCode = exitOf(verdict);
1280
1418
  }
1281
1419
 
1282
1420
  main().catch(e => {