evals-lab 0.4.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/README.md +125 -28
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +531 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +672 -61
- package/lab/run-evals.js +195 -57
- package/lab/server.py +681 -127
- package/lab/web/dist/assets/gallery-BFf9vis6.js +3 -0
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/main-DeeRLWnO.css +1 -0
- package/lab/web/dist/assets/main-LT0U2TYF.js +21 -0
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +59 -0
- package/lab/web/dist/assets/tokens-lq45aAPS.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-BkZTEix2.js +0 -21
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/run-evals.js
CHANGED
|
@@ -6,15 +6,19 @@
|
|
|
6
6
|
// what it did twice: a short human report, and a JSON one with every verdict
|
|
7
7
|
// in it. The exit code is the third statement and the coarsest:
|
|
8
8
|
//
|
|
9
|
-
// 0 every
|
|
10
|
-
// 1 the
|
|
11
|
-
// 2 the
|
|
12
|
-
// recorded reply was missing. NOT a pass. An eval
|
|
13
|
-
// because it never ran is the failure this whole
|
|
9
|
+
// 0 every eval passed
|
|
10
|
+
// 1 the run ran and at least one eval failed
|
|
11
|
+
// 2 the run did not run in full -- nothing is graded, an image or a
|
|
12
|
+
// recorded reply was missing, or it was cancelled. NOT a pass. An eval
|
|
13
|
+
// set that scores well because it never ran is the failure this whole
|
|
14
|
+
// ticket is about.
|
|
14
15
|
// 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
|
|
15
16
|
//
|
|
16
|
-
// The
|
|
17
|
-
//
|
|
17
|
+
// The same in every mode, --run and --rescore included, so CI can gate on it
|
|
18
|
+
// (#251): each eval is held to its own rule -- every item it reads passes, or
|
|
19
|
+
// a whole-run eval's verdict -- and --min-pass relaxes the per-item rule to a
|
|
20
|
+
// share. The report's `verdict` (pass, fail, incomplete) says the same as the
|
|
21
|
+
// code, and --junit writes it for a CI test panel.
|
|
18
22
|
//
|
|
19
23
|
// Ollama is firewalled to a handful of hosts, so a live run has to happen on
|
|
20
24
|
// one of them. That is a fact about the network and not something a flag here
|
|
@@ -64,7 +68,10 @@ const { execFileSync } = require("child_process");
|
|
|
64
68
|
const core = require("./evals-core.mjs");
|
|
65
69
|
|
|
66
70
|
const HERE = __dirname;
|
|
67
|
-
const ROOT =
|
|
71
|
+
const ROOT = HERE;
|
|
72
|
+
|
|
73
|
+
// What a report names as its runner: this file, from the lab's root.
|
|
74
|
+
const RUNNER = "run-evals.js";
|
|
68
75
|
|
|
69
76
|
// Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
|
|
70
77
|
// into 448px tiles and the tile COUNT is what costs. The same budget the tab
|
|
@@ -82,11 +89,12 @@ const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
|
|
|
82
89
|
// single-flight queue for good; `--timeout` can shorten it, never lengthen.
|
|
83
90
|
const REQUEST_CAP = 600;
|
|
84
91
|
|
|
85
|
-
const USAGE = `Usage: node
|
|
92
|
+
const USAGE = `Usage: node run-evals.js [options]
|
|
86
93
|
|
|
87
94
|
--model <id> the model to grade. Required for a live run.
|
|
88
95
|
--url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
|
|
89
|
-
Ollama on this machine. A key comes from
|
|
96
|
+
Ollama on this machine. A key comes from
|
|
97
|
+
$EVALSLAB_API_KEY, never argv.
|
|
90
98
|
--prompt <text> the prompt to grade, as a template: its tokens resolve
|
|
91
99
|
under --tokens. Required without --pipeline: a dataset
|
|
92
100
|
holds no prompt (the lab's Prompt library does).
|
|
@@ -103,7 +111,8 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
103
111
|
--pipeline <file> a run document (docs/pipeline-model.md) with one
|
|
104
112
|
scenario and a graded eval, graded case by case against
|
|
105
113
|
the set. Instead of --prompt, --model and --url. Each
|
|
106
|
-
profile's key comes from $
|
|
114
|
+
profile's key comes from $EVALSLAB_API_KEY_<SLUG> -- its
|
|
115
|
+
id's, in a file whose profile has no slug -- never the
|
|
107
116
|
file; its content.files, when not empty, is the file
|
|
108
117
|
list a case has to be in, as --files says.
|
|
109
118
|
--samples <dir> where the images are. Default: the lab's samples/.
|
|
@@ -153,6 +162,12 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
153
162
|
alone. No model, no files, no network.
|
|
154
163
|
--json <path|-> write the machine-readable report. "-" means stdout, and
|
|
155
164
|
sends the human report to stderr.
|
|
165
|
+
--min-pass <0..1> an eval whose items are read one by one passes when at
|
|
166
|
+
least this share of them pass. It relaxes each eval's
|
|
167
|
+
own rule -- every item passes -- and never tightens it;
|
|
168
|
+
a whole-run eval keeps its own verdict.
|
|
169
|
+
--junit <file> write JUnit XML: one testsuite per target and eval, one
|
|
170
|
+
testcase per item, a failure carrying its reason.
|
|
156
171
|
`;
|
|
157
172
|
|
|
158
173
|
// NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
|
|
@@ -192,7 +207,7 @@ function parseArgs(argv) {
|
|
|
192
207
|
replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
193
208
|
timeout: null, progress: false, cancelFile: "",
|
|
194
209
|
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
195
|
-
rescore: false, tokens: "", plugins: "" };
|
|
210
|
+
rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
|
|
196
211
|
for (let i = 0; i < argv.length; i++) {
|
|
197
212
|
const a = argv[i];
|
|
198
213
|
const value = () => {
|
|
@@ -225,6 +240,13 @@ function parseArgs(argv) {
|
|
|
225
240
|
case "--only": o.only = value(); break;
|
|
226
241
|
case "--rescore": o.rescore = true; break;
|
|
227
242
|
case "--json": o.json = value(); break;
|
|
243
|
+
case "--min-pass": {
|
|
244
|
+
const v = value(), n = Number(v);
|
|
245
|
+
if (v.trim() === "" || !(n >= 0 && n <= 1)) broken(`--min-pass is a share from 0 to 1, and ${v} is not one`);
|
|
246
|
+
o.minPass = n;
|
|
247
|
+
break;
|
|
248
|
+
}
|
|
249
|
+
case "--junit": o.junit = value(); break;
|
|
228
250
|
case "-h": case "--help":
|
|
229
251
|
process.stdout.write(USAGE);
|
|
230
252
|
throw new Stop("", EXIT.passed);
|
|
@@ -275,7 +297,7 @@ function readRun(file, dataset) {
|
|
|
275
297
|
broken(`${file}: ${e.message}`);
|
|
276
298
|
}
|
|
277
299
|
const named = isObject(doc) ? core.evalsDataset(doc) : null;
|
|
278
|
-
const bad = core.validatePipeline(doc, {
|
|
300
|
+
const bad = core.validatePipeline(doc, { groups: dataset && named ? [named] : [] });
|
|
279
301
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
280
302
|
return doc;
|
|
281
303
|
}
|
|
@@ -300,6 +322,11 @@ function readDataset(file) {
|
|
|
300
322
|
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
|
|
301
323
|
}
|
|
302
324
|
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
325
|
+
} else if (isObject(doc) && doc.format === "evals-lab/eval-group") {
|
|
326
|
+
// The lab's Export of an eval group (docs/pipeline-model.md §17), which
|
|
327
|
+
// began at version 7.
|
|
328
|
+
if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
|
|
329
|
+
doc = isObject(doc.group) ? doc.group.body : null;
|
|
303
330
|
}
|
|
304
331
|
if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
|
|
305
332
|
broken(`${file}: rules has to be null or { "rules": [...] }`);
|
|
@@ -315,6 +342,10 @@ function readDataset(file) {
|
|
|
315
342
|
return doc;
|
|
316
343
|
}
|
|
317
344
|
|
|
345
|
+
// A link's eval group, by reference: the body --dataset handed over, the one
|
|
346
|
+
// Library group a run reads, whatever id it names it by.
|
|
347
|
+
const groupOf = (dataset) => () => dataset ?? undefined;
|
|
348
|
+
|
|
318
349
|
// The cases a graded run is scored against, by the item each names: exactly,
|
|
319
350
|
// as the Source names it.
|
|
320
351
|
function gradedBy(dataset) {
|
|
@@ -505,7 +536,7 @@ function graderFor(run) {
|
|
|
505
536
|
const c = run.profiles?.[ref.id];
|
|
506
537
|
if (!c) return undefined;
|
|
507
538
|
if (!made.has(ref.id)) {
|
|
508
|
-
const link = reach(c, core.keyVar(ref.id));
|
|
539
|
+
const link = reach(c, core.keyVar(ref.id, c.slug));
|
|
509
540
|
made.set(ref.id, async prompt => {
|
|
510
541
|
const r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
511
542
|
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
@@ -613,18 +644,88 @@ function readUtf8(file) {
|
|
|
613
644
|
return fs.readFileSync(file).toString("utf8");
|
|
614
645
|
}
|
|
615
646
|
|
|
616
|
-
// Every eval's reading of each scenario,
|
|
617
|
-
//
|
|
618
|
-
//
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
647
|
+
// Every eval's reading of each scenario, exactly as the page reads it.
|
|
648
|
+
// [settled] is false for a run that stopped short, whose whole-run evals
|
|
649
|
+
// have not settled.
|
|
650
|
+
const outcomesOf = (run, items, settled) =>
|
|
651
|
+
core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
|
|
652
|
+
|
|
653
|
+
/**
|
|
654
|
+
* One eval's verdict on one target, for the build: its own rule -- every
|
|
655
|
+
* item it read passed, or a whole-run eval's verdict -- with [minPass] the
|
|
656
|
+
* share of items that is enough instead. It only relaxes: an eval whose
|
|
657
|
+
* every item passed passes whatever the share. One an earlier failure
|
|
658
|
+
* stopped is skipped -- that failure is the run's verdict already -- and one
|
|
659
|
+
* that had nothing to read, a dataset grading none of the run's items, says
|
|
660
|
+
* none: the run ran in full, which is what the server's status reads from
|
|
661
|
+
* the exit code, so it is not incomplete either.
|
|
662
|
+
*/
|
|
663
|
+
function evalVerdict(o, settled, minPass) {
|
|
664
|
+
if (o.skipped) return "skipped";
|
|
665
|
+
if (!settled) return "incomplete";
|
|
666
|
+
if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
|
|
667
|
+
if (!o.ran) return o.skippedItems ? "skipped" : "none";
|
|
668
|
+
if (o.passed === o.ran) return "pass";
|
|
669
|
+
return minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
|
|
673
|
+
// a per-item eval's counts, and the build's verdict on each.
|
|
674
|
+
const verdictsOf = (outcomes, settled, minPass) => outcomes.map(evals =>
|
|
675
|
+
Object.fromEntries(evals.map(o => [o.id, {
|
|
676
|
+
name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass),
|
|
623
677
|
...(o.whole
|
|
624
678
|
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
625
679
|
: { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
|
|
626
680
|
}])));
|
|
627
681
|
|
|
682
|
+
/** The run's verdict from its evals': a run that did not finish is
|
|
683
|
+
incomplete, whatever its evals read so far; then any eval failing fails it. */
|
|
684
|
+
function runVerdict(verdicts, complete) {
|
|
685
|
+
if (!complete) return "incomplete";
|
|
686
|
+
return verdicts.some(v => Object.values(v).some(e => e.verdict === "fail")) ? "fail" : "pass";
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
|
|
690
|
+
|
|
691
|
+
// ---- JUnit -----------------------------------------------------------------
|
|
692
|
+
// What a CI test panel reads, written by the core's junitXml.
|
|
693
|
+
|
|
694
|
+
/** Why a score failed, in a line: its failing metrics, else what it missed. */
|
|
695
|
+
function reasonOf(score) {
|
|
696
|
+
const off = (score.metrics || []).filter(m => !m.pass).map(m => `${m.label}: ${m.reason}`);
|
|
697
|
+
if (off.length) return off.join(" · ");
|
|
698
|
+
const missed = (score.missed || []).map(r => (Array.isArray(r) ? r.join(" or ") : r));
|
|
699
|
+
return missed.length ? `missed ${missed.join(", ")}` : "failed";
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
/** A --run's or a --rescore's suites: each target's evals over [items]. */
|
|
703
|
+
function runSuites(run, outcomes, items, settled, minPass) {
|
|
704
|
+
return outcomes.flatMap((evals, i) => evals.map(o => {
|
|
705
|
+
const name = `${core.targetLabel(run, i)} › ${o.label}`;
|
|
706
|
+
const v = evalVerdict(o, settled, minPass);
|
|
707
|
+
if (o.whole) {
|
|
708
|
+
// Settled over the run rather than item by item: one case, the run.
|
|
709
|
+
const c = { name: o.label };
|
|
710
|
+
if (v === "skipped") c.skipped = "an earlier eval failed";
|
|
711
|
+
else if (v === "none") c.skipped = "nothing to read";
|
|
712
|
+
else if (v === "incomplete") c.error = "the run did not finish";
|
|
713
|
+
else if (v === "fail") c.failure = o.verdict.detail || "failed";
|
|
714
|
+
return { name, cases: [c] };
|
|
715
|
+
}
|
|
716
|
+
return { name, cases: items.map((it, x) => {
|
|
717
|
+
const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
|
|
718
|
+
const read = o.items[x];
|
|
719
|
+
if (!it) c.error = "not run";
|
|
720
|
+
else if (it.unrun) c.error = it.unrun;
|
|
721
|
+
else if (!read) c.skipped = "not graded";
|
|
722
|
+
else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
|
|
723
|
+
else if (!read.pass) c.failure = reasonOf(read);
|
|
724
|
+
return c;
|
|
725
|
+
}) };
|
|
726
|
+
}));
|
|
727
|
+
}
|
|
728
|
+
|
|
628
729
|
// The run as the report restates it: each scenario's stages with the
|
|
629
730
|
// connection each one asked.
|
|
630
731
|
const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
@@ -678,7 +779,7 @@ async function runSnapshot(o) {
|
|
|
678
779
|
// profile's id.
|
|
679
780
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
680
781
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
681
|
-
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
782
|
+
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
682
783
|
return { stages, tokens, connections, links, calls, cells };
|
|
683
784
|
});
|
|
684
785
|
|
|
@@ -728,6 +829,12 @@ async function runSnapshot(o) {
|
|
|
728
829
|
break;
|
|
729
830
|
}
|
|
730
831
|
}
|
|
832
|
+
// A text file the Source does not hold is unrun, as an image is: the
|
|
833
|
+
// run is incomplete rather than broken.
|
|
834
|
+
if (item.kind === "text" && !item.bare && item.text == null && !fs.existsSync(path.join(o.source, item.name))) {
|
|
835
|
+
failed = `the file ${item.name} is not in the source`;
|
|
836
|
+
break;
|
|
837
|
+
}
|
|
731
838
|
const textOf = item.kind === "text" && !item.bare
|
|
732
839
|
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
733
840
|
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
@@ -736,7 +843,7 @@ async function runSnapshot(o) {
|
|
|
736
843
|
// text item a dataset grades is graded like an image.
|
|
737
844
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
738
845
|
production = core.productionOf(run, record);
|
|
739
|
-
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run) }) });
|
|
846
|
+
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run), group: groupOf(dataset) }) });
|
|
740
847
|
}
|
|
741
848
|
// Production's reply to the item, where its record holds one: what a
|
|
742
849
|
// metric compares with, kept so a re-score reads it again.
|
|
@@ -749,16 +856,25 @@ async function runSnapshot(o) {
|
|
|
749
856
|
}
|
|
750
857
|
if (progressFd != null) fs.closeSync(progressFd);
|
|
751
858
|
|
|
859
|
+
// Every item in, none of them unrun: only then has the run settled. A
|
|
860
|
+
// re-run of one item reads the rest from the results it was handed.
|
|
861
|
+
const ranItems = results.slice(0, total);
|
|
862
|
+
const complete = !cancelled && !unrun.length
|
|
863
|
+
&& ranItems.length === total && ranItems.every(it => it && !it.unrun);
|
|
864
|
+
const outcomes = outcomesOf(run, ranItems, complete);
|
|
865
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass);
|
|
752
866
|
const report = {
|
|
753
|
-
runner:
|
|
867
|
+
runner: RUNNER,
|
|
754
868
|
ranAt: new Date().toISOString(),
|
|
755
|
-
verdict:
|
|
869
|
+
verdict: runVerdict(verdicts, complete),
|
|
870
|
+
...(cancelled ? { cancelled: true } : {}),
|
|
871
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
756
872
|
run: {
|
|
757
873
|
name: run.name ?? null,
|
|
758
874
|
evals: run.evals,
|
|
759
|
-
verdicts
|
|
875
|
+
verdicts,
|
|
760
876
|
scenarios: scenariosOf(run),
|
|
761
|
-
items:
|
|
877
|
+
items: ranItems,
|
|
762
878
|
},
|
|
763
879
|
unrun,
|
|
764
880
|
};
|
|
@@ -767,14 +883,31 @@ async function runSnapshot(o) {
|
|
|
767
883
|
say(`${report.runner} — run ${report.run.name ?? ""}`);
|
|
768
884
|
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
769
885
|
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
770
|
-
say
|
|
886
|
+
sayVerdicts(say, run, verdicts);
|
|
887
|
+
say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
|
|
771
888
|
|
|
889
|
+
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
|
|
890
|
+
process.exitCode = exitOf(report.verdict);
|
|
891
|
+
}
|
|
892
|
+
|
|
893
|
+
/** Each target's evals and their verdicts, a line each. */
|
|
894
|
+
function sayVerdicts(say, run, verdicts) {
|
|
895
|
+
verdicts.forEach((evals, i) => {
|
|
896
|
+
for (const e of Object.values(evals)) {
|
|
897
|
+
const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
|
|
898
|
+
say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}`);
|
|
899
|
+
}
|
|
900
|
+
});
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
/** The JSON report, and the JUnit one from [suites], where each was asked for. */
|
|
904
|
+
function writeReports(o, report, suites) {
|
|
772
905
|
if (o.json) {
|
|
773
|
-
const
|
|
774
|
-
if (o.json === "-") process.stdout.write(
|
|
775
|
-
else fs.writeFileSync(o.json,
|
|
906
|
+
const text = JSON.stringify(report, null, 2);
|
|
907
|
+
if (o.json === "-") process.stdout.write(text + "\n");
|
|
908
|
+
else fs.writeFileSync(o.json, text + "\n");
|
|
776
909
|
}
|
|
777
|
-
|
|
910
|
+
if (o.junit) fs.writeFileSync(o.junit, core.junitXml(suites(), RUNNER));
|
|
778
911
|
}
|
|
779
912
|
|
|
780
913
|
/** Re-score a run's stored results against the dataset --dataset hands over,
|
|
@@ -807,39 +940,40 @@ async function runRescore(o){
|
|
|
807
940
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
808
941
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
809
942
|
const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
|
|
810
|
-
const more = { production: it.production ?? null, grader: graderFor(run) };
|
|
943
|
+
const more = { production: it.production ?? null, grader: graderFor(run), group: groupOf(dataset) };
|
|
811
944
|
return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
|
|
812
945
|
(isObject(side) && side.res)
|
|
813
946
|
? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
|
|
814
947
|
}));
|
|
815
948
|
|
|
949
|
+
// A stored run that stopped short re-scores as what it is: incomplete.
|
|
950
|
+
const complete = items.every(it => isObject(it) && !it.unrun);
|
|
951
|
+
const outcomes = outcomesOf(run, items, complete);
|
|
952
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass);
|
|
816
953
|
const report = {
|
|
817
|
-
runner:
|
|
954
|
+
runner: RUNNER,
|
|
818
955
|
ranAt: new Date().toISOString(),
|
|
819
|
-
verdict:
|
|
956
|
+
verdict: runVerdict(verdicts, complete),
|
|
957
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
820
958
|
rescored: true,
|
|
821
959
|
run: {
|
|
822
960
|
name: run.name ?? null,
|
|
823
961
|
evals: run.evals,
|
|
824
|
-
verdicts
|
|
962
|
+
verdicts,
|
|
825
963
|
scenarios: scenariosOf(run),
|
|
826
964
|
items,
|
|
827
965
|
},
|
|
828
966
|
};
|
|
829
967
|
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
if (o.json === "-") process.stdout.write(text + "\n");
|
|
833
|
-
else fs.writeFileSync(o.json, text + "\n");
|
|
834
|
-
}
|
|
835
|
-
process.exitCode = EXIT.passed;
|
|
968
|
+
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
|
|
969
|
+
process.exitCode = exitOf(report.verdict);
|
|
836
970
|
}
|
|
837
971
|
|
|
838
972
|
/**
|
|
839
973
|
* The run --prompt, --model and --url describe, as a run document: one job
|
|
840
974
|
* that sees the image and answers in the default kind, one scenario,
|
|
841
975
|
* graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
|
|
842
|
-
* connection, as it never did, and its key is $
|
|
976
|
+
* connection, as it never did, and its key is $EVALSLAB_API_KEY.
|
|
843
977
|
*/
|
|
844
978
|
function cliRun(o, dataset) {
|
|
845
979
|
let doc = core.blankPipeline();
|
|
@@ -857,8 +991,7 @@ function cliRun(o, dataset) {
|
|
|
857
991
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
858
992
|
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
859
993
|
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
860
|
-
doc.evals = [{ id: "cli", type: "
|
|
861
|
-
mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
|
|
994
|
+
doc.evals = [{ id: "cli", type: "group", name: "", continueOnFailure: true, group: { id: "cli", name: "" }, pin: null }];
|
|
862
995
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
863
996
|
return doc;
|
|
864
997
|
}
|
|
@@ -939,7 +1072,7 @@ async function main() {
|
|
|
939
1072
|
|
|
940
1073
|
// The run to grade. A --pipeline document is one scenario graded against
|
|
941
1074
|
// its dataset; without one it is the stage this script always ran -- the
|
|
942
|
-
// prompt, seeing the image, on --url with $
|
|
1075
|
+
// prompt, seeing the image, on --url with $EVALSLAB_API_KEY -- stated as
|
|
943
1076
|
// the same kind of document, so both go through one reading of it.
|
|
944
1077
|
if (!o.pipeline && !(o.prompt ?? "").trim()) {
|
|
945
1078
|
broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
|
|
@@ -1038,7 +1171,7 @@ async function main() {
|
|
|
1038
1171
|
if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
|
|
1039
1172
|
links = connections.map((c, i) => {
|
|
1040
1173
|
if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
|
|
1041
|
-
const to = reach(c, o.pipeline ? core.keyVar(c.id) : "
|
|
1174
|
+
const to = reach(c, o.pipeline ? core.keyVar(c.id, c.slug) : "EVALSLAB_API_KEY");
|
|
1042
1175
|
// Once here rather than once per image: a llama.cpp stage on a
|
|
1043
1176
|
// hosted model is a request that cannot be built at all.
|
|
1044
1177
|
try {
|
|
@@ -1186,10 +1319,12 @@ async function main() {
|
|
|
1186
1319
|
// passed, and 49 items absent with the fiftieth green is exactly how
|
|
1187
1320
|
// that happens.
|
|
1188
1321
|
const complete = graded.length > 0 && unrun.length === 0;
|
|
1189
|
-
const
|
|
1322
|
+
const enough = tally.passed === tally.ran
|
|
1323
|
+
|| (o.minPass != null && tally.passed / tally.ran >= o.minPass);
|
|
1324
|
+
const verdict = !complete ? "incomplete" : enough ? "pass" : "fail";
|
|
1190
1325
|
|
|
1191
1326
|
const report = {
|
|
1192
|
-
runner:
|
|
1327
|
+
runner: RUNNER,
|
|
1193
1328
|
ranAt: new Date().toISOString(),
|
|
1194
1329
|
dataset: inRepo(o.dataset),
|
|
1195
1330
|
prompt,
|
|
@@ -1217,6 +1352,7 @@ async function main() {
|
|
|
1217
1352
|
filesListed: snapshotSet.size } : {}),
|
|
1218
1353
|
...(o.source ? { source: inRepo(o.source) } : {}),
|
|
1219
1354
|
verdict,
|
|
1355
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
1220
1356
|
graded: (itemsOnly ?? graded).length,
|
|
1221
1357
|
ungraded: set.length - graded.length,
|
|
1222
1358
|
ran: tally.ran,
|
|
@@ -1269,14 +1405,16 @@ async function main() {
|
|
|
1269
1405
|
+ `so this is a partial run and not a result.`);
|
|
1270
1406
|
}
|
|
1271
1407
|
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1408
|
+
// The one eval this mode grades by, as a suite of its cases.
|
|
1409
|
+
const suites = () => {
|
|
1410
|
+
const j = Math.max(0, core.evalsOf(run).findIndex(t => core.evalsDataset({ evals: [t] })));
|
|
1411
|
+
return [{ name: `${core.targetLabel(run, 0)} › ${core.evalLabel(run, j)}`, cases: [
|
|
1412
|
+
...rows.map(r => ({ name: r.id, ...(r.pass ? {} : { failure: r.reasons.join(" · ") || "failed" }) })),
|
|
1413
|
+
...unrun.map(u => { const at = u.indexOf(": "); return { name: u.slice(0, at), error: u.slice(at + 2) }; }),
|
|
1414
|
+
] }];
|
|
1415
|
+
};
|
|
1416
|
+
writeReports(o, report, suites);
|
|
1417
|
+
process.exitCode = exitOf(verdict);
|
|
1280
1418
|
}
|
|
1281
1419
|
|
|
1282
1420
|
main().catch(e => {
|