evals-lab 0.4.1 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +81 -0
- package/README.md +125 -28
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +542 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +735 -86
- package/lab/kinds/list.mjs +2 -1
- package/lab/run-evals.js +354 -99
- package/lab/server.py +711 -135
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +3 -0
- package/lab/web/dist/assets/main-DDoeU6hq.css +1 -0
- package/lab/web/dist/assets/main-nz6Q4jVm.js +21 -0
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +59 -0
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-BkZTEix2.js +0 -21
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/run-evals.js
CHANGED
|
@@ -6,15 +6,19 @@
|
|
|
6
6
|
// what it did twice: a short human report, and a JSON one with every verdict
|
|
7
7
|
// in it. The exit code is the third statement and the coarsest:
|
|
8
8
|
//
|
|
9
|
-
// 0 every
|
|
10
|
-
// 1 the
|
|
11
|
-
// 2 the
|
|
12
|
-
// recorded reply was missing. NOT a pass. An eval
|
|
13
|
-
// because it never ran is the failure this whole
|
|
9
|
+
// 0 every eval passed
|
|
10
|
+
// 1 the run ran and at least one eval failed
|
|
11
|
+
// 2 the run did not run in full -- nothing is graded, an image or a
|
|
12
|
+
// recorded reply was missing, or it was cancelled. NOT a pass. An eval
|
|
13
|
+
// set that scores well because it never ran is the failure this whole
|
|
14
|
+
// ticket is about.
|
|
14
15
|
// 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
|
|
15
16
|
//
|
|
16
|
-
// The
|
|
17
|
-
//
|
|
17
|
+
// The same in every mode, --run and --rescore included, so CI can gate on it
|
|
18
|
+
// (#251): each eval is held to its own rule -- every item it reads passes, or
|
|
19
|
+
// a whole-run eval's verdict -- and --min-pass relaxes the per-item rule to a
|
|
20
|
+
// share. The report's `verdict` (pass, fail, incomplete) says the same as the
|
|
21
|
+
// code, and --junit writes it for a CI test panel.
|
|
18
22
|
//
|
|
19
23
|
// Ollama is firewalled to a handful of hosts, so a live run has to happen on
|
|
20
24
|
// one of them. That is a fact about the network and not something a flag here
|
|
@@ -64,7 +68,10 @@ const { execFileSync } = require("child_process");
|
|
|
64
68
|
const core = require("./evals-core.mjs");
|
|
65
69
|
|
|
66
70
|
const HERE = __dirname;
|
|
67
|
-
const ROOT =
|
|
71
|
+
const ROOT = HERE;
|
|
72
|
+
|
|
73
|
+
// What a report names as its runner: this file, from the lab's root.
|
|
74
|
+
const RUNNER = "run-evals.js";
|
|
68
75
|
|
|
69
76
|
// Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
|
|
70
77
|
// into 448px tiles and the tile COUNT is what costs. The same budget the tab
|
|
@@ -82,11 +89,12 @@ const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
|
|
|
82
89
|
// single-flight queue for good; `--timeout` can shorten it, never lengthen.
|
|
83
90
|
const REQUEST_CAP = 600;
|
|
84
91
|
|
|
85
|
-
const USAGE = `Usage: node
|
|
92
|
+
const USAGE = `Usage: node run-evals.js [options]
|
|
86
93
|
|
|
87
94
|
--model <id> the model to grade. Required for a live run.
|
|
88
95
|
--url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
|
|
89
|
-
Ollama on this machine. A key comes from
|
|
96
|
+
Ollama on this machine. A key comes from
|
|
97
|
+
$EVALSLAB_API_KEY, never argv.
|
|
90
98
|
--prompt <text> the prompt to grade, as a template: its tokens resolve
|
|
91
99
|
under --tokens. Required without --pipeline: a dataset
|
|
92
100
|
holds no prompt (the lab's Prompt library does).
|
|
@@ -103,17 +111,22 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
103
111
|
--pipeline <file> a run document (docs/pipeline-model.md) with one
|
|
104
112
|
scenario and a graded eval, graded case by case against
|
|
105
113
|
the set. Instead of --prompt, --model and --url. Each
|
|
106
|
-
profile's key comes from $
|
|
114
|
+
profile's key comes from $EVALSLAB_API_KEY_<SLUG> -- its
|
|
115
|
+
id's, in a file whose profile has no slug -- never the
|
|
107
116
|
file; its content.files, when not empty, is the file
|
|
108
117
|
list a case has to be in, as --files says.
|
|
109
118
|
--samples <dir> where the images are. Default: the lab's samples/.
|
|
110
|
-
--dataset <file> the
|
|
111
|
-
({"cases"}), or the file the lab's Export writes.
|
|
119
|
+
--dataset <file> the one eval group a graded run is scored against: its
|
|
120
|
+
body ({"cases"}), or the file the lab's Export writes.
|
|
112
121
|
Versions 1 to 3 are read too (a prompt they hold is
|
|
113
122
|
not used: --prompt names it); a version-2
|
|
114
123
|
body's rules are what an older run's jobs, and a run
|
|
115
124
|
without --pipeline, read replies under. Its cases grade.
|
|
116
|
-
Required for anything graded.
|
|
125
|
+
Required for anything graded (or --groups, for a --run).
|
|
126
|
+
--groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
|
|
127
|
+
grades against each one its evals link (§17), so a run
|
|
128
|
+
linking several groups is handed them here instead of the
|
|
129
|
+
one --dataset. For --run and --rescore; not with --dataset.
|
|
117
130
|
--source <dir> the same, named the way a run names it: the directory a
|
|
118
131
|
Source is stored at. --samples and --source are one
|
|
119
132
|
flag by two names; both together are refused.
|
|
@@ -153,6 +166,12 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
153
166
|
alone. No model, no files, no network.
|
|
154
167
|
--json <path|-> write the machine-readable report. "-" means stdout, and
|
|
155
168
|
sends the human report to stderr.
|
|
169
|
+
--min-pass <0..1> an eval whose items are read one by one passes when at
|
|
170
|
+
least this share of them pass. It relaxes each eval's
|
|
171
|
+
own rule -- every item passes -- and never tightens it;
|
|
172
|
+
a whole-run eval keeps its own verdict.
|
|
173
|
+
--junit <file> write JUnit XML: one testsuite per target and eval, one
|
|
174
|
+
testcase per item, a failure carrying its reason.
|
|
156
175
|
`;
|
|
157
176
|
|
|
158
177
|
// NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
|
|
@@ -189,10 +208,10 @@ function broken(msg) {
|
|
|
189
208
|
|
|
190
209
|
function parseArgs(argv) {
|
|
191
210
|
const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
|
|
192
|
-
replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
211
|
+
groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
193
212
|
timeout: null, progress: false, cancelFile: "",
|
|
194
213
|
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
195
|
-
rescore: false, tokens: "", plugins: "" };
|
|
214
|
+
rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
|
|
196
215
|
for (let i = 0; i < argv.length; i++) {
|
|
197
216
|
const a = argv[i];
|
|
198
217
|
const value = () => {
|
|
@@ -210,6 +229,7 @@ function parseArgs(argv) {
|
|
|
210
229
|
case "--pipeline": o.pipeline = value(); break;
|
|
211
230
|
case "--samples": o.samples = value(); break;
|
|
212
231
|
case "--dataset": o.dataset = value(); break;
|
|
232
|
+
case "--groups": o.groups = value(); break;
|
|
213
233
|
case "--replies": o.replies = value(); break;
|
|
214
234
|
case "--source": o.source = value(); break;
|
|
215
235
|
case "--files": o.files = value(); break;
|
|
@@ -225,6 +245,13 @@ function parseArgs(argv) {
|
|
|
225
245
|
case "--only": o.only = value(); break;
|
|
226
246
|
case "--rescore": o.rescore = true; break;
|
|
227
247
|
case "--json": o.json = value(); break;
|
|
248
|
+
case "--min-pass": {
|
|
249
|
+
const v = value(), n = Number(v);
|
|
250
|
+
if (v.trim() === "" || !(n >= 0 && n <= 1)) broken(`--min-pass is a share from 0 to 1, and ${v} is not one`);
|
|
251
|
+
o.minPass = n;
|
|
252
|
+
break;
|
|
253
|
+
}
|
|
254
|
+
case "--junit": o.junit = value(); break;
|
|
228
255
|
case "-h": case "--help":
|
|
229
256
|
process.stdout.write(USAGE);
|
|
230
257
|
throw new Stop("", EXIT.passed);
|
|
@@ -259,10 +286,10 @@ const inRepo = p => {
|
|
|
259
286
|
const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
|
|
260
287
|
const isText = v => typeof v === "string";
|
|
261
288
|
|
|
262
|
-
// [
|
|
263
|
-
//
|
|
264
|
-
//
|
|
265
|
-
function readRun(file,
|
|
289
|
+
// [gctx] is the grading context (gradingContext): the eval group bodies a run
|
|
290
|
+
// grades against -- one, as --dataset, or a body per group kept with the run,
|
|
291
|
+
// as --groups (§17). The run reads each link's body by its reference.
|
|
292
|
+
function readRun(file, gctx) {
|
|
266
293
|
let doc;
|
|
267
294
|
try {
|
|
268
295
|
// A run queued before the current version is read as one of today's:
|
|
@@ -270,12 +297,20 @@ function readRun(file, dataset) {
|
|
|
270
297
|
// submitted with. A v2 run's list jobs read their replies under the
|
|
271
298
|
// rules of the dataset it was graded against -- the body handed over.
|
|
272
299
|
doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
|
|
273
|
-
{ rulesFor: () => rulesOf(
|
|
300
|
+
{ rulesFor: () => rulesOf(gctx.primary()) });
|
|
274
301
|
} catch (e) {
|
|
275
302
|
broken(`${file}: ${e.message}`);
|
|
276
303
|
}
|
|
277
|
-
|
|
278
|
-
|
|
304
|
+
// Every Library group the evals name -- a link's group, a private group's
|
|
305
|
+
// Cases from -- each as validatePipeline reads one. Pins were resolved at
|
|
306
|
+
// submit (the reference carries `n`), so they are not checked again here.
|
|
307
|
+
const groups = [], seen = new Set();
|
|
308
|
+
for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
|
|
309
|
+
if (seen.has(ref.id)) continue;
|
|
310
|
+
seen.add(ref.id);
|
|
311
|
+
groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
|
|
312
|
+
}
|
|
313
|
+
const bad = core.validatePipeline(doc, { groups });
|
|
279
314
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
280
315
|
return doc;
|
|
281
316
|
}
|
|
@@ -283,9 +318,25 @@ function readRun(file, dataset) {
|
|
|
283
318
|
// The rules a version-2 body held, by the body readDataset made of it: what
|
|
284
319
|
// an older run's jobs read replies under. A body of today's holds none.
|
|
285
320
|
const heldRules = new WeakMap();
|
|
286
|
-
const rulesOf = (
|
|
321
|
+
const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
|
|
322
|
+
|
|
323
|
+
// An eval group's body as one of today's, with its version-2 rules taken
|
|
324
|
+
// first: a bare body, or the body the lab's Export wraps, unwrapped by its
|
|
325
|
+
// caller. [where] names it in a refusal.
|
|
326
|
+
function bodyFrom(doc, where) {
|
|
327
|
+
if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
|
|
328
|
+
broken(`${where}: rules has to be null or { "rules": [...] }`);
|
|
329
|
+
}
|
|
330
|
+
const rules = core.datasetRules(doc);
|
|
331
|
+
const body = core.upgradeDatasetBody(doc);
|
|
332
|
+
if (!isObject(body) || !Array.isArray(body.cases)) {
|
|
333
|
+
broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
|
|
334
|
+
}
|
|
335
|
+
if (rules) heldRules.set(body, rules);
|
|
336
|
+
return body;
|
|
337
|
+
}
|
|
287
338
|
|
|
288
|
-
// The dataset --dataset names:
|
|
339
|
+
// The dataset --dataset names: an eval group's body, or the file the lab's
|
|
289
340
|
// Export writes for one, as one of today's.
|
|
290
341
|
function readDataset(file) {
|
|
291
342
|
if (!file) return null;
|
|
@@ -300,26 +351,57 @@ function readDataset(file) {
|
|
|
300
351
|
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
|
|
301
352
|
}
|
|
302
353
|
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
354
|
+
} else if (isObject(doc) && doc.format === "evals-lab/eval-group") {
|
|
355
|
+
// The lab's Export of an eval group (docs/pipeline-model.md §17), which
|
|
356
|
+
// began at version 7.
|
|
357
|
+
if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
|
|
358
|
+
doc = isObject(doc.group) ? doc.group.body : null;
|
|
303
359
|
}
|
|
304
|
-
|
|
305
|
-
|
|
360
|
+
return bodyFrom(doc, file);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
// The bodies a run kept, as --groups hands them (§17): a JSON object of
|
|
364
|
+
// `<id>@<n>` to body -- the group's id and the version it graded with, its id
|
|
365
|
+
// alone for a run from before the lab numbered them -- each read as today's.
|
|
366
|
+
function readGroups(file) {
|
|
367
|
+
let raw;
|
|
368
|
+
try {
|
|
369
|
+
raw = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
370
|
+
} catch (e) {
|
|
371
|
+
broken(`${file}: ${e.message}`);
|
|
306
372
|
}
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
const
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
373
|
+
if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
|
|
374
|
+
const bodies = new Map();
|
|
375
|
+
for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
|
|
376
|
+
return bodies;
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// The key a run keeps a group's body under, mirroring the server's group_key:
|
|
380
|
+
// `<id>@<n>`, or the id alone for a reference from before versions.
|
|
381
|
+
const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
|
|
382
|
+
|
|
383
|
+
// The eval groups a --run grades against: one body, as --dataset, or a body
|
|
384
|
+
// per group, as --groups -- never both. `resolve` gives a link's body by its
|
|
385
|
+
// reference; `primary` the one a single-group run names, for a v2 run's rules.
|
|
386
|
+
function gradingContext(o) {
|
|
387
|
+
if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
|
|
388
|
+
if (o.groups) {
|
|
389
|
+
const bodies = readGroups(o.groups);
|
|
390
|
+
const first = bodies.values().next().value ?? null;
|
|
391
|
+
return { single: null, bodies, primary: () => first,
|
|
392
|
+
resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
|
|
313
393
|
}
|
|
314
|
-
|
|
315
|
-
return
|
|
394
|
+
const single = readDataset(o.dataset);
|
|
395
|
+
return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
|
|
316
396
|
}
|
|
317
397
|
|
|
318
398
|
// The cases a graded run is scored against, by the item each names: exactly,
|
|
319
|
-
// as the Source names it.
|
|
320
|
-
|
|
399
|
+
// as the Source names it. Built from the one group a single-group run names,
|
|
400
|
+
// for an eval type of its own that reads a case (a plugin's); a `group` eval
|
|
401
|
+
// reads each group's own case, by the item's name, as it grades.
|
|
402
|
+
function gradedBy(body) {
|
|
321
403
|
const out = new Map();
|
|
322
|
-
for (const c of core.gradedSetFrom(
|
|
404
|
+
for (const c of core.gradedSetFrom(body || {})) {
|
|
323
405
|
if (!c.todo) out.set(c.item, c);
|
|
324
406
|
}
|
|
325
407
|
return out;
|
|
@@ -505,7 +587,7 @@ function graderFor(run) {
|
|
|
505
587
|
const c = run.profiles?.[ref.id];
|
|
506
588
|
if (!c) return undefined;
|
|
507
589
|
if (!made.has(ref.id)) {
|
|
508
|
-
const link = reach(c, core.keyVar(ref.id));
|
|
590
|
+
const link = reach(c, core.keyVar(ref.id, c.slug));
|
|
509
591
|
made.set(ref.id, async prompt => {
|
|
510
592
|
const r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
511
593
|
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
@@ -613,17 +695,138 @@ function readUtf8(file) {
|
|
|
613
695
|
return fs.readFileSync(file).toString("utf8");
|
|
614
696
|
}
|
|
615
697
|
|
|
616
|
-
// Every eval's reading of each scenario,
|
|
617
|
-
//
|
|
618
|
-
//
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
698
|
+
// Every eval's reading of each scenario, exactly as the page reads it.
|
|
699
|
+
// [settled] is false for a run that stopped short, whose whole-run evals
|
|
700
|
+
// have not settled.
|
|
701
|
+
const outcomesOf = (run, items, settled) =>
|
|
702
|
+
core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
|
|
703
|
+
|
|
704
|
+
// A link's Whole run verdict, per Target, keyed by the eval's id: a Library
|
|
705
|
+
// group's `run` metrics read over the replies so far (readGroupRun). A private
|
|
706
|
+
// group's Whole run is scenarioEvals' own (it holds the body); a link's body
|
|
707
|
+
// is held apart, in --groups, so reading it beside the link's items is the
|
|
708
|
+
// runner's (§17). [resolve] gives a link's body by its reference.
|
|
709
|
+
function wholeRunsOf(run, items, resolve) {
|
|
710
|
+
const kind = core.lastKind(run);
|
|
711
|
+
return core.targetsOf(run).map((_, i) => {
|
|
712
|
+
const out = {};
|
|
713
|
+
const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
|
|
714
|
+
for (const t of core.evalsOf(run)) {
|
|
715
|
+
// A private whole-run group is already scenarioEvals' verdict; here is
|
|
716
|
+
// the run beside a link's (or private group's) own items.
|
|
717
|
+
if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
|
|
718
|
+
const body = core.groupOf(t, { group: resolve });
|
|
719
|
+
if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
|
|
720
|
+
}
|
|
721
|
+
return out;
|
|
722
|
+
});
|
|
723
|
+
}
|
|
724
|
+
|
|
725
|
+
/**
|
|
726
|
+
* One eval's verdict on one target, for the build: its own rule -- every
|
|
727
|
+
* item it read passed, or a whole-run eval's verdict -- with [minPass] the
|
|
728
|
+
* share of items that is enough instead. It only relaxes: an eval whose
|
|
729
|
+
* every item passed passes whatever the share. One an earlier failure
|
|
730
|
+
* stopped is skipped -- that failure is the run's verdict already -- and one
|
|
731
|
+
* that had nothing to read, a group grading none of the run's items, says
|
|
732
|
+
* none: the run ran in full, which is what the server's status reads from
|
|
733
|
+
* the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
|
|
734
|
+
* run verdict, where its group reads one beside its items (§17): the group
|
|
735
|
+
* passes when both its items and its Whole run do, and either failing fails it.
|
|
736
|
+
*/
|
|
737
|
+
function evalVerdict(o, settled, minPass, wholeRun) {
|
|
738
|
+
if (o.skipped) return "skipped";
|
|
739
|
+
if (!settled) return "incomplete";
|
|
740
|
+
if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
|
|
741
|
+
const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
|
|
742
|
+
: o.passed === o.ran ? "pass"
|
|
743
|
+
: minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
|
|
744
|
+
if (!wholeRun) return item;
|
|
745
|
+
const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
|
|
746
|
+
if (item === "fail" || whole === "fail") return "fail";
|
|
747
|
+
if (item === "pass" || whole === "pass") return "pass";
|
|
748
|
+
return item === "skipped" || whole === "skipped" ? "skipped" : "none";
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
|
|
752
|
+
// per-item eval's counts -- and, where a link reads a Whole run beside its
|
|
753
|
+
// items, that verdict too -- with the build's verdict on each. [wholes] is
|
|
754
|
+
// wholeRunsOf, one map per target.
|
|
755
|
+
const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
|
|
756
|
+
Object.fromEntries(evals.map(o => {
|
|
757
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
758
|
+
return [o.id, {
|
|
759
|
+
name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
|
|
760
|
+
...(o.whole
|
|
761
|
+
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
762
|
+
: { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
|
|
763
|
+
...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
|
|
764
|
+
}];
|
|
765
|
+
})));
|
|
766
|
+
|
|
767
|
+
/** Each Target's verdict under the overall pass rule (§17), from its groups'
|
|
768
|
+
verdicts: a group that passed counts, one that failed counts against, and
|
|
769
|
+
one that read nothing or was skipped is neither. */
|
|
770
|
+
const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
|
|
771
|
+
Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
|
|
772
|
+
|
|
773
|
+
/** The run's verdict from its evals' and the overall rule: a run that did not
|
|
774
|
+
finish is incomplete, whatever its evals read so far; then it passes when
|
|
775
|
+
every Target does under the rule. */
|
|
776
|
+
function runVerdict(verdicts, complete, pass) {
|
|
777
|
+
if (!complete) return "incomplete";
|
|
778
|
+
return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
|
|
779
|
+
}
|
|
780
|
+
|
|
781
|
+
const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
|
|
782
|
+
|
|
783
|
+
// ---- JUnit -----------------------------------------------------------------
|
|
784
|
+
// What a CI test panel reads, written by the core's junitXml.
|
|
785
|
+
|
|
786
|
+
/** Why a score failed, in a line: its failing metrics, else what it missed. */
|
|
787
|
+
function reasonOf(score) {
|
|
788
|
+
const off = (score.metrics || []).filter(m => !m.pass).map(m => `${m.label}: ${m.reason}`);
|
|
789
|
+
if (off.length) return off.join(" · ");
|
|
790
|
+
const missed = (score.missed || []).map(r => (Array.isArray(r) ? r.join(" or ") : r));
|
|
791
|
+
return missed.length ? `missed ${missed.join(", ")}` : "failed";
|
|
792
|
+
}
|
|
793
|
+
|
|
794
|
+
/** A --run's or a --rescore's suites: each target's evals over [items], with a
|
|
795
|
+
Whole run case where a link reads one beside its items ([wholes]). */
|
|
796
|
+
function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
|
|
797
|
+
return outcomes.flatMap((evals, i) => evals.map(o => {
|
|
798
|
+
const name = `${core.targetLabel(run, i)} › ${o.label}`;
|
|
799
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
800
|
+
const v = evalVerdict(o, settled, minPass, wr);
|
|
801
|
+
if (o.whole) {
|
|
802
|
+
// Settled over the run rather than item by item: one case, the run.
|
|
803
|
+
const c = { name: o.label };
|
|
804
|
+
if (v === "skipped") c.skipped = "an earlier eval failed";
|
|
805
|
+
else if (v === "none") c.skipped = "nothing to read";
|
|
806
|
+
else if (v === "incomplete") c.error = "the run did not finish";
|
|
807
|
+
else if (v === "fail") c.failure = o.verdict.detail || "failed";
|
|
808
|
+
return { name, cases: [c] };
|
|
809
|
+
}
|
|
810
|
+
const cases = items.map((it, x) => {
|
|
811
|
+
const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
|
|
812
|
+
const read = o.items[x];
|
|
813
|
+
if (!it) c.error = "not run";
|
|
814
|
+
else if (it.unrun) c.error = it.unrun;
|
|
815
|
+
else if (!read) c.skipped = "not graded";
|
|
816
|
+
else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
|
|
817
|
+
else if (!read.pass) c.failure = reasonOf(read);
|
|
818
|
+
return c;
|
|
819
|
+
});
|
|
820
|
+
if (wr) {
|
|
821
|
+
const c = { name: "Whole run" };
|
|
822
|
+
if (!settled) c.error = "the run did not finish";
|
|
823
|
+
else if (!wr.ran) c.skipped = "nothing to read";
|
|
824
|
+
else if (!wr.pass) c.failure = wr.detail || "failed";
|
|
825
|
+
cases.push(c);
|
|
826
|
+
}
|
|
827
|
+
return { name, cases };
|
|
828
|
+
}));
|
|
829
|
+
}
|
|
627
830
|
|
|
628
831
|
// The run as the report restates it: each scenario's stages with the
|
|
629
832
|
// connection each one asked.
|
|
@@ -639,8 +842,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
|
639
842
|
});
|
|
640
843
|
|
|
641
844
|
async function runSnapshot(o) {
|
|
642
|
-
const
|
|
643
|
-
const run = readRun(o.run,
|
|
845
|
+
const gctx = gradingContext(o);
|
|
846
|
+
const run = readRun(o.run, gctx);
|
|
644
847
|
// One item a pipeline holds itself (Text, or Prompt only's bare one), or
|
|
645
848
|
// the Source's files.
|
|
646
849
|
const content = core.contentOf(run);
|
|
@@ -652,8 +855,9 @@ async function runSnapshot(o) {
|
|
|
652
855
|
broken(`${o.run} names no files and no text, so there is nothing to run`);
|
|
653
856
|
}
|
|
654
857
|
|
|
655
|
-
// The
|
|
656
|
-
|
|
858
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
859
|
+
// reads a case; a `group` eval reads each group's case by the item's name.
|
|
860
|
+
const graded = gradedBy(gctx.primary());
|
|
657
861
|
|
|
658
862
|
const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
|
|
659
863
|
: text.trim() ? [{ name: null, kind: "text", text }]
|
|
@@ -678,7 +882,7 @@ async function runSnapshot(o) {
|
|
|
678
882
|
// profile's id.
|
|
679
883
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
680
884
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
681
|
-
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
885
|
+
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
682
886
|
return { stages, tokens, connections, links, calls, cells };
|
|
683
887
|
});
|
|
684
888
|
|
|
@@ -728,6 +932,12 @@ async function runSnapshot(o) {
|
|
|
728
932
|
break;
|
|
729
933
|
}
|
|
730
934
|
}
|
|
935
|
+
// A text file the Source does not hold is unrun, as an image is: the
|
|
936
|
+
// run is incomplete rather than broken.
|
|
937
|
+
if (item.kind === "text" && !item.bare && item.text == null && !fs.existsSync(path.join(o.source, item.name))) {
|
|
938
|
+
failed = `the file ${item.name} is not in the source`;
|
|
939
|
+
break;
|
|
940
|
+
}
|
|
731
941
|
const textOf = item.kind === "text" && !item.bare
|
|
732
942
|
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
733
943
|
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
@@ -736,7 +946,8 @@ async function runSnapshot(o) {
|
|
|
736
946
|
// text item a dataset grades is graded like an image.
|
|
737
947
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
738
948
|
production = core.productionOf(run, record);
|
|
739
|
-
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
949
|
+
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
950
|
+
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
|
|
740
951
|
}
|
|
741
952
|
// Production's reply to the item, where its record holds one: what a
|
|
742
953
|
// metric compares with, kept so a re-score reads it again.
|
|
@@ -749,16 +960,27 @@ async function runSnapshot(o) {
|
|
|
749
960
|
}
|
|
750
961
|
if (progressFd != null) fs.closeSync(progressFd);
|
|
751
962
|
|
|
963
|
+
// Every item in, none of them unrun: only then has the run settled. A
|
|
964
|
+
// re-run of one item reads the rest from the results it was handed.
|
|
965
|
+
const ranItems = results.slice(0, total);
|
|
966
|
+
const complete = !cancelled && !unrun.length
|
|
967
|
+
&& ranItems.length === total && ranItems.every(it => it && !it.unrun);
|
|
968
|
+
const outcomes = outcomesOf(run, ranItems, complete);
|
|
969
|
+
const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
|
|
970
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
752
971
|
const report = {
|
|
753
|
-
runner:
|
|
972
|
+
runner: RUNNER,
|
|
754
973
|
ranAt: new Date().toISOString(),
|
|
755
|
-
verdict:
|
|
974
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
975
|
+
...(cancelled ? { cancelled: true } : {}),
|
|
976
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
756
977
|
run: {
|
|
757
978
|
name: run.name ?? null,
|
|
758
979
|
evals: run.evals,
|
|
759
|
-
verdicts
|
|
980
|
+
verdicts,
|
|
981
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
760
982
|
scenarios: scenariosOf(run),
|
|
761
|
-
items:
|
|
983
|
+
items: ranItems,
|
|
762
984
|
},
|
|
763
985
|
unrun,
|
|
764
986
|
};
|
|
@@ -767,14 +989,37 @@ async function runSnapshot(o) {
|
|
|
767
989
|
say(`${report.runner} — run ${report.run.name ?? ""}`);
|
|
768
990
|
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
769
991
|
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
770
|
-
say
|
|
992
|
+
sayVerdicts(say, run, verdicts, report.run.pass);
|
|
993
|
+
say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
|
|
994
|
+
|
|
995
|
+
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
|
|
996
|
+
process.exitCode = exitOf(report.verdict);
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
/** Each target's groups and their verdicts, a line each, and the overall rule
|
|
1000
|
+
beneath them where [pass] gives a figure per target. */
|
|
1001
|
+
function sayVerdicts(say, run, verdicts, pass) {
|
|
1002
|
+
verdicts.forEach((evals, i) => {
|
|
1003
|
+
for (const e of Object.values(evals)) {
|
|
1004
|
+
const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
|
|
1005
|
+
const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
|
|
1006
|
+
say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
|
|
1007
|
+
}
|
|
1008
|
+
if (pass) {
|
|
1009
|
+
const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
|
|
1010
|
+
say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
|
|
1011
|
+
}
|
|
1012
|
+
});
|
|
1013
|
+
}
|
|
771
1014
|
|
|
1015
|
+
/** The JSON report, and the JUnit one from [suites], where each was asked for. */
|
|
1016
|
+
function writeReports(o, report, suites) {
|
|
772
1017
|
if (o.json) {
|
|
773
|
-
const
|
|
774
|
-
if (o.json === "-") process.stdout.write(
|
|
775
|
-
else fs.writeFileSync(o.json,
|
|
1018
|
+
const text = JSON.stringify(report, null, 2);
|
|
1019
|
+
if (o.json === "-") process.stdout.write(text + "\n");
|
|
1020
|
+
else fs.writeFileSync(o.json, text + "\n");
|
|
776
1021
|
}
|
|
777
|
-
|
|
1022
|
+
if (o.junit) fs.writeFileSync(o.junit, core.junitXml(suites(), RUNNER));
|
|
778
1023
|
}
|
|
779
1024
|
|
|
780
1025
|
/** Re-score a run's stored results against the dataset --dataset hands over,
|
|
@@ -783,8 +1028,8 @@ async function runSnapshot(o) {
|
|
|
783
1028
|
* a --run's, with the stored results as its items, and a whole-run eval's
|
|
784
1029
|
* verdict settled the same way. */
|
|
785
1030
|
async function runRescore(o){
|
|
786
|
-
const
|
|
787
|
-
const run = readRun(o.run,
|
|
1031
|
+
const gctx = gradingContext(o);
|
|
1032
|
+
const run = readRun(o.run, gctx);
|
|
788
1033
|
let results;
|
|
789
1034
|
try {
|
|
790
1035
|
results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
|
|
@@ -796,50 +1041,53 @@ async function runRescore(o){
|
|
|
796
1041
|
}
|
|
797
1042
|
results = core.upgradeResults(results);
|
|
798
1043
|
|
|
799
|
-
// The
|
|
800
|
-
//
|
|
801
|
-
// say what the
|
|
802
|
-
const graded = gradedBy(
|
|
1044
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
1045
|
+
// reads a case. A file no group grades keeps its reply and has no score --
|
|
1046
|
+
// honestly: nothing else would say what the groups cover.
|
|
1047
|
+
const graded = gradedBy(gctx.primary());
|
|
803
1048
|
|
|
804
1049
|
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
805
|
-
// An eval that reads every item (
|
|
1050
|
+
// An eval that reads every item (a group) re-reads one with no case
|
|
806
1051
|
// too, against the production reply the item kept.
|
|
807
1052
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
808
1053
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
809
1054
|
const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
|
|
810
|
-
const more = { production: it.production ?? null, grader: graderFor(run) };
|
|
1055
|
+
const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
|
|
811
1056
|
return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
|
|
812
1057
|
(isObject(side) && side.res)
|
|
813
1058
|
? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
|
|
814
1059
|
}));
|
|
815
1060
|
|
|
1061
|
+
// A stored run that stopped short re-scores as what it is: incomplete.
|
|
1062
|
+
const complete = items.every(it => isObject(it) && !it.unrun);
|
|
1063
|
+
const outcomes = outcomesOf(run, items, complete);
|
|
1064
|
+
const wholes = wholeRunsOf(run, items, gctx.resolve);
|
|
1065
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
816
1066
|
const report = {
|
|
817
|
-
runner:
|
|
1067
|
+
runner: RUNNER,
|
|
818
1068
|
ranAt: new Date().toISOString(),
|
|
819
|
-
verdict:
|
|
1069
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
1070
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
820
1071
|
rescored: true,
|
|
821
1072
|
run: {
|
|
822
1073
|
name: run.name ?? null,
|
|
823
1074
|
evals: run.evals,
|
|
824
|
-
verdicts
|
|
1075
|
+
verdicts,
|
|
1076
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
825
1077
|
scenarios: scenariosOf(run),
|
|
826
1078
|
items,
|
|
827
1079
|
},
|
|
828
1080
|
};
|
|
829
1081
|
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
if (o.json === "-") process.stdout.write(text + "\n");
|
|
833
|
-
else fs.writeFileSync(o.json, text + "\n");
|
|
834
|
-
}
|
|
835
|
-
process.exitCode = EXIT.passed;
|
|
1082
|
+
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
|
|
1083
|
+
process.exitCode = exitOf(report.verdict);
|
|
836
1084
|
}
|
|
837
1085
|
|
|
838
1086
|
/**
|
|
839
1087
|
* The run --prompt, --model and --url describe, as a run document: one job
|
|
840
1088
|
* that sees the image and answers in the default kind, one scenario,
|
|
841
1089
|
* graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
|
|
842
|
-
* connection, as it never did, and its key is $
|
|
1090
|
+
* connection, as it never did, and its key is $EVALSLAB_API_KEY.
|
|
843
1091
|
*/
|
|
844
1092
|
function cliRun(o, dataset) {
|
|
845
1093
|
let doc = core.blankPipeline();
|
|
@@ -857,8 +1105,7 @@ function cliRun(o, dataset) {
|
|
|
857
1105
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
858
1106
|
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
859
1107
|
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
860
|
-
doc.evals = [{ id: "cli", type: "
|
|
861
|
-
mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
|
|
1108
|
+
doc.evals = [{ id: "cli", type: "group", name: "", continueOnFailure: true, group: { id: "cli", name: "" }, pin: null }];
|
|
862
1109
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
863
1110
|
return doc;
|
|
864
1111
|
}
|
|
@@ -933,19 +1180,22 @@ async function main() {
|
|
|
933
1180
|
}
|
|
934
1181
|
|
|
935
1182
|
// The dataset the run is graded against. Everything below grades, so there
|
|
936
|
-
// is no run without one.
|
|
1183
|
+
// is no run without one. --groups hands a --run the bodies it kept; a
|
|
1184
|
+
// --pipeline or --prompt grades one scenario against one, named by --dataset.
|
|
1185
|
+
if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
|
|
937
1186
|
if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
|
|
938
1187
|
const dataset = readDataset(o.dataset);
|
|
1188
|
+
const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
|
|
939
1189
|
|
|
940
1190
|
// The run to grade. A --pipeline document is one scenario graded against
|
|
941
1191
|
// its dataset; without one it is the stage this script always ran -- the
|
|
942
|
-
// prompt, seeing the image, on --url with $
|
|
1192
|
+
// prompt, seeing the image, on --url with $EVALSLAB_API_KEY -- stated as
|
|
943
1193
|
// the same kind of document, so both go through one reading of it.
|
|
944
1194
|
if (!o.pipeline && !(o.prompt ?? "").trim()) {
|
|
945
1195
|
broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
|
|
946
1196
|
+ "the lab's Prompt library does.");
|
|
947
1197
|
}
|
|
948
|
-
const run = o.pipeline ? readRun(o.pipeline,
|
|
1198
|
+
const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
|
|
949
1199
|
if (o.pipeline) {
|
|
950
1200
|
if (core.targetsOf(run).length !== 1) {
|
|
951
1201
|
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
|
@@ -1038,7 +1288,7 @@ async function main() {
|
|
|
1038
1288
|
if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
|
|
1039
1289
|
links = connections.map((c, i) => {
|
|
1040
1290
|
if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
|
|
1041
|
-
const to = reach(c, o.pipeline ? core.keyVar(c.id) : "
|
|
1291
|
+
const to = reach(c, o.pipeline ? core.keyVar(c.id, c.slug) : "EVALSLAB_API_KEY");
|
|
1042
1292
|
// Once here rather than once per image: a llama.cpp stage on a
|
|
1043
1293
|
// hosted model is a request that cannot be built at all.
|
|
1044
1294
|
try {
|
|
@@ -1186,10 +1436,12 @@ async function main() {
|
|
|
1186
1436
|
// passed, and 49 items absent with the fiftieth green is exactly how
|
|
1187
1437
|
// that happens.
|
|
1188
1438
|
const complete = graded.length > 0 && unrun.length === 0;
|
|
1189
|
-
const
|
|
1439
|
+
const enough = tally.passed === tally.ran
|
|
1440
|
+
|| (o.minPass != null && tally.passed / tally.ran >= o.minPass);
|
|
1441
|
+
const verdict = !complete ? "incomplete" : enough ? "pass" : "fail";
|
|
1190
1442
|
|
|
1191
1443
|
const report = {
|
|
1192
|
-
runner:
|
|
1444
|
+
runner: RUNNER,
|
|
1193
1445
|
ranAt: new Date().toISOString(),
|
|
1194
1446
|
dataset: inRepo(o.dataset),
|
|
1195
1447
|
prompt,
|
|
@@ -1217,6 +1469,7 @@ async function main() {
|
|
|
1217
1469
|
filesListed: snapshotSet.size } : {}),
|
|
1218
1470
|
...(o.source ? { source: inRepo(o.source) } : {}),
|
|
1219
1471
|
verdict,
|
|
1472
|
+
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
1220
1473
|
graded: (itemsOnly ?? graded).length,
|
|
1221
1474
|
ungraded: set.length - graded.length,
|
|
1222
1475
|
ran: tally.ran,
|
|
@@ -1269,14 +1522,16 @@ async function main() {
|
|
|
1269
1522
|
+ `so this is a partial run and not a result.`);
|
|
1270
1523
|
}
|
|
1271
1524
|
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1525
|
+
// The one eval this mode grades by, as a suite of its cases.
|
|
1526
|
+
const suites = () => {
|
|
1527
|
+
const j = Math.max(0, core.evalsOf(run).findIndex(t => core.evalsDataset({ evals: [t] })));
|
|
1528
|
+
return [{ name: `${core.targetLabel(run, 0)} › ${core.evalLabel(run, j)}`, cases: [
|
|
1529
|
+
...rows.map(r => ({ name: r.id, ...(r.pass ? {} : { failure: r.reasons.join(" · ") || "failed" }) })),
|
|
1530
|
+
...unrun.map(u => { const at = u.indexOf(": "); return { name: u.slice(0, at), error: u.slice(at + 2) }; }),
|
|
1531
|
+
] }];
|
|
1532
|
+
};
|
|
1533
|
+
writeReports(o, report, suites);
|
|
1534
|
+
process.exitCode = exitOf(verdict);
|
|
1280
1535
|
}
|
|
1281
1536
|
|
|
1282
1537
|
main().catch(e => {
|