evals-lab 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +77 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +353 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/kinds/list.mjs +2 -1
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +220 -83
- package/lab/server.py +617 -176
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/lab/run-evals.js
CHANGED
|
@@ -116,13 +116,17 @@ const USAGE = `Usage: node run-evals.js [options]
|
|
|
116
116
|
file; its content.files, when not empty, is the file
|
|
117
117
|
list a case has to be in, as --files says.
|
|
118
118
|
--samples <dir> where the images are. Default: the lab's samples/.
|
|
119
|
-
--dataset <file> the
|
|
120
|
-
({"cases"}), or the file the lab's Export writes.
|
|
119
|
+
--dataset <file> the one eval group a graded run is scored against: its
|
|
120
|
+
body ({"cases"}), or the file the lab's Export writes.
|
|
121
121
|
Versions 1 to 3 are read too (a prompt they hold is
|
|
122
122
|
not used: --prompt names it); a version-2
|
|
123
123
|
body's rules are what an older run's jobs, and a run
|
|
124
124
|
without --pipeline, read replies under. Its cases grade.
|
|
125
|
-
Required for anything graded.
|
|
125
|
+
Required for anything graded (or --groups, for a --run).
|
|
126
|
+
--groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
|
|
127
|
+
grades against each one its evals link (§17), so a run
|
|
128
|
+
linking several groups is handed them here instead of the
|
|
129
|
+
one --dataset. For --run and --rescore; not with --dataset.
|
|
126
130
|
--source <dir> the same, named the way a run names it: the directory a
|
|
127
131
|
Source is stored at. --samples and --source are one
|
|
128
132
|
flag by two names; both together are refused.
|
|
@@ -204,7 +208,7 @@ function broken(msg) {
|
|
|
204
208
|
|
|
205
209
|
function parseArgs(argv) {
|
|
206
210
|
const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
|
|
207
|
-
replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
211
|
+
groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
208
212
|
timeout: null, progress: false, cancelFile: "",
|
|
209
213
|
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
210
214
|
rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
|
|
@@ -225,6 +229,7 @@ function parseArgs(argv) {
|
|
|
225
229
|
case "--pipeline": o.pipeline = value(); break;
|
|
226
230
|
case "--samples": o.samples = value(); break;
|
|
227
231
|
case "--dataset": o.dataset = value(); break;
|
|
232
|
+
case "--groups": o.groups = value(); break;
|
|
228
233
|
case "--replies": o.replies = value(); break;
|
|
229
234
|
case "--source": o.source = value(); break;
|
|
230
235
|
case "--files": o.files = value(); break;
|
|
@@ -281,10 +286,10 @@ const inRepo = p => {
|
|
|
281
286
|
const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
|
|
282
287
|
const isText = v => typeof v === "string";
|
|
283
288
|
|
|
284
|
-
// [
|
|
285
|
-
//
|
|
286
|
-
//
|
|
287
|
-
function readRun(file,
|
|
289
|
+
// [gctx] is the grading context (gradingContext): the eval group bodies a run
|
|
290
|
+
// grades against -- one, as --dataset, or a body per group kept with the run,
|
|
291
|
+
// as --groups (§17). The run reads each link's body by its reference.
|
|
292
|
+
function readRun(file, gctx) {
|
|
288
293
|
let doc;
|
|
289
294
|
try {
|
|
290
295
|
// A run queued before the current version is read as one of today's:
|
|
@@ -292,12 +297,20 @@ function readRun(file, dataset) {
|
|
|
292
297
|
// submitted with. A v2 run's list jobs read their replies under the
|
|
293
298
|
// rules of the dataset it was graded against -- the body handed over.
|
|
294
299
|
doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
|
|
295
|
-
{ rulesFor: () => rulesOf(
|
|
300
|
+
{ rulesFor: () => rulesOf(gctx.primary()) });
|
|
296
301
|
} catch (e) {
|
|
297
302
|
broken(`${file}: ${e.message}`);
|
|
298
303
|
}
|
|
299
|
-
|
|
300
|
-
|
|
304
|
+
// Every Library group the evals name -- a link's group, a private group's
|
|
305
|
+
// Cases from -- each as validatePipeline reads one. Pins were resolved at
|
|
306
|
+
// submit (the reference carries `n`), so they are not checked again here.
|
|
307
|
+
const groups = [], seen = new Set();
|
|
308
|
+
for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
|
|
309
|
+
if (seen.has(ref.id)) continue;
|
|
310
|
+
seen.add(ref.id);
|
|
311
|
+
groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
|
|
312
|
+
}
|
|
313
|
+
const bad = core.validatePipeline(doc, { groups });
|
|
301
314
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
302
315
|
return doc;
|
|
303
316
|
}
|
|
@@ -305,9 +318,25 @@ function readRun(file, dataset) {
|
|
|
305
318
|
// The rules a version-2 body held, by the body readDataset made of it: what
|
|
306
319
|
// an older run's jobs read replies under. A body of today's holds none.
|
|
307
320
|
const heldRules = new WeakMap();
|
|
308
|
-
const rulesOf = (
|
|
321
|
+
const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
|
|
322
|
+
|
|
323
|
+
// An eval group's body as one of today's, with its version-2 rules taken
|
|
324
|
+
// first: a bare body, or the body the lab's Export wraps, unwrapped by its
|
|
325
|
+
// caller. [where] names it in a refusal.
|
|
326
|
+
function bodyFrom(doc, where) {
|
|
327
|
+
if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
|
|
328
|
+
broken(`${where}: rules has to be null or { "rules": [...] }`);
|
|
329
|
+
}
|
|
330
|
+
const rules = core.datasetRules(doc);
|
|
331
|
+
const body = core.upgradeDatasetBody(doc);
|
|
332
|
+
if (!isObject(body) || !Array.isArray(body.cases)) {
|
|
333
|
+
broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
|
|
334
|
+
}
|
|
335
|
+
if (rules) heldRules.set(body, rules);
|
|
336
|
+
return body;
|
|
337
|
+
}
|
|
309
338
|
|
|
310
|
-
// The dataset --dataset names:
|
|
339
|
+
// The dataset --dataset names: an eval group's body, or the file the lab's
|
|
311
340
|
// Export writes for one, as one of today's.
|
|
312
341
|
function readDataset(file) {
|
|
313
342
|
if (!file) return null;
|
|
@@ -328,29 +357,51 @@ function readDataset(file) {
|
|
|
328
357
|
if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
|
|
329
358
|
doc = isObject(doc.group) ? doc.group.body : null;
|
|
330
359
|
}
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
360
|
+
return bodyFrom(doc, file);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
// The bodies a run kept, as --groups hands them (§17): a JSON object of
|
|
364
|
+
// `<id>@<n>` to body -- the group's id and the version it graded with, its id
|
|
365
|
+
// alone for a run from before the lab numbered them -- each read as today's.
|
|
366
|
+
function readGroups(file) {
|
|
367
|
+
let raw;
|
|
368
|
+
try {
|
|
369
|
+
raw = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
370
|
+
} catch (e) {
|
|
371
|
+
broken(`${file}: ${e.message}`);
|
|
340
372
|
}
|
|
341
|
-
if (
|
|
342
|
-
|
|
373
|
+
if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
|
|
374
|
+
const bodies = new Map();
|
|
375
|
+
for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
|
|
376
|
+
return bodies;
|
|
343
377
|
}
|
|
344
378
|
|
|
345
|
-
//
|
|
346
|
-
//
|
|
347
|
-
const
|
|
379
|
+
// The key a run keeps a group's body under, mirroring the server's group_key:
|
|
380
|
+
// `<id>@<n>`, or the id alone for a reference from before versions.
|
|
381
|
+
const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
|
|
382
|
+
|
|
383
|
+
// The eval groups a --run grades against: one body, as --dataset, or a body
|
|
384
|
+
// per group, as --groups -- never both. `resolve` gives a link's body by its
|
|
385
|
+
// reference; `primary` the one a single-group run names, for a v2 run's rules.
|
|
386
|
+
function gradingContext(o) {
|
|
387
|
+
if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
|
|
388
|
+
if (o.groups) {
|
|
389
|
+
const bodies = readGroups(o.groups);
|
|
390
|
+
const first = bodies.values().next().value ?? null;
|
|
391
|
+
return { single: null, bodies, primary: () => first,
|
|
392
|
+
resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
|
|
393
|
+
}
|
|
394
|
+
const single = readDataset(o.dataset);
|
|
395
|
+
return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
|
|
396
|
+
}
|
|
348
397
|
|
|
349
398
|
// The cases a graded run is scored against, by the item each names: exactly,
|
|
350
|
-
// as the Source names it.
|
|
351
|
-
|
|
399
|
+
// as the Source names it. Built from the one group a single-group run names,
|
|
400
|
+
// for an eval type of its own that reads a case (a plugin's); a `group` eval
|
|
401
|
+
// reads each group's own case, by the item's name, as it grades.
|
|
402
|
+
function gradedBy(body) {
|
|
352
403
|
const out = new Map();
|
|
353
|
-
for (const c of core.gradedSetFrom(
|
|
404
|
+
for (const c of core.gradedSetFrom(body || {})) {
|
|
354
405
|
if (!c.todo) out.set(c.item, c);
|
|
355
406
|
}
|
|
356
407
|
return out;
|
|
@@ -499,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
|
|
|
499
550
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
500
551
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
501
552
|
}
|
|
553
|
+
// The Recorded target replays each call's recorded reply, read as the flow
|
|
554
|
+
// reads one (like send, but sending nothing); it does not go through the
|
|
555
|
+
// model Read-as wrapper below -- the recorded body is read directly.
|
|
556
|
+
if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
|
|
557
|
+
return () => {
|
|
558
|
+
const r = record ?? textRecord(text);
|
|
559
|
+
const result = r && r.result;
|
|
560
|
+
if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
|
|
561
|
+
try {
|
|
562
|
+
const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
|
|
563
|
+
{ scope: r.scope || {}, now: r.at ?? null });
|
|
564
|
+
return { raw: read.raw, said: read.said, ms: 0, conn: null };
|
|
565
|
+
} catch (e) {
|
|
566
|
+
return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
|
|
567
|
+
}
|
|
568
|
+
};
|
|
569
|
+
}
|
|
502
570
|
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
503
|
-
? async sent => core.localAnswer(c, text, sent)
|
|
571
|
+
? async sent => core.localAnswer(c, text, sent, record)
|
|
504
572
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
505
573
|
if (!step?.readAs || !step?.step) return call;
|
|
506
574
|
return async (sent, url) => {
|
|
@@ -650,40 +718,81 @@ function readUtf8(file) {
|
|
|
650
718
|
const outcomesOf = (run, items, settled) =>
|
|
651
719
|
core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
|
|
652
720
|
|
|
721
|
+
// A link's Whole run verdict, per Target, keyed by the eval's id: a Library
|
|
722
|
+
// group's `run` metrics read over the replies so far (readGroupRun). A private
|
|
723
|
+
// group's Whole run is scenarioEvals' own (it holds the body); a link's body
|
|
724
|
+
// is held apart, in --groups, so reading it beside the link's items is the
|
|
725
|
+
// runner's (§17). [resolve] gives a link's body by its reference.
|
|
726
|
+
function wholeRunsOf(run, items, resolve) {
|
|
727
|
+
const kind = core.lastKind(run);
|
|
728
|
+
return core.targetsOf(run).map((_, i) => {
|
|
729
|
+
const out = {};
|
|
730
|
+
const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
|
|
731
|
+
for (const t of core.evalsOf(run)) {
|
|
732
|
+
// A private whole-run group is already scenarioEvals' verdict; here is
|
|
733
|
+
// the run beside a link's (or private group's) own items.
|
|
734
|
+
if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
|
|
735
|
+
const body = core.groupOf(t, { group: resolve });
|
|
736
|
+
if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
|
|
737
|
+
}
|
|
738
|
+
return out;
|
|
739
|
+
});
|
|
740
|
+
}
|
|
741
|
+
|
|
653
742
|
/**
|
|
654
743
|
* One eval's verdict on one target, for the build: its own rule -- every
|
|
655
744
|
* item it read passed, or a whole-run eval's verdict -- with [minPass] the
|
|
656
745
|
* share of items that is enough instead. It only relaxes: an eval whose
|
|
657
746
|
* every item passed passes whatever the share. One an earlier failure
|
|
658
747
|
* stopped is skipped -- that failure is the run's verdict already -- and one
|
|
659
|
-
* that had nothing to read, a
|
|
748
|
+
* that had nothing to read, a group grading none of the run's items, says
|
|
660
749
|
* none: the run ran in full, which is what the server's status reads from
|
|
661
|
-
* the exit code, so it is not incomplete either.
|
|
750
|
+
* the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
|
|
751
|
+
* run verdict, where its group reads one beside its items (§17): the group
|
|
752
|
+
* passes when both its items and its Whole run do, and either failing fails it.
|
|
662
753
|
*/
|
|
663
|
-
function evalVerdict(o, settled, minPass) {
|
|
754
|
+
function evalVerdict(o, settled, minPass, wholeRun) {
|
|
664
755
|
if (o.skipped) return "skipped";
|
|
665
756
|
if (!settled) return "incomplete";
|
|
666
757
|
if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
758
|
+
const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
|
|
759
|
+
: o.passed === o.ran ? "pass"
|
|
760
|
+
: minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
|
|
761
|
+
if (!wholeRun) return item;
|
|
762
|
+
const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
|
|
763
|
+
if (item === "fail" || whole === "fail") return "fail";
|
|
764
|
+
if (item === "pass" || whole === "pass") return "pass";
|
|
765
|
+
return item === "skipped" || whole === "skipped" ? "skipped" : "none";
|
|
670
766
|
}
|
|
671
767
|
|
|
672
|
-
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
|
|
673
|
-
//
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
768
|
+
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
|
|
769
|
+
// per-item eval's counts -- and, where a link reads a Whole run beside its
|
|
770
|
+
// items, that verdict too -- with the build's verdict on each. [wholes] is
|
|
771
|
+
// wholeRunsOf, one map per target.
|
|
772
|
+
const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
|
|
773
|
+
Object.fromEntries(evals.map(o => {
|
|
774
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
775
|
+
return [o.id, {
|
|
776
|
+
name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
|
|
777
|
+
...(o.whole
|
|
778
|
+
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
779
|
+
: { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
|
|
780
|
+
...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
|
|
781
|
+
}];
|
|
782
|
+
})));
|
|
783
|
+
|
|
784
|
+
/** Each Target's verdict under the overall pass rule (§17), from its groups'
|
|
785
|
+
verdicts: a group that passed counts, one that failed counts against, and
|
|
786
|
+
one that read nothing or was skipped is neither. */
|
|
787
|
+
const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
|
|
788
|
+
Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
|
|
789
|
+
|
|
790
|
+
/** The run's verdict from its evals' and the overall rule: a run that did not
|
|
791
|
+
finish is incomplete, whatever its evals read so far; then it passes when
|
|
792
|
+
every Target does under the rule. */
|
|
793
|
+
function runVerdict(verdicts, complete, pass) {
|
|
685
794
|
if (!complete) return "incomplete";
|
|
686
|
-
return verdicts
|
|
795
|
+
return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
|
|
687
796
|
}
|
|
688
797
|
|
|
689
798
|
const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
|
|
@@ -699,11 +808,13 @@ function reasonOf(score) {
|
|
|
699
808
|
return missed.length ? `missed ${missed.join(", ")}` : "failed";
|
|
700
809
|
}
|
|
701
810
|
|
|
702
|
-
/** A --run's or a --rescore's suites: each target's evals over [items]
|
|
703
|
-
|
|
811
|
+
/** A --run's or a --rescore's suites: each target's evals over [items], with a
|
|
812
|
+
Whole run case where a link reads one beside its items ([wholes]). */
|
|
813
|
+
function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
|
|
704
814
|
return outcomes.flatMap((evals, i) => evals.map(o => {
|
|
705
815
|
const name = `${core.targetLabel(run, i)} › ${o.label}`;
|
|
706
|
-
const
|
|
816
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
817
|
+
const v = evalVerdict(o, settled, minPass, wr);
|
|
707
818
|
if (o.whole) {
|
|
708
819
|
// Settled over the run rather than item by item: one case, the run.
|
|
709
820
|
const c = { name: o.label };
|
|
@@ -713,7 +824,7 @@ function runSuites(run, outcomes, items, settled, minPass) {
|
|
|
713
824
|
else if (v === "fail") c.failure = o.verdict.detail || "failed";
|
|
714
825
|
return { name, cases: [c] };
|
|
715
826
|
}
|
|
716
|
-
|
|
827
|
+
const cases = items.map((it, x) => {
|
|
717
828
|
const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
|
|
718
829
|
const read = o.items[x];
|
|
719
830
|
if (!it) c.error = "not run";
|
|
@@ -722,7 +833,15 @@ function runSuites(run, outcomes, items, settled, minPass) {
|
|
|
722
833
|
else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
|
|
723
834
|
else if (!read.pass) c.failure = reasonOf(read);
|
|
724
835
|
return c;
|
|
725
|
-
})
|
|
836
|
+
});
|
|
837
|
+
if (wr) {
|
|
838
|
+
const c = { name: "Whole run" };
|
|
839
|
+
if (!settled) c.error = "the run did not finish";
|
|
840
|
+
else if (!wr.ran) c.skipped = "nothing to read";
|
|
841
|
+
else if (!wr.pass) c.failure = wr.detail || "failed";
|
|
842
|
+
cases.push(c);
|
|
843
|
+
}
|
|
844
|
+
return { name, cases };
|
|
726
845
|
}));
|
|
727
846
|
}
|
|
728
847
|
|
|
@@ -740,8 +859,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
|
740
859
|
});
|
|
741
860
|
|
|
742
861
|
async function runSnapshot(o) {
|
|
743
|
-
const
|
|
744
|
-
const run = readRun(o.run,
|
|
862
|
+
const gctx = gradingContext(o);
|
|
863
|
+
const run = readRun(o.run, gctx);
|
|
745
864
|
// One item a pipeline holds itself (Text, or Prompt only's bare one), or
|
|
746
865
|
// the Source's files.
|
|
747
866
|
const content = core.contentOf(run);
|
|
@@ -753,8 +872,9 @@ async function runSnapshot(o) {
|
|
|
753
872
|
broken(`${o.run} names no files and no text, so there is nothing to run`);
|
|
754
873
|
}
|
|
755
874
|
|
|
756
|
-
// The
|
|
757
|
-
|
|
875
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
876
|
+
// reads a case; a `group` eval reads each group's case by the item's name.
|
|
877
|
+
const graded = gradedBy(gctx.primary());
|
|
758
878
|
|
|
759
879
|
const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
|
|
760
880
|
: text.trim() ? [{ name: null, kind: "text", text }]
|
|
@@ -780,7 +900,10 @@ async function runSnapshot(o) {
|
|
|
780
900
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
781
901
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
782
902
|
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
783
|
-
|
|
903
|
+
// The Recorded target replays the recorded reply, so a metric that compares
|
|
904
|
+
// with the recorded reply has nothing to say of it (n/a).
|
|
905
|
+
const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
|
|
906
|
+
return { stages, tokens, connections, links, calls, cells, recordedTarget };
|
|
784
907
|
});
|
|
785
908
|
|
|
786
909
|
let results = [];
|
|
@@ -842,8 +965,9 @@ async function runSnapshot(o) {
|
|
|
842
965
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
843
966
|
// text item a dataset grades is graded like an image.
|
|
844
967
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
845
|
-
production = core.
|
|
846
|
-
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
968
|
+
production = core.recordedReplyOf(run, record);
|
|
969
|
+
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
970
|
+
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
|
|
847
971
|
}
|
|
848
972
|
// Production's reply to the item, where its record holds one: what a
|
|
849
973
|
// metric compares with, kept so a re-score reads it again.
|
|
@@ -862,17 +986,19 @@ async function runSnapshot(o) {
|
|
|
862
986
|
const complete = !cancelled && !unrun.length
|
|
863
987
|
&& ranItems.length === total && ranItems.every(it => it && !it.unrun);
|
|
864
988
|
const outcomes = outcomesOf(run, ranItems, complete);
|
|
865
|
-
const
|
|
989
|
+
const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
|
|
990
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
866
991
|
const report = {
|
|
867
992
|
runner: RUNNER,
|
|
868
993
|
ranAt: new Date().toISOString(),
|
|
869
|
-
verdict: runVerdict(verdicts, complete),
|
|
994
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
870
995
|
...(cancelled ? { cancelled: true } : {}),
|
|
871
996
|
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
872
997
|
run: {
|
|
873
998
|
name: run.name ?? null,
|
|
874
999
|
evals: run.evals,
|
|
875
1000
|
verdicts,
|
|
1001
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
876
1002
|
scenarios: scenariosOf(run),
|
|
877
1003
|
items: ranItems,
|
|
878
1004
|
},
|
|
@@ -883,19 +1009,25 @@ async function runSnapshot(o) {
|
|
|
883
1009
|
say(`${report.runner} — run ${report.run.name ?? ""}`);
|
|
884
1010
|
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
885
1011
|
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
886
|
-
sayVerdicts(say, run, verdicts);
|
|
1012
|
+
sayVerdicts(say, run, verdicts, report.run.pass);
|
|
887
1013
|
say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
|
|
888
1014
|
|
|
889
|
-
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
|
|
1015
|
+
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
|
|
890
1016
|
process.exitCode = exitOf(report.verdict);
|
|
891
1017
|
}
|
|
892
1018
|
|
|
893
|
-
/** Each target's
|
|
894
|
-
|
|
1019
|
+
/** Each target's groups and their verdicts, a line each, and the overall rule
|
|
1020
|
+
beneath them where [pass] gives a figure per target. */
|
|
1021
|
+
function sayVerdicts(say, run, verdicts, pass) {
|
|
895
1022
|
verdicts.forEach((evals, i) => {
|
|
896
1023
|
for (const e of Object.values(evals)) {
|
|
897
1024
|
const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
|
|
898
|
-
|
|
1025
|
+
const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
|
|
1026
|
+
say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
|
|
1027
|
+
}
|
|
1028
|
+
if (pass) {
|
|
1029
|
+
const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
|
|
1030
|
+
say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
|
|
899
1031
|
}
|
|
900
1032
|
});
|
|
901
1033
|
}
|
|
@@ -916,8 +1048,8 @@ function writeReports(o, report, suites) {
|
|
|
916
1048
|
* a --run's, with the stored results as its items, and a whole-run eval's
|
|
917
1049
|
* verdict settled the same way. */
|
|
918
1050
|
async function runRescore(o){
|
|
919
|
-
const
|
|
920
|
-
const run = readRun(o.run,
|
|
1051
|
+
const gctx = gradingContext(o);
|
|
1052
|
+
const run = readRun(o.run, gctx);
|
|
921
1053
|
let results;
|
|
922
1054
|
try {
|
|
923
1055
|
results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
|
|
@@ -929,18 +1061,18 @@ async function runRescore(o){
|
|
|
929
1061
|
}
|
|
930
1062
|
results = core.upgradeResults(results);
|
|
931
1063
|
|
|
932
|
-
// The
|
|
933
|
-
//
|
|
934
|
-
// say what the
|
|
935
|
-
const graded = gradedBy(
|
|
1064
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
1065
|
+
// reads a case. A file no group grades keeps its reply and has no score --
|
|
1066
|
+
// honestly: nothing else would say what the groups cover.
|
|
1067
|
+
const graded = gradedBy(gctx.primary());
|
|
936
1068
|
|
|
937
1069
|
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
938
|
-
// An eval that reads every item (
|
|
1070
|
+
// An eval that reads every item (a group) re-reads one with no case
|
|
939
1071
|
// too, against the production reply the item kept.
|
|
940
1072
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
941
1073
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
942
1074
|
const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
|
|
943
|
-
const more = { production: it.production ?? null, grader: graderFor(run), group:
|
|
1075
|
+
const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
|
|
944
1076
|
return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
|
|
945
1077
|
(isObject(side) && side.res)
|
|
946
1078
|
? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
|
|
@@ -949,23 +1081,25 @@ async function runRescore(o){
|
|
|
949
1081
|
// A stored run that stopped short re-scores as what it is: incomplete.
|
|
950
1082
|
const complete = items.every(it => isObject(it) && !it.unrun);
|
|
951
1083
|
const outcomes = outcomesOf(run, items, complete);
|
|
952
|
-
const
|
|
1084
|
+
const wholes = wholeRunsOf(run, items, gctx.resolve);
|
|
1085
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
953
1086
|
const report = {
|
|
954
1087
|
runner: RUNNER,
|
|
955
1088
|
ranAt: new Date().toISOString(),
|
|
956
|
-
verdict: runVerdict(verdicts, complete),
|
|
1089
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
957
1090
|
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
958
1091
|
rescored: true,
|
|
959
1092
|
run: {
|
|
960
1093
|
name: run.name ?? null,
|
|
961
1094
|
evals: run.evals,
|
|
962
1095
|
verdicts,
|
|
1096
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
963
1097
|
scenarios: scenariosOf(run),
|
|
964
1098
|
items,
|
|
965
1099
|
},
|
|
966
1100
|
};
|
|
967
1101
|
|
|
968
|
-
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
|
|
1102
|
+
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
|
|
969
1103
|
process.exitCode = exitOf(report.verdict);
|
|
970
1104
|
}
|
|
971
1105
|
|
|
@@ -1066,9 +1200,12 @@ async function main() {
|
|
|
1066
1200
|
}
|
|
1067
1201
|
|
|
1068
1202
|
// The dataset the run is graded against. Everything below grades, so there
|
|
1069
|
-
// is no run without one.
|
|
1203
|
+
// is no run without one. --groups hands a --run the bodies it kept; a
|
|
1204
|
+
// --pipeline or --prompt grades one scenario against one, named by --dataset.
|
|
1205
|
+
if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
|
|
1070
1206
|
if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
|
|
1071
1207
|
const dataset = readDataset(o.dataset);
|
|
1208
|
+
const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
|
|
1072
1209
|
|
|
1073
1210
|
// The run to grade. A --pipeline document is one scenario graded against
|
|
1074
1211
|
// its dataset; without one it is the stage this script always ran -- the
|
|
@@ -1078,7 +1215,7 @@ async function main() {
|
|
|
1078
1215
|
broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
|
|
1079
1216
|
+ "the lab's Prompt library does.");
|
|
1080
1217
|
}
|
|
1081
|
-
const run = o.pipeline ? readRun(o.pipeline,
|
|
1218
|
+
const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
|
|
1082
1219
|
if (o.pipeline) {
|
|
1083
1220
|
if (core.targetsOf(run).length !== 1) {
|
|
1084
1221
|
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|