evals-lab 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +69 -31
- package/lab/kinds/list.mjs +2 -1
- package/lab/run-evals.js +197 -80
- package/lab/server.py +34 -12
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +3 -0
- package/lab/web/dist/assets/main-DDoeU6hq.css +1 -0
- package/lab/web/dist/assets/main-nz6Q4jVm.js +21 -0
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +59 -0
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/lab/run-evals.js
CHANGED
|
@@ -116,13 +116,17 @@ const USAGE = `Usage: node run-evals.js [options]
|
|
|
116
116
|
file; its content.files, when not empty, is the file
|
|
117
117
|
list a case has to be in, as --files says.
|
|
118
118
|
--samples <dir> where the images are. Default: the lab's samples/.
|
|
119
|
-
--dataset <file> the
|
|
120
|
-
({"cases"}), or the file the lab's Export writes.
|
|
119
|
+
--dataset <file> the one eval group a graded run is scored against: its
|
|
120
|
+
body ({"cases"}), or the file the lab's Export writes.
|
|
121
121
|
Versions 1 to 3 are read too (a prompt they hold is
|
|
122
122
|
not used: --prompt names it); a version-2
|
|
123
123
|
body's rules are what an older run's jobs, and a run
|
|
124
124
|
without --pipeline, read replies under. Its cases grade.
|
|
125
|
-
Required for anything graded.
|
|
125
|
+
Required for anything graded (or --groups, for a --run).
|
|
126
|
+
--groups <file> the eval group bodies a --run kept, by "<id>@<n>": a run
|
|
127
|
+
grades against each one its evals link (§17), so a run
|
|
128
|
+
linking several groups is handed them here instead of the
|
|
129
|
+
one --dataset. For --run and --rescore; not with --dataset.
|
|
126
130
|
--source <dir> the same, named the way a run names it: the directory a
|
|
127
131
|
Source is stored at. --samples and --source are one
|
|
128
132
|
flag by two names; both together are refused.
|
|
@@ -204,7 +208,7 @@ function broken(msg) {
|
|
|
204
208
|
|
|
205
209
|
function parseArgs(argv) {
|
|
206
210
|
const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
|
|
207
|
-
replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
211
|
+
groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
208
212
|
timeout: null, progress: false, cancelFile: "",
|
|
209
213
|
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
210
214
|
rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
|
|
@@ -225,6 +229,7 @@ function parseArgs(argv) {
|
|
|
225
229
|
case "--pipeline": o.pipeline = value(); break;
|
|
226
230
|
case "--samples": o.samples = value(); break;
|
|
227
231
|
case "--dataset": o.dataset = value(); break;
|
|
232
|
+
case "--groups": o.groups = value(); break;
|
|
228
233
|
case "--replies": o.replies = value(); break;
|
|
229
234
|
case "--source": o.source = value(); break;
|
|
230
235
|
case "--files": o.files = value(); break;
|
|
@@ -281,10 +286,10 @@ const inRepo = p => {
|
|
|
281
286
|
const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
|
|
282
287
|
const isText = v => typeof v === "string";
|
|
283
288
|
|
|
284
|
-
// [
|
|
285
|
-
//
|
|
286
|
-
//
|
|
287
|
-
function readRun(file,
|
|
289
|
+
// [gctx] is the grading context (gradingContext): the eval group bodies a run
|
|
290
|
+
// grades against -- one, as --dataset, or a body per group kept with the run,
|
|
291
|
+
// as --groups (§17). The run reads each link's body by its reference.
|
|
292
|
+
function readRun(file, gctx) {
|
|
288
293
|
let doc;
|
|
289
294
|
try {
|
|
290
295
|
// A run queued before the current version is read as one of today's:
|
|
@@ -292,12 +297,20 @@ function readRun(file, dataset) {
|
|
|
292
297
|
// submitted with. A v2 run's list jobs read their replies under the
|
|
293
298
|
// rules of the dataset it was graded against -- the body handed over.
|
|
294
299
|
doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
|
|
295
|
-
{ rulesFor: () => rulesOf(
|
|
300
|
+
{ rulesFor: () => rulesOf(gctx.primary()) });
|
|
296
301
|
} catch (e) {
|
|
297
302
|
broken(`${file}: ${e.message}`);
|
|
298
303
|
}
|
|
299
|
-
|
|
300
|
-
|
|
304
|
+
// Every Library group the evals name -- a link's group, a private group's
|
|
305
|
+
// Cases from -- each as validatePipeline reads one. Pins were resolved at
|
|
306
|
+
// submit (the reference carries `n`), so they are not checked again here.
|
|
307
|
+
const groups = [], seen = new Set();
|
|
308
|
+
for (const ref of (isObject(doc) ? core.evalsOf(doc).map(core.casesRef).filter(Boolean) : [])) {
|
|
309
|
+
if (seen.has(ref.id)) continue;
|
|
310
|
+
seen.add(ref.id);
|
|
311
|
+
groups.push({ id: ref.id, name: ref.name, source: gctx.resolve(ref)?.source ?? null });
|
|
312
|
+
}
|
|
313
|
+
const bad = core.validatePipeline(doc, { groups });
|
|
301
314
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
302
315
|
return doc;
|
|
303
316
|
}
|
|
@@ -305,9 +318,25 @@ function readRun(file, dataset) {
|
|
|
305
318
|
// The rules a version-2 body held, by the body readDataset made of it: what
|
|
306
319
|
// an older run's jobs read replies under. A body of today's holds none.
|
|
307
320
|
const heldRules = new WeakMap();
|
|
308
|
-
const rulesOf = (
|
|
321
|
+
const rulesOf = (body) => (body && heldRules.get(body)) ?? null;
|
|
322
|
+
|
|
323
|
+
// An eval group's body as one of today's, with its version-2 rules taken
|
|
324
|
+
// first: a bare body, or the body the lab's Export wraps, unwrapped by its
|
|
325
|
+
// caller. [where] names it in a refusal.
|
|
326
|
+
function bodyFrom(doc, where) {
|
|
327
|
+
if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
|
|
328
|
+
broken(`${where}: rules has to be null or { "rules": [...] }`);
|
|
329
|
+
}
|
|
330
|
+
const rules = core.datasetRules(doc);
|
|
331
|
+
const body = core.upgradeDatasetBody(doc);
|
|
332
|
+
if (!isObject(body) || !Array.isArray(body.cases)) {
|
|
333
|
+
broken(`${where} is not an eval group: its body has a cases list, or it is the lab's Export of one`);
|
|
334
|
+
}
|
|
335
|
+
if (rules) heldRules.set(body, rules);
|
|
336
|
+
return body;
|
|
337
|
+
}
|
|
309
338
|
|
|
310
|
-
// The dataset --dataset names:
|
|
339
|
+
// The dataset --dataset names: an eval group's body, or the file the lab's
|
|
311
340
|
// Export writes for one, as one of today's.
|
|
312
341
|
function readDataset(file) {
|
|
313
342
|
if (!file) return null;
|
|
@@ -328,29 +357,51 @@ function readDataset(file) {
|
|
|
328
357
|
if (doc.version !== 7) broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads version 7`);
|
|
329
358
|
doc = isObject(doc.group) ? doc.group.body : null;
|
|
330
359
|
}
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
360
|
+
return bodyFrom(doc, file);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
// The bodies a run kept, as --groups hands them (§17): a JSON object of
|
|
364
|
+
// `<id>@<n>` to body -- the group's id and the version it graded with, its id
|
|
365
|
+
// alone for a run from before the lab numbered them -- each read as today's.
|
|
366
|
+
function readGroups(file) {
|
|
367
|
+
let raw;
|
|
368
|
+
try {
|
|
369
|
+
raw = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
370
|
+
} catch (e) {
|
|
371
|
+
broken(`${file}: ${e.message}`);
|
|
340
372
|
}
|
|
341
|
-
if (
|
|
342
|
-
|
|
373
|
+
if (!isObject(raw)) broken(`${file}: the kept groups are a JSON object of "<id>@<n>": body`);
|
|
374
|
+
const bodies = new Map();
|
|
375
|
+
for (const [key, body] of Object.entries(raw)) bodies.set(key, bodyFrom(body, `${file} [${key}]`));
|
|
376
|
+
return bodies;
|
|
343
377
|
}
|
|
344
378
|
|
|
345
|
-
//
|
|
346
|
-
//
|
|
347
|
-
const
|
|
379
|
+
// The key a run keeps a group's body under, mirroring the server's group_key:
|
|
380
|
+
// `<id>@<n>`, or the id alone for a reference from before versions.
|
|
381
|
+
const groupKey = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
|
|
382
|
+
|
|
383
|
+
// The eval groups a --run grades against: one body, as --dataset, or a body
|
|
384
|
+
// per group, as --groups -- never both. `resolve` gives a link's body by its
|
|
385
|
+
// reference; `primary` the one a single-group run names, for a v2 run's rules.
|
|
386
|
+
function gradingContext(o) {
|
|
387
|
+
if (o.groups && o.dataset) broken("--groups and --dataset name the groups two ways; pass one.");
|
|
388
|
+
if (o.groups) {
|
|
389
|
+
const bodies = readGroups(o.groups);
|
|
390
|
+
const first = bodies.values().next().value ?? null;
|
|
391
|
+
return { single: null, bodies, primary: () => first,
|
|
392
|
+
resolve: (ref) => bodies.get(groupKey(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined };
|
|
393
|
+
}
|
|
394
|
+
const single = readDataset(o.dataset);
|
|
395
|
+
return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
|
|
396
|
+
}
|
|
348
397
|
|
|
349
398
|
// The cases a graded run is scored against, by the item each names: exactly,
|
|
350
|
-
// as the Source names it.
|
|
351
|
-
|
|
399
|
+
// as the Source names it. Built from the one group a single-group run names,
|
|
400
|
+
// for an eval type of its own that reads a case (a plugin's); a `group` eval
|
|
401
|
+
// reads each group's own case, by the item's name, as it grades.
|
|
402
|
+
function gradedBy(body) {
|
|
352
403
|
const out = new Map();
|
|
353
|
-
for (const c of core.gradedSetFrom(
|
|
404
|
+
for (const c of core.gradedSetFrom(body || {})) {
|
|
354
405
|
if (!c.todo) out.set(c.item, c);
|
|
355
406
|
}
|
|
356
407
|
return out;
|
|
@@ -650,40 +701,81 @@ function readUtf8(file) {
|
|
|
650
701
|
const outcomesOf = (run, items, settled) =>
|
|
651
702
|
core.targetsOf(run).map((_, i) => core.scenarioEvals(run, i, items, settled));
|
|
652
703
|
|
|
704
|
+
// A link's Whole run verdict, per Target, keyed by the eval's id: a Library
|
|
705
|
+
// group's `run` metrics read over the replies so far (readGroupRun). A private
|
|
706
|
+
// group's Whole run is scenarioEvals' own (it holds the body); a link's body
|
|
707
|
+
// is held apart, in --groups, so reading it beside the link's items is the
|
|
708
|
+
// runner's (§17). [resolve] gives a link's body by its reference.
|
|
709
|
+
function wholeRunsOf(run, items, resolve) {
|
|
710
|
+
const kind = core.lastKind(run);
|
|
711
|
+
return core.targetsOf(run).map((_, i) => {
|
|
712
|
+
const out = {};
|
|
713
|
+
const ress = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i]?.res : undefined));
|
|
714
|
+
for (const t of core.evalsOf(run)) {
|
|
715
|
+
// A private whole-run group is already scenarioEvals' verdict; here is
|
|
716
|
+
// the run beside a link's (or private group's) own items.
|
|
717
|
+
if (t.type !== "group" || core.isWholeRun(core.EVAL_TYPES.group, t)) continue;
|
|
718
|
+
const body = core.groupOf(t, { group: resolve });
|
|
719
|
+
if (body && Array.isArray(body.run) && body.run.length) out[t.id] = core.readGroupRun(body, ress, kind);
|
|
720
|
+
}
|
|
721
|
+
return out;
|
|
722
|
+
});
|
|
723
|
+
}
|
|
724
|
+
|
|
653
725
|
/**
|
|
654
726
|
* One eval's verdict on one target, for the build: its own rule -- every
|
|
655
727
|
* item it read passed, or a whole-run eval's verdict -- with [minPass] the
|
|
656
728
|
* share of items that is enough instead. It only relaxes: an eval whose
|
|
657
729
|
* every item passed passes whatever the share. One an earlier failure
|
|
658
730
|
* stopped is skipped -- that failure is the run's verdict already -- and one
|
|
659
|
-
* that had nothing to read, a
|
|
731
|
+
* that had nothing to read, a group grading none of the run's items, says
|
|
660
732
|
* none: the run ran in full, which is what the server's status reads from
|
|
661
|
-
* the exit code, so it is not incomplete either.
|
|
733
|
+
* the exit code, so it is not incomplete either. [wholeRun] is a link's Whole
|
|
734
|
+
* run verdict, where its group reads one beside its items (§17): the group
|
|
735
|
+
* passes when both its items and its Whole run do, and either failing fails it.
|
|
662
736
|
*/
|
|
663
|
-
function evalVerdict(o, settled, minPass) {
|
|
737
|
+
function evalVerdict(o, settled, minPass, wholeRun) {
|
|
664
738
|
if (o.skipped) return "skipped";
|
|
665
739
|
if (!settled) return "incomplete";
|
|
666
740
|
if (o.whole) return !o.verdict?.ran ? "none" : o.verdict.pass ? "pass" : "fail";
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
741
|
+
const item = !o.ran ? (o.skippedItems ? "skipped" : "none")
|
|
742
|
+
: o.passed === o.ran ? "pass"
|
|
743
|
+
: minPass != null && o.passed / o.ran >= minPass ? "pass" : "fail";
|
|
744
|
+
if (!wholeRun) return item;
|
|
745
|
+
const whole = !wholeRun.ran ? "none" : wholeRun.pass ? "pass" : "fail";
|
|
746
|
+
if (item === "fail" || whole === "fail") return "fail";
|
|
747
|
+
if (item === "pass" || whole === "pass") return "pass";
|
|
748
|
+
return item === "skipped" || whole === "skipped" ? "skipped" : "none";
|
|
670
749
|
}
|
|
671
750
|
|
|
672
|
-
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict,
|
|
673
|
-
//
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
751
|
+
// Each target's evals, keyed by the eval's id: a whole-run eval's verdict, a
|
|
752
|
+
// per-item eval's counts -- and, where a link reads a Whole run beside its
|
|
753
|
+
// items, that verdict too -- with the build's verdict on each. [wholes] is
|
|
754
|
+
// wholeRunsOf, one map per target.
|
|
755
|
+
const verdictsOf = (outcomes, settled, minPass, wholes = []) => outcomes.map((evals, i) =>
|
|
756
|
+
Object.fromEntries(evals.map(o => {
|
|
757
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
758
|
+
return [o.id, {
|
|
759
|
+
name: o.label, skipped: o.skipped, verdict: evalVerdict(o, settled, minPass, wr),
|
|
760
|
+
...(o.whole
|
|
761
|
+
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
762
|
+
: { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems,
|
|
763
|
+
...(wr ? { wholeRun: { pass: wr.pass, detail: wr.detail, ran: wr.ran } } : {}) }),
|
|
764
|
+
}];
|
|
765
|
+
})));
|
|
766
|
+
|
|
767
|
+
/** Each Target's verdict under the overall pass rule (§17), from its groups'
|
|
768
|
+
verdicts: a group that passed counts, one that failed counts against, and
|
|
769
|
+
one that read nothing or was skipped is neither. */
|
|
770
|
+
const passTargets = (verdicts, pass) => verdicts.map(v => core.passVerdict(pass,
|
|
771
|
+
Object.values(v).map(e => (e.verdict === "pass" ? true : e.verdict === "fail" ? false : null))));
|
|
772
|
+
|
|
773
|
+
/** The run's verdict from its evals' and the overall rule: a run that did not
|
|
774
|
+
finish is incomplete, whatever its evals read so far; then it passes when
|
|
775
|
+
every Target does under the rule. */
|
|
776
|
+
function runVerdict(verdicts, complete, pass) {
|
|
685
777
|
if (!complete) return "incomplete";
|
|
686
|
-
return verdicts
|
|
778
|
+
return passTargets(verdicts, pass).every(Boolean) ? "pass" : "fail";
|
|
687
779
|
}
|
|
688
780
|
|
|
689
781
|
const exitOf = verdict => ({ pass: EXIT.passed, fail: EXIT.failed })[verdict] ?? EXIT.incomplete;
|
|
@@ -699,11 +791,13 @@ function reasonOf(score) {
|
|
|
699
791
|
return missed.length ? `missed ${missed.join(", ")}` : "failed";
|
|
700
792
|
}
|
|
701
793
|
|
|
702
|
-
/** A --run's or a --rescore's suites: each target's evals over [items]
|
|
703
|
-
|
|
794
|
+
/** A --run's or a --rescore's suites: each target's evals over [items], with a
|
|
795
|
+
Whole run case where a link reads one beside its items ([wholes]). */
|
|
796
|
+
function runSuites(run, outcomes, items, settled, minPass, wholes = []) {
|
|
704
797
|
return outcomes.flatMap((evals, i) => evals.map(o => {
|
|
705
798
|
const name = `${core.targetLabel(run, i)} › ${o.label}`;
|
|
706
|
-
const
|
|
799
|
+
const wr = wholes[i] && wholes[i][o.id];
|
|
800
|
+
const v = evalVerdict(o, settled, minPass, wr);
|
|
707
801
|
if (o.whole) {
|
|
708
802
|
// Settled over the run rather than item by item: one case, the run.
|
|
709
803
|
const c = { name: o.label };
|
|
@@ -713,7 +807,7 @@ function runSuites(run, outcomes, items, settled, minPass) {
|
|
|
713
807
|
else if (v === "fail") c.failure = o.verdict.detail || "failed";
|
|
714
808
|
return { name, cases: [c] };
|
|
715
809
|
}
|
|
716
|
-
|
|
810
|
+
const cases = items.map((it, x) => {
|
|
717
811
|
const c = { name: it?.name ?? (items.length > 1 ? `item ${x + 1}` : "the text") };
|
|
718
812
|
const read = o.items[x];
|
|
719
813
|
if (!it) c.error = "not run";
|
|
@@ -722,7 +816,15 @@ function runSuites(run, outcomes, items, settled, minPass) {
|
|
|
722
816
|
else if (core.isSkipped(read)) c.skipped = "an earlier eval failed";
|
|
723
817
|
else if (!read.pass) c.failure = reasonOf(read);
|
|
724
818
|
return c;
|
|
725
|
-
})
|
|
819
|
+
});
|
|
820
|
+
if (wr) {
|
|
821
|
+
const c = { name: "Whole run" };
|
|
822
|
+
if (!settled) c.error = "the run did not finish";
|
|
823
|
+
else if (!wr.ran) c.skipped = "nothing to read";
|
|
824
|
+
else if (!wr.pass) c.failure = wr.detail || "failed";
|
|
825
|
+
cases.push(c);
|
|
826
|
+
}
|
|
827
|
+
return { name, cases };
|
|
726
828
|
}));
|
|
727
829
|
}
|
|
728
830
|
|
|
@@ -740,8 +842,8 @@ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
|
740
842
|
});
|
|
741
843
|
|
|
742
844
|
async function runSnapshot(o) {
|
|
743
|
-
const
|
|
744
|
-
const run = readRun(o.run,
|
|
845
|
+
const gctx = gradingContext(o);
|
|
846
|
+
const run = readRun(o.run, gctx);
|
|
745
847
|
// One item a pipeline holds itself (Text, or Prompt only's bare one), or
|
|
746
848
|
// the Source's files.
|
|
747
849
|
const content = core.contentOf(run);
|
|
@@ -753,8 +855,9 @@ async function runSnapshot(o) {
|
|
|
753
855
|
broken(`${o.run} names no files and no text, so there is nothing to run`);
|
|
754
856
|
}
|
|
755
857
|
|
|
756
|
-
// The
|
|
757
|
-
|
|
858
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
859
|
+
// reads a case; a `group` eval reads each group's case by the item's name.
|
|
860
|
+
const graded = gradedBy(gctx.primary());
|
|
758
861
|
|
|
759
862
|
const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
|
|
760
863
|
: text.trim() ? [{ name: null, kind: "text", text }]
|
|
@@ -843,7 +946,8 @@ async function runSnapshot(o) {
|
|
|
843
946
|
// text item a dataset grades is graded like an image.
|
|
844
947
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
845
948
|
production = core.productionOf(run, record);
|
|
846
|
-
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
949
|
+
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
950
|
+
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
|
|
847
951
|
}
|
|
848
952
|
// Production's reply to the item, where its record holds one: what a
|
|
849
953
|
// metric compares with, kept so a re-score reads it again.
|
|
@@ -862,17 +966,19 @@ async function runSnapshot(o) {
|
|
|
862
966
|
const complete = !cancelled && !unrun.length
|
|
863
967
|
&& ranItems.length === total && ranItems.every(it => it && !it.unrun);
|
|
864
968
|
const outcomes = outcomesOf(run, ranItems, complete);
|
|
865
|
-
const
|
|
969
|
+
const wholes = wholeRunsOf(run, ranItems, gctx.resolve);
|
|
970
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
866
971
|
const report = {
|
|
867
972
|
runner: RUNNER,
|
|
868
973
|
ranAt: new Date().toISOString(),
|
|
869
|
-
verdict: runVerdict(verdicts, complete),
|
|
974
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
870
975
|
...(cancelled ? { cancelled: true } : {}),
|
|
871
976
|
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
872
977
|
run: {
|
|
873
978
|
name: run.name ?? null,
|
|
874
979
|
evals: run.evals,
|
|
875
980
|
verdicts,
|
|
981
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
876
982
|
scenarios: scenariosOf(run),
|
|
877
983
|
items: ranItems,
|
|
878
984
|
},
|
|
@@ -883,19 +989,25 @@ async function runSnapshot(o) {
|
|
|
883
989
|
say(`${report.runner} — run ${report.run.name ?? ""}`);
|
|
884
990
|
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
885
991
|
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
886
|
-
sayVerdicts(say, run, verdicts);
|
|
992
|
+
sayVerdicts(say, run, verdicts, report.run.pass);
|
|
887
993
|
say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled · " : ""}${report.verdict}`);
|
|
888
994
|
|
|
889
|
-
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass));
|
|
995
|
+
writeReports(o, report, () => runSuites(run, outcomes, ranItems, complete, o.minPass, wholes));
|
|
890
996
|
process.exitCode = exitOf(report.verdict);
|
|
891
997
|
}
|
|
892
998
|
|
|
893
|
-
/** Each target's
|
|
894
|
-
|
|
999
|
+
/** Each target's groups and their verdicts, a line each, and the overall rule
|
|
1000
|
+
beneath them where [pass] gives a figure per target. */
|
|
1001
|
+
function sayVerdicts(say, run, verdicts, pass) {
|
|
895
1002
|
verdicts.forEach((evals, i) => {
|
|
896
1003
|
for (const e of Object.values(evals)) {
|
|
897
1004
|
const n = e.ran != null ? ` · ${e.passed} of ${e.ran} passed` : e.detail ? ` · ${e.detail}` : "";
|
|
898
|
-
|
|
1005
|
+
const whole = e.wholeRun ? ` · whole run ${e.wholeRun.pass ? "passed" : "failed"}` : "";
|
|
1006
|
+
say(`${e.verdict.padEnd(10)} ${core.targetLabel(run, i)} › ${e.name}${n}${whole}`);
|
|
1007
|
+
}
|
|
1008
|
+
if (pass) {
|
|
1009
|
+
const rule = pass.rule.mode === "atLeast" ? `at least ${pass.rule.count}` : "all groups";
|
|
1010
|
+
say(`${(pass.targets[i] ? "pass" : "fail").padEnd(10)} ${core.targetLabel(run, i)} › overall (${rule})`);
|
|
899
1011
|
}
|
|
900
1012
|
});
|
|
901
1013
|
}
|
|
@@ -916,8 +1028,8 @@ function writeReports(o, report, suites) {
|
|
|
916
1028
|
* a --run's, with the stored results as its items, and a whole-run eval's
|
|
917
1029
|
* verdict settled the same way. */
|
|
918
1030
|
async function runRescore(o){
|
|
919
|
-
const
|
|
920
|
-
const run = readRun(o.run,
|
|
1031
|
+
const gctx = gradingContext(o);
|
|
1032
|
+
const run = readRun(o.run, gctx);
|
|
921
1033
|
let results;
|
|
922
1034
|
try {
|
|
923
1035
|
results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
|
|
@@ -929,18 +1041,18 @@ async function runRescore(o){
|
|
|
929
1041
|
}
|
|
930
1042
|
results = core.upgradeResults(results);
|
|
931
1043
|
|
|
932
|
-
// The
|
|
933
|
-
//
|
|
934
|
-
// say what the
|
|
935
|
-
const graded = gradedBy(
|
|
1044
|
+
// The one group a single-group run names, for an eval type of its own that
|
|
1045
|
+
// reads a case. A file no group grades keeps its reply and has no score --
|
|
1046
|
+
// honestly: nothing else would say what the groups cover.
|
|
1047
|
+
const graded = gradedBy(gctx.primary());
|
|
936
1048
|
|
|
937
1049
|
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
938
|
-
// An eval that reads every item (
|
|
1050
|
+
// An eval that reads every item (a group) re-reads one with no case
|
|
939
1051
|
// too, against the production reply the item kept.
|
|
940
1052
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
941
1053
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
942
1054
|
const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
|
|
943
|
-
const more = { production: it.production ?? null, grader: graderFor(run), group:
|
|
1055
|
+
const more = { production: it.production ?? null, grader: graderFor(run), group: gctx.resolve, item: it.name ?? null };
|
|
944
1056
|
return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
|
|
945
1057
|
(isObject(side) && side.res)
|
|
946
1058
|
? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
|
|
@@ -949,23 +1061,25 @@ async function runRescore(o){
|
|
|
949
1061
|
// A stored run that stopped short re-scores as what it is: incomplete.
|
|
950
1062
|
const complete = items.every(it => isObject(it) && !it.unrun);
|
|
951
1063
|
const outcomes = outcomesOf(run, items, complete);
|
|
952
|
-
const
|
|
1064
|
+
const wholes = wholeRunsOf(run, items, gctx.resolve);
|
|
1065
|
+
const verdicts = verdictsOf(outcomes, complete, o.minPass, wholes);
|
|
953
1066
|
const report = {
|
|
954
1067
|
runner: RUNNER,
|
|
955
1068
|
ranAt: new Date().toISOString(),
|
|
956
|
-
verdict: runVerdict(verdicts, complete),
|
|
1069
|
+
verdict: runVerdict(verdicts, complete, run.pass),
|
|
957
1070
|
...(o.minPass != null ? { minPass: o.minPass } : {}),
|
|
958
1071
|
rescored: true,
|
|
959
1072
|
run: {
|
|
960
1073
|
name: run.name ?? null,
|
|
961
1074
|
evals: run.evals,
|
|
962
1075
|
verdicts,
|
|
1076
|
+
pass: { rule: run.pass ?? { mode: "all" }, targets: passTargets(verdicts, run.pass) },
|
|
963
1077
|
scenarios: scenariosOf(run),
|
|
964
1078
|
items,
|
|
965
1079
|
},
|
|
966
1080
|
};
|
|
967
1081
|
|
|
968
|
-
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass));
|
|
1082
|
+
writeReports(o, report, () => runSuites(run, outcomes, items, complete, o.minPass, wholes));
|
|
969
1083
|
process.exitCode = exitOf(report.verdict);
|
|
970
1084
|
}
|
|
971
1085
|
|
|
@@ -1066,9 +1180,12 @@ async function main() {
|
|
|
1066
1180
|
}
|
|
1067
1181
|
|
|
1068
1182
|
// The dataset the run is graded against. Everything below grades, so there
|
|
1069
|
-
// is no run without one.
|
|
1183
|
+
// is no run without one. --groups hands a --run the bodies it kept; a
|
|
1184
|
+
// --pipeline or --prompt grades one scenario against one, named by --dataset.
|
|
1185
|
+
if (o.groups) broken("--groups hands the bodies a --run kept; grade a --pipeline or --prompt against --dataset.");
|
|
1070
1186
|
if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
|
|
1071
1187
|
const dataset = readDataset(o.dataset);
|
|
1188
|
+
const gctx = { single: dataset, bodies: null, primary: () => dataset, resolve: () => dataset ?? undefined };
|
|
1072
1189
|
|
|
1073
1190
|
// The run to grade. A --pipeline document is one scenario graded against
|
|
1074
1191
|
// its dataset; without one it is the stage this script always ran -- the
|
|
@@ -1078,7 +1195,7 @@ async function main() {
|
|
|
1078
1195
|
broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
|
|
1079
1196
|
+ "the lab's Prompt library does.");
|
|
1080
1197
|
}
|
|
1081
|
-
const run = o.pipeline ? readRun(o.pipeline,
|
|
1198
|
+
const run = o.pipeline ? readRun(o.pipeline, gctx) : cliRun(o, dataset);
|
|
1082
1199
|
if (o.pipeline) {
|
|
1083
1200
|
if (core.targetsOf(run).length !== 1) {
|
|
1084
1201
|
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
package/lab/server.py
CHANGED
|
@@ -809,7 +809,7 @@ class Sources:
|
|
|
809
809
|
out = []
|
|
810
810
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
811
811
|
db.row_factory = sqlite3.Row
|
|
812
|
-
for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
|
|
812
|
+
for r in db.execute("SELECT id, name, system, bytes, type, created_at FROM sources "
|
|
813
813
|
"ORDER BY system DESC, name COLLATE NOCASE"):
|
|
814
814
|
if r["system"]:
|
|
815
815
|
files = self._sample_files()
|
|
@@ -817,10 +817,13 @@ class Sources:
|
|
|
817
817
|
"type": r["type"],
|
|
818
818
|
"files": len(files), "bytes": sum(f["bytes"] for f in files)})
|
|
819
819
|
else:
|
|
820
|
-
n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
|
|
821
|
-
|
|
820
|
+
n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
|
|
821
|
+
(r["id"],)).fetchone()
|
|
822
|
+
# When it last changed: made, or a file added -- what a
|
|
823
|
+
# picker orders its recent Sources by.
|
|
822
824
|
out.append({"id": r["id"], "name": r["name"], "system": False,
|
|
823
|
-
"type": r["type"], "files": n, "bytes": r["bytes"]
|
|
825
|
+
"type": r["type"], "files": n, "bytes": r["bytes"],
|
|
826
|
+
"changed": max(filter(None, (r["created_at"], last)))})
|
|
824
827
|
return out
|
|
825
828
|
|
|
826
829
|
def get(self, sid) -> dict:
|
|
@@ -1754,7 +1757,7 @@ def scenario_ref(sc, i):
|
|
|
1754
1757
|
what version 4's upgrade gives one, by position."""
|
|
1755
1758
|
sid = sc.get("id") if isinstance(sc, dict) else None
|
|
1756
1759
|
name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
|
|
1757
|
-
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
|
|
1760
|
+
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
|
|
1758
1761
|
|
|
1759
1762
|
|
|
1760
1763
|
class Prompts:
|
|
@@ -3882,6 +3885,27 @@ class Queue:
|
|
|
3882
3885
|
(rundir / "dataset.json").write_text(json.dumps(body))
|
|
3883
3886
|
return ["--dataset", str(rundir / "dataset.json")], None
|
|
3884
3887
|
|
|
3888
|
+
def _grading_args(self, run, rundir):
|
|
3889
|
+
"""The worker's eval group bodies: --dataset for a run that grades
|
|
3890
|
+
against one group, which keeps its single-body path and old-row
|
|
3891
|
+
pinning, and --groups for a run that links several (#233) -- every body
|
|
3892
|
+
it kept, by `<id>@<n>`, written beside the run document. Returns
|
|
3893
|
+
(args, None) or (None, why)."""
|
|
3894
|
+
refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
|
|
3895
|
+
keys = {group_key(r) for r in refs if r is not None}
|
|
3896
|
+
if len(keys) <= 1:
|
|
3897
|
+
return self._dataset_args(run, rundir)
|
|
3898
|
+
kept = self.groups(run["id"], raw=True)
|
|
3899
|
+
if kept is None:
|
|
3900
|
+
return None, "the run kept no eval group bodies"
|
|
3901
|
+
missing = sorted({(r.get("name") or r.get("id")) for r in refs
|
|
3902
|
+
if r is not None and group_key(r) not in kept})
|
|
3903
|
+
if missing:
|
|
3904
|
+
return None, f"the run kept no body of the eval group {', '.join(missing)}"
|
|
3905
|
+
rundir.mkdir(parents=True, exist_ok=True)
|
|
3906
|
+
(rundir / "groups.json").write_text(json.dumps(kept))
|
|
3907
|
+
return ["--groups", str(rundir / "groups.json")], None
|
|
3908
|
+
|
|
3885
3909
|
def _behind(self, rid):
|
|
3886
3910
|
"""How many submissions stand between this one and the worker, by
|
|
3887
3911
|
submit time -- what a waiting form names when it says what it is
|
|
@@ -4020,7 +4044,7 @@ class Queue:
|
|
|
4020
4044
|
return None, (403, err)
|
|
4021
4045
|
if NODE is None:
|
|
4022
4046
|
return None, (500, "node is not installed, so nothing can run")
|
|
4023
|
-
dataset, err = self.
|
|
4047
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4024
4048
|
if err:
|
|
4025
4049
|
return None, (409, err)
|
|
4026
4050
|
plugins, err = self._plugin_args(run)
|
|
@@ -4086,7 +4110,7 @@ class Queue:
|
|
|
4086
4110
|
return None, (500, "node is not installed, so nothing can run")
|
|
4087
4111
|
rundir = self.dir / run["id"]
|
|
4088
4112
|
rundir.mkdir(parents=True, exist_ok=True)
|
|
4089
|
-
dataset, err = self.
|
|
4113
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4090
4114
|
if err:
|
|
4091
4115
|
return None, (409, err)
|
|
4092
4116
|
plugins, err = self._plugin_args(run)
|
|
@@ -4202,7 +4226,7 @@ class Queue:
|
|
|
4202
4226
|
return self._finish(rid, "failed", error=err)
|
|
4203
4227
|
if NODE is None:
|
|
4204
4228
|
return self._finish(rid, "failed", error="node is not installed, so nothing can run")
|
|
4205
|
-
dataset, err = self.
|
|
4229
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4206
4230
|
if err:
|
|
4207
4231
|
return self._finish(rid, "failed", error=err)
|
|
4208
4232
|
plugins, err = self._plugin_args(run)
|
|
@@ -5687,10 +5711,8 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5687
5711
|
return self._json(400, {"error": f"the run cannot be queued: {body}"})
|
|
5688
5712
|
ref["n"], ref["version"] = n, fingerprint(body)
|
|
5689
5713
|
groups[group_key(ref)] = body
|
|
5690
|
-
# The worker
|
|
5691
|
-
|
|
5692
|
-
return self._json(400, {"error": "the run cannot be queued: its evals grade against "
|
|
5693
|
-
"one version of one Library eval group at a time"})
|
|
5714
|
+
# The worker reads a body per group now (#233), so a run may link
|
|
5715
|
+
# several; each body is kept with the row under `<id>@<n>`.
|
|
5694
5716
|
return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
|
|
5695
5717
|
|
|
5696
5718
|
# ---- Datasets ----------------------------------------------------------
|