evals-lab 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +69 -31
- package/lab/kinds/list.mjs +2 -1
- package/lab/run-evals.js +197 -80
- package/lab/server.py +34 -12
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +3 -0
- package/lab/web/dist/assets/main-DDoeU6hq.css +1 -0
- package/lab/web/dist/assets/main-nz6Q4jVm.js +21 -0
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +59 -0
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,43 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.6.0
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Library › Evals (was Library › Datasets) lists every eval group: its
|
|
13
|
+
Source, cases, scoring, and the pipelines that link it. A group's page
|
|
14
|
+
holds its Grades, Every item, Whole run and cases.
|
|
15
|
+
- Runs › Evals links eval groups to a pipeline. Link eval group picks from
|
|
16
|
+
the Library and marks the groups made for this content. A link follows a
|
|
17
|
+
group's latest version, or is pinned to one. A pipeline's own group is
|
|
18
|
+
edited in place, and Move to Library shares it.
|
|
19
|
+
- A pipeline can link several eval groups. Its pass rule says whether every
|
|
20
|
+
one must pass, or at least a number. Results, History and `evals-lab run`
|
|
21
|
+
all grade by it.
|
|
22
|
+
- An Undo button in the header takes back the last 20 actions, newest
|
|
23
|
+
first, until the page reloads.
|
|
24
|
+
- Choose pipeline lists the 5 used most recently, then the rest, searchable
|
|
25
|
+
and sortable.
|
|
26
|
+
- A Profile, Source or Grader field opens a picker, most recent first.
|
|
27
|
+
|
|
28
|
+
### Changed
|
|
29
|
+
|
|
30
|
+
- Results shows a row per eval group, with each Target's share and verdict,
|
|
31
|
+
then the overall verdict. History's Outcome is the overall verdict.
|
|
32
|
+
- A dialog is a draft: nothing reaches the lab until Save & Close. Cancel,
|
|
33
|
+
× and Esc discard it.
|
|
34
|
+
- An unnamed target is Target A, B, C.
|
|
35
|
+
- The run bar sits under the pipeline's header. It says "12 of 42" while a
|
|
36
|
+
run goes and "42 in 2m" when it is done.
|
|
37
|
+
- Verdicts show as a check or a cross, in Results and History.
|
|
38
|
+
- A metric's "How it counts" is Options, and Grades is Reads: Parsed reply
|
|
39
|
+
or Raw reply.
|
|
40
|
+
|
|
41
|
+
### Fixed
|
|
42
|
+
|
|
43
|
+
- Add profile, then ×, added a profile.
|
|
44
|
+
|
|
8
45
|
## 0.5.0
|
|
9
46
|
|
|
10
47
|
### Upgrade notes
|
package/bin/run.js
CHANGED
|
@@ -487,10 +487,21 @@ async function run(argv, { lab, env = process.env, stdout = process.stdout, stde
|
|
|
487
487
|
const args = [path.join(lab, "run-evals.js"), "--run", path.join(tmp, "run.json"),
|
|
488
488
|
"--json", path.join(tmp, "report.json")];
|
|
489
489
|
if (items) args.push("--source", items);
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
490
|
+
// The eval group bodies the run grades against: its one, as --dataset, or
|
|
491
|
+
// a body per group it links, as --groups (#233), keyed by the id its
|
|
492
|
+
// reference carries (a bundle's refs record no version).
|
|
493
|
+
const ids = [...new Set(core.evalsOf(read.run).map(core.casesRef).filter(Boolean).map(r => r.id))];
|
|
494
|
+
if (ids.length === 1 && read.datasets[ids[0]]) {
|
|
495
|
+
fs.writeFileSync(path.join(tmp, "dataset.json"), JSON.stringify(read.datasets[ids[0]]));
|
|
493
496
|
args.push("--dataset", path.join(tmp, "dataset.json"));
|
|
497
|
+
} else if (ids.length > 1) {
|
|
498
|
+
const groups = {};
|
|
499
|
+
for (const id of ids) {
|
|
500
|
+
if (!read.datasets[id]) throw new Refused(`${o.bundle}: the bundle holds no eval group ${id}`);
|
|
501
|
+
groups[id] = read.datasets[id];
|
|
502
|
+
}
|
|
503
|
+
fs.writeFileSync(path.join(tmp, "groups.json"), JSON.stringify(groups));
|
|
504
|
+
args.push("--groups", path.join(tmp, "groups.json"));
|
|
494
505
|
}
|
|
495
506
|
if (plugins.length) args.push("--plugins", path.join(dir, "plugins"));
|
|
496
507
|
if (o.minPass != null) args.push("--min-pass", o.minPass);
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.6.0 (2026.10.05-430)
|
package/lab/evals-core.mjs
CHANGED
|
@@ -775,16 +775,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
775
775
|
|
|
776
776
|
// ---- the registries ----
|
|
777
777
|
|
|
778
|
-
/** An option a modifier or an eval type exposes for editing.
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
778
|
+
/** An option a modifier or an eval type exposes for editing. `hint` says
|
|
779
|
+
what it does in one short line, under its control, where the label
|
|
780
|
+
cannot: a switch's effect, not why it exists. */
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
|
|
782
784
|
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
|
|
786
788
|
|
|
787
|
-
|
|
789
|
+
|
|
788
790
|
|
|
789
791
|
/** What an output kind made of a reply. */
|
|
790
792
|
|
|
@@ -966,6 +968,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
966
968
|
|
|
967
969
|
|
|
968
970
|
|
|
971
|
+
|
|
972
|
+
|
|
973
|
+
|
|
974
|
+
|
|
969
975
|
|
|
970
976
|
|
|
971
977
|
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
@@ -3487,10 +3493,21 @@ EVAL_TYPES.group = {
|
|
|
3487
3493
|
read: async (t, kase, res, more) => {
|
|
3488
3494
|
const group = groupOf(t, more);
|
|
3489
3495
|
const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
|
|
3490
|
-
|
|
3496
|
+
const ref = casesRef(t);
|
|
3497
|
+
// The item's case in this eval's own group: resolved from the group whose
|
|
3498
|
+
// cases it reads where the runner names the item (grading several groups
|
|
3499
|
+
// at once, --groups), else the case handed in (a single group).
|
|
3500
|
+
const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
|
|
3501
|
+
return readGroup({ ...group, grader } , ref ? own : null, res, more);
|
|
3491
3502
|
},
|
|
3492
3503
|
};
|
|
3493
3504
|
|
|
3505
|
+
/** The case for an item in a group's body, by the name its file has, or null:
|
|
3506
|
+
a non-todo case whose item matches, as a Library group reads one. */
|
|
3507
|
+
function caseIn(body , item ) {
|
|
3508
|
+
return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
|
|
3509
|
+
}
|
|
3510
|
+
|
|
3494
3511
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3495
3512
|
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3496
3513
|
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
@@ -3901,9 +3918,15 @@ function targetsOf(doc
|
|
|
3901
3918
|
return Array.isArray(doc?.targets) ? doc .targets : [];
|
|
3902
3919
|
}
|
|
3903
3920
|
|
|
3904
|
-
/** Target [i]'s
|
|
3921
|
+
/** Target [i]'s letter, A for the first: the same in every job, so a
|
|
3922
|
+
Target reads as one column through them. A number past Z. */
|
|
3923
|
+
function targetLetter(i ) {
|
|
3924
|
+
return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
|
|
3925
|
+
}
|
|
3926
|
+
|
|
3927
|
+
/** Target [i]'s name, or its letter's: "Target A". */
|
|
3905
3928
|
function targetLabel(doc , i ) {
|
|
3906
|
-
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i
|
|
3929
|
+
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
|
|
3907
3930
|
}
|
|
3908
3931
|
|
|
3909
3932
|
/** What target [i] sends in job [k]: its step there. */
|
|
@@ -4150,11 +4173,8 @@ STEP_TYPES.evals = {
|
|
|
4150
4173
|
+ `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
|
|
4151
4174
|
}
|
|
4152
4175
|
});
|
|
4153
|
-
// The worker
|
|
4154
|
-
//
|
|
4155
|
-
// group (#233).
|
|
4156
|
-
const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
|
|
4157
|
-
if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
|
|
4176
|
+
// The worker reads a body per group now (#233), so the evals may grade
|
|
4177
|
+
// against several Library groups at once; the queue keeps each body.
|
|
4158
4178
|
},
|
|
4159
4179
|
};
|
|
4160
4180
|
|
|
@@ -5390,22 +5410,40 @@ function scenarioEvals(run , i , items
|
|
|
5390
5410
|
});
|
|
5391
5411
|
}
|
|
5392
5412
|
|
|
5413
|
+
/** One eval group's verdict over a scenario (§17): a whole-run group's own
|
|
5414
|
+
verdict, or a per-item group's "every item it read passed"; null where it
|
|
5415
|
+
was skipped, or had nothing to read yet. */
|
|
5416
|
+
function groupPass(o ) {
|
|
5417
|
+
if (o.skipped) return null;
|
|
5418
|
+
if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
|
|
5419
|
+
return o.ran ? o.passed === o.ran : null;
|
|
5420
|
+
}
|
|
5421
|
+
|
|
5393
5422
|
/** A scenario's pass or fail over every eval that read it: null where none
|
|
5394
5423
|
has anything to say yet. */
|
|
5395
5424
|
function scenarioPasses(outcomes ) {
|
|
5396
|
-
|
|
5397
|
-
|
|
5398
|
-
|
|
5399
|
-
|
|
5400
|
-
|
|
5401
|
-
|
|
5402
|
-
|
|
5403
|
-
|
|
5404
|
-
|
|
5405
|
-
|
|
5406
|
-
|
|
5407
|
-
|
|
5408
|
-
return
|
|
5425
|
+
const passes = outcomes.map(groupPass);
|
|
5426
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5427
|
+
return !passes.includes(false);
|
|
5428
|
+
}
|
|
5429
|
+
|
|
5430
|
+
/** A scenario's overall verdict under a run's pass rule (§17): each linked
|
|
5431
|
+
group's verdict folded together the way `pass` says -- every group passes,
|
|
5432
|
+
or at least a number of them. Null where no group has a verdict yet, so a
|
|
5433
|
+
run with no evals, or one still grading, reads as it does without a rule. */
|
|
5434
|
+
function overallVerdict(outcomes , pass ) {
|
|
5435
|
+
const passes = outcomes.map(groupPass);
|
|
5436
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5437
|
+
return passVerdict(pass, passes);
|
|
5438
|
+
}
|
|
5439
|
+
|
|
5440
|
+
/** Whether a run passes a Target under its overall pass rule (§17): `all`
|
|
5441
|
+
passes when no linked group fails, `atLeast` when at least `count` pass.
|
|
5442
|
+
[passes] is each group's pass (true), fail (false), or neither (null: it
|
|
5443
|
+
graded nothing, or an earlier group skipped it). */
|
|
5444
|
+
function passVerdict(pass , passes ) {
|
|
5445
|
+
if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
|
|
5446
|
+
return !passes.includes(false);
|
|
5409
5447
|
}
|
|
5410
5448
|
|
|
5411
5449
|
/**
|
|
@@ -5610,10 +5648,10 @@ export {
|
|
|
5610
5648
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5611
5649
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5612
5650
|
contentOf, withContent, replyOf,
|
|
5613
|
-
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5651
|
+
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5614
5652
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5615
5653
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5616
|
-
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
5654
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|
|
5617
5655
|
evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
|
|
5618
5656
|
pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
|
|
5619
5657
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
package/lab/kinds/list.mjs
CHANGED
|
@@ -336,7 +336,8 @@ const LIST_KIND = {
|
|
|
336
336
|
settings: [
|
|
337
337
|
{ key: "parse", label: "Parse as", type: "select", choices: [
|
|
338
338
|
{ value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
|
|
339
|
-
{ key: "preamble", label: "Ignore preamble", type: "checkbox"
|
|
339
|
+
{ key: "preamble", label: "Ignore preamble", type: "checkbox",
|
|
340
|
+
hint: "Discard any text in <think> tags and any text before the first colon" },
|
|
340
341
|
],
|
|
341
342
|
settingDefaults: () => ({ parse: "csv", preamble: false }),
|
|
342
343
|
validateSettings(out, at, bad) {
|