evals-lab 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -5,6 +5,43 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
5
5
  upgrade can rewrite what the lab keeps in your data directory, and an older
6
6
  version cannot always read it back.
7
7
 
8
+ ## 0.6.0
9
+
10
+ ### Added
11
+
12
+ - Library › Evals (was Library › Datasets) lists every eval group: its
13
+ Source, cases, scoring, and the pipelines that link it. A group's page
14
+ holds its Grades, Every item, Whole run and cases.
15
+ - Runs › Evals links eval groups to a pipeline. Link eval group picks from
16
+ the Library and marks the groups made for this content. A link follows a
17
+ group's latest version, or is pinned to one. A pipeline's own group is
18
+ edited in place, and Move to Library shares it.
19
+ - A pipeline can link several eval groups. Its pass rule says whether every
20
+ one must pass, or at least a number. Results, History and `evals-lab run`
21
+ all grade by it.
22
+ - An Undo button in the header takes back the last 20 actions, newest
23
+ first, until the page reloads.
24
+ - Choose pipeline lists the 5 used most recently, then the rest, searchable
25
+ and sortable.
26
+ - A Profile, Source or Grader field opens a picker, most recent first.
27
+
28
+ ### Changed
29
+
30
+ - Results shows a row per eval group, with each Target's share and verdict,
31
+ then the overall verdict. History's Outcome is the overall verdict.
32
+ - A dialog is a draft: nothing reaches the lab until Save & Close. Cancel,
33
+ × and Esc discard it.
34
+ - An unnamed target is Target A, B, C.
35
+ - The run bar sits under the pipeline's header. It says "12 of 42" while a
36
+ run goes and "42 in 2m" when it is done.
37
+ - Verdicts show as a check or a cross, in Results and History.
38
+ - A metric's "How it counts" is Options, and Grades is Reads: Parsed reply
39
+ or Raw reply.
40
+
41
+ ### Fixed
42
+
43
+ - Add profile, then ×, added a profile.
44
+
8
45
  ## 0.5.0
9
46
 
10
47
  ### Upgrade notes
package/bin/run.js CHANGED
@@ -487,10 +487,21 @@ async function run(argv, { lab, env = process.env, stdout = process.stdout, stde
487
487
  const args = [path.join(lab, "run-evals.js"), "--run", path.join(tmp, "run.json"),
488
488
  "--json", path.join(tmp, "report.json")];
489
489
  if (items) args.push("--source", items);
490
- const ref = core.evalsDataset(read.run);
491
- if (ref && read.datasets[ref.id]) {
492
- fs.writeFileSync(path.join(tmp, "dataset.json"), JSON.stringify(read.datasets[ref.id]));
490
+ // The eval group bodies the run grades against: its one, as --dataset, or
491
+ // a body per group it links, as --groups (#233), keyed by the id its
492
+ // reference carries (a bundle's refs record no version).
493
+ const ids = [...new Set(core.evalsOf(read.run).map(core.casesRef).filter(Boolean).map(r => r.id))];
494
+ if (ids.length === 1 && read.datasets[ids[0]]) {
495
+ fs.writeFileSync(path.join(tmp, "dataset.json"), JSON.stringify(read.datasets[ids[0]]));
493
496
  args.push("--dataset", path.join(tmp, "dataset.json"));
497
+ } else if (ids.length > 1) {
498
+ const groups = {};
499
+ for (const id of ids) {
500
+ if (!read.datasets[id]) throw new Refused(`${o.bundle}: the bundle holds no eval group ${id}`);
501
+ groups[id] = read.datasets[id];
502
+ }
503
+ fs.writeFileSync(path.join(tmp, "groups.json"), JSON.stringify(groups));
504
+ args.push("--groups", path.join(tmp, "groups.json"));
494
505
  }
495
506
  if (plugins.length) args.push("--plugins", path.join(dir, "plugins"));
496
507
  if (o.minPass != null) args.push("--min-pass", o.minPass);
package/lab/VERSION CHANGED
@@ -1 +1 @@
1
- 0.5.0 (2026.10.04-397)
1
+ 0.6.0 (2026.10.05-430)
@@ -775,16 +775,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
775
775
 
776
776
  // ---- the registries ----
777
777
 
778
- /** An option a modifier or an eval type exposes for editing. */
779
-
780
-
781
-
778
+ /** An option a modifier or an eval type exposes for editing. `hint` says
779
+ what it does in one short line, under its control, where the label
780
+ cannot: a switch's effect, not why it exists. */
781
+
782
+
783
+
782
784
 
783
-
784
-
785
-
785
+
786
+
787
+
786
788
 
787
-
789
+
788
790
 
789
791
  /** What an output kind made of a reply. */
790
792
 
@@ -966,6 +968,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
966
968
 
967
969
 
968
970
 
971
+
972
+
973
+
974
+
969
975
 
970
976
 
971
977
  /** A job's stages, in the order they run (pipeline-model §16). A target's
@@ -3487,10 +3493,21 @@ EVAL_TYPES.group = {
3487
3493
  read: async (t, kase, res, more) => {
3488
3494
  const group = groupOf(t, more);
3489
3495
  const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
3490
- return readGroup({ ...group, grader } , casesRef(t) ? kase : null, res, more);
3496
+ const ref = casesRef(t);
3497
+ // The item's case in this eval's own group: resolved from the group whose
3498
+ // cases it reads where the runner names the item (grading several groups
3499
+ // at once, --groups), else the case handed in (a single group).
3500
+ const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
3501
+ return readGroup({ ...group, grader } , ref ? own : null, res, more);
3491
3502
  },
3492
3503
  };
3493
3504
 
3505
+ /** The case for an item in a group's body, by the name its file has, or null:
3506
+ a non-todo case whose item matches, as a Library group reads one. */
3507
+ function caseIn(body , item ) {
3508
+ return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
3509
+ }
3510
+
3494
3511
  // ---- the lab's own scorers, as metrics ----------------------------------------
3495
3512
  // What the Single Test checks, one metric each, so an eval of it converts to
3496
3513
  // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
@@ -3901,9 +3918,15 @@ function targetsOf(doc
3901
3918
  return Array.isArray(doc?.targets) ? doc .targets : [];
3902
3919
  }
3903
3920
 
3904
- /** Target [i]'s name, or the number it has always shown. */
3921
+ /** Target [i]'s letter, A for the first: the same in every job, so a
3922
+ Target reads as one column through them. A number past Z. */
3923
+ function targetLetter(i ) {
3924
+ return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
3925
+ }
3926
+
3927
+ /** Target [i]'s name, or its letter's: "Target A". */
3905
3928
  function targetLabel(doc , i ) {
3906
- return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
3929
+ return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
3907
3930
  }
3908
3931
 
3909
3932
  /** What target [i] sends in job [k]: its step there. */
@@ -4150,11 +4173,8 @@ STEP_TYPES.evals = {
4150
4173
  + `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
4151
4174
  }
4152
4175
  });
4153
- // The worker is handed one Library group's body to grade against
4154
- // (server-side-runs §4) until it reads the body the queue keeps per
4155
- // group (#233).
4156
- const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
4157
- if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
4176
+ // The worker reads a body per group now (#233), so the evals may grade
4177
+ // against several Library groups at once; the queue keeps each body.
4158
4178
  },
4159
4179
  };
4160
4180
 
@@ -5390,22 +5410,40 @@ function scenarioEvals(run , i , items
5390
5410
  });
5391
5411
  }
5392
5412
 
5413
+ /** One eval group's verdict over a scenario (§17): a whole-run group's own
5414
+ verdict, or a per-item group's "every item it read passed"; null where it
5415
+ was skipped, or had nothing to read yet. */
5416
+ function groupPass(o ) {
5417
+ if (o.skipped) return null;
5418
+ if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
5419
+ return o.ran ? o.passed === o.ran : null;
5420
+ }
5421
+
5393
5422
  /** A scenario's pass or fail over every eval that read it: null where none
5394
5423
  has anything to say yet. */
5395
5424
  function scenarioPasses(outcomes ) {
5396
- let said = false;
5397
- for (const o of outcomes) {
5398
- if (o.skipped) continue;
5399
- if (o.whole) {
5400
- if (!o.verdict?.ran) continue;
5401
- said = true;
5402
- if (!o.verdict.pass) return false;
5403
- } else if (o.ran) {
5404
- said = true;
5405
- if (o.passed < o.ran) return false;
5406
- }
5407
- }
5408
- return said ? true : null;
5425
+ const passes = outcomes.map(groupPass);
5426
+ if (!passes.some((p) => p != null)) return null;
5427
+ return !passes.includes(false);
5428
+ }
5429
+
5430
+ /** A scenario's overall verdict under a run's pass rule (§17): each linked
5431
+ group's verdict folded together the way `pass` says -- every group passes,
5432
+ or at least a number of them. Null where no group has a verdict yet, so a
5433
+ run with no evals, or one still grading, reads as it does without a rule. */
5434
+ function overallVerdict(outcomes , pass ) {
5435
+ const passes = outcomes.map(groupPass);
5436
+ if (!passes.some((p) => p != null)) return null;
5437
+ return passVerdict(pass, passes);
5438
+ }
5439
+
5440
+ /** Whether a run passes a Target under its overall pass rule (§17): `all`
5441
+ passes when no linked group fails, `atLeast` when at least `count` pass.
5442
+ [passes] is each group's pass (true), fail (false), or neither (null: it
5443
+ graded nothing, or an earlier group skipped it). */
5444
+ function passVerdict(pass , passes ) {
5445
+ if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
5446
+ return !passes.includes(false);
5409
5447
  }
5410
5448
 
5411
5449
  /**
@@ -5610,10 +5648,10 @@ export {
5610
5648
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5611
5649
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5612
5650
  contentOf, withContent, replyOf,
5613
- targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5651
+ targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5614
5652
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
5615
5653
  blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5616
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
5654
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
5617
5655
  evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
5618
5656
  pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
5619
5657
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
@@ -336,7 +336,8 @@ const LIST_KIND = {
336
336
  settings: [
337
337
  { key: "parse", label: "Parse as", type: "select", choices: [
338
338
  { value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
339
- { key: "preamble", label: "Ignore preamble", type: "checkbox" },
339
+ { key: "preamble", label: "Ignore preamble", type: "checkbox",
340
+ hint: "Discard any text in <think> tags and any text before the first colon" },
340
341
  ],
341
342
  settingDefaults: () => ({ parse: "csv", preamble: false }),
342
343
  validateSettings(out, at, bad) {