evals-lab 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,7 +26,7 @@
26
26
  // WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
27
27
  // script has not, so `prepare()` stays in the page and the script shells out
28
28
  // to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
29
- // the script writes JSON. Both read the same verdict out of `scoreCase`.
29
+ // the script writes JSON. Both read the same verdict out of the same Metrics.
30
30
  //
31
31
  // `runner-check.js` runs one set through two of the callers and asserts the
32
32
  // verdicts and the totals are identical, because sharing a file is a claim
@@ -232,8 +232,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
232
232
 
233
233
 
234
234
 
235
- /** What every test carries whatever its type: an id minted once and never
236
- shown, an optional name ("Test 2" when blank), and whether the tests after
235
+ /** What every eval carries whatever its type: an id minted once and never
236
+ shown, an optional name ("Eval 2" when blank), and whether the evals after
237
237
  it still read what it failed on (docs/pipeline-model.md §3). */
238
238
 
239
239
 
@@ -260,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
260
260
 
261
261
 
262
262
 
263
- /** Checks of every reply (TEST_TYPES.metrics): its own for every item, and
263
+ /** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
264
264
  a case's own for its item; all must pass, or weighted points reach the
265
265
  threshold. A model-graded one asks the grader. */
266
266
 
@@ -275,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
275
275
 
276
276
 
277
277
 
278
- /** A test, from version 9: Metrics. A Single Test or a Graded set is what an
278
+ /** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
279
279
  older document held (upgradePipeline reads it converted). */
280
280
 
281
281
 
282
- /** A pipeline's tests, in the order they read a run. */
282
+ /** A pipeline's evals, in the order they read a run. */
283
283
 
284
284
 
285
285
 
@@ -482,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
482
482
 
483
483
  // ---- grading ----
484
484
 
485
- /** A graded case, as cases.json writes one.
486
-
487
- The item a case grades is named by `filename`: a dataset joins on a file
488
- name, and nothing about that is an image -- the next dataset addresses
489
- a row of a spreadsheet or a line of a log by the same key. The field had
490
- an older name while the lab was one app's bench, and it is still *read*:
491
- a set re-synced from that app's repository arrives spelled that way, and
492
- a case that silently stopped matching would be worse than one that reads
493
- both.
494
- `caseFile` is the one reader; `evalsJson` is the one writer, and it writes
495
- `filename`. */
485
+ /** A graded case (dataset body version 5): an Item and the Metrics its
486
+ reply is held to.
487
+
488
+ The item is named by `item`: a dataset joins on an item's name in the
489
+ Source it grades, exactly, and nothing about that is an image -- the
490
+ next dataset addresses a row of a spreadsheet or a line of a log by the
491
+ same key. Version 4 called it `filename`, and the lab's first app had a
492
+ name of its own for it; `upgradeDatasetBody` reads both as `item`.
493
+
494
+ A case's metrics are what a good answer is: each a metric as a
495
+ pipeline's Metrics eval holds one, added to the eval's own for this item
496
+ when an eval names the dataset. One with weight 0 is watched -- reported,
497
+ never scored. */
496
498
 
497
499
 
498
-
499
-
500
-
500
+
501
+
501
502
 
502
-
503
-
504
-
505
-
506
-
507
-
508
-
509
-
510
-
511
-
503
+
504
+
512
505
 
513
-
514
-
515
-
516
-
517
506
 
518
507
 
519
508
 
520
- /** A graded set: cases.json. */
509
+ /** A graded set: a dataset's body, or anything holding its cases. */
521
510
 
522
511
 
523
512
 
524
513
 
525
514
 
526
- /** One row the Datasets tab's Cases group draws, as gradedSetFrom builds it:
527
- a case of a set validateEvals accepts, so it has its id and file. */
515
+ /** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
516
+ a case of a set validateEvals accepts, so it has its id and item. */
528
517
 
529
518
 
530
-
519
+
531
520
 
532
521
 
533
522
 
534
523
  /** A requirement as a score reports it: a term, or the group it was. */
535
524
 
536
525
 
537
- /** How a graded case read a reply: the same shape both engines agree on. */
538
-
539
-
540
-
541
-
542
-
543
-
544
-
545
-
546
-
547
-
548
-
549
-
550
-
551
-
552
-
553
-
554
526
  /** A run's totals, summed across cases. */
555
527
 
556
528
 
@@ -559,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
559
531
 
560
532
 
561
533
 
562
- /** The whole-run verdict of a test type, where it has one of its own. */
534
+ /** The whole-run verdict of an eval type, where it has one of its own. */
563
535
 
564
536
 
565
537
 
@@ -570,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
570
542
 
571
543
 
572
544
 
573
- /** One thing a test asserts, as its type names it: "Exact" "outdoor". */
545
+ /** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
574
546
 
575
547
 
576
548
 
@@ -599,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
599
571
 
600
572
 
601
573
 
602
- /** A per-item test's score as a stored row carries it. */
574
+ /** A per-item eval's score as a stored row carries it. */
603
575
 
604
576
 
605
577
 
@@ -678,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
678
650
  more entry, and no reader changes. */
679
651
 
680
652
 
653
+
654
+
655
+
681
656
 
682
657
 
683
658
 
@@ -698,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
698
673
 
699
674
 
700
675
 
701
- /** What an earlier test that failed and does not continue leaves a later one. */
676
+ /** What an earlier eval that failed and does not continue leaves a later one. */
702
677
 
703
678
 
704
679
 
705
680
 
706
- /** One test's reading of one scenario of a run (scenarioTests). */
681
+ /** One eval's reading of one scenario of a run (scenarioEvals). */
707
682
 
708
683
 
709
684
 
@@ -760,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
760
735
 
761
736
  // ---- the registries ----
762
737
 
763
- /** An option a modifier or a test type exposes for editing. */
738
+ /** An option a modifier or an eval type exposes for editing. */
764
739
 
765
740
 
766
741
 
@@ -820,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
820
795
 
821
796
 
822
797
 
823
-
798
+
824
799
 
825
800
 
826
801
 
@@ -896,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
896
871
 
897
872
 
898
873
 
899
- /** A test type: what it checks of a document, and how it scores a run. */
874
+ /** An eval type: what it checks of a document, and how it scores a run. */
900
875
 
901
876
 
902
-
877
+
903
878
 
904
879
 
905
880
 
@@ -909,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
909
884
 
910
885
 
911
886
 
912
-
887
+
913
888
 
914
889
 
915
890
 
916
-
891
+
917
892
 
918
893
 
919
894
 
@@ -934,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
934
909
 
935
910
 
936
911
 
937
- /** What a test's `read` is handed besides the reply: production's reply to
912
+ /** What an eval's `read` is handed besides the reply: production's reply to
938
913
  the item, and a grader, where the run has them. */
939
914
 
940
915
 
@@ -1004,42 +979,151 @@ const SLOTS = ["content", "target", "responses"];
1004
979
  * submitted against, so nothing grades from a file.
1005
980
  */
1006
981
 
1007
-
982
+
983
+
984
+
985
+
986
+
987
+
988
+
989
+
990
+
991
+
992
+
993
+
994
+
995
+
996
+
997
+
1008
998
 
1009
999
 
1010
1000
 
1001
+ /** An eval group's scoring (docs/pipeline-model.md §17). */
1002
+
1003
+
1004
+
1005
+
1006
+
1007
+ /** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
1008
+ named its Source; 4 and earlier, neither. */
1009
+ const DATASET_BODY_VERSION = 7 ;
1010
+
1011
+ /** The metrics whose Ignore case version 6 made mean what it says for a
1012
+ reply read as a list, as well as one read as text. */
1013
+ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1014
+
1011
1015
  /**
1012
- * [body] as this version of a dataset (4), from any earlier one. Version 1
1016
+ * [body] as this version of a dataset (7), from any earlier one. Version 1
1013
1017
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
1014
1018
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
1015
1019
  * Version 2 held `rules`, which clean a job's answer and so belong to the
1016
1020
  * job (docs/pipeline-model.md §13): they leave, and the terms a case
1017
1021
  * watches for are its `watch`. Version 3 held the `prompt` a new scenario
1018
- * started from, which the Prompt library holds now: it leaves. Every reader of a body calls this: the runner, the page, a
1019
- * run's kept copy. A reader that needs a version-2 body's rules -- to upgrade
1020
- * a pipeline graded against it -- takes them first (`datasetRules`).
1021
- * Anything else comes back as it was.
1022
+ * started from, which the Prompt library holds now: it leaves. Version 4's
1023
+ * case named its item `filename` and said what a good answer is in
1024
+ * expectations; each becomes the metric it is (`caseMetrics`), `why` is the
1025
+ * `note`, `traits` go, and the body names no Source yet. Version 5 matched
1026
+ * a list's items ignoring case whatever a Contains metric's Ignore case
1027
+ * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1028
+ * the body says its version. Version 6 is a group's cases alone: version 7
1029
+ * adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
1030
+ * stored row is read that way rather than rewritten. Every reader of a
1031
+ * body calls this: the runner, the page, a run's kept copy. A reader that
1032
+ * needs a version-2 body's rules -- to upgrade a pipeline graded against
1033
+ * it -- takes them first (`datasetRules`). Pure: the same body gives the
1034
+ * same answer, and server.py's upgrade_body is its twin. Anything else
1035
+ * comes back as it was.
1022
1036
  */
1023
1037
  function upgradeDatasetBody (body ) {
1024
1038
  if (!isObj(body)) return body;
1025
- if (!("imageCases" in body || "rules" in body)) {
1026
- if (!("prompt" in body)) return body;
1027
- const { prompt: _library, ...rest } = body ;
1028
- return rest ;
1029
- }
1039
+ if (body.version === DATASET_BODY_VERSION) return body;
1030
1040
  const b = body ;
1031
- // vocab: the names older versions gave these fields
1032
- const RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch" }; // vocab: as above
1033
- const renamed = (c ) => {
1034
- if (!isObj(c)) return c;
1035
- const out = {};
1036
- for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
1037
- return out;
1038
- };
1041
+ // A body saying any other version is one this lab does not read.
1042
+ if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
1043
+ // A body that names its Source, even as null, is version 5.
1044
+ if ("source" in b) {
1045
+ return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
1046
+ }
1039
1047
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
1040
1048
  // Not a body of any version -- a copy kept while a dataset was an overlay.
1041
1049
  if (!raw) return body;
1042
- return { cases: (canonicalCases({ cases: raw.map(renamed) }) ).cases };
1050
+ return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
1051
+ }
1052
+
1053
+ /** A version-6 body as a version-7 eval group: scored All, the lab's
1054
+ grader, and no metrics of its own for every item or the whole run --
1055
+ what a Metrics eval naming the dataset with none of its own graded, which
1056
+ is what every Graded set converted to. */
1057
+ function groupOfV6(b ) {
1058
+ const { version: _v, ...rest } = b;
1059
+ return { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null, every: [], run: [],
1060
+ ...rest } ;
1061
+ }
1062
+
1063
+ /** A version-5 case as a version-6 one: each Contains metric says Ignore
1064
+ case, as version 5 matched a list's items whatever it said. A reply read
1065
+ as text did mind its setting, but a case's metrics were written for a
1066
+ list -- each one converted from version 4, and each the case form made. */
1067
+ function caseOfV5(c ) {
1068
+ if (!isObj(c) || !Array.isArray(c.metrics)) return c;
1069
+ return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
1070
+ ? { ...m, ignoreCase: true } : m)) };
1071
+ }
1072
+
1073
+ // vocab: the names older versions gave these fields
1074
+ const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
1075
+ /** What a version-4 case said, which its metrics say now. */
1076
+ const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
1077
+
1078
+ /** One case of any earlier version as a version-5 one: its id, item, todo
1079
+ and note, its expectations as metrics ahead of any metrics it held, and
1080
+ anything else it carried kept as it was. */
1081
+ function caseOfV4(c ) {
1082
+ if (!isObj(c)) return c;
1083
+ const was = {};
1084
+ for (const [k, v] of Object.entries(c)) {
1085
+ const key = CASE_RENAMED[k] ?? k;
1086
+ // A case naming its item both ways keeps the newer key's.
1087
+ if (!(key in was) || key === k) was[key] = v;
1088
+ }
1089
+ if (isStr(was.item)) delete was.filename;
1090
+ const out = {};
1091
+ if ("id" in was) out.id = was.id;
1092
+ out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
1093
+ out.todo = was.todo === true;
1094
+ out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
1095
+ out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
1096
+ for (const [k, v] of Object.entries(was)) {
1097
+ if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
1098
+ }
1099
+ return out;
1100
+ }
1101
+
1102
+ /** A version-4 case's expectations as the metrics that say the same:
1103
+ `expect` is Contains all, each `anyOf` group a Contains any, each
1104
+ forbidden term a Contains turned round with the `allow` phrases that
1105
+ excuse it, the count bounds an Item count, `discarded` the Discarded
1106
+ metric, and each watched term a Contains any of weight 0 -- reported,
1107
+ never scored. The order is the one scoreCase read them in, so a reason
1108
+ reads in the order it did. */
1109
+ function caseMetrics(c ) {
1110
+ const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
1111
+ if (c.discarded === true) return [{ type: "discarded" }];
1112
+ const out = [];
1113
+ const expect = list(c.expect), allow = list(c.allow);
1114
+ if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
1115
+ for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
1116
+ if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
1117
+ }
1118
+ for (const t of list(c.forbid)) {
1119
+ // An exception excuses only the forbidden term inside it.
1120
+ const except = allow.filter(a => termIn([a], t));
1121
+ out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
1122
+ }
1123
+ const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
1124
+ if (min != null || max != null) out.push({ type: "item-count", min, max });
1125
+ for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
1126
+ return out;
1043
1127
  }
1044
1128
 
1045
1129
  /** The rules a version-1 or version-2 dataset body held, or null: what a
@@ -1055,6 +1139,9 @@ function datasetRules(body ) {
1055
1139
 
1056
1140
 
1057
1141
 
1142
+
1143
+
1144
+
1058
1145
 
1059
1146
 
1060
1147
 
@@ -1322,8 +1409,10 @@ function textPrompt(instruction , text ) {
1322
1409
  }
1323
1410
 
1324
1411
  // Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
1325
- // work in, so that "New Zealand." and "new zealand" are the same two words.
1326
- const words = (s ) => String(s||"").toLowerCase().match(/[\p{L}\p{N}]+/gu) || [];
1412
+ // work in, so that "New Zealand." and "new zealand" are the same two words --
1413
+ // unless [keepCase], for a metric whose Ignore case is off.
1414
+ const words = (s , keepCase = false) =>
1415
+ (keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
1327
1416
 
1328
1417
  // The request Tagger builds, field for field -- when the target reads those
1329
1418
  // fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
@@ -2039,11 +2128,12 @@ function budgetLabel(target ) {
2039
2128
  // A term is present when its words appear in some item, in order and adjacent:
2040
2129
  // "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
2041
2130
  // looser scores "new zealand" against "zealandia" and flatters every run.
2042
- function termIn(items , term ) {
2043
- const t = words(term);
2131
+ // Letter case counts only where a metric's Ignore case is off.
2132
+ function termIn(items , term , ignoreCase = true) {
2133
+ const t = words(term, !ignoreCase);
2044
2134
  if (!t.length) return false;
2045
2135
  return items.some(item => {
2046
- const w = words(item);
2136
+ const w = words(item, !ignoreCase);
2047
2137
  for (let i = 0; i + t.length <= w.length; i++) {
2048
2138
  if (t.every((x, j) => w[i + j] === x)) return true;
2049
2139
  }
@@ -2070,153 +2160,55 @@ function forbiddenIn(items , term , allow
2070
2160
  * check green. `evals-check.js` calls this directly.
2071
2161
  */
2072
2162
  function gradedSetFrom(ev ) {
2073
- return (ev.cases || []).map(c => ({ ...c, filename: caseFile(c), half: "cases" }) );
2163
+ return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
2074
2164
  }
2075
2165
 
2076
- /**
2077
- * The file a case grades, whichever of the two keys names it.
2078
- *
2079
- * One reader, so that accepting the older spelling is a fact about this
2080
- * function rather than a branch every caller carries. Everything that joins a
2081
- * case to an item -- the runner, the Datasets tab, a mapping against a Source
2082
- * -- goes through here.
2083
- */
2084
- function caseFile(kase ) { // vocab: the older spelling
2085
- const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
2086
- return typeof name === "string" ? name : "";
2166
+ /** The item a case grades: its `item`, or "" for none. The one reader, so
2167
+ everything that joins a case to an item -- the runner, the Library, a
2168
+ run's Results -- joins on the same key, exactly. */
2169
+ function caseItem(kase ) {
2170
+ return isStr(kase?.item) ? kase .item : "";
2087
2171
  }
2088
2172
 
2089
- /**
2090
- * A graded set with every case naming its file as `filename`.
2091
- *
2092
- * What `evalsJson` writes, so the committed bytes carry one key and a set
2093
- * re-synced from an app's own repository is normalised the first time it is
2094
- * exported. The key takes the place the older one held, so normalising a set
2095
- * changes the spelling of
2096
- * one key and not the order of any.
2097
- */
2098
- function canonicalCases(ev ) {
2099
- const OLD = "photo"; // vocab: the older spelling of filename
2100
- const set = ev ;
2101
- const cases = set && typeof set === "object" && Array.isArray(set.cases) ? set.cases : null;
2102
- if (!cases || !cases.some(c => c && typeof c === "object" && OLD in (c ))) return ev;
2103
- return {
2104
- ...set,
2105
- cases: cases.map(c => {
2106
- if (!c || typeof c !== "object" || !(OLD in (c ))) return c;
2107
- const out = {};
2108
- for (const [k, v] of Object.entries(c )) {
2109
- if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
2110
- else out[k] = v;
2111
- }
2112
- return out;
2113
- }),
2114
- };
2115
- }
2173
+ /** One graded case's reading of a reply, now: its own metrics, scored as a
2174
+ Metrics eval scores them (all must pass; weight 0 is watched, never
2175
+ scored), with what they found, missed and invented carried up, and
2176
+ `watch` the readings of weight 0. What the page draws and the runner's
2177
+ case-by-case report writes. A model-graded metric needs a grader and the
2178
+ wait for one, so here it reads as not met: the runner's `--run` asks it. */
2179
+
2180
+
2181
+
2182
+
2183
+
2184
+
2185
+
2186
+
2187
+
2188
+
2189
+
2190
+
2191
+
2192
+
2116
2193
 
2117
- /**
2118
- * One graded case, one result.
2119
- *
2120
- * Every expectation is a group, and a plain `expect` term is a group of one:
2121
- * the requirements are the `expect` terms and the `anyOf` groups alike, and
2122
- * the score is found requirements over all of them plus `forbid`. A count
2123
- * bound stays pass/fail: an item that produced two perfect terms when five
2124
- * were wanted has not done what was asked.
2125
- *
2126
- * Pure on purpose, and returning `reasons` as plain sentences rather than
2127
- * markup. The dashboard was the first consumer; `run-evals.js` is the second
2128
- * and CI the third, and neither can reach into a page for a rendered cell.
2129
- * Issue #248 wants a failure written out as a task an agent can act on.
2130
- */
2131
- function scoreCase(kase , res ) {
2132
- // A case that expects its answer discarded passes on a discard and on
2133
- // nothing else: the answer the job threw away is the finding.
2134
- if (kase.discarded === true) {
2135
- const thrown = !!res.error && res.error.startsWith("discarded: ");
2136
- const n = (res.terms || []).length;
2137
- return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
2138
- unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
2139
- reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
2140
- }
2141
- // A discarded answer is a failed item: whatever the pipeline produced
2142
- // along the way does not count.
2143
- const discarded = !!res.error;
2144
- const terms = discarded ? [] : (res.terms || []);
2145
- const expect = kase.expect || [], forbid = kase.forbid || [];
2146
-
2147
- // `missed` stays populated for a discarded reply even though it reads as
2148
- // vacuous, because the score depends on it: docs/datasets.md has such a reply
2149
- // scoring zero against everything the entry asked for, groups included, and
2150
- // `addToTally` gets there through `found + missed + invented`. Clear it and
2151
- // a run that discarded every item would contribute nothing to the
2152
- // total instead of contributing a nought, which flatters it.
2153
- //
2154
- // One member of a group is enough. Without this rule the set manufactures
2155
- // failures out of synonyms.
2156
- //
2157
- // A requirement reads back as a term when it had one member and as the
2158
- // group itself where a synonym list was allowed, so `missed` carries just
2159
- // enough to state the reason: "FAIL: Missed dog" for a plain expect term,
2160
- // "none of dog / puppy" for a group.
2161
- //
2162
- // Nothing satisfies a group when nothing was stored, so a discarded reply
2163
- // misses every requirement rather than none -- the same cast that makes its
2164
- // count bounds not breached by one. `unmet` is a finding only now that
2165
- // `missed` names the unsatisfied groups itself, but the run report has
2166
- // always carried it, so it stays.
2167
- const met = (g ) => g.some(t => termIn(terms, t));
2168
- const spoken = (g ) => g.length === 1 ? g[0] : g;
2169
- const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
2170
- const found = requirements.filter(met).map(spoken);
2171
- const missed = requirements.filter(g => !met(g)).map(spoken);
2172
- const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
2173
- const unmet = discarded ? []
2174
- : (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
2175
-
2176
- const n = terms.length;
2177
- // A null bound is not checked -- and neither is a bound on a reply that was
2178
- // discarded. Nothing was counted out and found wanting there; the reply was
2179
- // thrown away before it had a count, and saying "0 items, wanted at least 4"
2180
- // states a second failure that never happened.
2181
- const under = !discarded && kase.minCount != null && n < kase.minCount;
2182
- const over = !discarded && kase.maxCount != null && n > kase.maxCount;
2183
-
2184
- const denom = requirements.length + invented.length;
2185
- // A case that passes is one whose whole expectation was met, not one that
2186
- // scored well: an unsatisfied group is in `missed` alongside any term
2187
- // missed, so pass needs nothing more than the terms already covered.
2188
- const pass = !discarded && !missed.length && !invented.length && !under && !over;
2189
-
2190
- // `found` and `missed` are groups, not terms, so the reasons word them one
2191
- // by one: a bare term under one heading, a group on its own line.
2192
- const reasons = [];
2193
- if (discarded) {
2194
- // Everything else would be derived from this one fact -- there are no
2195
- // terms -- and would bury it. `scoreHtml` above suppresses `missed` for
2196
- // the same reason: the reason that matters is already on the row.
2197
- reasons.push(`Error - ${res.error}`);
2198
- } else {
2199
- const plain = missed.filter(t => typeof t === "string");
2200
- if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
2201
- for (const g of missed) {
2202
- if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
2194
+ function readCase(kase , res , plain = false) {
2195
+ const input = metricInput(res , kase, { plain });
2196
+ const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
2197
+ const r = readMetric(m, input, {});
2198
+ if (r && typeof (r ).then === "function") {
2199
+ return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
2200
+ pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
2203
2201
  }
2204
- if (invented.length) reasons.push(`invented ${invented.join(", ")}`);
2205
- if (under) reasons.push(`${n} items, wanted at least ${kase.minCount}`);
2206
- if (over) reasons.push(`${n} items, wanted at most ${kase.maxCount}`);
2207
- }
2208
-
2209
- // Observed rather than scored: a dataset that wants to know whether some
2210
- // terms turn up, without grading on them, lists them as `watch`.
2211
- const watch = kase.watch || [];
2212
- // `null`, not 1, when there are no requirements and nothing forbidden turned
2213
- // up: there is no score to report. Returning 1 there printed "fail 100%"
2214
- // beside an entry that asked for nothing -- a shape `evals-check.js` allows
2215
- // even though every graded case now names an expectation.
2216
- return { pass, score: denom ? found.length / denom : null, discarded,
2217
- found, missed, invented, unmet, under, over, count: n,
2218
- watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
2219
- reasons };
2202
+ return r ? [r ] : [];
2203
+ });
2204
+ const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
2205
+ const counted = metrics.filter(r => r.weight !== 0);
2206
+ return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
2207
+ discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
2208
+ // A reply the job threw away fails on that one fact, which would
2209
+ // be buried under everything that derives from it.
2210
+ reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
2211
+ watch: metrics.filter(r => r.weight === 0) };
2220
2212
  }
2221
2213
 
2222
2214
  /**
@@ -2257,11 +2249,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
2257
2249
  // docs/datasets.md: an item naming more terms should weigh more than
2258
2250
  // one naming fewer, and averaging percentages lets the easiest case carry the
2259
2251
  // score.
2260
- function addToTally(t , s ) {
2252
+ function addToTally(t , s ) {
2253
+ const found = s.found?.length ?? 0;
2261
2254
  t.ran++;
2262
2255
  if (s.pass) t.passed++;
2263
- t.found += s.found.length;
2264
- t.of += s.found.length + s.missed.length + s.invented.length;
2256
+ t.found += found;
2257
+ t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
2265
2258
  }
2266
2259
 
2267
2260
  // The headline percentage, in one place because it is a number and this file
@@ -2276,7 +2269,7 @@ function tallyPercent(t ) {
2276
2269
  // A whole-run assertion with no graded set: All of / Any of / None of, a
2277
2270
  // parse, a count, an exact reply and a length bound. It lives in the shared
2278
2271
  // core because it is a pass/fail the tab, the worker and History all have to
2279
- // agree on -- the same one-definition rule that keeps `scoreCase` here.
2272
+ // agree on -- the same one-definition rule that keeps the case reader here.
2280
2273
  function parseCount(raw , parse ) {
2281
2274
  // An unparsed reply is one result, not no result. It used to return null,
2282
2275
  // which made a Count of 1 fail as "null results" against a reply that
@@ -2666,10 +2659,10 @@ function applyModifiers (list , kind ,
2666
2659
  // queue and run-evals.js alike, and every one of them reads it through the
2667
2660
  // functions below rather than through a translation of its own.
2668
2661
  //
2669
- // The lab is generic, so what a reply is, what a test scores and how a value
2662
+ // The lab is generic, so what a reply is, what an eval scores and how a value
2670
2663
  // is changed are registry entries. The ones here are the lab's own, and so
2671
2664
  // are kinds/list.ts's -- the List kind and its modifiers. A dataset
2672
- // registers nothing: it is data a graded test names by id, and a run carries.
2665
+ // registers nothing: it is data a graded eval names by id, and a run carries.
2673
2666
 
2674
2667
  // 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
2675
2668
  // version 1 kept them as two maps (tokens: { values, blocks }).
@@ -2685,14 +2678,16 @@ function applyModifiers (list , kind ,
2685
2678
  // 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
2686
2679
  // 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
2687
2680
  // Call (Prompt: the image flag and token mappings), Read Reply (the output).
2688
- // 9: a test is Metrics. A Single Test and a Graded set are read as the Metrics
2681
+ // 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
2689
2682
  // they convert to (LEGACY_TESTS' toMetrics, proven equal by
2690
2683
  // metrics-parity-check.js).
2691
2684
  // 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
2692
2685
  // Content, the flow step, Attach Image, Token mappings) and Responses (Read
2693
2686
  // as, Response Format Validation, modifiers) -- and its call is gone: each
2694
2687
  // scenario is a target, whose own step in each job is what it sends there.
2695
- const PIPELINE_VERSION = 10 ;
2688
+ // 11: `tests` are `evals`: the key renames and nothing in an eval changes,
2689
+ // so a stored result's scores, keyed by eval id, read as they did.
2690
+ const PIPELINE_VERSION = 12 ;
2696
2691
 
2697
2692
  // Plain objects, so an entry is added by assignment and a reader never needs
2698
2693
  // to know which registered it.
@@ -2700,7 +2695,7 @@ const STEP_TYPES = Object.create(null);
2700
2695
  const CONTENT_TYPES = Object.create(null);
2701
2696
  const OUTPUT_KINDS = Object.create(null);
2702
2697
  const MODIFIERS = Object.create(null);
2703
- const TEST_TYPES = Object.create(null);
2698
+ const EVAL_TYPES = Object.create(null);
2704
2699
  const METRICS = Object.create(null);
2705
2700
  const SOURCE_TYPES = Object.create(null);
2706
2701
 
@@ -2714,11 +2709,15 @@ let defaultOutputKind = "text";
2714
2709
  const defaultKind = () => defaultOutputKind;
2715
2710
  const kindOf = (st ) => st.kind ?? defaultOutputKind;
2716
2711
 
2712
+ /** A module's eval types, under either spelling (Kinds.testTypes). */
2713
+ const evalTypesOf = (k ) =>
2714
+ ({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
2715
+
2717
2716
  /** A module's entries into the registries, in one call. */
2718
2717
  function registerKinds(k ) {
2719
2718
  Object.assign(OUTPUT_KINDS, k.outputKinds || {});
2720
2719
  Object.assign(MODIFIERS, k.modifiers || {});
2721
- Object.assign(TEST_TYPES, k.testTypes || {});
2720
+ Object.assign(EVAL_TYPES, evalTypesOf(k));
2722
2721
  Object.assign(SOURCE_TYPES, k.sourceTypes || {});
2723
2722
  Object.assign(METRICS, k.metrics || {});
2724
2723
  if (k.defaultKind) defaultOutputKind = k.defaultKind;
@@ -2749,7 +2748,7 @@ function pluginHost(pluginId ) {
2749
2748
  registerKinds(k) {
2750
2749
  taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
2751
2750
  taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
2752
- taken(TEST_TYPES, "test type", Object.keys(k.testTypes || {}));
2751
+ taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
2753
2752
  taken(METRICS, "metric", Object.keys(k.metrics || {}));
2754
2753
  // The server keeps its own copy of the Source types, to refuse a row of
2755
2754
  // one it does not know, and a plugin never reaches the server's code.
@@ -3015,11 +3014,9 @@ LEGACY_TESTS.single = {
3015
3014
  },
3016
3015
  };
3017
3016
 
3018
- // A dataset's cases, scored item by item with scoreCase. The test
3019
- // references the dataset the way a pipeline references a Source, and the run
3020
- // carries the body it was submitted against. It scores the terms a value
3021
- // yields, so it accepts every kind that yields any: plain text has none, and
3022
- // would fail every case.
3017
+ // A dataset's cases, scored item by item. The eval references the dataset
3018
+ // the way a pipeline references a Source, and the run carries the body it
3019
+ // was submitted against.
3023
3020
  LEGACY_TESTS.graded = {
3024
3021
  label: "Graded set",
3025
3022
  fields: ["type", "dataset"],
@@ -3027,16 +3024,15 @@ LEGACY_TESTS.graded = {
3027
3024
  validate(t, ctx, bad){
3028
3025
  const d = t.dataset;
3029
3026
  if (!isRef(d) || (d.version != null && !isStr(d.version))) {
3030
- return void bad.push("a graded test has to name its dataset as { id, name }");
3027
+ return void bad.push("a graded eval has to name its dataset as { id, name }");
3031
3028
  }
3032
3029
  if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
3033
3030
  },
3034
- score: (t, kase, res) => scoreCase(kase, res),
3035
3031
  };
3036
3032
 
3037
3033
  // Metrics: checks of each reply, deterministic or model-graded, from the
3038
3034
  // METRICS registry (metrics/builtin.ts registers the lab's own), with the
3039
- // test's own list for every item and a case's `metrics` for its item. Scored
3035
+ // eval's own list for every item and a case's `metrics` for its item. Scored
3040
3036
  // all-must-pass -- every metric passes -- or weighted: points, each metric's
3041
3037
  // score times its weight, against a threshold (#150's points, a negative
3042
3038
  // weight taking them away).
@@ -3045,6 +3041,13 @@ LEGACY_TESTS.graded = {
3045
3041
  const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
3046
3042
  const SCORING_MODES = ["all", "weighted"];
3047
3043
 
3044
+ /** A metric's list option -- Contains all's values, Contains's exceptions --
3045
+ one entry a line, or a list of them. */
3046
+ function metricLines(v ) {
3047
+ const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
3048
+ return all.map(x => x.trim()).filter(Boolean);
3049
+ }
3050
+
3048
3051
  /** What is wrong with a list of metrics, as sentences naming [at]. */
3049
3052
  function metricsProblems(list , at , bad ) {
3050
3053
  if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
@@ -3114,14 +3117,16 @@ function readMetric(m , input , ctx )
3114
3117
  } catch (e) { return timed(failed(e)); }
3115
3118
  }
3116
3119
 
3117
- /** A test's score from its metrics' readings, the way [mode] says: all must
3120
+ /** An eval's score from its metrics' readings, the way [mode] says: all must
3118
3121
  pass, or weighted points against [threshold]. What a case metric found,
3119
3122
  missed and invented is carried up, for a run's totals. Null for none. */
3120
3123
  function scoreOf(metrics , mode = "all", threshold = null) {
3121
3124
  if (!metrics.length) return null;
3122
- const detail = metrics.some(r => r.found || r.missed || r.invented) ? {
3123
- found: metrics.flatMap(r => r.found ?? []), missed: metrics.flatMap(r => r.missed ?? []),
3124
- invented: metrics.flatMap(r => r.invented ?? []) } : {};
3125
+ // A watched reading (weight 0) is reported, and counts toward nothing.
3126
+ const scored = metrics.filter(r => r.weight !== 0);
3127
+ const detail = scored.some(r => r.found || r.missed || r.invented) ? {
3128
+ found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
3129
+ invented: scored.flatMap(r => r.invented ?? []) } : {};
3125
3130
  if (mode === "weighted") {
3126
3131
  const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
3127
3132
  return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
@@ -3164,7 +3169,7 @@ function readRun(list , run ) {
3164
3169
  });
3165
3170
  }
3166
3171
 
3167
- TEST_TYPES.metrics = {
3172
+ EVAL_TYPES.metrics = {
3168
3173
  label: "Metrics",
3169
3174
  description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3170
3175
  fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
@@ -3188,9 +3193,9 @@ TEST_TYPES.metrics = {
3188
3193
  if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
3189
3194
  if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
3190
3195
  if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
3191
- // The test's own grader, or the lab's.
3196
+ // The eval's own grader, or the lab's.
3192
3197
  const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
3193
- if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a Target profile the lab's grader");
3198
+ if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
3194
3199
  // A grader is asked words, and needs a model to ask.
3195
3200
  const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3196
3201
  if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
@@ -3205,59 +3210,35 @@ TEST_TYPES.metrics = {
3205
3210
  want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3206
3211
  .filter(Boolean).join(" ") })),
3207
3212
  profiles: t => (isRef(t.grader) ? [t.grader] : []),
3208
- // A run carries the lab's grader on a test that names none and may ask one:
3213
+ // A run carries the lab's grader on an eval that names none and may ask one:
3209
3214
  // a model-graded metric of its own, or a case's, over a dataset.
3210
3215
  resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
3211
3216
  ? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
3212
3217
  wholeRun: t => t.over === "run",
3213
3218
  // Over the whole run: every metric over the replies together, or each alone.
3214
- verdict(t, ress, kind){
3215
- const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
3216
- const metrics = readRun(t.metrics || [], run);
3217
- const s = scoreOf(metrics, t.mode, t.threshold);
3218
- const off = metrics.filter(r => !r.pass);
3219
- return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
3220
- ran: run.replies.length,
3221
- checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3222
- },
3223
- // Rule m{x} is the test's own metric x, read in that place on every item.
3219
+ verdict: (t, ress, kind) => wholeRunVerdict(t.metrics || [], ress, kind, t.mode, t.threshold),
3220
+ // Rule m{x} is the eval's own metric x, read in that place on every item.
3224
3221
  // A score stored before Metrics -- a Graded set's, read as its conversion --
3225
3222
  // has no readings of its own: its one metric's reading is the score's.
3226
3223
  ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3227
3224
  : Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
3228
- // Every item, with its case's own metrics where it has a case.
3229
- read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((kase?.metrics ) || [])],
3225
+ // Every item, with its case's own metrics where the eval names the
3226
+ // dataset and the item has a case there.
3227
+ read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
3230
3228
  metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3231
3229
  };
3232
3230
 
3233
3231
  // ---- the lab's own scorers, as metrics ----------------------------------------
3234
- // What the Graded set and the Single Test check, one metric each, so a test of
3235
- // either converts to Metrics that read a run exactly as it did
3236
- // (metrics-parity-check.js). Here rather than in metrics/builtin.ts because
3237
- // each is the core's own matcher.
3238
-
3239
- // An item's case, as the Graded set scores it: its expectations, forbidden
3240
- // terms and count bounds, with what it found, missed and invented kept.
3241
- METRICS.case = {
3242
- label: "Matches its case",
3243
- description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
3244
- perItem: true,
3245
- needsTerms: true,
3246
- options: [],
3247
- defaults: () => ({}),
3248
- score(input) {
3249
- if (!input.kase) return null;
3250
- const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
3251
- return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
3252
- found: s.found, missed: s.missed, invented: s.invented };
3253
- },
3254
- };
3232
+ // What the Single Test checks, one metric each, so an eval of it converts to
3233
+ // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
3234
+ // rather than in metrics/builtin.ts because each is the core's own matcher.
3255
3235
 
3256
3236
  // Items the reply holds -- all of them, or any -- matched as the lab matches
3257
3237
  // a term; a kind that yields none is matched in the replies' text instead,
3258
3238
  // ignoring case, as the Single Test did.
3259
3239
  METRICS["has-items"] = {
3260
3240
  label: "Has items",
3241
+ family: "The reply's text",
3261
3242
  description: "Passes when the reply holds all, or any, of the items listed.",
3262
3243
  options: [{ key: "values", label: "Items", type: "textarea" },
3263
3244
  { key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
@@ -3277,6 +3258,7 @@ METRICS["has-items"] = {
3277
3258
  // A reply's length in characters, once trimmed.
3278
3259
  METRICS.length = {
3279
3260
  label: "Length",
3261
+ family: "The reply's text",
3280
3262
  description: "Compares the reply's length in characters.",
3281
3263
  options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
3282
3264
  { key: "n", label: "Characters", type: "number" }],
@@ -3292,6 +3274,7 @@ METRICS.length = {
3292
3274
  // as the count says.
3293
3275
  METRICS["parse-count"] = {
3294
3276
  label: "Parses",
3277
+ family: "The result",
3295
3278
  description: "Passes when the reply reads as the format chosen, holding the number of results set.",
3296
3279
  // Unformatted is one result a reply, as the Single Test counted it.
3297
3280
  options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
@@ -3310,12 +3293,13 @@ METRICS["parse-count"] = {
3310
3293
  },
3311
3294
  };
3312
3295
 
3313
- /** What every test carries, kept across a conversion. */
3296
+ /** What every eval carries, kept across a conversion. */
3314
3297
  const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
3315
3298
 
3316
- // A Graded set is a Metrics test over the same dataset, its one metric the case.
3299
+ // A Graded set is a Metrics eval over the same dataset, with none of its own:
3300
+ // each case's metrics are what it scored (dataset-parity-check.js).
3317
3301
  LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
3318
- dataset: t.dataset, over: "item", metrics: [{ type: "case" }] });
3302
+ dataset: t.dataset, over: "item", metrics: [] });
3319
3303
 
3320
3304
  // A Single Test is Metrics over the whole run: its lists over the run's items
3321
3305
  // together, and exact, length and parse over each reply alone.
@@ -3333,7 +3317,39 @@ LEGACY_TESTS.single .toMetrics = t => {
3333
3317
  return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
3334
3318
  };
3335
3319
 
3336
- /** The grader a Metrics test names, reached through what the runner hands it. */
3320
+ /** [list] read over a whole run's replies so far, scored the way [mode]
3321
+ says: a Metrics eval's verdict over the run, and an eval group's Whole run. */
3322
+ function wholeRunVerdict(list , ress , kind ,
3323
+ mode = "all", threshold = null) {
3324
+ const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
3325
+ const metrics = readRun(list, run);
3326
+ const s = scoreOf(metrics, mode, threshold);
3327
+ const off = metrics.filter(r => !r.pass);
3328
+ return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
3329
+ ran: run.replies.length,
3330
+ checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3331
+ }
3332
+
3333
+ /**
3334
+ * An eval group's verdict on one item (docs/pipeline-model.md §17): its
3335
+ * Every item metrics and [kase]'s own -- the item's case in the group, or
3336
+ * null where it has none -- read of the reply under the group's scoring.
3337
+ * The reading a Metrics eval naming the group as its dataset gives, which
3338
+ * groups-check.js holds it to. Null where no metric had anything to read.
3339
+ */
3340
+ function readGroup(group , kase , res , more = {}) {
3341
+ return readMetrics([...group.every, ...(kase?.metrics ?? [])], metricInput(res, kase, more), graderCtx(group, more),
3342
+ group.scoring.mode, group.scoring.threshold);
3343
+ }
3344
+
3345
+ /** An eval group's Whole run verdict, from the replies so far: its `run`
3346
+ metrics under its scoring. Null for a group with no Whole run metrics,
3347
+ which has nothing to say of a run. */
3348
+ function readGroupRun(group , ress , kind = null) {
3349
+ return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
3350
+ }
3351
+
3352
+ /** The grader a Metrics eval names, reached through what the runner hands it. */
3337
3353
  function graderCtx(t , more ) {
3338
3354
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3339
3355
  return ask ? { ask } : {};
@@ -3343,12 +3359,14 @@ function graderCtx(t , more ) {
3343
3359
  for one with none set: what History prints beside its name. */
3344
3360
  function metricSummary(m ) {
3345
3361
  const entry = METRICS[m.type];
3346
- // Each option as it reads in the editor: a choice's label, a box that is
3347
- // ticked by its own label, text by its first line.
3362
+ // Each option as it reads in the editor: a choice's label, a box by its
3363
+ // own label where it is not as a new metric has it (ticked, or "off"),
3364
+ // text by its first line.
3365
+ const fresh = entry?.defaults() ?? {};
3348
3366
  const said = (entry?.options || []).map(o => {
3349
3367
  const v = m[o.key];
3350
3368
  if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3351
- if (o.type === "checkbox") return v ? o.label.toLowerCase() : "";
3369
+ if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
3352
3370
  // Text that runs to lines (a schema) reads as its first words, run together.
3353
3371
  return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3354
3372
  }).filter(Boolean);
@@ -3531,7 +3549,7 @@ STEP_TYPES.readAs = {
3531
3549
  // Response Format Validation: the kind a reply is read as.
3532
3550
  STEP_TYPES.readReply = {
3533
3551
  label: "Read Reply", slot: "responses", rank: 1,
3534
- description: "How the reply is read before tests and later jobs see it.",
3552
+ description: "How the reply is read before evals and later jobs see it.",
3535
3553
  in: "text", out: step => step.out?.kind,
3536
3554
  // Read inside the stage: runPipeline parses each reply as it comes back.
3537
3555
  apply: "runPipeline",
@@ -3787,7 +3805,7 @@ STEP_TYPES.job = {
3787
3805
  if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
3788
3806
  const steps = c.steps;
3789
3807
  // A job's steps are the entries with a stage; one without (a job, the
3790
- // tests) or of no type the lab has is not one.
3808
+ // evals) or of no type the lab has is not one.
3791
3809
  const unknown = steps.find(st => !slotOf(st));
3792
3810
  if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
3793
3811
  if (steps.some(st => slotOf(st) === "target")) {
@@ -3808,54 +3826,55 @@ STEP_TYPES.job = {
3808
3826
  }
3809
3827
  },
3810
3828
  };
3811
- // What a test carries besides its type's own fields.
3812
- const TEST_FIELDS = ["id", "name", "continueOnFailure"];
3829
+ // What an eval carries besides its type's own fields.
3830
+ const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
3813
3831
 
3814
- /** Test [j]'s name, or the number it has always shown. */
3815
- const testLabel = (doc , j ) =>
3816
- (doc.tests?.[j]?.name || "").trim() || `Test ${j + 1}`;
3832
+ /** Eval [j]'s name, or the number it has always shown. */
3833
+ const evalLabel = (doc , j ) =>
3834
+ (doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
3817
3835
 
3818
- /** A test type that settles over the whole run rather than item by item. */
3836
+ /** An eval type that settles over the whole run rather than item by item. */
3819
3837
  const isWholeRun = (type , t ) =>
3820
3838
  type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
3821
3839
 
3822
3840
  /**
3823
- * A document's tests as a list, whatever version wrote it: a queue row keeps
3824
- * the run document it was submitted with, so a run from before version 6
3825
- * holds one test or null, read here as the upgrade reads it (testsList).
3841
+ * A document's evals as a list, whatever version wrote it: a queue row keeps
3842
+ * the run document it was submitted with, so a run from before version 11
3843
+ * spells them `tests`, and one from before version 6 holds one or null, read
3844
+ * here as the upgrade reads it (testsList, evalsKey).
3826
3845
  */
3827
- function testsOf(doc ) {
3828
- const t = doc?.tests;
3846
+ function evalsOf(doc ) {
3847
+ const t = doc?.evals ?? doc?.tests;
3829
3848
  if (Array.isArray(t)) return t ;
3830
3849
  return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3831
3850
  }
3832
3851
 
3833
- /** The dataset a document's tests grade against, where one does: a run
3852
+ /** The dataset a document's evals grade against, where one does: a run
3834
3853
  grades against one (validatePipeline says so), so the first names it. */
3835
- function testsDataset(doc ) {
3836
- const t = testsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3854
+ function evalsDataset(doc ) {
3855
+ const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3837
3856
  return t && "dataset" in t ? t.dataset : null;
3838
3857
  }
3839
3858
 
3840
- STEP_TYPES.tests = {
3859
+ STEP_TYPES.evals = {
3841
3860
  in: "results", out: "verdict",
3842
3861
  apply: "score",
3843
- validate(tests, ctx, bad, lastKind, doc){
3844
- if (!Array.isArray(tests)) return void bad.push("tests has to be a list, empty for an unscored run");
3862
+ validate(evals, ctx, bad, lastKind, doc){
3863
+ if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
3845
3864
  const seen = new Set ();
3846
- tests.forEach((t , j ) => {
3847
- const at = testLabel(doc, j);
3848
- const type = isObj(t) && TEST_TYPES[t.type];
3865
+ evals.forEach((t , j ) => {
3866
+ const at = evalLabel(doc, j);
3867
+ const type = isObj(t) && EVAL_TYPES[t.type];
3849
3868
  if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
3850
3869
  const before = bad.length;
3851
3870
  if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
3852
- else if (seen.has(t.id)) bad.push(`${at} has the id of another test`);
3871
+ else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
3853
3872
  else seen.add(t.id);
3854
3873
  if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
3855
3874
  if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
3856
- onlyFields(t, at, [...type.fields, ...TEST_FIELDS], bad);
3875
+ onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
3857
3876
  type.validate(t, ctx, bad);
3858
- // Which kinds a test scores is only worth saying of a test that is whole.
3877
+ // Which kinds an eval scores is only worth saying of an eval that is whole.
3859
3878
  if (bad.length > before) return;
3860
3879
  const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
3861
3880
  if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
@@ -3864,14 +3883,14 @@ STEP_TYPES.tests = {
3864
3883
  }
3865
3884
  });
3866
3885
  // A run is handed one dataset's body to grade against (server-side-runs §4).
3867
- const named = new Set(tests.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3868
- if (named.size > 1) bad.push("the tests grade against one dataset at a time");
3886
+ const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3887
+ if (named.size > 1) bad.push("the evals grade against one dataset at a time");
3869
3888
  },
3870
3889
  };
3871
3890
 
3872
3891
  // ---- the document -------------------------------------------------------------
3873
3892
 
3874
- const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "tests"];
3893
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
3875
3894
  // What resolving adds, and nothing else: the profiles it resolved to and the
3876
3895
  // run's own comment, which belongs to the run and never to the pipeline.
3877
3896
  const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
@@ -3888,7 +3907,7 @@ function versionProblem(doc ) {
3888
3907
  /** What an upgrade from version 2 needs from outside the document: the rules
3889
3908
  of the dataset a pipeline was graded against, which the retired
3890
3909
  version-2 list kind read every reply under. Without them a job is upgraded with no
3891
- rules, as a run with no graded test parsed. */
3910
+ rules, as a run with no graded eval parsed. */
3892
3911
 
3893
3912
 
3894
3913
 
@@ -3913,7 +3932,8 @@ function versionProblem(doc ) {
3913
3932
  * as it was, for versionProblem to name. A copy: the caller's document is not
3914
3933
  * touched. From version 5, its test -- or none -- becomes a list of one (or
3915
3934
  * none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
3916
- * renames and every job's `type` becomes `"job"`.
3935
+ * renames and every job's `type` becomes `"job"`. From version 10, its
3936
+ * `tests` are `evals` (evalsKey).
3917
3937
  */
3918
3938
  /** Every id a current document must carry, minted for the ones [doc] lacks.
3919
3939
  An id it already has is kept. */
@@ -3949,6 +3969,9 @@ function profileRefs(doc ) {
3949
3969
  }
3950
3970
  }
3951
3971
 
3972
+ /** [doc] with its profile references cut (profileRefs), for chaining. */
3973
+ const cutRefs = (doc ) => { profileRefs(doc); return doc; };
3974
+
3952
3975
  /** Whether any profile reference in a pipeline holds more than { id, name }. */
3953
3976
  function fatProfileRef(doc ) {
3954
3977
  const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
@@ -3967,9 +3990,9 @@ function testsList(doc ) {
3967
3990
  return doc;
3968
3991
  }
3969
3992
 
3970
- /** A new test of [type] at the lab's defaults, named [name], continuing on failure. */
3971
- function newTest(type , name = "", fields = {}) {
3972
- const own = TEST_TYPES[type]?.defaults?.() ?? { type };
3993
+ /** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
3994
+ function newEval(type , name = "", fields = {}) {
3995
+ const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
3973
3996
  return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
3974
3997
  }
3975
3998
 
@@ -4067,6 +4090,29 @@ function targetsFromScenarios(next ) {
4067
4090
  return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
4068
4091
  }
4069
4092
 
4093
+ /** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
4094
+ upgraded document reads the same, key for key, and each eval is as it
4095
+ was. Every version before 11 passes through here. */
4096
+ function evalsKey(next ) {
4097
+ next.version = PIPELINE_VERSION;
4098
+ // `tests` is the spelling of every version before 11.
4099
+ if (!("tests" in next)) return next;
4100
+ return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
4101
+ .map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
4102
+ }
4103
+
4104
+ /** Version 11 to 12: a Contains metric's Ignore case holds where the reply
4105
+ is matched item by item, as it does where it is matched as text, and it
4106
+ is kept as written. Until version 11's last hours (#199) every Contains
4107
+ metric matched the reply's text and minded its Ignore case, so what a
4108
+ stored metric says is what its author meant; only the item-by-item
4109
+ matching #199 added, case-blind for a few hours, read it otherwise.
4110
+ Every version before 12 ends here. */
4111
+ function caseAsWritten(next ) {
4112
+ next.version = PIPELINE_VERSION;
4113
+ return next;
4114
+ }
4115
+
4070
4116
  function upgradePipeline (doc , ctx = {}) {
4071
4117
  // A current document is read as it is, but for a profile reference the Runs
4072
4118
  // tab saved whole (see profileRefs), which is cut back, and a step on a
@@ -4077,11 +4123,29 @@ function upgradePipeline (doc , ctx = {}) {
4077
4123
  out = clone(out) ;
4078
4124
  profileRefs(out);
4079
4125
  }
4080
- return localSteps(out, ctx) ;
4126
+ return localSteps(withoutCaseMetric(out), ctx) ;
4081
4127
  }
4082
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9].includes(doc.version )) return doc;
4083
- // Every version before 10 reads as version 9 first, then as 10.
4084
- return localSteps(targetsFromScenarios(nineOf(clone(doc) , ctx)), ctx) ;
4128
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
4129
+ if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
4130
+ // Every version before 10 reads as version 9 first, then as 10, then as 11
4131
+ // and 12.
4132
+ // A version-10 document is cut as a current one was (profileRefs); the
4133
+ // earlier ones are cut on their way through nineOf.
4134
+ const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4135
+ return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
4136
+ }
4137
+
4138
+ /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
4139
+ are metrics now (dataset body version 5), which an eval naming the
4140
+ dataset adds for each item, so the case's own metrics carry what that
4141
+ metric scored. Read so at every version, the current one included, as a
4142
+ document saved before it went still holds it. The same document where
4143
+ none does. */
4144
+ function withoutCaseMetric(doc ) {
4145
+ const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
4146
+ if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
4147
+ return { ...doc, evals: doc.evals.map((t ) => (has(t)
4148
+ ? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
4085
4149
  }
4086
4150
 
4087
4151
  /** [doc] with each step asked of a profile whose type a target step stands
@@ -4165,7 +4229,7 @@ function tokenMappingsFromV1(set ) {
4165
4229
  function blankPipeline(opts = {}) {
4166
4230
  const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
4167
4231
  return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
4168
- jobs: [job], targets: [], tests: [] };
4232
+ jobs: [job], targets: [], evals: [] };
4169
4233
  }
4170
4234
 
4171
4235
  /**
@@ -4303,7 +4367,7 @@ function validatePipeline(input , ctx = {}) {
4303
4367
  : `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
4304
4368
  }
4305
4369
  });
4306
- STEP_TYPES.tests .validate(doc.tests, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4370
+ STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4307
4371
  if (bad.length) return bad;
4308
4372
 
4309
4373
  // Asked before anything is sent, so a misspelt token costs nothing and
@@ -4400,9 +4464,9 @@ function profileIds(doc )
4400
4464
  if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4401
4465
  }
4402
4466
  }
4403
- // Then the ones a test asks (a grader), so a run carries them too.
4404
- for (const t of testsOf(doc)) {
4405
- for (const ref of TEST_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4467
+ // Then the ones an eval asks (a grader), so a run carries them too.
4468
+ for (const t of evalsOf(doc)) {
4469
+ for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4406
4470
  }
4407
4471
  return ids;
4408
4472
  }
@@ -4415,9 +4479,9 @@ function profileIds(doc )
4415
4479
  function resolvePipeline(doc ,
4416
4480
  ctx = {}) {
4417
4481
  const run = clone(doc) ;
4418
- // What the lab supplies a test -- its grader -- before the profiles it
4482
+ // What the lab supplies an eval -- its grader -- before the profiles it
4419
4483
  // asks are carried.
4420
- run.tests = (Array.isArray(run.tests) ? run.tests : []).map((t ) => TEST_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4484
+ run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4421
4485
  run.profiles = {};
4422
4486
  for (const id of profileIds(run)) {
4423
4487
  const p = ctx.profiles?.(id);
@@ -4426,7 +4490,7 @@ function resolvePipeline(doc ,
4426
4490
  const content = contentOf(run) ;
4427
4491
  if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4428
4492
  if (ctx.datasetVersion) {
4429
- for (const t of run.tests || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4493
+ for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4430
4494
  }
4431
4495
  if (isStr(ctx.comment)) run.comment = ctx.comment;
4432
4496
  return run;
@@ -4440,7 +4504,7 @@ function pipelineOfRun(run , ctx = {}) {
4440
4504
  delete doc.plugins;
4441
4505
  const content = contentOf(doc) ;
4442
4506
  if (content) { delete content.files; delete content.revs; }
4443
- for (const t of doc.tests || []) if (isObj(t.dataset)) delete t.dataset.version;
4507
+ for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
4444
4508
  return doc;
4445
4509
  }
4446
4510
 
@@ -4498,7 +4562,7 @@ function importPipeline(input , ctx = {}) {
4498
4562
  if (st?.profile) st.profile = remap(st.profile, "profile");
4499
4563
  }
4500
4564
  }
4501
- for (const t of Array.isArray(next.tests) ? next.tests : []) {
4565
+ for (const t of Array.isArray(next.evals) ? next.evals : []) {
4502
4566
  if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4503
4567
  }
4504
4568
  // Its references are this lab's now, so a step asking this lab's Echo
@@ -4530,7 +4594,7 @@ function mintIds(doc ) {
4530
4594
  for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
4531
4595
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4532
4596
  }
4533
- for (const t of Array.isArray(doc.tests) ? doc.tests : []) {
4597
+ for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4534
4598
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4535
4599
  }
4536
4600
  return doc;
@@ -4608,14 +4672,14 @@ function modifierSummary(m ) {
4608
4672
  return said.length ? said.join(", ") : "on";
4609
4673
  }
4610
4674
 
4611
- // ---- the tests, in order (docs/pipeline-model.md §3) ---------------------------
4612
- // Tests only read a run: none changes what a later one sees, and none stops
4675
+ // ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
4676
+ // Evals only read a run: none changes what a later one sees, and none stops
4613
4677
  // the model being sent the next item. What order changes is Continue on
4614
- // failure. A per-item test that fails and does not continue stops the tests
4678
+ // failure. A per-item eval that fails and does not continue stops the evals
4615
4679
  // after it for that item alone -- they read Skipped there, and a whole-run
4616
- // test after it pools the items it did not stop. A whole-run test settles
4680
+ // eval after it pools the items it did not stop. A whole-run eval settles
4617
4681
  // once every item is in, and one that fails then and does not continue
4618
- // leaves every test after it Skipped.
4682
+ // leaves every eval after it Skipped.
4619
4683
 
4620
4684
  const SKIPPED = Object.freeze({ skipped: true });
4621
4685
  const isSkipped = (s ) => isObj(s) && s.skipped === true;
@@ -4623,16 +4687,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
4623
4687
  const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
4624
4688
 
4625
4689
  /**
4626
- * One reply's scores under a run's per-item tests, by test id, in order: a
4627
- * test with no case to score leaves no entry, and one after a failure that
4690
+ * One reply's scores under a run's per-item evals, by eval id, in order: a
4691
+ * eval with no case to score leaves no entry, and one after a failure that
4628
4692
  * does not continue reads Skipped. Null where nothing was scored.
4629
4693
  */
4630
- function itemScores(run , kase , res ) {
4694
+ function itemScores(run , kase , res ) {
4631
4695
  if (!kase) return null;
4632
- const out = {};
4696
+ const out = {};
4633
4697
  let stopped = false;
4634
- for (const t of testsOf(run)) {
4635
- const type = TEST_TYPES[t.type];
4698
+ for (const t of evalsOf(run)) {
4699
+ const type = EVAL_TYPES[t.type];
4636
4700
  if (!type?.score) continue;
4637
4701
  if (stopped) { out[t.id] = SKIPPED; continue; }
4638
4702
  const s = type.score(t, kase, res);
@@ -4650,19 +4714,19 @@ function productionOf(run , record )
4650
4714
  }
4651
4715
 
4652
4716
  /**
4653
- * itemScores for the runner, which can wait: a test that reads every item
4717
+ * itemScores for the runner, which can wait: an eval that reads every item
4654
4718
  * (`read`: the Metrics, which may ask a grader) scores one with no case too.
4655
4719
  */
4656
4720
  async function itemScoresAsync(run , kase , res ,
4657
4721
  more = {}) {
4658
4722
  const out = {};
4659
4723
  let stopped = false;
4660
- for (const t of testsOf(run)) {
4661
- const type = TEST_TYPES[t.type];
4724
+ for (const t of evalsOf(run)) {
4725
+ const type = EVAL_TYPES[t.type];
4662
4726
  if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
4663
4727
  if (stopped) { out[t.id] = SKIPPED; continue; }
4664
- // A test with nothing to read on this item leaves no entry, as a graded
4665
- // test does on an item with no case.
4728
+ // An eval with nothing to read on this item leaves no entry, as a graded
4729
+ // eval does on an item with no case.
4666
4730
  const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
4667
4731
  : type.score (t, kase , res);
4668
4732
  if (!s) continue;
@@ -4673,22 +4737,22 @@ async function itemScoresAsync(run , kase ,
4673
4737
  }
4674
4738
 
4675
4739
  /**
4676
- * Every test's reading of scenario [i] of a run, in the run's order, from
4740
+ * Every eval's reading of scenario [i] of a run, in the run's order, from
4677
4741
  * the items so far. [settled] says every item is in: only then has a
4678
- * whole-run test settled, so only then does its failure skip the tests
4742
+ * whole-run eval settled, so only then does its failure skip the evals
4679
4743
  * after it.
4680
4744
  */
4681
- function scenarioTests(run , i , items ,
4745
+ function scenarioEvals(run , i , items ,
4682
4746
  settled = true) {
4683
4747
  const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
4684
- // Which items a per-item test has stopped so far, for the tests after it.
4748
+ // Which items a per-item eval has stopped so far, for the evals after it.
4685
4749
  const stopped = cells.map(() => false);
4686
4750
  let skipRest = false;
4687
- const tests = testsOf(run);
4688
- return tests.map((t, j) => {
4689
- const type = TEST_TYPES[t.type];
4751
+ const evals = evalsOf(run);
4752
+ return evals.map((t, j) => {
4753
+ const type = EVAL_TYPES[t.type];
4690
4754
  const whole = isWholeRun(type, t);
4691
- const base = { id: t.id, label: testLabel({ tests }, j), whole, skipped: skipRest,
4755
+ const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
4692
4756
  verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
4693
4757
  ran: 0, passed: 0, skippedItems: 0 };
4694
4758
  if (skipRest) {
@@ -4715,7 +4779,7 @@ function scenarioTests(run , i , items
4715
4779
  });
4716
4780
  }
4717
4781
 
4718
- /** A scenario's pass or fail over every test that read it: null where none
4782
+ /** A scenario's pass or fail over every eval that read it: null where none
4719
4783
  has anything to say yet. */
4720
4784
  function scenarioPasses(outcomes ) {
4721
4785
  let said = false;
@@ -4756,95 +4820,95 @@ function validateEvals(ev , files
4756
4820
  return ["the graded set has to be a JSON object with a `cases` list"];
4757
4821
  }
4758
4822
  if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
4823
+ if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
4824
+ // An eval group's own sections (version 7): what a Metrics eval held.
4825
+ if (ev.scoring != null) {
4826
+ const sc = ev.scoring;
4827
+ if (!isObj(sc) || !SCORING_MODES.includes(sc.mode)) bad.push(`a group is scored ${SCORING_MODES.join(" or ")}`);
4828
+ else if (sc.mode === "weighted" && typeof sc.threshold !== "number") bad.push("a group scored in points needs Pass at: the points an item has to reach");
4829
+ else if (sc.threshold != null && typeof sc.threshold !== "number") bad.push("a group's Pass at has to be a number");
4830
+ }
4831
+ if (ev.grader != null && !isRef(ev.grader)) bad.push("a group names its grader as { id, name }, or null");
4832
+ if (ev.every != null) metricsProblems(ev.every, "Every item", bad);
4833
+ if (ev.run != null) {
4834
+ metricsProblems(ev.run, "Whole run", bad);
4835
+ // Read once over every reply: nothing that reads one item, or asks a grader.
4836
+ for (const m of Array.isArray(ev.run) ? ev.run : []) {
4837
+ const entry = isObj(m) ? METRICS[m.type] : undefined;
4838
+ if (entry?.graded) bad.push(`Whole run: ${entry.label} is model-graded, so it cannot read a whole run`);
4839
+ else if (entry?.perItem) bad.push(`Whole run: ${entry.label} reads one item at a time, so it cannot read a whole run`);
4840
+ }
4841
+ }
4759
4842
  if (bad.length) return bad;
4760
4843
 
4761
4844
  const graded = ev.cases || [];
4762
4845
  // Identity first: every rule below reports which case is at fault, so a
4763
4846
  // case with no usable id makes the rest of the report unreadable.
4764
- const seen = new Map ();
4847
+ const seen = new Set ();
4765
4848
  for (const c of graded) {
4766
- {
4767
- if (!c || typeof c !== "object" || Array.isArray(c)) {
4768
- bad.push("cases holds something that is not a case");
4769
- continue;
4770
- }
4771
- if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
4772
- else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4773
- else seen.set(c.id, "cases");
4774
- if (!caseFile(c).trim()) {
4775
- bad.push(`${c.id || "a case"} names no file`);
4776
- }
4777
- for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
4778
- if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
4779
- }
4780
- for (const key of ["minCount", "maxCount"]) {
4781
- if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
4782
- }
4783
- if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
4784
- // A case's own metrics, which a Metrics test adds to its own for this item.
4785
- if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4849
+ if (!isObj(c)) {
4850
+ bad.push("cases holds something that is not a case");
4851
+ continue;
4786
4852
  }
4853
+ if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
4854
+ else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4855
+ else seen.add(c.id);
4856
+ if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
4857
+ if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
4858
+ if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
4859
+ if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4787
4860
  }
4788
4861
  if (bad.length) return bad;
4789
4862
 
4790
- // A term the scorer cannot match is a term that can never be produced, so
4791
- // the case is unpassable however well the model answers.
4863
+ // What a case says, read from its metrics: a term the scorer cannot match
4864
+ // can never be produced, so the case is unpassable however well the model
4865
+ // answers -- and the same goes for a bound no reply can meet.
4792
4866
  const matchable = (term ) => termIn([String(term).trim()], term);
4793
-
4867
+ const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
4794
4868
  for (const c of graded) {
4795
4869
  if (c.todo) continue;
4796
4870
  const say = (m ) => bad.push(`${c.id}: ${m}`);
4797
- const expect = c.expect || [], forbid = c.forbid || [], anyOf = c.anyOf || [];
4798
-
4799
- if (c.discarded === true) {
4871
+ const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
4872
+ if (!scored.length) { say("graded but states nothing to expect"); continue; }
4873
+ if (scored.some(m => m.type === "discarded" && !m.not)) {
4800
4874
  // What is discarded holds nothing, so there is nothing else to expect.
4801
- if (expect.length || anyOf.length || forbid.length || c.minCount != null || c.maxCount != null) {
4802
- say("expects its answer discarded, and states something the answer should hold too");
4803
- }
4875
+ if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
4804
4876
  continue;
4805
4877
  }
4806
- const metrics = Array.isArray(c.metrics) ? c.metrics : [];
4807
- if (!expect.length && !anyOf.length && !metrics.length) say("graded but states nothing to expect");
4808
- for (const t of expect) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4809
- for (const g of anyOf) {
4810
- if (!Array.isArray(g)) { say("an anyOf group has to be a list of terms"); continue; }
4811
- if (!g.length) say("an anyOf group is empty, so nothing can satisfy it");
4812
- for (const t of g) if (!matchable(t)) say(`offers ${t}, which its own scorer cannot match`);
4813
- }
4814
- // An exception excuses only the forbidden terms inside it, so one that
4815
- // holds none of them changes nothing and reads as if it did.
4816
- for (const a of c.allow || []) {
4817
- if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
4878
+ const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
4879
+ const forbids = scored.filter(m => m.not && m.type === "contains");
4880
+ for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4881
+ // An exception excuses only the forbidden term inside it, so one that
4882
+ // holds it not changes nothing and reads as if it did.
4883
+ for (const m of forbids) {
4884
+ for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
4818
4885
  }
4819
4886
  // A term on both lists cannot be produced and cannot be withheld.
4820
- for (const t of [...expect, ...anyOf.flat()]) {
4821
- if (forbid.includes(t)) say(`${t} is both expected and forbidden`);
4822
- }
4823
- const { minCount: lo, maxCount: hi } = c;
4824
- if (lo != null && hi != null && lo > hi) say(`minCount ${lo} is above maxCount ${hi}`);
4825
- if (lo != null && lo < 1) say(`minCount ${lo} is not a bound`);
4826
- // The mirror of the floor: a ceiling below one says no answer is
4827
- // acceptable, which is an entry that can never pass rather than a
4828
- // strict one.
4829
- if (hi != null && hi < 1) say(`maxCount ${hi} leaves no answer that could pass`);
4830
- // A case that names more distinct things than the reply may carry, in
4831
- // the scorer's own counting of a thing (#486: an expect term is one, an
4832
- // anyOf group is one), can never pass however well the model answers.
4833
- const things = expect.length + anyOf.length;
4834
- if (hi != null && things > hi) {
4835
- say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
4887
+ const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
4888
+ for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
4889
+ for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
4890
+ const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
4891
+ if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
4892
+ // A ceiling below one says no answer is acceptable, which is a case
4893
+ // that can never pass rather than a strict one.
4894
+ if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
4895
+ // A case that names more distinct things than the reply may carry, in
4896
+ // the scorer's own counting of a thing (#486: each term Contains all
4897
+ // wants is one, each Contains any is one), can never pass.
4898
+ const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
4899
+ if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
4836
4900
  }
4837
4901
  }
4838
4902
 
4839
- // One file, one set of expectations. A file here twice is two sets of
4840
- // them, graded separately, and both would be listed.
4903
+ // One item, one case. An item here twice is two cases of it, graded
4904
+ // separately, and both would be listed.
4841
4905
  const where = new Map ();
4842
4906
  for (const c of graded) {
4843
- const name = caseFile(c);
4907
+ const name = caseItem(c);
4844
4908
  const counted = where.get(name);
4845
4909
  if (counted) {
4846
- bad.push(`${name} is graded twice — one file, `
4847
- + `two sets of expectations. Grade it once.`);
4910
+ bad.push(`${name} is graded twice — one item, `
4911
+ + `two cases. Grade it once.`);
4848
4912
  } else where.set(name, true);
4849
4913
  }
4850
4914
  const gradedAt = (name ) => where.has(name);
@@ -4865,7 +4929,7 @@ function validateEvals(ev , files
4865
4929
  for (const f of shown) {
4866
4930
  if (!gradedAt(f)) {
4867
4931
  bad.push(`${f} is in the Source and this dataset does not grade it, `
4868
- + `so the Evals tab does not list it`);
4932
+ + `so the Cases view does not list it`);
4869
4933
  }
4870
4934
  }
4871
4935
  return bad;
@@ -4887,7 +4951,9 @@ function evalsWarnings(ev , prompt ) {
4887
4951
  const want = Number(n), warn = [];
4888
4952
  for (const c of ev.cases || []) {
4889
4953
  if (!c || c.todo) continue;
4890
- const things = (c.expect || []).length + (c.anyOf || []).length;
4954
+ const things = (Array.isArray(c.metrics) ? c.metrics : [])
4955
+ .filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
4956
+ .reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
4891
4957
  if (things > want) {
4892
4958
  warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
4893
4959
  }
@@ -4906,7 +4972,7 @@ function evalsWarnings(ev , prompt ) {
4906
4972
  * `evals-check.js` asserts the round trip against the real file.
4907
4973
  */
4908
4974
  function evalsJson(ev ) {
4909
- return JSON.stringify(canonicalCases(ev), null, 2) + "\n";
4975
+ return JSON.stringify(ev, null, 2) + "\n";
4910
4976
  }
4911
4977
 
4912
4978
 
@@ -4920,14 +4986,14 @@ registerMetrics({ registerKinds });
4920
4986
  export {
4921
4987
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
4922
4988
  loopReplyError, preparedSize,
4923
- termIn, forbiddenIn, scoreCase, gradedSetFrom, caseFile, canonicalCases, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4989
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4924
4990
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
4925
4991
  CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4926
4992
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
4927
4993
  tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
4928
4994
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
4929
4995
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
4930
- PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, TEST_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4996
+ PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4931
4997
  SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4932
4998
  registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
4933
4999
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
@@ -4935,9 +5001,9 @@ CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWord
4935
5001
  contentOf, withContent, replyOf,
4936
5002
  targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
4937
5003
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
4938
- blankPipeline, upgradePipeline, fatProfileRef, newId, newTest, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4939
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioTests, scenarioPasses, isSkipped, failedScore,
4940
- testLabel, testsOf, testsDataset, isWholeRun, TEST_FIELDS,
5004
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5005
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
5006
+ evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
4941
5007
  pipelineToYaml, importPipeline, yamlToPipeline,
4942
5008
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
4943
5009
  };