evals-lab 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,7 +26,7 @@
26
26
  // WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
27
27
  // script has not, so `prepare()` stays in the page and the script shells out
28
28
  // to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
29
- // the script writes JSON. Both read the same verdict out of `scoreCase`.
29
+ // the script writes JSON. Both read the same verdict out of the same Metrics.
30
30
  //
31
31
  // `runner-check.js` runs one set through two of the callers and asserts the
32
32
  // verdicts and the totals are identical, because sharing a file is a claim
@@ -232,8 +232,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
232
232
 
233
233
 
234
234
 
235
- /** What every test carries whatever its type: an id minted once and never
236
- shown, an optional name ("Test 2" when blank), and whether the tests after
235
+ /** What every eval carries whatever its type: an id minted once and never
236
+ shown, an optional name ("Eval 2" when blank), and whether the evals after
237
237
  it still read what it failed on (docs/pipeline-model.md §3). */
238
238
 
239
239
 
@@ -260,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
260
260
 
261
261
 
262
262
 
263
- /** Checks of every reply (TEST_TYPES.metrics): its own for every item, and
263
+ /** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
264
264
  a case's own for its item; all must pass, or weighted points reach the
265
265
  threshold. A model-graded one asks the grader. */
266
266
 
@@ -275,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
275
275
 
276
276
 
277
277
 
278
- /** A test, from version 9: Metrics. A Single Test or a Graded set is what an
278
+ /** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
279
279
  older document held (upgradePipeline reads it converted). */
280
280
 
281
281
 
282
- /** A pipeline's tests, in the order they read a run. */
282
+ /** A pipeline's evals, in the order they read a run. */
283
283
 
284
284
 
285
285
 
@@ -482,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
482
482
 
483
483
  // ---- grading ----
484
484
 
485
- /** A graded case, as cases.json writes one.
486
-
487
- The item a case grades is named by `filename`: a dataset joins on a file
488
- name, and nothing about that is an image -- the next dataset addresses
489
- a row of a spreadsheet or a line of a log by the same key. The field had
490
- an older name while the lab was one app's bench, and it is still *read*:
491
- a set re-synced from that app's repository arrives spelled that way, and
492
- a case that silently stopped matching would be worse than one that reads
493
- both.
494
- `caseFile` is the one reader; `evalsJson` is the one writer, and it writes
495
- `filename`. */
485
+ /** A graded case (dataset body version 5): an Item and the Metrics its
486
+ reply is held to.
487
+
488
+ The item is named by `item`: a dataset joins on an item's name in the
489
+ Source it grades, exactly, and nothing about that is an image -- the
490
+ next dataset addresses a row of a spreadsheet or a line of a log by the
491
+ same key. Version 4 called it `filename`, and the lab's first app had a
492
+ name of its own for it; `upgradeDatasetBody` reads both as `item`.
493
+
494
+ A case's metrics are what a good answer is: each a metric as a
495
+ pipeline's Metrics eval holds one, added to the eval's own for this item
496
+ when an eval names the dataset. One with weight 0 is watched -- reported,
497
+ never scored. */
496
498
 
497
499
 
498
-
499
-
500
-
500
+
501
+
501
502
 
502
-
503
-
504
-
505
-
506
-
507
-
508
-
509
-
510
-
511
-
503
+
504
+
512
505
 
513
-
514
-
515
-
516
-
517
506
 
518
507
 
519
508
 
520
- /** A graded set: cases.json. */
509
+ /** A graded set: a dataset's body, or anything holding its cases. */
521
510
 
522
511
 
523
512
 
524
513
 
525
514
 
526
- /** One row the Datasets tab's Cases group draws, as gradedSetFrom builds it:
527
- a case of a set validateEvals accepts, so it has its id and file. */
515
+ /** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
516
+ a case of a set validateEvals accepts, so it has its id and item. */
528
517
 
529
518
 
530
-
519
+
531
520
 
532
521
 
533
522
 
534
523
  /** A requirement as a score reports it: a term, or the group it was. */
535
524
 
536
525
 
537
- /** How a graded case read a reply: the same shape both engines agree on. */
538
-
539
-
540
-
541
-
542
-
543
-
544
-
545
-
546
-
547
-
548
-
549
-
550
-
551
-
552
-
553
-
554
526
  /** A run's totals, summed across cases. */
555
527
 
556
528
 
@@ -559,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
559
531
 
560
532
 
561
533
 
562
- /** The whole-run verdict of a test type, where it has one of its own. */
534
+ /** The whole-run verdict of an eval type, where it has one of its own. */
563
535
 
564
536
 
565
537
 
@@ -570,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
570
542
 
571
543
 
572
544
 
573
- /** One thing a test asserts, as its type names it: "Exact" "outdoor". */
545
+ /** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
574
546
 
575
547
 
576
548
 
@@ -599,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
599
571
 
600
572
 
601
573
 
602
- /** A per-item test's score as a stored row carries it. */
574
+ /** A per-item eval's score as a stored row carries it. */
603
575
 
604
576
 
605
577
 
@@ -678,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
678
650
  more entry, and no reader changes. */
679
651
 
680
652
 
653
+
654
+
655
+
681
656
 
682
657
 
683
658
 
@@ -698,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
698
673
 
699
674
 
700
675
 
701
- /** What an earlier test that failed and does not continue leaves a later one. */
676
+ /** What an earlier eval that failed and does not continue leaves a later one. */
702
677
 
703
678
 
704
679
 
705
680
 
706
- /** One test's reading of one scenario of a run (scenarioTests). */
681
+ /** One eval's reading of one scenario of a run (scenarioEvals). */
707
682
 
708
683
 
709
684
 
@@ -760,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
760
735
 
761
736
  // ---- the registries ----
762
737
 
763
- /** An option a modifier or a test type exposes for editing. */
738
+ /** An option a modifier or an eval type exposes for editing. */
764
739
 
765
740
 
766
741
 
@@ -820,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
820
795
 
821
796
 
822
797
 
823
-
798
+
824
799
 
825
800
 
826
801
 
@@ -896,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
896
871
 
897
872
 
898
873
 
899
- /** A test type: what it checks of a document, and how it scores a run. */
874
+ /** An eval type: what it checks of a document, and how it scores a run. */
900
875
 
901
876
 
902
-
877
+
903
878
 
904
879
 
905
880
 
@@ -909,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
909
884
 
910
885
 
911
886
 
912
-
887
+
913
888
 
914
889
 
915
890
 
916
-
891
+
917
892
 
918
893
 
919
894
 
@@ -934,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
934
909
 
935
910
 
936
911
 
937
- /** What a test's `read` is handed besides the reply: production's reply to
912
+ /** What an eval's `read` is handed besides the reply: production's reply to
938
913
  the item, and a grader, where the run has them. */
939
914
 
940
915
 
@@ -1004,42 +979,122 @@ const SLOTS = ["content", "target", "responses"];
1004
979
  * submitted against, so nothing grades from a file.
1005
980
  */
1006
981
 
982
+
983
+
984
+
985
+
986
+
987
+
1007
988
 
1008
989
 
1009
990
 
1010
991
 
992
+ /** The dataset body's version: 6 marks itself; 5 named its Source; 4 and
993
+ earlier, neither. */
994
+ const DATASET_BODY_VERSION = 6 ;
995
+
996
+ /** The metrics whose Ignore case version 6 made mean what it says for a
997
+ reply read as a list, as well as one read as text. */
998
+ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
999
+
1011
1000
  /**
1012
- * [body] as this version of a dataset (4), from any earlier one. Version 1
1001
+ * [body] as this version of a dataset (6), from any earlier one. Version 1
1013
1002
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
1014
1003
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
1015
1004
  * Version 2 held `rules`, which clean a job's answer and so belong to the
1016
1005
  * job (docs/pipeline-model.md §13): they leave, and the terms a case
1017
1006
  * watches for are its `watch`. Version 3 held the `prompt` a new scenario
1018
- * started from, which the Prompt library holds now: it leaves. Every reader of a body calls this: the runner, the page, a
1019
- * run's kept copy. A reader that needs a version-2 body's rules -- to upgrade
1020
- * a pipeline graded against it -- takes them first (`datasetRules`).
1021
- * Anything else comes back as it was.
1007
+ * started from, which the Prompt library holds now: it leaves. Version 4's
1008
+ * case named its item `filename` and said what a good answer is in
1009
+ * expectations; each becomes the metric it is (`caseMetrics`), `why` is the
1010
+ * `note`, `traits` go, and the body names no Source yet. Version 5 matched
1011
+ * a list's items ignoring case whatever a Contains metric's Ignore case
1012
+ * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1013
+ * the body says its version. Every reader of a
1014
+ * body calls this: the runner, the page, a run's kept copy. A reader that
1015
+ * needs a version-2 body's rules -- to upgrade a pipeline graded against
1016
+ * it -- takes them first (`datasetRules`). Pure: the same body gives the
1017
+ * same answer, and server.py's upgrade_body is its twin. Anything else
1018
+ * comes back as it was.
1022
1019
  */
1023
1020
  function upgradeDatasetBody (body ) {
1024
1021
  if (!isObj(body)) return body;
1025
- if (!("imageCases" in body || "rules" in body)) {
1026
- if (!("prompt" in body)) return body;
1027
- const { prompt: _library, ...rest } = body ;
1028
- return rest ;
1029
- }
1022
+ if (body.version === DATASET_BODY_VERSION) return body;
1030
1023
  const b = body ;
1031
- // vocab: the names older versions gave these fields
1032
- const RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch" }; // vocab: as above
1033
- const renamed = (c ) => {
1034
- if (!isObj(c)) return c;
1035
- const out = {};
1036
- for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
1037
- return out;
1038
- };
1024
+ // A body that names its Source, even as null, is version 5.
1025
+ if ("source" in b) {
1026
+ return { version: DATASET_BODY_VERSION, ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) };
1027
+ }
1039
1028
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
1040
1029
  // Not a body of any version -- a copy kept while a dataset was an overlay.
1041
1030
  if (!raw) return body;
1042
- return { cases: (canonicalCases({ cases: raw.map(renamed) }) ).cases };
1031
+ return { version: DATASET_BODY_VERSION, source: null, cases: raw.map(caseOfV4).map(caseOfV5) };
1032
+ }
1033
+
1034
+ /** A version-5 case as a version-6 one: each Contains metric says Ignore
1035
+ case, as version 5 matched a list's items whatever it said. A reply read
1036
+ as text did mind its setting, but a case's metrics were written for a
1037
+ list -- each one converted from version 4, and each the case form made. */
1038
+ function caseOfV5(c ) {
1039
+ if (!isObj(c) || !Array.isArray(c.metrics)) return c;
1040
+ return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
1041
+ ? { ...m, ignoreCase: true } : m)) };
1042
+ }
1043
+
1044
+ // vocab: the names older versions gave these fields
1045
+ const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
1046
+ /** What a version-4 case said, which its metrics say now. */
1047
+ const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
1048
+
1049
+ /** One case of any earlier version as a version-5 one: its id, item, todo
1050
+ and note, its expectations as metrics ahead of any metrics it held, and
1051
+ anything else it carried kept as it was. */
1052
+ function caseOfV4(c ) {
1053
+ if (!isObj(c)) return c;
1054
+ const was = {};
1055
+ for (const [k, v] of Object.entries(c)) {
1056
+ const key = CASE_RENAMED[k] ?? k;
1057
+ // A case naming its item both ways keeps the newer key's.
1058
+ if (!(key in was) || key === k) was[key] = v;
1059
+ }
1060
+ if (isStr(was.item)) delete was.filename;
1061
+ const out = {};
1062
+ if ("id" in was) out.id = was.id;
1063
+ out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
1064
+ out.todo = was.todo === true;
1065
+ out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
1066
+ out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
1067
+ for (const [k, v] of Object.entries(was)) {
1068
+ if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
1069
+ }
1070
+ return out;
1071
+ }
1072
+
1073
+ /** A version-4 case's expectations as the metrics that say the same:
1074
+ `expect` is Contains all, each `anyOf` group a Contains any, each
1075
+ forbidden term a Contains turned round with the `allow` phrases that
1076
+ excuse it, the count bounds an Item count, `discarded` the Discarded
1077
+ metric, and each watched term a Contains any of weight 0 -- reported,
1078
+ never scored. The order is the one scoreCase read them in, so a reason
1079
+ reads in the order it did. */
1080
+ function caseMetrics(c ) {
1081
+ const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
1082
+ if (c.discarded === true) return [{ type: "discarded" }];
1083
+ const out = [];
1084
+ const expect = list(c.expect), allow = list(c.allow);
1085
+ if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
1086
+ for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
1087
+ if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
1088
+ }
1089
+ for (const t of list(c.forbid)) {
1090
+ // An exception excuses only the forbidden term inside it.
1091
+ const except = allow.filter(a => termIn([a], t));
1092
+ out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
1093
+ }
1094
+ const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
1095
+ if (min != null || max != null) out.push({ type: "item-count", min, max });
1096
+ for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
1097
+ return out;
1043
1098
  }
1044
1099
 
1045
1100
  /** The rules a version-1 or version-2 dataset body held, or null: what a
@@ -1055,6 +1110,9 @@ function datasetRules(body ) {
1055
1110
 
1056
1111
 
1057
1112
 
1113
+
1114
+
1115
+
1058
1116
 
1059
1117
 
1060
1118
 
@@ -1322,8 +1380,10 @@ function textPrompt(instruction , text ) {
1322
1380
  }
1323
1381
 
1324
1382
  // Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
1325
- // work in, so that "New Zealand." and "new zealand" are the same two words.
1326
- const words = (s ) => String(s||"").toLowerCase().match(/[\p{L}\p{N}]+/gu) || [];
1383
+ // work in, so that "New Zealand." and "new zealand" are the same two words --
1384
+ // unless [keepCase], for a metric whose Ignore case is off.
1385
+ const words = (s , keepCase = false) =>
1386
+ (keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
1327
1387
 
1328
1388
  // The request Tagger builds, field for field -- when the target reads those
1329
1389
  // fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
@@ -2039,11 +2099,12 @@ function budgetLabel(target ) {
2039
2099
  // A term is present when its words appear in some item, in order and adjacent:
2040
2100
  // "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
2041
2101
  // looser scores "new zealand" against "zealandia" and flatters every run.
2042
- function termIn(items , term ) {
2043
- const t = words(term);
2102
+ // Letter case counts only where a metric's Ignore case is off.
2103
+ function termIn(items , term , ignoreCase = true) {
2104
+ const t = words(term, !ignoreCase);
2044
2105
  if (!t.length) return false;
2045
2106
  return items.some(item => {
2046
- const w = words(item);
2107
+ const w = words(item, !ignoreCase);
2047
2108
  for (let i = 0; i + t.length <= w.length; i++) {
2048
2109
  if (t.every((x, j) => w[i + j] === x)) return true;
2049
2110
  }
@@ -2070,153 +2131,55 @@ function forbiddenIn(items , term , allow
2070
2131
  * check green. `evals-check.js` calls this directly.
2071
2132
  */
2072
2133
  function gradedSetFrom(ev ) {
2073
- return (ev.cases || []).map(c => ({ ...c, filename: caseFile(c), half: "cases" }) );
2134
+ return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
2074
2135
  }
2075
2136
 
2076
- /**
2077
- * The file a case grades, whichever of the two keys names it.
2078
- *
2079
- * One reader, so that accepting the older spelling is a fact about this
2080
- * function rather than a branch every caller carries. Everything that joins a
2081
- * case to an item -- the runner, the Datasets tab, a mapping against a Source
2082
- * -- goes through here.
2083
- */
2084
- function caseFile(kase ) { // vocab: the older spelling
2085
- const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
2086
- return typeof name === "string" ? name : "";
2137
+ /** The item a case grades: its `item`, or "" for none. The one reader, so
2138
+ everything that joins a case to an item -- the runner, the Library, a
2139
+ run's Results -- joins on the same key, exactly. */
2140
+ function caseItem(kase ) {
2141
+ return isStr(kase?.item) ? kase .item : "";
2087
2142
  }
2088
2143
 
2089
- /**
2090
- * A graded set with every case naming its file as `filename`.
2091
- *
2092
- * What `evalsJson` writes, so the committed bytes carry one key and a set
2093
- * re-synced from an app's own repository is normalised the first time it is
2094
- * exported. The key takes the place the older one held, so normalising a set
2095
- * changes the spelling of
2096
- * one key and not the order of any.
2097
- */
2098
- function canonicalCases(ev ) {
2099
- const OLD = "photo"; // vocab: the older spelling of filename
2100
- const set = ev ;
2101
- const cases = set && typeof set === "object" && Array.isArray(set.cases) ? set.cases : null;
2102
- if (!cases || !cases.some(c => c && typeof c === "object" && OLD in (c ))) return ev;
2103
- return {
2104
- ...set,
2105
- cases: cases.map(c => {
2106
- if (!c || typeof c !== "object" || !(OLD in (c ))) return c;
2107
- const out = {};
2108
- for (const [k, v] of Object.entries(c )) {
2109
- if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
2110
- else out[k] = v;
2111
- }
2112
- return out;
2113
- }),
2114
- };
2115
- }
2144
+ /** One graded case's reading of a reply, now: its own metrics, scored as a
2145
+ Metrics eval scores them (all must pass; weight 0 is watched, never
2146
+ scored), with what they found, missed and invented carried up, and
2147
+ `watch` the readings of weight 0. What the page draws and the runner's
2148
+ case-by-case report writes. A model-graded metric needs a grader and the
2149
+ wait for one, so here it reads as not met: the runner's `--run` asks it. */
2150
+
2151
+
2152
+
2153
+
2154
+
2155
+
2156
+
2157
+
2158
+
2159
+
2160
+
2161
+
2162
+
2163
+
2116
2164
 
2117
- /**
2118
- * One graded case, one result.
2119
- *
2120
- * Every expectation is a group, and a plain `expect` term is a group of one:
2121
- * the requirements are the `expect` terms and the `anyOf` groups alike, and
2122
- * the score is found requirements over all of them plus `forbid`. A count
2123
- * bound stays pass/fail: an item that produced two perfect terms when five
2124
- * were wanted has not done what was asked.
2125
- *
2126
- * Pure on purpose, and returning `reasons` as plain sentences rather than
2127
- * markup. The dashboard was the first consumer; `run-evals.js` is the second
2128
- * and CI the third, and neither can reach into a page for a rendered cell.
2129
- * Issue #248 wants a failure written out as a task an agent can act on.
2130
- */
2131
- function scoreCase(kase , res ) {
2132
- // A case that expects its answer discarded passes on a discard and on
2133
- // nothing else: the answer the job threw away is the finding.
2134
- if (kase.discarded === true) {
2135
- const thrown = !!res.error && res.error.startsWith("discarded: ");
2136
- const n = (res.terms || []).length;
2137
- return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
2138
- unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
2139
- reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
2140
- }
2141
- // A discarded answer is a failed item: whatever the pipeline produced
2142
- // along the way does not count.
2143
- const discarded = !!res.error;
2144
- const terms = discarded ? [] : (res.terms || []);
2145
- const expect = kase.expect || [], forbid = kase.forbid || [];
2146
-
2147
- // `missed` stays populated for a discarded reply even though it reads as
2148
- // vacuous, because the score depends on it: docs/datasets.md has such a reply
2149
- // scoring zero against everything the entry asked for, groups included, and
2150
- // `addToTally` gets there through `found + missed + invented`. Clear it and
2151
- // a run that discarded every item would contribute nothing to the
2152
- // total instead of contributing a nought, which flatters it.
2153
- //
2154
- // One member of a group is enough. Without this rule the set manufactures
2155
- // failures out of synonyms.
2156
- //
2157
- // A requirement reads back as a term when it had one member and as the
2158
- // group itself where a synonym list was allowed, so `missed` carries just
2159
- // enough to state the reason: "FAIL: Missed dog" for a plain expect term,
2160
- // "none of dog / puppy" for a group.
2161
- //
2162
- // Nothing satisfies a group when nothing was stored, so a discarded reply
2163
- // misses every requirement rather than none -- the same cast that makes its
2164
- // count bounds not breached by one. `unmet` is a finding only now that
2165
- // `missed` names the unsatisfied groups itself, but the run report has
2166
- // always carried it, so it stays.
2167
- const met = (g ) => g.some(t => termIn(terms, t));
2168
- const spoken = (g ) => g.length === 1 ? g[0] : g;
2169
- const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
2170
- const found = requirements.filter(met).map(spoken);
2171
- const missed = requirements.filter(g => !met(g)).map(spoken);
2172
- const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
2173
- const unmet = discarded ? []
2174
- : (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
2175
-
2176
- const n = terms.length;
2177
- // A null bound is not checked -- and neither is a bound on a reply that was
2178
- // discarded. Nothing was counted out and found wanting there; the reply was
2179
- // thrown away before it had a count, and saying "0 items, wanted at least 4"
2180
- // states a second failure that never happened.
2181
- const under = !discarded && kase.minCount != null && n < kase.minCount;
2182
- const over = !discarded && kase.maxCount != null && n > kase.maxCount;
2183
-
2184
- const denom = requirements.length + invented.length;
2185
- // A case that passes is one whose whole expectation was met, not one that
2186
- // scored well: an unsatisfied group is in `missed` alongside any term
2187
- // missed, so pass needs nothing more than the terms already covered.
2188
- const pass = !discarded && !missed.length && !invented.length && !under && !over;
2189
-
2190
- // `found` and `missed` are groups, not terms, so the reasons word them one
2191
- // by one: a bare term under one heading, a group on its own line.
2192
- const reasons = [];
2193
- if (discarded) {
2194
- // Everything else would be derived from this one fact -- there are no
2195
- // terms -- and would bury it. `scoreHtml` above suppresses `missed` for
2196
- // the same reason: the reason that matters is already on the row.
2197
- reasons.push(`Error - ${res.error}`);
2198
- } else {
2199
- const plain = missed.filter(t => typeof t === "string");
2200
- if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
2201
- for (const g of missed) {
2202
- if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
2165
+ function readCase(kase , res , plain = false) {
2166
+ const input = metricInput(res , kase, { plain });
2167
+ const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
2168
+ const r = readMetric(m, input, {});
2169
+ if (r && typeof (r ).then === "function") {
2170
+ return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
2171
+ pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
2203
2172
  }
2204
- if (invented.length) reasons.push(`invented ${invented.join(", ")}`);
2205
- if (under) reasons.push(`${n} items, wanted at least ${kase.minCount}`);
2206
- if (over) reasons.push(`${n} items, wanted at most ${kase.maxCount}`);
2207
- }
2208
-
2209
- // Observed rather than scored: a dataset that wants to know whether some
2210
- // terms turn up, without grading on them, lists them as `watch`.
2211
- const watch = kase.watch || [];
2212
- // `null`, not 1, when there are no requirements and nothing forbidden turned
2213
- // up: there is no score to report. Returning 1 there printed "fail 100%"
2214
- // beside an entry that asked for nothing -- a shape `evals-check.js` allows
2215
- // even though every graded case now names an expectation.
2216
- return { pass, score: denom ? found.length / denom : null, discarded,
2217
- found, missed, invented, unmet, under, over, count: n,
2218
- watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
2219
- reasons };
2173
+ return r ? [r ] : [];
2174
+ });
2175
+ const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
2176
+ const counted = metrics.filter(r => r.weight !== 0);
2177
+ return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
2178
+ discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
2179
+ // A reply the job threw away fails on that one fact, which would
2180
+ // be buried under everything that derives from it.
2181
+ reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
2182
+ watch: metrics.filter(r => r.weight === 0) };
2220
2183
  }
2221
2184
 
2222
2185
  /**
@@ -2257,11 +2220,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
2257
2220
  // docs/datasets.md: an item naming more terms should weigh more than
2258
2221
  // one naming fewer, and averaging percentages lets the easiest case carry the
2259
2222
  // score.
2260
- function addToTally(t , s ) {
2223
+ function addToTally(t , s ) {
2224
+ const found = s.found?.length ?? 0;
2261
2225
  t.ran++;
2262
2226
  if (s.pass) t.passed++;
2263
- t.found += s.found.length;
2264
- t.of += s.found.length + s.missed.length + s.invented.length;
2227
+ t.found += found;
2228
+ t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
2265
2229
  }
2266
2230
 
2267
2231
  // The headline percentage, in one place because it is a number and this file
@@ -2276,7 +2240,7 @@ function tallyPercent(t ) {
2276
2240
  // A whole-run assertion with no graded set: All of / Any of / None of, a
2277
2241
  // parse, a count, an exact reply and a length bound. It lives in the shared
2278
2242
  // core because it is a pass/fail the tab, the worker and History all have to
2279
- // agree on -- the same one-definition rule that keeps `scoreCase` here.
2243
+ // agree on -- the same one-definition rule that keeps the case reader here.
2280
2244
  function parseCount(raw , parse ) {
2281
2245
  // An unparsed reply is one result, not no result. It used to return null,
2282
2246
  // which made a Count of 1 fail as "null results" against a reply that
@@ -2666,10 +2630,10 @@ function applyModifiers (list , kind ,
2666
2630
  // queue and run-evals.js alike, and every one of them reads it through the
2667
2631
  // functions below rather than through a translation of its own.
2668
2632
  //
2669
- // The lab is generic, so what a reply is, what a test scores and how a value
2633
+ // The lab is generic, so what a reply is, what an eval scores and how a value
2670
2634
  // is changed are registry entries. The ones here are the lab's own, and so
2671
2635
  // are kinds/list.ts's -- the List kind and its modifiers. A dataset
2672
- // registers nothing: it is data a graded test names by id, and a run carries.
2636
+ // registers nothing: it is data a graded eval names by id, and a run carries.
2673
2637
 
2674
2638
  // 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
2675
2639
  // version 1 kept them as two maps (tokens: { values, blocks }).
@@ -2685,14 +2649,16 @@ function applyModifiers (list , kind ,
2685
2649
  // 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
2686
2650
  // 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
2687
2651
  // Call (Prompt: the image flag and token mappings), Read Reply (the output).
2688
- // 9: a test is Metrics. A Single Test and a Graded set are read as the Metrics
2652
+ // 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
2689
2653
  // they convert to (LEGACY_TESTS' toMetrics, proven equal by
2690
2654
  // metrics-parity-check.js).
2691
2655
  // 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
2692
2656
  // Content, the flow step, Attach Image, Token mappings) and Responses (Read
2693
2657
  // as, Response Format Validation, modifiers) -- and its call is gone: each
2694
2658
  // scenario is a target, whose own step in each job is what it sends there.
2695
- const PIPELINE_VERSION = 10 ;
2659
+ // 11: `tests` are `evals`: the key renames and nothing in an eval changes,
2660
+ // so a stored result's scores, keyed by eval id, read as they did.
2661
+ const PIPELINE_VERSION = 12 ;
2696
2662
 
2697
2663
  // Plain objects, so an entry is added by assignment and a reader never needs
2698
2664
  // to know which registered it.
@@ -2700,7 +2666,7 @@ const STEP_TYPES = Object.create(null);
2700
2666
  const CONTENT_TYPES = Object.create(null);
2701
2667
  const OUTPUT_KINDS = Object.create(null);
2702
2668
  const MODIFIERS = Object.create(null);
2703
- const TEST_TYPES = Object.create(null);
2669
+ const EVAL_TYPES = Object.create(null);
2704
2670
  const METRICS = Object.create(null);
2705
2671
  const SOURCE_TYPES = Object.create(null);
2706
2672
 
@@ -2714,11 +2680,15 @@ let defaultOutputKind = "text";
2714
2680
  const defaultKind = () => defaultOutputKind;
2715
2681
  const kindOf = (st ) => st.kind ?? defaultOutputKind;
2716
2682
 
2683
+ /** A module's eval types, under either spelling (Kinds.testTypes). */
2684
+ const evalTypesOf = (k ) =>
2685
+ ({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
2686
+
2717
2687
  /** A module's entries into the registries, in one call. */
2718
2688
  function registerKinds(k ) {
2719
2689
  Object.assign(OUTPUT_KINDS, k.outputKinds || {});
2720
2690
  Object.assign(MODIFIERS, k.modifiers || {});
2721
- Object.assign(TEST_TYPES, k.testTypes || {});
2691
+ Object.assign(EVAL_TYPES, evalTypesOf(k));
2722
2692
  Object.assign(SOURCE_TYPES, k.sourceTypes || {});
2723
2693
  Object.assign(METRICS, k.metrics || {});
2724
2694
  if (k.defaultKind) defaultOutputKind = k.defaultKind;
@@ -2749,7 +2719,7 @@ function pluginHost(pluginId ) {
2749
2719
  registerKinds(k) {
2750
2720
  taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
2751
2721
  taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
2752
- taken(TEST_TYPES, "test type", Object.keys(k.testTypes || {}));
2722
+ taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
2753
2723
  taken(METRICS, "metric", Object.keys(k.metrics || {}));
2754
2724
  // The server keeps its own copy of the Source types, to refuse a row of
2755
2725
  // one it does not know, and a plugin never reaches the server's code.
@@ -3015,11 +2985,9 @@ LEGACY_TESTS.single = {
3015
2985
  },
3016
2986
  };
3017
2987
 
3018
- // A dataset's cases, scored item by item with scoreCase. The test
3019
- // references the dataset the way a pipeline references a Source, and the run
3020
- // carries the body it was submitted against. It scores the terms a value
3021
- // yields, so it accepts every kind that yields any: plain text has none, and
3022
- // would fail every case.
2988
+ // A dataset's cases, scored item by item. The eval references the dataset
2989
+ // the way a pipeline references a Source, and the run carries the body it
2990
+ // was submitted against.
3023
2991
  LEGACY_TESTS.graded = {
3024
2992
  label: "Graded set",
3025
2993
  fields: ["type", "dataset"],
@@ -3027,16 +2995,15 @@ LEGACY_TESTS.graded = {
3027
2995
  validate(t, ctx, bad){
3028
2996
  const d = t.dataset;
3029
2997
  if (!isRef(d) || (d.version != null && !isStr(d.version))) {
3030
- return void bad.push("a graded test has to name its dataset as { id, name }");
2998
+ return void bad.push("a graded eval has to name its dataset as { id, name }");
3031
2999
  }
3032
3000
  if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
3033
3001
  },
3034
- score: (t, kase, res) => scoreCase(kase, res),
3035
3002
  };
3036
3003
 
3037
3004
  // Metrics: checks of each reply, deterministic or model-graded, from the
3038
3005
  // METRICS registry (metrics/builtin.ts registers the lab's own), with the
3039
- // test's own list for every item and a case's `metrics` for its item. Scored
3006
+ // eval's own list for every item and a case's `metrics` for its item. Scored
3040
3007
  // all-must-pass -- every metric passes -- or weighted: points, each metric's
3041
3008
  // score times its weight, against a threshold (#150's points, a negative
3042
3009
  // weight taking them away).
@@ -3045,6 +3012,13 @@ LEGACY_TESTS.graded = {
3045
3012
  const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
3046
3013
  const SCORING_MODES = ["all", "weighted"];
3047
3014
 
3015
+ /** A metric's list option -- Contains all's values, Contains's exceptions --
3016
+ one entry a line, or a list of them. */
3017
+ function metricLines(v ) {
3018
+ const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
3019
+ return all.map(x => x.trim()).filter(Boolean);
3020
+ }
3021
+
3048
3022
  /** What is wrong with a list of metrics, as sentences naming [at]. */
3049
3023
  function metricsProblems(list , at , bad ) {
3050
3024
  if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
@@ -3114,14 +3088,16 @@ function readMetric(m , input , ctx )
3114
3088
  } catch (e) { return timed(failed(e)); }
3115
3089
  }
3116
3090
 
3117
- /** A test's score from its metrics' readings, the way [mode] says: all must
3091
+ /** An eval's score from its metrics' readings, the way [mode] says: all must
3118
3092
  pass, or weighted points against [threshold]. What a case metric found,
3119
3093
  missed and invented is carried up, for a run's totals. Null for none. */
3120
3094
  function scoreOf(metrics , mode = "all", threshold = null) {
3121
3095
  if (!metrics.length) return null;
3122
- const detail = metrics.some(r => r.found || r.missed || r.invented) ? {
3123
- found: metrics.flatMap(r => r.found ?? []), missed: metrics.flatMap(r => r.missed ?? []),
3124
- invented: metrics.flatMap(r => r.invented ?? []) } : {};
3096
+ // A watched reading (weight 0) is reported, and counts toward nothing.
3097
+ const scored = metrics.filter(r => r.weight !== 0);
3098
+ const detail = scored.some(r => r.found || r.missed || r.invented) ? {
3099
+ found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
3100
+ invented: scored.flatMap(r => r.invented ?? []) } : {};
3125
3101
  if (mode === "weighted") {
3126
3102
  const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
3127
3103
  return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
@@ -3164,7 +3140,7 @@ function readRun(list , run ) {
3164
3140
  });
3165
3141
  }
3166
3142
 
3167
- TEST_TYPES.metrics = {
3143
+ EVAL_TYPES.metrics = {
3168
3144
  label: "Metrics",
3169
3145
  description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3170
3146
  fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
@@ -3188,9 +3164,9 @@ TEST_TYPES.metrics = {
3188
3164
  if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
3189
3165
  if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
3190
3166
  if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
3191
- // The test's own grader, or the lab's.
3167
+ // The eval's own grader, or the lab's.
3192
3168
  const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
3193
- if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a Target profile the lab's grader");
3169
+ if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
3194
3170
  // A grader is asked words, and needs a model to ask.
3195
3171
  const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3196
3172
  if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
@@ -3205,7 +3181,7 @@ TEST_TYPES.metrics = {
3205
3181
  want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3206
3182
  .filter(Boolean).join(" ") })),
3207
3183
  profiles: t => (isRef(t.grader) ? [t.grader] : []),
3208
- // A run carries the lab's grader on a test that names none and may ask one:
3184
+ // A run carries the lab's grader on an eval that names none and may ask one:
3209
3185
  // a model-graded metric of its own, or a case's, over a dataset.
3210
3186
  resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
3211
3187
  ? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
@@ -3220,44 +3196,28 @@ TEST_TYPES.metrics = {
3220
3196
  ran: run.replies.length,
3221
3197
  checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3222
3198
  },
3223
- // Rule m{x} is the test's own metric x, read in that place on every item.
3199
+ // Rule m{x} is the eval's own metric x, read in that place on every item.
3224
3200
  // A score stored before Metrics -- a Graded set's, read as its conversion --
3225
3201
  // has no readings of its own: its one metric's reading is the score's.
3226
3202
  ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3227
3203
  : Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
3228
- // Every item, with its case's own metrics where it has a case.
3229
- read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((kase?.metrics ) || [])],
3204
+ // Every item, with its case's own metrics where the eval names the
3205
+ // dataset and the item has a case there.
3206
+ read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
3230
3207
  metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3231
3208
  };
3232
3209
 
3233
3210
  // ---- the lab's own scorers, as metrics ----------------------------------------
3234
- // What the Graded set and the Single Test check, one metric each, so a test of
3235
- // either converts to Metrics that read a run exactly as it did
3236
- // (metrics-parity-check.js). Here rather than in metrics/builtin.ts because
3237
- // each is the core's own matcher.
3238
-
3239
- // An item's case, as the Graded set scores it: its expectations, forbidden
3240
- // terms and count bounds, with what it found, missed and invented kept.
3241
- METRICS.case = {
3242
- label: "Matches its case",
3243
- description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
3244
- perItem: true,
3245
- needsTerms: true,
3246
- options: [],
3247
- defaults: () => ({}),
3248
- score(input) {
3249
- if (!input.kase) return null;
3250
- const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
3251
- return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
3252
- found: s.found, missed: s.missed, invented: s.invented };
3253
- },
3254
- };
3211
+ // What the Single Test checks, one metric each, so an eval of it converts to
3212
+ // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
3213
+ // rather than in metrics/builtin.ts because each is the core's own matcher.
3255
3214
 
3256
3215
  // Items the reply holds -- all of them, or any -- matched as the lab matches
3257
3216
  // a term; a kind that yields none is matched in the replies' text instead,
3258
3217
  // ignoring case, as the Single Test did.
3259
3218
  METRICS["has-items"] = {
3260
3219
  label: "Has items",
3220
+ family: "The reply's text",
3261
3221
  description: "Passes when the reply holds all, or any, of the items listed.",
3262
3222
  options: [{ key: "values", label: "Items", type: "textarea" },
3263
3223
  { key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
@@ -3277,6 +3237,7 @@ METRICS["has-items"] = {
3277
3237
  // A reply's length in characters, once trimmed.
3278
3238
  METRICS.length = {
3279
3239
  label: "Length",
3240
+ family: "The reply's text",
3280
3241
  description: "Compares the reply's length in characters.",
3281
3242
  options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
3282
3243
  { key: "n", label: "Characters", type: "number" }],
@@ -3292,6 +3253,7 @@ METRICS.length = {
3292
3253
  // as the count says.
3293
3254
  METRICS["parse-count"] = {
3294
3255
  label: "Parses",
3256
+ family: "The result",
3295
3257
  description: "Passes when the reply reads as the format chosen, holding the number of results set.",
3296
3258
  // Unformatted is one result a reply, as the Single Test counted it.
3297
3259
  options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
@@ -3310,12 +3272,13 @@ METRICS["parse-count"] = {
3310
3272
  },
3311
3273
  };
3312
3274
 
3313
- /** What every test carries, kept across a conversion. */
3275
+ /** What every eval carries, kept across a conversion. */
3314
3276
  const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
3315
3277
 
3316
- // A Graded set is a Metrics test over the same dataset, its one metric the case.
3278
+ // A Graded set is a Metrics eval over the same dataset, with none of its own:
3279
+ // each case's metrics are what it scored (dataset-parity-check.js).
3317
3280
  LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
3318
- dataset: t.dataset, over: "item", metrics: [{ type: "case" }] });
3281
+ dataset: t.dataset, over: "item", metrics: [] });
3319
3282
 
3320
3283
  // A Single Test is Metrics over the whole run: its lists over the run's items
3321
3284
  // together, and exact, length and parse over each reply alone.
@@ -3333,7 +3296,7 @@ LEGACY_TESTS.single .toMetrics = t => {
3333
3296
  return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
3334
3297
  };
3335
3298
 
3336
- /** The grader a Metrics test names, reached through what the runner hands it. */
3299
+ /** The grader a Metrics eval names, reached through what the runner hands it. */
3337
3300
  function graderCtx(t , more ) {
3338
3301
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3339
3302
  return ask ? { ask } : {};
@@ -3343,12 +3306,14 @@ function graderCtx(t , more ) {
3343
3306
  for one with none set: what History prints beside its name. */
3344
3307
  function metricSummary(m ) {
3345
3308
  const entry = METRICS[m.type];
3346
- // Each option as it reads in the editor: a choice's label, a box that is
3347
- // ticked by its own label, text by its first line.
3309
+ // Each option as it reads in the editor: a choice's label, a box by its
3310
+ // own label where it is not as a new metric has it (ticked, or "off"),
3311
+ // text by its first line.
3312
+ const fresh = entry?.defaults() ?? {};
3348
3313
  const said = (entry?.options || []).map(o => {
3349
3314
  const v = m[o.key];
3350
3315
  if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3351
- if (o.type === "checkbox") return v ? o.label.toLowerCase() : "";
3316
+ if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
3352
3317
  // Text that runs to lines (a schema) reads as its first words, run together.
3353
3318
  return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3354
3319
  }).filter(Boolean);
@@ -3531,7 +3496,7 @@ STEP_TYPES.readAs = {
3531
3496
  // Response Format Validation: the kind a reply is read as.
3532
3497
  STEP_TYPES.readReply = {
3533
3498
  label: "Read Reply", slot: "responses", rank: 1,
3534
- description: "How the reply is read before tests and later jobs see it.",
3499
+ description: "How the reply is read before evals and later jobs see it.",
3535
3500
  in: "text", out: step => step.out?.kind,
3536
3501
  // Read inside the stage: runPipeline parses each reply as it comes back.
3537
3502
  apply: "runPipeline",
@@ -3787,7 +3752,7 @@ STEP_TYPES.job = {
3787
3752
  if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
3788
3753
  const steps = c.steps;
3789
3754
  // A job's steps are the entries with a stage; one without (a job, the
3790
- // tests) or of no type the lab has is not one.
3755
+ // evals) or of no type the lab has is not one.
3791
3756
  const unknown = steps.find(st => !slotOf(st));
3792
3757
  if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
3793
3758
  if (steps.some(st => slotOf(st) === "target")) {
@@ -3808,54 +3773,55 @@ STEP_TYPES.job = {
3808
3773
  }
3809
3774
  },
3810
3775
  };
3811
- // What a test carries besides its type's own fields.
3812
- const TEST_FIELDS = ["id", "name", "continueOnFailure"];
3776
+ // What an eval carries besides its type's own fields.
3777
+ const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
3813
3778
 
3814
- /** Test [j]'s name, or the number it has always shown. */
3815
- const testLabel = (doc , j ) =>
3816
- (doc.tests?.[j]?.name || "").trim() || `Test ${j + 1}`;
3779
+ /** Eval [j]'s name, or the number it has always shown. */
3780
+ const evalLabel = (doc , j ) =>
3781
+ (doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
3817
3782
 
3818
- /** A test type that settles over the whole run rather than item by item. */
3783
+ /** An eval type that settles over the whole run rather than item by item. */
3819
3784
  const isWholeRun = (type , t ) =>
3820
3785
  type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
3821
3786
 
3822
3787
  /**
3823
- * A document's tests as a list, whatever version wrote it: a queue row keeps
3824
- * the run document it was submitted with, so a run from before version 6
3825
- * holds one test or null, read here as the upgrade reads it (testsList).
3788
+ * A document's evals as a list, whatever version wrote it: a queue row keeps
3789
+ * the run document it was submitted with, so a run from before version 11
3790
+ * spells them `tests`, and one from before version 6 holds one or null, read
3791
+ * here as the upgrade reads it (testsList, evalsKey).
3826
3792
  */
3827
- function testsOf(doc ) {
3828
- const t = doc?.tests;
3793
+ function evalsOf(doc ) {
3794
+ const t = doc?.evals ?? doc?.tests;
3829
3795
  if (Array.isArray(t)) return t ;
3830
3796
  return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3831
3797
  }
3832
3798
 
3833
- /** The dataset a document's tests grade against, where one does: a run
3799
+ /** The dataset a document's evals grade against, where one does: a run
3834
3800
  grades against one (validatePipeline says so), so the first names it. */
3835
- function testsDataset(doc ) {
3836
- const t = testsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3801
+ function evalsDataset(doc ) {
3802
+ const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3837
3803
  return t && "dataset" in t ? t.dataset : null;
3838
3804
  }
3839
3805
 
3840
- STEP_TYPES.tests = {
3806
+ STEP_TYPES.evals = {
3841
3807
  in: "results", out: "verdict",
3842
3808
  apply: "score",
3843
- validate(tests, ctx, bad, lastKind, doc){
3844
- if (!Array.isArray(tests)) return void bad.push("tests has to be a list, empty for an unscored run");
3809
+ validate(evals, ctx, bad, lastKind, doc){
3810
+ if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
3845
3811
  const seen = new Set ();
3846
- tests.forEach((t , j ) => {
3847
- const at = testLabel(doc, j);
3848
- const type = isObj(t) && TEST_TYPES[t.type];
3812
+ evals.forEach((t , j ) => {
3813
+ const at = evalLabel(doc, j);
3814
+ const type = isObj(t) && EVAL_TYPES[t.type];
3849
3815
  if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
3850
3816
  const before = bad.length;
3851
3817
  if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
3852
- else if (seen.has(t.id)) bad.push(`${at} has the id of another test`);
3818
+ else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
3853
3819
  else seen.add(t.id);
3854
3820
  if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
3855
3821
  if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
3856
- onlyFields(t, at, [...type.fields, ...TEST_FIELDS], bad);
3822
+ onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
3857
3823
  type.validate(t, ctx, bad);
3858
- // Which kinds a test scores is only worth saying of a test that is whole.
3824
+ // Which kinds an eval scores is only worth saying of an eval that is whole.
3859
3825
  if (bad.length > before) return;
3860
3826
  const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
3861
3827
  if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
@@ -3864,14 +3830,14 @@ STEP_TYPES.tests = {
3864
3830
  }
3865
3831
  });
3866
3832
  // A run is handed one dataset's body to grade against (server-side-runs §4).
3867
- const named = new Set(tests.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3868
- if (named.size > 1) bad.push("the tests grade against one dataset at a time");
3833
+ const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3834
+ if (named.size > 1) bad.push("the evals grade against one dataset at a time");
3869
3835
  },
3870
3836
  };
3871
3837
 
3872
3838
  // ---- the document -------------------------------------------------------------
3873
3839
 
3874
- const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "tests"];
3840
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
3875
3841
  // What resolving adds, and nothing else: the profiles it resolved to and the
3876
3842
  // run's own comment, which belongs to the run and never to the pipeline.
3877
3843
  const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
@@ -3888,7 +3854,7 @@ function versionProblem(doc ) {
3888
3854
  /** What an upgrade from version 2 needs from outside the document: the rules
3889
3855
  of the dataset a pipeline was graded against, which the retired
3890
3856
  version-2 list kind read every reply under. Without them a job is upgraded with no
3891
- rules, as a run with no graded test parsed. */
3857
+ rules, as a run with no graded eval parsed. */
3892
3858
 
3893
3859
 
3894
3860
 
@@ -3913,7 +3879,8 @@ function versionProblem(doc ) {
3913
3879
  * as it was, for versionProblem to name. A copy: the caller's document is not
3914
3880
  * touched. From version 5, its test -- or none -- becomes a list of one (or
3915
3881
  * none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
3916
- * renames and every job's `type` becomes `"job"`.
3882
+ * renames and every job's `type` becomes `"job"`. From version 10, its
3883
+ * `tests` are `evals` (evalsKey).
3917
3884
  */
3918
3885
  /** Every id a current document must carry, minted for the ones [doc] lacks.
3919
3886
  An id it already has is kept. */
@@ -3949,6 +3916,9 @@ function profileRefs(doc ) {
3949
3916
  }
3950
3917
  }
3951
3918
 
3919
+ /** [doc] with its profile references cut (profileRefs), for chaining. */
3920
+ const cutRefs = (doc ) => { profileRefs(doc); return doc; };
3921
+
3952
3922
  /** Whether any profile reference in a pipeline holds more than { id, name }. */
3953
3923
  function fatProfileRef(doc ) {
3954
3924
  const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
@@ -3967,9 +3937,9 @@ function testsList(doc ) {
3967
3937
  return doc;
3968
3938
  }
3969
3939
 
3970
- /** A new test of [type] at the lab's defaults, named [name], continuing on failure. */
3971
- function newTest(type , name = "", fields = {}) {
3972
- const own = TEST_TYPES[type]?.defaults?.() ?? { type };
3940
+ /** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
3941
+ function newEval(type , name = "", fields = {}) {
3942
+ const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
3973
3943
  return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
3974
3944
  }
3975
3945
 
@@ -4067,6 +4037,29 @@ function targetsFromScenarios(next ) {
4067
4037
  return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
4068
4038
  }
4069
4039
 
4040
+ /** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
4041
+ upgraded document reads the same, key for key, and each eval is as it
4042
+ was. Every version before 11 passes through here. */
4043
+ function evalsKey(next ) {
4044
+ next.version = PIPELINE_VERSION;
4045
+ // `tests` is the spelling of every version before 11.
4046
+ if (!("tests" in next)) return next;
4047
+ return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
4048
+ .map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
4049
+ }
4050
+
4051
+ /** Version 11 to 12: a Contains metric's Ignore case holds where the reply
4052
+ is matched item by item, as it does where it is matched as text, and it
4053
+ is kept as written. Until version 11's last hours (#199) every Contains
4054
+ metric matched the reply's text and minded its Ignore case, so what a
4055
+ stored metric says is what its author meant; only the item-by-item
4056
+ matching #199 added, case-blind for a few hours, read it otherwise.
4057
+ Every version before 12 ends here. */
4058
+ function caseAsWritten(next ) {
4059
+ next.version = PIPELINE_VERSION;
4060
+ return next;
4061
+ }
4062
+
4070
4063
  function upgradePipeline (doc , ctx = {}) {
4071
4064
  // A current document is read as it is, but for a profile reference the Runs
4072
4065
  // tab saved whole (see profileRefs), which is cut back, and a step on a
@@ -4077,11 +4070,29 @@ function upgradePipeline (doc , ctx = {}) {
4077
4070
  out = clone(out) ;
4078
4071
  profileRefs(out);
4079
4072
  }
4080
- return localSteps(out, ctx) ;
4073
+ return localSteps(withoutCaseMetric(out), ctx) ;
4081
4074
  }
4082
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9].includes(doc.version )) return doc;
4083
- // Every version before 10 reads as version 9 first, then as 10.
4084
- return localSteps(targetsFromScenarios(nineOf(clone(doc) , ctx)), ctx) ;
4075
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
4076
+ if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
4077
+ // Every version before 10 reads as version 9 first, then as 10, then as 11
4078
+ // and 12.
4079
+ // A version-10 document is cut as a current one was (profileRefs); the
4080
+ // earlier ones are cut on their way through nineOf.
4081
+ const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4082
+ return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
4083
+ }
4084
+
4085
+ /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
4086
+ are metrics now (dataset body version 5), which an eval naming the
4087
+ dataset adds for each item, so the case's own metrics carry what that
4088
+ metric scored. Read so at every version, the current one included, as a
4089
+ document saved before it went still holds it. The same document where
4090
+ none does. */
4091
+ function withoutCaseMetric(doc ) {
4092
+ const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
4093
+ if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
4094
+ return { ...doc, evals: doc.evals.map((t ) => (has(t)
4095
+ ? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
4085
4096
  }
4086
4097
 
4087
4098
  /** [doc] with each step asked of a profile whose type a target step stands
@@ -4165,7 +4176,7 @@ function tokenMappingsFromV1(set ) {
4165
4176
  function blankPipeline(opts = {}) {
4166
4177
  const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
4167
4178
  return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
4168
- jobs: [job], targets: [], tests: [] };
4179
+ jobs: [job], targets: [], evals: [] };
4169
4180
  }
4170
4181
 
4171
4182
  /**
@@ -4303,7 +4314,7 @@ function validatePipeline(input , ctx = {}) {
4303
4314
  : `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
4304
4315
  }
4305
4316
  });
4306
- STEP_TYPES.tests .validate(doc.tests, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4317
+ STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4307
4318
  if (bad.length) return bad;
4308
4319
 
4309
4320
  // Asked before anything is sent, so a misspelt token costs nothing and
@@ -4400,9 +4411,9 @@ function profileIds(doc )
4400
4411
  if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4401
4412
  }
4402
4413
  }
4403
- // Then the ones a test asks (a grader), so a run carries them too.
4404
- for (const t of testsOf(doc)) {
4405
- for (const ref of TEST_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4414
+ // Then the ones an eval asks (a grader), so a run carries them too.
4415
+ for (const t of evalsOf(doc)) {
4416
+ for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4406
4417
  }
4407
4418
  return ids;
4408
4419
  }
@@ -4415,9 +4426,9 @@ function profileIds(doc )
4415
4426
  function resolvePipeline(doc ,
4416
4427
  ctx = {}) {
4417
4428
  const run = clone(doc) ;
4418
- // What the lab supplies a test -- its grader -- before the profiles it
4429
+ // What the lab supplies an eval -- its grader -- before the profiles it
4419
4430
  // asks are carried.
4420
- run.tests = (Array.isArray(run.tests) ? run.tests : []).map((t ) => TEST_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4431
+ run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4421
4432
  run.profiles = {};
4422
4433
  for (const id of profileIds(run)) {
4423
4434
  const p = ctx.profiles?.(id);
@@ -4426,7 +4437,7 @@ function resolvePipeline(doc ,
4426
4437
  const content = contentOf(run) ;
4427
4438
  if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4428
4439
  if (ctx.datasetVersion) {
4429
- for (const t of run.tests || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4440
+ for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4430
4441
  }
4431
4442
  if (isStr(ctx.comment)) run.comment = ctx.comment;
4432
4443
  return run;
@@ -4440,7 +4451,7 @@ function pipelineOfRun(run , ctx = {}) {
4440
4451
  delete doc.plugins;
4441
4452
  const content = contentOf(doc) ;
4442
4453
  if (content) { delete content.files; delete content.revs; }
4443
- for (const t of doc.tests || []) if (isObj(t.dataset)) delete t.dataset.version;
4454
+ for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
4444
4455
  return doc;
4445
4456
  }
4446
4457
 
@@ -4498,7 +4509,7 @@ function importPipeline(input , ctx = {}) {
4498
4509
  if (st?.profile) st.profile = remap(st.profile, "profile");
4499
4510
  }
4500
4511
  }
4501
- for (const t of Array.isArray(next.tests) ? next.tests : []) {
4512
+ for (const t of Array.isArray(next.evals) ? next.evals : []) {
4502
4513
  if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4503
4514
  }
4504
4515
  // Its references are this lab's now, so a step asking this lab's Echo
@@ -4530,7 +4541,7 @@ function mintIds(doc ) {
4530
4541
  for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
4531
4542
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4532
4543
  }
4533
- for (const t of Array.isArray(doc.tests) ? doc.tests : []) {
4544
+ for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4534
4545
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4535
4546
  }
4536
4547
  return doc;
@@ -4608,14 +4619,14 @@ function modifierSummary(m ) {
4608
4619
  return said.length ? said.join(", ") : "on";
4609
4620
  }
4610
4621
 
4611
- // ---- the tests, in order (docs/pipeline-model.md §3) ---------------------------
4612
- // Tests only read a run: none changes what a later one sees, and none stops
4622
+ // ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
4623
+ // Evals only read a run: none changes what a later one sees, and none stops
4613
4624
  // the model being sent the next item. What order changes is Continue on
4614
- // failure. A per-item test that fails and does not continue stops the tests
4625
+ // failure. A per-item eval that fails and does not continue stops the evals
4615
4626
  // after it for that item alone -- they read Skipped there, and a whole-run
4616
- // test after it pools the items it did not stop. A whole-run test settles
4627
+ // eval after it pools the items it did not stop. A whole-run eval settles
4617
4628
  // once every item is in, and one that fails then and does not continue
4618
- // leaves every test after it Skipped.
4629
+ // leaves every eval after it Skipped.
4619
4630
 
4620
4631
  const SKIPPED = Object.freeze({ skipped: true });
4621
4632
  const isSkipped = (s ) => isObj(s) && s.skipped === true;
@@ -4623,16 +4634,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
4623
4634
  const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
4624
4635
 
4625
4636
  /**
4626
- * One reply's scores under a run's per-item tests, by test id, in order: a
4627
- * test with no case to score leaves no entry, and one after a failure that
4637
+ * One reply's scores under a run's per-item evals, by eval id, in order: a
4638
+ * eval with no case to score leaves no entry, and one after a failure that
4628
4639
  * does not continue reads Skipped. Null where nothing was scored.
4629
4640
  */
4630
- function itemScores(run , kase , res ) {
4641
+ function itemScores(run , kase , res ) {
4631
4642
  if (!kase) return null;
4632
- const out = {};
4643
+ const out = {};
4633
4644
  let stopped = false;
4634
- for (const t of testsOf(run)) {
4635
- const type = TEST_TYPES[t.type];
4645
+ for (const t of evalsOf(run)) {
4646
+ const type = EVAL_TYPES[t.type];
4636
4647
  if (!type?.score) continue;
4637
4648
  if (stopped) { out[t.id] = SKIPPED; continue; }
4638
4649
  const s = type.score(t, kase, res);
@@ -4650,19 +4661,19 @@ function productionOf(run , record )
4650
4661
  }
4651
4662
 
4652
4663
  /**
4653
- * itemScores for the runner, which can wait: a test that reads every item
4664
+ * itemScores for the runner, which can wait: an eval that reads every item
4654
4665
  * (`read`: the Metrics, which may ask a grader) scores one with no case too.
4655
4666
  */
4656
4667
  async function itemScoresAsync(run , kase , res ,
4657
4668
  more = {}) {
4658
4669
  const out = {};
4659
4670
  let stopped = false;
4660
- for (const t of testsOf(run)) {
4661
- const type = TEST_TYPES[t.type];
4671
+ for (const t of evalsOf(run)) {
4672
+ const type = EVAL_TYPES[t.type];
4662
4673
  if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
4663
4674
  if (stopped) { out[t.id] = SKIPPED; continue; }
4664
- // A test with nothing to read on this item leaves no entry, as a graded
4665
- // test does on an item with no case.
4675
+ // An eval with nothing to read on this item leaves no entry, as a graded
4676
+ // eval does on an item with no case.
4666
4677
  const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
4667
4678
  : type.score (t, kase , res);
4668
4679
  if (!s) continue;
@@ -4673,22 +4684,22 @@ async function itemScoresAsync(run , kase ,
4673
4684
  }
4674
4685
 
4675
4686
  /**
4676
- * Every test's reading of scenario [i] of a run, in the run's order, from
4687
+ * Every eval's reading of scenario [i] of a run, in the run's order, from
4677
4688
  * the items so far. [settled] says every item is in: only then has a
4678
- * whole-run test settled, so only then does its failure skip the tests
4689
+ * whole-run eval settled, so only then does its failure skip the evals
4679
4690
  * after it.
4680
4691
  */
4681
- function scenarioTests(run , i , items ,
4692
+ function scenarioEvals(run , i , items ,
4682
4693
  settled = true) {
4683
4694
  const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
4684
- // Which items a per-item test has stopped so far, for the tests after it.
4695
+ // Which items a per-item eval has stopped so far, for the evals after it.
4685
4696
  const stopped = cells.map(() => false);
4686
4697
  let skipRest = false;
4687
- const tests = testsOf(run);
4688
- return tests.map((t, j) => {
4689
- const type = TEST_TYPES[t.type];
4698
+ const evals = evalsOf(run);
4699
+ return evals.map((t, j) => {
4700
+ const type = EVAL_TYPES[t.type];
4690
4701
  const whole = isWholeRun(type, t);
4691
- const base = { id: t.id, label: testLabel({ tests }, j), whole, skipped: skipRest,
4702
+ const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
4692
4703
  verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
4693
4704
  ran: 0, passed: 0, skippedItems: 0 };
4694
4705
  if (skipRest) {
@@ -4715,7 +4726,7 @@ function scenarioTests(run , i , items
4715
4726
  });
4716
4727
  }
4717
4728
 
4718
- /** A scenario's pass or fail over every test that read it: null where none
4729
+ /** A scenario's pass or fail over every eval that read it: null where none
4719
4730
  has anything to say yet. */
4720
4731
  function scenarioPasses(outcomes ) {
4721
4732
  let said = false;
@@ -4756,95 +4767,77 @@ function validateEvals(ev , files
4756
4767
  return ["the graded set has to be a JSON object with a `cases` list"];
4757
4768
  }
4758
4769
  if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
4770
+ if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
4759
4771
  if (bad.length) return bad;
4760
4772
 
4761
4773
  const graded = ev.cases || [];
4762
4774
  // Identity first: every rule below reports which case is at fault, so a
4763
4775
  // case with no usable id makes the rest of the report unreadable.
4764
- const seen = new Map ();
4776
+ const seen = new Set ();
4765
4777
  for (const c of graded) {
4766
- {
4767
- if (!c || typeof c !== "object" || Array.isArray(c)) {
4768
- bad.push("cases holds something that is not a case");
4769
- continue;
4770
- }
4771
- if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
4772
- else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4773
- else seen.set(c.id, "cases");
4774
- if (!caseFile(c).trim()) {
4775
- bad.push(`${c.id || "a case"} names no file`);
4776
- }
4777
- for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
4778
- if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
4779
- }
4780
- for (const key of ["minCount", "maxCount"]) {
4781
- if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
4782
- }
4783
- if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
4784
- // A case's own metrics, which a Metrics test adds to its own for this item.
4785
- if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4778
+ if (!isObj(c)) {
4779
+ bad.push("cases holds something that is not a case");
4780
+ continue;
4786
4781
  }
4782
+ if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
4783
+ else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4784
+ else seen.add(c.id);
4785
+ if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
4786
+ if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
4787
+ if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
4788
+ if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4787
4789
  }
4788
4790
  if (bad.length) return bad;
4789
4791
 
4790
- // A term the scorer cannot match is a term that can never be produced, so
4791
- // the case is unpassable however well the model answers.
4792
+ // What a case says, read from its metrics: a term the scorer cannot match
4793
+ // can never be produced, so the case is unpassable however well the model
4794
+ // answers -- and the same goes for a bound no reply can meet.
4792
4795
  const matchable = (term ) => termIn([String(term).trim()], term);
4793
-
4796
+ const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
4794
4797
  for (const c of graded) {
4795
4798
  if (c.todo) continue;
4796
4799
  const say = (m ) => bad.push(`${c.id}: ${m}`);
4797
- const expect = c.expect || [], forbid = c.forbid || [], anyOf = c.anyOf || [];
4798
-
4799
- if (c.discarded === true) {
4800
+ const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
4801
+ if (!scored.length) { say("graded but states nothing to expect"); continue; }
4802
+ if (scored.some(m => m.type === "discarded" && !m.not)) {
4800
4803
  // What is discarded holds nothing, so there is nothing else to expect.
4801
- if (expect.length || anyOf.length || forbid.length || c.minCount != null || c.maxCount != null) {
4802
- say("expects its answer discarded, and states something the answer should hold too");
4803
- }
4804
+ if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
4804
4805
  continue;
4805
4806
  }
4806
- const metrics = Array.isArray(c.metrics) ? c.metrics : [];
4807
- if (!expect.length && !anyOf.length && !metrics.length) say("graded but states nothing to expect");
4808
- for (const t of expect) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4809
- for (const g of anyOf) {
4810
- if (!Array.isArray(g)) { say("an anyOf group has to be a list of terms"); continue; }
4811
- if (!g.length) say("an anyOf group is empty, so nothing can satisfy it");
4812
- for (const t of g) if (!matchable(t)) say(`offers ${t}, which its own scorer cannot match`);
4813
- }
4814
- // An exception excuses only the forbidden terms inside it, so one that
4815
- // holds none of them changes nothing and reads as if it did.
4816
- for (const a of c.allow || []) {
4817
- if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
4807
+ const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
4808
+ const forbids = scored.filter(m => m.not && m.type === "contains");
4809
+ for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4810
+ // An exception excuses only the forbidden term inside it, so one that
4811
+ // holds it not changes nothing and reads as if it did.
4812
+ for (const m of forbids) {
4813
+ for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
4818
4814
  }
4819
4815
  // A term on both lists cannot be produced and cannot be withheld.
4820
- for (const t of [...expect, ...anyOf.flat()]) {
4821
- if (forbid.includes(t)) say(`${t} is both expected and forbidden`);
4822
- }
4823
- const { minCount: lo, maxCount: hi } = c;
4824
- if (lo != null && hi != null && lo > hi) say(`minCount ${lo} is above maxCount ${hi}`);
4825
- if (lo != null && lo < 1) say(`minCount ${lo} is not a bound`);
4826
- // The mirror of the floor: a ceiling below one says no answer is
4827
- // acceptable, which is an entry that can never pass rather than a
4828
- // strict one.
4829
- if (hi != null && hi < 1) say(`maxCount ${hi} leaves no answer that could pass`);
4830
- // A case that names more distinct things than the reply may carry, in
4831
- // the scorer's own counting of a thing (#486: an expect term is one, an
4832
- // anyOf group is one), can never pass however well the model answers.
4833
- const things = expect.length + anyOf.length;
4834
- if (hi != null && things > hi) {
4835
- say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
4816
+ const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
4817
+ for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
4818
+ for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
4819
+ const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
4820
+ if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
4821
+ // A ceiling below one says no answer is acceptable, which is a case
4822
+ // that can never pass rather than a strict one.
4823
+ if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
4824
+ // A case that names more distinct things than the reply may carry, in
4825
+ // the scorer's own counting of a thing (#486: each term Contains all
4826
+ // wants is one, each Contains any is one), can never pass.
4827
+ const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
4828
+ if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
4836
4829
  }
4837
4830
  }
4838
4831
 
4839
- // One file, one set of expectations. A file here twice is two sets of
4840
- // them, graded separately, and both would be listed.
4832
+ // One item, one case. An item here twice is two cases of it, graded
4833
+ // separately, and both would be listed.
4841
4834
  const where = new Map ();
4842
4835
  for (const c of graded) {
4843
- const name = caseFile(c);
4836
+ const name = caseItem(c);
4844
4837
  const counted = where.get(name);
4845
4838
  if (counted) {
4846
- bad.push(`${name} is graded twice — one file, `
4847
- + `two sets of expectations. Grade it once.`);
4839
+ bad.push(`${name} is graded twice — one item, `
4840
+ + `two cases. Grade it once.`);
4848
4841
  } else where.set(name, true);
4849
4842
  }
4850
4843
  const gradedAt = (name ) => where.has(name);
@@ -4865,7 +4858,7 @@ function validateEvals(ev , files
4865
4858
  for (const f of shown) {
4866
4859
  if (!gradedAt(f)) {
4867
4860
  bad.push(`${f} is in the Source and this dataset does not grade it, `
4868
- + `so the Evals tab does not list it`);
4861
+ + `so the Cases view does not list it`);
4869
4862
  }
4870
4863
  }
4871
4864
  return bad;
@@ -4887,7 +4880,9 @@ function evalsWarnings(ev , prompt ) {
4887
4880
  const want = Number(n), warn = [];
4888
4881
  for (const c of ev.cases || []) {
4889
4882
  if (!c || c.todo) continue;
4890
- const things = (c.expect || []).length + (c.anyOf || []).length;
4883
+ const things = (Array.isArray(c.metrics) ? c.metrics : [])
4884
+ .filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
4885
+ .reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
4891
4886
  if (things > want) {
4892
4887
  warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
4893
4888
  }
@@ -4906,7 +4901,7 @@ function evalsWarnings(ev , prompt ) {
4906
4901
  * `evals-check.js` asserts the round trip against the real file.
4907
4902
  */
4908
4903
  function evalsJson(ev ) {
4909
- return JSON.stringify(canonicalCases(ev), null, 2) + "\n";
4904
+ return JSON.stringify(ev, null, 2) + "\n";
4910
4905
  }
4911
4906
 
4912
4907
 
@@ -4920,14 +4915,14 @@ registerMetrics({ registerKinds });
4920
4915
  export {
4921
4916
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
4922
4917
  loopReplyError, preparedSize,
4923
- termIn, forbiddenIn, scoreCase, gradedSetFrom, caseFile, canonicalCases, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4918
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4924
4919
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
4925
4920
  CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4926
4921
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
4927
4922
  tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
4928
4923
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
4929
4924
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
4930
- PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, TEST_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4925
+ PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4931
4926
  SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4932
4927
  registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
4933
4928
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
@@ -4935,9 +4930,9 @@ CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWord
4935
4930
  contentOf, withContent, replyOf,
4936
4931
  targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
4937
4932
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
4938
- blankPipeline, upgradePipeline, fatProfileRef, newId, newTest, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4939
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioTests, scenarioPasses, isSkipped, failedScore,
4940
- testLabel, testsOf, testsDataset, isWholeRun, TEST_FIELDS,
4933
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4934
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
4935
+ evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
4941
4936
  pipelineToYaml, importPipeline, yamlToPipeline,
4942
4937
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
4943
4938
  };