evals-lab 0.1.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,7 +26,7 @@
26
26
  // WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
27
27
  // script has not, so `prepare()` stays in the page and the script shells out
28
28
  // to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
29
- // the script writes JSON. Both read the same verdict out of `scoreCase`.
29
+ // the script writes JSON. Both read the same verdict out of the same Metrics.
30
30
  //
31
31
  // `runner-check.js` runs one set through two of the callers and asserts the
32
32
  // verdicts and the totals are identical, because sharing a file is a claim
@@ -94,49 +94,76 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
94
94
 
95
95
 
96
96
 
97
- /** A Call that prompts a model: each scenario's cell says with what words
98
- and whom to ask; the step says whether the item's image goes too, and the
99
- token mappings the words are resolved under. */
97
+ /** The Power Automate step a job's records are calls of (job 1, Content):
98
+ the action's name its expressions read it by, the loop item() means,
99
+ and the HTTP API its replies speak -- how production's reply is read. */
100
+
101
+
102
+
103
+
104
+
105
+
106
+
107
+ /** The item's image goes beside every target's words in this job. */
108
+
109
+
110
+
111
+
112
+ /** The token mappings a job's words are resolved under. */
113
+
114
+
115
+
116
+
117
+
118
+ /** What the flow does to the reply before it reads it -- its Parse JSON's
119
+ content -- as an expression over body('<step>'). First of Responses. */
100
120
 
101
121
 
102
-
103
-
122
+
104
123
 
105
124
 
106
- /** How a job's reply is read: its output kind, and the modifiers applied. */
125
+ /** Response Format Validation: the output kind a reply is read as, and that
126
+ kind's own settings. A job with none reads its reply as text. */
107
127
 
108
128
 
109
-
129
+
110
130
 
111
131
 
112
- /** A Call that sends a Power Automate step's request (docs/power-automate.md
113
- § The HTTP Request step): the flow's own template -- method, path, query
114
- and body, Workflow Definition Language expressions and all -- evaluated
115
- against each item's record. The Setup profile gives the address and the
116
- key; each scenario's cell the words, model, settings and prefill the
117
- HTTP API puts in the body. */
118
-
132
+ /** One change applied to the parsed reply, after the Format Validation
133
+ whose kind accepts it. */
134
+
135
+
136
+
137
+
138
+
139
+ /** A job's steps: its Content stage, then its Responses (§16). */
140
+
141
+
142
+
143
+ /** A target step that prompts a model: its words, resolved under the job's
144
+ token mappings, and a profile of its own where it names one. */
145
+
146
+
147
+
148
+
149
+ /** A target step that sends a Power Automate step's request
150
+ (docs/power-automate.md § The HTTP Request step): the flow's own template
151
+ -- method, path, query and body, Workflow Definition Language expressions
152
+ and all -- with the target's words and fields placed in the body by its
153
+ HTTP API, evaluated against each item's record. The profile gives the
154
+ address and the key. */
155
+
119
156
 
120
-
121
-
122
157
 
123
158
 
124
159
 
125
160
 
126
161
 
127
162
 
128
-
129
-
130
-
131
-
132
-
133
-
134
163
 
135
164
 
136
-
137
-
138
- /** A job is its steps, in fixed slots (pipeline-model §3): an Attach Content
139
- on job 1 only, one Call, then Read Reply. */
165
+ /** A job is its steps, staged (pipeline-model §16): Content, then
166
+ Responses. What each target sends in it is the target's own. */
140
167
 
141
168
 
142
169
 
@@ -170,13 +197,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
170
197
 
171
198
 
172
199
 
173
- /** One cell of a scenario: what to ask in that job, and whom. */
200
+ /** What a target's step says beside its type: what to ask in that job,
201
+ and whom, where it names a profile of its own. */
174
202
 
175
203
 
176
204
 
177
-
178
205
 
179
-
206
+
207
+
180
208
 
181
209
 
182
210
 
@@ -187,17 +215,25 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
187
215
 
188
216
 
189
217
 
190
-
191
-
218
+ /** What a target sends in one job: one of the target step types. */
219
+
220
+
221
+ /** A target (pipeline-model §16): one column through every job, compared
222
+ side by side in Results -- what version 9 called a scenario. Its step in
223
+ each job says what it sends there. */
224
+
225
+
192
226
 
193
227
 
194
228
 
195
-
196
-
229
+
230
+
231
+
232
+
197
233
 
198
234
 
199
- /** What every test carries whatever its type: an id minted once and never
200
- shown, an optional name ("Test 2" when blank), and whether the tests after
235
+ /** What every eval carries whatever its type: an id minted once and never
236
+ shown, an optional name ("Eval 2" when blank), and whether the evals after
201
237
  it still read what it failed on (docs/pipeline-model.md §3). */
202
238
 
203
239
 
@@ -224,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
224
260
 
225
261
 
226
262
 
227
- /** Checks of every reply (TEST_TYPES.metrics): its own for every item, and
263
+ /** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
228
264
  a case's own for its item; all must pass, or weighted points reach the
229
265
  threshold. A model-graded one asks the grader. */
230
266
 
@@ -239,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
239
275
 
240
276
 
241
277
 
242
- /** A test, from version 9: Metrics. A Single Test or a Graded set is what an
278
+ /** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
243
279
  older document held (upgradePipeline reads it converted). */
244
280
 
245
281
 
246
- /** A pipeline's tests, in the order they read a run. */
282
+ /** A pipeline's evals, in the order they read a run. */
247
283
 
248
284
 
249
285
 
@@ -255,7 +291,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
255
291
 
256
292
 
257
293
 
258
-
294
+
259
295
 
260
296
 
261
297
 
@@ -323,7 +359,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
323
359
  /** A pipeline read from YAML, references remapped to this lab. */
324
360
 
325
361
 
326
-
362
+
327
363
 
328
364
 
329
365
 
@@ -335,6 +371,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
335
371
 
336
372
 
337
373
 
374
+
375
+
376
+
338
377
 
339
378
 
340
379
  /** An import's outcome: the remapped pipeline, or one sentence refusing it. */
@@ -417,12 +456,23 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
417
456
 
418
457
 
419
458
 
459
+
460
+
461
+
462
+
463
+
464
+
465
+ /** What an expression token reads: a record's scope, its time, the loop. */
466
+
467
+
468
+
469
+
420
470
 
421
471
 
422
472
  /** A stage of a run as a transport asks it: stagesFor's answer. */
423
473
 
424
474
 
425
-
475
+
426
476
 
427
477
 
428
478
 
@@ -432,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
432
482
 
433
483
  // ---- grading ----
434
484
 
435
- /** A graded case, as cases.json writes one.
436
-
437
- The item a case grades is named by `filename`: a dataset joins on a file
438
- name, and nothing about that is an image -- the next dataset addresses
439
- a row of a spreadsheet or a line of a log by the same key. The field had
440
- an older name while the lab was one app's bench, and it is still *read*:
441
- a set re-synced from that app's repository arrives spelled that way, and
442
- a case that silently stopped matching would be worse than one that reads
443
- both.
444
- `caseFile` is the one reader; `evalsJson` is the one writer, and it writes
445
- `filename`. */
485
+ /** A graded case (dataset body version 5): an Item and the Metrics its
486
+ reply is held to.
487
+
488
+ The item is named by `item`: a dataset joins on an item's name in the
489
+ Source it grades, exactly, and nothing about that is an image -- the
490
+ next dataset addresses a row of a spreadsheet or a line of a log by the
491
+ same key. Version 4 called it `filename`, and the lab's first app had a
492
+ name of its own for it; `upgradeDatasetBody` reads both as `item`.
493
+
494
+ A case's metrics are what a good answer is: each a metric as a
495
+ pipeline's Metrics eval holds one, added to the eval's own for this item
496
+ when an eval names the dataset. One with weight 0 is watched -- reported,
497
+ never scored. */
446
498
 
447
499
 
448
-
449
-
450
-
500
+
501
+
451
502
 
452
-
453
-
454
-
455
-
456
-
457
-
458
-
459
-
460
-
461
-
503
+
504
+
462
505
 
463
-
464
-
465
-
466
-
467
506
 
468
507
 
469
508
 
470
- /** A graded set: cases.json. */
509
+ /** A graded set: a dataset's body, or anything holding its cases. */
471
510
 
472
511
 
473
512
 
474
513
 
475
514
 
476
- /** One row the Datasets tab's Cases group draws, as gradedSetFrom builds it:
477
- a case of a set validateEvals accepts, so it has its id and file. */
515
+ /** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
516
+ a case of a set validateEvals accepts, so it has its id and item. */
478
517
 
479
518
 
480
-
519
+
481
520
 
482
521
 
483
522
 
484
523
  /** A requirement as a score reports it: a term, or the group it was. */
485
524
 
486
525
 
487
- /** How a graded case read a reply: the same shape both engines agree on. */
488
-
489
-
490
-
491
-
492
-
493
-
494
-
495
-
496
-
497
-
498
-
499
-
500
-
501
-
502
-
503
-
504
526
  /** A run's totals, summed across cases. */
505
527
 
506
528
 
@@ -509,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
509
531
 
510
532
 
511
533
 
512
- /** The whole-run verdict of a test type, where it has one of its own. */
534
+ /** The whole-run verdict of an eval type, where it has one of its own. */
513
535
 
514
536
 
515
537
 
@@ -520,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
520
542
 
521
543
 
522
544
 
523
- /** One thing a test asserts, as its type names it: "Exact" "outdoor". */
545
+ /** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
524
546
 
525
547
 
526
548
 
@@ -549,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
549
571
 
550
572
 
551
573
 
552
- /** A per-item test's score as a stored row carries it. */
574
+ /** A per-item eval's score as a stored row carries it. */
553
575
 
554
576
 
555
577
 
@@ -628,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
628
650
  more entry, and no reader changes. */
629
651
 
630
652
 
653
+
654
+
655
+
631
656
 
632
657
 
633
658
 
@@ -648,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
648
673
 
649
674
 
650
675
 
651
- /** What an earlier test that failed and does not continue leaves a later one. */
676
+ /** What an earlier eval that failed and does not continue leaves a later one. */
652
677
 
653
678
 
654
679
 
655
680
 
656
- /** One test's reading of one scenario of a run (scenarioTests). */
681
+ /** One eval's reading of one scenario of a run (scenarioEvals). */
657
682
 
658
683
 
659
684
 
@@ -710,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
710
735
 
711
736
  // ---- the registries ----
712
737
 
713
- /** An option a modifier or a test type exposes for editing. */
738
+ /** An option a modifier or an eval type exposes for editing. */
714
739
 
715
740
 
716
741
 
@@ -770,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
770
795
 
771
796
 
772
797
 
773
-
798
+
774
799
 
775
800
 
776
801
 
@@ -846,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
846
871
 
847
872
 
848
873
 
849
- /** A test type: what it checks of a document, and how it scores a run. */
874
+ /** An eval type: what it checks of a document, and how it scores a run. */
850
875
 
851
876
 
852
-
877
+
853
878
 
854
879
 
855
880
 
@@ -859,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
859
884
 
860
885
 
861
886
 
862
-
887
+
863
888
 
864
889
 
865
890
 
866
-
891
+
867
892
 
868
893
 
869
894
 
@@ -884,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
884
909
 
885
910
 
886
911
 
887
- /** What a test's `read` is handed besides the reply: production's reply to
912
+ /** What an eval's `read` is handed besides the reply: production's reply to
888
913
  the item, and a grader, where the run has them. */
889
914
 
890
915
 
@@ -894,45 +919,54 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
894
919
 
895
920
 
896
921
 
897
- /** A job's slots, in the order they run. */
898
-
899
- const SLOTS = ["content", "call", "reply"];
922
+ /** A job's stages, in the order they run (pipeline-model §16). A target's
923
+ step is in no job: it is the target's, in that job's place. */
924
+
925
+ const SLOTS = ["content", "target", "responses"];
900
926
 
901
927
 
902
928
 
903
929
 
904
930
 
905
-  
906
-
907
-
931
+  
932
+
933
+
934
+
908
935
 
909
-
910
936
 
911
937
 
938
+
939
+
940
+
941
+
942
+
912
943
 
913
944
 
914
945
 
915
946
 
916
-
947
+
948
+
949
+
950
+
917
951
 
952
+
953
+
954
+
918
955
 
919
956
 
920
957
 
921
-
922
-
923
-
924
958
 
959
+
925
960
 
926
-
961
+
927
962
 
963
+
928
964
 
929
-
930
-
965
+
966
+
931
967
 
932
-
968
+
933
969
 
934
-
935
-
936
970
 
937
971
 
938
972
 
@@ -945,42 +979,122 @@ const SLOTS = ["content", "call", "reply"];
945
979
  * submitted against, so nothing grades from a file.
946
980
  */
947
981
 
982
+
983
+
984
+
985
+
986
+
987
+
948
988
 
949
989
 
950
990
 
951
991
 
992
+ /** The dataset body's version: 6 marks itself; 5 named its Source; 4 and
993
+ earlier, neither. */
994
+ const DATASET_BODY_VERSION = 6 ;
995
+
996
+ /** The metrics whose Ignore case version 6 made mean what it says for a
997
+ reply read as a list, as well as one read as text. */
998
+ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
999
+
952
1000
  /**
953
- * [body] as this version of a dataset (4), from any earlier one. Version 1
1001
+ * [body] as this version of a dataset (6), from any earlier one. Version 1
954
1002
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
955
1003
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
956
1004
  * Version 2 held `rules`, which clean a job's answer and so belong to the
957
1005
  * job (docs/pipeline-model.md §13): they leave, and the terms a case
958
1006
  * watches for are its `watch`. Version 3 held the `prompt` a new scenario
959
- * started from, which the Prompt library holds now: it leaves. Every reader of a body calls this: the runner, the page, a
960
- * run's kept copy. A reader that needs a version-2 body's rules -- to upgrade
961
- * a pipeline graded against it -- takes them first (`datasetRules`).
962
- * Anything else comes back as it was.
1007
+ * started from, which the Prompt library holds now: it leaves. Version 4's
1008
+ * case named its item `filename` and said what a good answer is in
1009
+ * expectations; each becomes the metric it is (`caseMetrics`), `why` is the
1010
+ * `note`, `traits` go, and the body names no Source yet. Version 5 matched
1011
+ * a list's items ignoring case whatever a Contains metric's Ignore case
1012
+ * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1013
+ * the body says its version. Every reader of a
1014
+ * body calls this: the runner, the page, a run's kept copy. A reader that
1015
+ * needs a version-2 body's rules -- to upgrade a pipeline graded against
1016
+ * it -- takes them first (`datasetRules`). Pure: the same body gives the
1017
+ * same answer, and server.py's upgrade_body is its twin. Anything else
1018
+ * comes back as it was.
963
1019
  */
964
1020
  function upgradeDatasetBody (body ) {
965
1021
  if (!isObj(body)) return body;
966
- if (!("imageCases" in body || "rules" in body)) {
967
- if (!("prompt" in body)) return body;
968
- const { prompt: _library, ...rest } = body ;
969
- return rest ;
970
- }
1022
+ if (body.version === DATASET_BODY_VERSION) return body;
971
1023
  const b = body ;
972
- // vocab: the names older versions gave these fields
973
- const RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch" }; // vocab: as above
974
- const renamed = (c ) => {
975
- if (!isObj(c)) return c;
976
- const out = {};
977
- for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
978
- return out;
979
- };
1024
+ // A body that names its Source, even as null, is version 5.
1025
+ if ("source" in b) {
1026
+ return { version: DATASET_BODY_VERSION, ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) };
1027
+ }
980
1028
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
981
1029
  // Not a body of any version -- a copy kept while a dataset was an overlay.
982
1030
  if (!raw) return body;
983
- return { cases: (canonicalCases({ cases: raw.map(renamed) }) ).cases };
1031
+ return { version: DATASET_BODY_VERSION, source: null, cases: raw.map(caseOfV4).map(caseOfV5) };
1032
+ }
1033
+
1034
+ /** A version-5 case as a version-6 one: each Contains metric says Ignore
1035
+ case, as version 5 matched a list's items whatever it said. A reply read
1036
+ as text did mind its setting, but a case's metrics were written for a
1037
+ list -- each one converted from version 4, and each the case form made. */
1038
+ function caseOfV5(c ) {
1039
+ if (!isObj(c) || !Array.isArray(c.metrics)) return c;
1040
+ return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
1041
+ ? { ...m, ignoreCase: true } : m)) };
1042
+ }
1043
+
1044
+ // vocab: the names older versions gave these fields
1045
+ const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
1046
+ /** What a version-4 case said, which its metrics say now. */
1047
+ const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
1048
+
1049
+ /** One case of any earlier version as a version-5 one: its id, item, todo
1050
+ and note, its expectations as metrics ahead of any metrics it held, and
1051
+ anything else it carried kept as it was. */
1052
+ function caseOfV4(c ) {
1053
+ if (!isObj(c)) return c;
1054
+ const was = {};
1055
+ for (const [k, v] of Object.entries(c)) {
1056
+ const key = CASE_RENAMED[k] ?? k;
1057
+ // A case naming its item both ways keeps the newer key's.
1058
+ if (!(key in was) || key === k) was[key] = v;
1059
+ }
1060
+ if (isStr(was.item)) delete was.filename;
1061
+ const out = {};
1062
+ if ("id" in was) out.id = was.id;
1063
+ out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
1064
+ out.todo = was.todo === true;
1065
+ out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
1066
+ out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
1067
+ for (const [k, v] of Object.entries(was)) {
1068
+ if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
1069
+ }
1070
+ return out;
1071
+ }
1072
+
1073
+ /** A version-4 case's expectations as the metrics that say the same:
1074
+ `expect` is Contains all, each `anyOf` group a Contains any, each
1075
+ forbidden term a Contains turned round with the `allow` phrases that
1076
+ excuse it, the count bounds an Item count, `discarded` the Discarded
1077
+ metric, and each watched term a Contains any of weight 0 -- reported,
1078
+ never scored. The order is the one scoreCase read them in, so a reason
1079
+ reads in the order it did. */
1080
+ function caseMetrics(c ) {
1081
+ const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
1082
+ if (c.discarded === true) return [{ type: "discarded" }];
1083
+ const out = [];
1084
+ const expect = list(c.expect), allow = list(c.allow);
1085
+ if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
1086
+ for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
1087
+ if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
1088
+ }
1089
+ for (const t of list(c.forbid)) {
1090
+ // An exception excuses only the forbidden term inside it.
1091
+ const except = allow.filter(a => termIn([a], t));
1092
+ out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
1093
+ }
1094
+ const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
1095
+ if (min != null || max != null) out.push({ type: "item-count", min, max });
1096
+ for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
1097
+ return out;
984
1098
  }
985
1099
 
986
1100
  /** The rules a version-1 or version-2 dataset body held, or null: what a
@@ -996,6 +1110,9 @@ function datasetRules(body ) {
996
1110
 
997
1111
 
998
1112
 
1113
+
1114
+
1115
+
999
1116
 
1000
1117
 
1001
1118
 
@@ -1023,6 +1140,10 @@ function datasetRules(body ) {
1023
1140
 
1024
1141
 
1025
1142
 
1143
+
1144
+
1145
+
1146
+
1026
1147
 
1027
1148
 
1028
1149
 
@@ -1067,7 +1188,7 @@ function datasetRules(body ) {
1067
1188
 
1068
1189
 
1069
1190
  /** What a Call hands its connection. */
1070
-
1191
+
1071
1192
 
1072
1193
  /** What the core hands a registry module (kinds/list.ts) to register with. */
1073
1194
 
@@ -1120,7 +1241,9 @@ function mappingsFor(prompt ) {
1120
1241
 
1121
1242
 
1122
1243
 
1123
-
1244
+
1245
+
1246
+
1124
1247
 
1125
1248
 
1126
1249
 
@@ -1157,6 +1280,32 @@ TOKEN_TYPES.block = {
1157
1280
  validate(m, at, bad){ if (typeof m.enabled !== "boolean") bad.push(`${at}: a block's enabled has to be true or false`); },
1158
1281
  };
1159
1282
 
1283
+ // Expression: a Power Automate expression, evaluated against the item's
1284
+ // record -- `@triggerBody()?['note']`, or text with `@{…}` in it -- so a
1285
+ // prompt can say what the flow's own request says (docs/pipeline-model.md
1286
+ // §16 › Targets).
1287
+ TOKEN_TYPES.expression = {
1288
+ label: "Expression",
1289
+ fields: ["value"],
1290
+ defaults: () => ({ value: "" }),
1291
+ notation: name => `{${name}}`,
1292
+ names: name => [name],
1293
+ order: 1,
1294
+ apply(prompt, m, record){
1295
+ const token = `{${m.name}}`;
1296
+ if (!prompt.includes(token)) return prompt;
1297
+ if (!record) throw new WdlError(`{${m.name}} reads the item's record, and this item is not one`);
1298
+ try {
1299
+ return prompt.replaceAll(token, asText(evaluate(m.value ?? "", { scope: record.scope || {}, now: record.at ?? null, loop: record.loop ?? null })));
1300
+ } catch (e) {
1301
+ throw new WdlError(`{${m.name}}: ${(e ).message}`);
1302
+ }
1303
+ },
1304
+ validate(m, at, bad){
1305
+ if (!isStr(m.value)) bad.push(`${at}: an expression's value has to be text`);
1306
+ },
1307
+ };
1308
+
1160
1309
  /** A mapping of [type] named [name], holding its type's defaults. */
1161
1310
  function tokenMapping(name , type ) {
1162
1311
  return { name, type, ...(TOKEN_TYPES[type]?.defaults() ?? {}) };
@@ -1186,12 +1335,12 @@ function tokenMappingsProblems(list , at , bad ) {
1186
1335
  * The tokens are a parameter and not a global, because only one of the three
1187
1336
  * callers has a localStorage to have loaded a set from.
1188
1337
  */
1189
- function resolvePrompt(tpl , tokens ) {
1338
+ function resolvePrompt(tpl , tokens , record = null) {
1190
1339
  const order = (m ) => TOKEN_TYPES[m.type]?.order ?? Infinity;
1191
1340
  let out = tpl;
1192
1341
  for (const m of [...tokens].sort((a, b) => order(a) - order(b))) {
1193
1342
  const type = TOKEN_TYPES[m.type];
1194
- if (type) out = type.apply(out, m);
1343
+ if (type) out = type.apply(out, m, record);
1195
1344
  }
1196
1345
  // Dropping a block leaves the spaces that surrounded it.
1197
1346
  return out.replace(/[ \t]{2,}/g, " ").trim();
@@ -1231,8 +1380,10 @@ function textPrompt(instruction , text ) {
1231
1380
  }
1232
1381
 
1233
1382
  // Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
1234
- // work in, so that "New Zealand." and "new zealand" are the same two words.
1235
- const words = (s ) => String(s||"").toLowerCase().match(/[\p{L}\p{N}]+/gu) || [];
1383
+ // work in, so that "New Zealand." and "new zealand" are the same two words --
1384
+ // unless [keepCase], for a metric whose Ignore case is off.
1385
+ const words = (s , keepCase = false) =>
1386
+ (keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
1236
1387
 
1237
1388
  // The request Tagger builds, field for field -- when the target reads those
1238
1389
  // fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
@@ -1376,8 +1527,10 @@ CONNECTION_TYPES["ollama-cloud"] = ollamaType("ollama-cloud", "Ollama (Cloud)",
1376
1527
  // each scenario holds one. Only job 1 can use it: a later job's item is the
1377
1528
  // job before's answer, not a file.
1378
1529
  const sendsNothing = () => { throw new Error("Echo answers from the item and sends no request"); };
1530
+ // Echo is a target step now (STEP_TYPES.echo), with no profile; the type
1531
+ // stays so a stored run, or a profile made before, still reads.
1379
1532
  CONNECTION_TYPES.echo = {
1380
- id: "echo", label: "Echo", settings: [], keyless: true,
1533
+ id: "echo", label: "Echo", settings: [], keyless: true, picker: false,
1381
1534
  description: "No model: answers with each text item itself, to grade replies recorded earlier.",
1382
1535
  local: (item, sent) => item.text ?? sent,
1383
1536
  request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
@@ -1547,16 +1700,19 @@ CONNECTION_TYPES["llama.cpp"] = {
1547
1700
 
1548
1701
 
1549
1702
 
1550
-
1703
+
1551
1704
 
1552
-
1705
+
1553
1706
 
1554
-
1707
+
1555
1708
 
1556
1709
 
1557
1710
 
1558
1711
 
1559
1712
 
1713
+
1714
+
1715
+
1560
1716
 
1561
1717
 
1562
1718
 
@@ -1622,6 +1778,7 @@ HTTP_APIS.anthropic = {
1622
1778
  const r = j ;
1623
1779
  return { raw: (r?.content || []).filter(b => b?.type === "text").map(b => b.text).join(""), finishReason: r?.stop_reason ?? null };
1624
1780
  },
1781
+ wrap: text => ({ type: "message", role: "assistant", content: [{ type: "text", text }], stop_reason: "end_turn" }),
1625
1782
  withPrefill(j, prefill) {
1626
1783
  const r = j ;
1627
1784
  if (!isObj(r) || !Array.isArray(r.content)) return j;
@@ -1667,6 +1824,7 @@ const openAiApi = (label , matches ) =>
1667
1824
  return { ...b, messages: msgs };
1668
1825
  },
1669
1826
  reply: chatReply,
1827
+ wrap: text => ({ object: "chat.completion", choices: [{ index: 0, message: { role: "assistant", content: text }, finish_reason: "stop" }] }),
1670
1828
  });
1671
1829
  HTTP_APIS.openai = openAiApi("OpenAI", u => u.hostname === "api.openai.com");
1672
1830
  HTTP_APIS["azure-openai"] = openAiApi("Azure OpenAI", u => u.hostname.endsWith(".openai.azure.com"));
@@ -1684,6 +1842,9 @@ HTTP_APIS.raw = {
1684
1842
  try { return JSON.parse(cell.prompt); } catch { throw new Error("the body is not JSON"); }
1685
1843
  },
1686
1844
  reply: j => ({ raw: isStr(j) ? j : JSON.stringify(j), finishReason: null }),
1845
+ // A body the flow reads is JSON where it parses as JSON, as the HTTP
1846
+ // action hands it on, and text otherwise.
1847
+ wrap: text => { try { return JSON.parse(text); } catch { return text; } },
1687
1848
  };
1688
1849
 
1689
1850
  /** The HTTP API a flow's request to [uri] speaks: the first entry that
@@ -1938,11 +2099,12 @@ function budgetLabel(target ) {
1938
2099
  // A term is present when its words appear in some item, in order and adjacent:
1939
2100
  // "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
1940
2101
  // looser scores "new zealand" against "zealandia" and flatters every run.
1941
- function termIn(items , term ) {
1942
- const t = words(term);
2102
+ // Letter case counts only where a metric's Ignore case is off.
2103
+ function termIn(items , term , ignoreCase = true) {
2104
+ const t = words(term, !ignoreCase);
1943
2105
  if (!t.length) return false;
1944
2106
  return items.some(item => {
1945
- const w = words(item);
2107
+ const w = words(item, !ignoreCase);
1946
2108
  for (let i = 0; i + t.length <= w.length; i++) {
1947
2109
  if (t.every((x, j) => w[i + j] === x)) return true;
1948
2110
  }
@@ -1969,153 +2131,55 @@ function forbiddenIn(items , term , allow
1969
2131
  * check green. `evals-check.js` calls this directly.
1970
2132
  */
1971
2133
  function gradedSetFrom(ev ) {
1972
- return (ev.cases || []).map(c => ({ ...c, filename: caseFile(c), half: "cases" }) );
2134
+ return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
1973
2135
  }
1974
2136
 
1975
- /**
1976
- * The file a case grades, whichever of the two keys names it.
1977
- *
1978
- * One reader, so that accepting the older spelling is a fact about this
1979
- * function rather than a branch every caller carries. Everything that joins a
1980
- * case to an item -- the runner, the Datasets tab, a mapping against a Source
1981
- * -- goes through here.
1982
- */
1983
- function caseFile(kase ) { // vocab: the older spelling
1984
- const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
1985
- return typeof name === "string" ? name : "";
2137
+ /** The item a case grades: its `item`, or "" for none. The one reader, so
2138
+ everything that joins a case to an item -- the runner, the Library, a
2139
+ run's Results -- joins on the same key, exactly. */
2140
+ function caseItem(kase ) {
2141
+ return isStr(kase?.item) ? kase .item : "";
1986
2142
  }
1987
2143
 
1988
- /**
1989
- * A graded set with every case naming its file as `filename`.
1990
- *
1991
- * What `evalsJson` writes, so the committed bytes carry one key and a set
1992
- * re-synced from an app's own repository is normalised the first time it is
1993
- * exported. The key takes the place the older one held, so normalising a set
1994
- * changes the spelling of
1995
- * one key and not the order of any.
1996
- */
1997
- function canonicalCases(ev ) {
1998
- const OLD = "photo"; // vocab: the older spelling of filename
1999
- const set = ev ;
2000
- const cases = set && typeof set === "object" && Array.isArray(set.cases) ? set.cases : null;
2001
- if (!cases || !cases.some(c => c && typeof c === "object" && OLD in (c ))) return ev;
2002
- return {
2003
- ...set,
2004
- cases: cases.map(c => {
2005
- if (!c || typeof c !== "object" || !(OLD in (c ))) return c;
2006
- const out = {};
2007
- for (const [k, v] of Object.entries(c )) {
2008
- if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
2009
- else out[k] = v;
2010
- }
2011
- return out;
2012
- }),
2013
- };
2014
- }
2144
+ /** One graded case's reading of a reply, now: its own metrics, scored as a
2145
+ Metrics eval scores them (all must pass; weight 0 is watched, never
2146
+ scored), with what they found, missed and invented carried up, and
2147
+ `watch` the readings of weight 0. What the page draws and the runner's
2148
+ case-by-case report writes. A model-graded metric needs a grader and the
2149
+ wait for one, so here it reads as not met: the runner's `--run` asks it. */
2150
+
2151
+
2152
+
2153
+
2154
+
2155
+
2156
+
2157
+
2158
+
2159
+
2160
+
2161
+
2162
+
2163
+
2015
2164
 
2016
- /**
2017
- * One graded case, one result.
2018
- *
2019
- * Every expectation is a group, and a plain `expect` term is a group of one:
2020
- * the requirements are the `expect` terms and the `anyOf` groups alike, and
2021
- * the score is found requirements over all of them plus `forbid`. A count
2022
- * bound stays pass/fail: an item that produced two perfect terms when five
2023
- * were wanted has not done what was asked.
2024
- *
2025
- * Pure on purpose, and returning `reasons` as plain sentences rather than
2026
- * markup. The dashboard was the first consumer; `run-evals.js` is the second
2027
- * and CI the third, and neither can reach into a page for a rendered cell.
2028
- * Issue #248 wants a failure written out as a task an agent can act on.
2029
- */
2030
- function scoreCase(kase , res ) {
2031
- // A case that expects its answer discarded passes on a discard and on
2032
- // nothing else: the answer the job threw away is the finding.
2033
- if (kase.discarded === true) {
2034
- const thrown = !!res.error && res.error.startsWith("discarded: ");
2035
- const n = (res.terms || []).length;
2036
- return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
2037
- unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
2038
- reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
2039
- }
2040
- // A discarded answer is a failed item: whatever the pipeline produced
2041
- // along the way does not count.
2042
- const discarded = !!res.error;
2043
- const terms = discarded ? [] : (res.terms || []);
2044
- const expect = kase.expect || [], forbid = kase.forbid || [];
2045
-
2046
- // `missed` stays populated for a discarded reply even though it reads as
2047
- // vacuous, because the score depends on it: docs/datasets.md has such a reply
2048
- // scoring zero against everything the entry asked for, groups included, and
2049
- // `addToTally` gets there through `found + missed + invented`. Clear it and
2050
- // a run that discarded every item would contribute nothing to the
2051
- // total instead of contributing a nought, which flatters it.
2052
- //
2053
- // One member of a group is enough. Without this rule the set manufactures
2054
- // failures out of synonyms.
2055
- //
2056
- // A requirement reads back as a term when it had one member and as the
2057
- // group itself where a synonym list was allowed, so `missed` carries just
2058
- // enough to state the reason: "FAIL: Missed dog" for a plain expect term,
2059
- // "none of dog / puppy" for a group.
2060
- //
2061
- // Nothing satisfies a group when nothing was stored, so a discarded reply
2062
- // misses every requirement rather than none -- the same cast that makes its
2063
- // count bounds not breached by one. `unmet` is a finding only now that
2064
- // `missed` names the unsatisfied groups itself, but the run report has
2065
- // always carried it, so it stays.
2066
- const met = (g ) => g.some(t => termIn(terms, t));
2067
- const spoken = (g ) => g.length === 1 ? g[0] : g;
2068
- const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
2069
- const found = requirements.filter(met).map(spoken);
2070
- const missed = requirements.filter(g => !met(g)).map(spoken);
2071
- const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
2072
- const unmet = discarded ? []
2073
- : (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
2074
-
2075
- const n = terms.length;
2076
- // A null bound is not checked -- and neither is a bound on a reply that was
2077
- // discarded. Nothing was counted out and found wanting there; the reply was
2078
- // thrown away before it had a count, and saying "0 items, wanted at least 4"
2079
- // states a second failure that never happened.
2080
- const under = !discarded && kase.minCount != null && n < kase.minCount;
2081
- const over = !discarded && kase.maxCount != null && n > kase.maxCount;
2082
-
2083
- const denom = requirements.length + invented.length;
2084
- // A case that passes is one whose whole expectation was met, not one that
2085
- // scored well: an unsatisfied group is in `missed` alongside any term
2086
- // missed, so pass needs nothing more than the terms already covered.
2087
- const pass = !discarded && !missed.length && !invented.length && !under && !over;
2088
-
2089
- // `found` and `missed` are groups, not terms, so the reasons word them one
2090
- // by one: a bare term under one heading, a group on its own line.
2091
- const reasons = [];
2092
- if (discarded) {
2093
- // Everything else would be derived from this one fact -- there are no
2094
- // terms -- and would bury it. `scoreHtml` above suppresses `missed` for
2095
- // the same reason: the reason that matters is already on the row.
2096
- reasons.push(`Error - ${res.error}`);
2097
- } else {
2098
- const plain = missed.filter(t => typeof t === "string");
2099
- if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
2100
- for (const g of missed) {
2101
- if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
2165
+ function readCase(kase , res , plain = false) {
2166
+ const input = metricInput(res , kase, { plain });
2167
+ const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
2168
+ const r = readMetric(m, input, {});
2169
+ if (r && typeof (r ).then === "function") {
2170
+ return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
2171
+ pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
2102
2172
  }
2103
- if (invented.length) reasons.push(`invented ${invented.join(", ")}`);
2104
- if (under) reasons.push(`${n} items, wanted at least ${kase.minCount}`);
2105
- if (over) reasons.push(`${n} items, wanted at most ${kase.maxCount}`);
2106
- }
2107
-
2108
- // Observed rather than scored: a dataset that wants to know whether some
2109
- // terms turn up, without grading on them, lists them as `watch`.
2110
- const watch = kase.watch || [];
2111
- // `null`, not 1, when there are no requirements and nothing forbidden turned
2112
- // up: there is no score to report. Returning 1 there printed "fail 100%"
2113
- // beside an entry that asked for nothing -- a shape `evals-check.js` allows
2114
- // even though every graded case now names an expectation.
2115
- return { pass, score: denom ? found.length / denom : null, discarded,
2116
- found, missed, invented, unmet, under, over, count: n,
2117
- watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
2118
- reasons };
2173
+ return r ? [r ] : [];
2174
+ });
2175
+ const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
2176
+ const counted = metrics.filter(r => r.weight !== 0);
2177
+ return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
2178
+ discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
2179
+ // A reply the job threw away fails on that one fact, which would
2180
+ // be buried under everything that derives from it.
2181
+ reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
2182
+ watch: metrics.filter(r => r.weight === 0) };
2119
2183
  }
2120
2184
 
2121
2185
  /**
@@ -2156,11 +2220,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
2156
2220
  // docs/datasets.md: an item naming more terms should weigh more than
2157
2221
  // one naming fewer, and averaging percentages lets the easiest case carry the
2158
2222
  // score.
2159
- function addToTally(t , s ) {
2223
+ function addToTally(t , s ) {
2224
+ const found = s.found?.length ?? 0;
2160
2225
  t.ran++;
2161
2226
  if (s.pass) t.passed++;
2162
- t.found += s.found.length;
2163
- t.of += s.found.length + s.missed.length + s.invented.length;
2227
+ t.found += found;
2228
+ t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
2164
2229
  }
2165
2230
 
2166
2231
  // The headline percentage, in one place because it is a number and this file
@@ -2175,7 +2240,7 @@ function tallyPercent(t ) {
2175
2240
  // A whole-run assertion with no graded set: All of / Any of / None of, a
2176
2241
  // parse, a count, an exact reply and a length bound. It lives in the shared
2177
2242
  // core because it is a pass/fail the tab, the worker and History all have to
2178
- // agree on -- the same one-definition rule that keeps `scoreCase` here.
2243
+ // agree on -- the same one-definition rule that keeps the case reader here.
2179
2244
  function parseCount(raw , parse ) {
2180
2245
  // An unparsed reply is one result, not no result. It used to return null,
2181
2246
  // which made a Count of 1 fail as "null results" against a reply that
@@ -2432,7 +2497,7 @@ function jobProblem(stages , call
2432
2497
  // 1 the way `textPrompt` says, and any later stage that places {text}.
2433
2498
  async function runPipeline(stages , dataUrl , call ,
2434
2499
  opts = {}) {
2435
- const { tokens = TOKEN_DEFAULTS, mode, text = null } = opts;
2500
+ const { tokens = TOKEN_DEFAULTS, mode, text = null, record = null } = opts;
2436
2501
  const transcript = [];
2437
2502
  const problem = jobProblem(stages, call, tokens, text);
2438
2503
  if (problem) return { error: problem, terms: [], transcript, ms: 0 };
@@ -2466,8 +2531,15 @@ async function runPipeline(stages , dataUrl , cal
2466
2531
  forwarded.set("reply", prev.reply);
2467
2532
  for (const [name, v] of Object.entries(prev.rendered)) forwarded.set(name, v);
2468
2533
  }
2469
- // A verbatim stage's words are the call's own template, sent as written.
2470
- const instruction = st.verbatim ? st.text : resolvePrompt(st.text, tokenSet(tokens, i));
2534
+ // A verbatim stage's words are its target step's own template, sent as written.
2535
+ // An expression token that cannot be read is the stage's error, said
2536
+ // before anything is sent.
2537
+ let instruction ;
2538
+ try { instruction = st.verbatim ? st.text : resolvePrompt(st.text, tokenSet(tokens, i), record); }
2539
+ catch (e) {
2540
+ if (!(e instanceof WdlError)) throw e;
2541
+ return result({ error: named(e.message) });
2542
+ }
2471
2543
  const fill = (to ) => instruction.replace(TOKEN, (m, name ) =>
2472
2544
  forwarded.has(name) ? to(forwarded.get(name) ) : m);
2473
2545
  const sent = st.verbatim ? instruction : i === 0 && text != null ? textPrompt(instruction, text) : fill(v => v);
@@ -2558,10 +2630,10 @@ function applyModifiers (list , kind ,
2558
2630
  // queue and run-evals.js alike, and every one of them reads it through the
2559
2631
  // functions below rather than through a translation of its own.
2560
2632
  //
2561
- // The lab is generic, so what a reply is, what a test scores and how a value
2633
+ // The lab is generic, so what a reply is, what an eval scores and how a value
2562
2634
  // is changed are registry entries. The ones here are the lab's own, and so
2563
2635
  // are kinds/list.ts's -- the List kind and its modifiers. A dataset
2564
- // registers nothing: it is data a graded test names by id, and a run carries.
2636
+ // registers nothing: it is data a graded eval names by id, and a run carries.
2565
2637
 
2566
2638
  // 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
2567
2639
  // version 1 kept them as two maps (tokens: { values, blocks }).
@@ -2577,10 +2649,16 @@ function applyModifiers (list , kind ,
2577
2649
  // 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
2578
2650
  // 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
2579
2651
  // Call (Prompt: the image flag and token mappings), Read Reply (the output).
2580
- // 9: a test is Metrics. A Single Test and a Graded set are read as the Metrics
2652
+ // 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
2581
2653
  // they convert to (LEGACY_TESTS' toMetrics, proven equal by
2582
2654
  // metrics-parity-check.js).
2583
- const PIPELINE_VERSION = 9 ;
2655
+ // 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
2656
+ // Content, the flow step, Attach Image, Token mappings) and Responses (Read
2657
+ // as, Response Format Validation, modifiers) -- and its call is gone: each
2658
+ // scenario is a target, whose own step in each job is what it sends there.
2659
+ // 11: `tests` are `evals`: the key renames and nothing in an eval changes,
2660
+ // so a stored result's scores, keyed by eval id, read as they did.
2661
+ const PIPELINE_VERSION = 12 ;
2584
2662
 
2585
2663
  // Plain objects, so an entry is added by assignment and a reader never needs
2586
2664
  // to know which registered it.
@@ -2588,7 +2666,7 @@ const STEP_TYPES = Object.create(null);
2588
2666
  const CONTENT_TYPES = Object.create(null);
2589
2667
  const OUTPUT_KINDS = Object.create(null);
2590
2668
  const MODIFIERS = Object.create(null);
2591
- const TEST_TYPES = Object.create(null);
2669
+ const EVAL_TYPES = Object.create(null);
2592
2670
  const METRICS = Object.create(null);
2593
2671
  const SOURCE_TYPES = Object.create(null);
2594
2672
 
@@ -2602,11 +2680,15 @@ let defaultOutputKind = "text";
2602
2680
  const defaultKind = () => defaultOutputKind;
2603
2681
  const kindOf = (st ) => st.kind ?? defaultOutputKind;
2604
2682
 
2683
+ /** A module's eval types, under either spelling (Kinds.testTypes). */
2684
+ const evalTypesOf = (k ) =>
2685
+ ({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
2686
+
2605
2687
  /** A module's entries into the registries, in one call. */
2606
2688
  function registerKinds(k ) {
2607
2689
  Object.assign(OUTPUT_KINDS, k.outputKinds || {});
2608
2690
  Object.assign(MODIFIERS, k.modifiers || {});
2609
- Object.assign(TEST_TYPES, k.testTypes || {});
2691
+ Object.assign(EVAL_TYPES, evalTypesOf(k));
2610
2692
  Object.assign(SOURCE_TYPES, k.sourceTypes || {});
2611
2693
  Object.assign(METRICS, k.metrics || {});
2612
2694
  if (k.defaultKind) defaultOutputKind = k.defaultKind;
@@ -2637,7 +2719,7 @@ function pluginHost(pluginId ) {
2637
2719
  registerKinds(k) {
2638
2720
  taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
2639
2721
  taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
2640
- taken(TEST_TYPES, "test type", Object.keys(k.testTypes || {}));
2722
+ taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
2641
2723
  taken(METRICS, "metric", Object.keys(k.metrics || {}));
2642
2724
  // The server keeps its own copy of the Source types, to refuse a row of
2643
2725
  // one it does not know, and a plugin never reaches the server's code.
@@ -2741,8 +2823,7 @@ function encoderQuality(quality ) {
2741
2823
  // always shown, so a sentence names what the reader sees.
2742
2824
  const jobLabel = (doc , k ) =>
2743
2825
  (doc.jobs?.[k]?.name || "").trim() || `Job ${k + 1}`;
2744
- const scenarioLabel = (doc , i ) =>
2745
- (doc.scenarios?.[i]?.name || "").trim() || `Scenario ${i + 1}`;
2826
+ const scenarioLabel = (doc , i ) => targetLabel(doc, i);
2746
2827
  // The runner calls a job's link a stage; the model calls it a job.
2747
2828
  const jobWords = (s ) => s && String(s).replace(/\bstage[ ](\d+)/g, "job $1");
2748
2829
 
@@ -2850,7 +2931,7 @@ CONTENT_TYPES.text = {
2850
2931
  // being the recorded reply.
2851
2932
  CONTENT_TYPES.prompt = {
2852
2933
  label: "Prompt only",
2853
- description: "No item: each scenario's prompt is sent on its own.",
2934
+ description: "No item: each target's prompt is sent on its own.",
2854
2935
  inline: true,
2855
2936
  bare: true,
2856
2937
  fields: ["type"],
@@ -2904,11 +2985,9 @@ LEGACY_TESTS.single = {
2904
2985
  },
2905
2986
  };
2906
2987
 
2907
- // A dataset's cases, scored item by item with scoreCase. The test
2908
- // references the dataset the way a pipeline references a Source, and the run
2909
- // carries the body it was submitted against. It scores the terms a value
2910
- // yields, so it accepts every kind that yields any: plain text has none, and
2911
- // would fail every case.
2988
+ // A dataset's cases, scored item by item. The eval references the dataset
2989
+ // the way a pipeline references a Source, and the run carries the body it
2990
+ // was submitted against.
2912
2991
  LEGACY_TESTS.graded = {
2913
2992
  label: "Graded set",
2914
2993
  fields: ["type", "dataset"],
@@ -2916,16 +2995,15 @@ LEGACY_TESTS.graded = {
2916
2995
  validate(t, ctx, bad){
2917
2996
  const d = t.dataset;
2918
2997
  if (!isRef(d) || (d.version != null && !isStr(d.version))) {
2919
- return void bad.push("a graded test has to name its dataset as { id, name }");
2998
+ return void bad.push("a graded eval has to name its dataset as { id, name }");
2920
2999
  }
2921
3000
  if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
2922
3001
  },
2923
- score: (t, kase, res) => scoreCase(kase, res),
2924
3002
  };
2925
3003
 
2926
3004
  // Metrics: checks of each reply, deterministic or model-graded, from the
2927
3005
  // METRICS registry (metrics/builtin.ts registers the lab's own), with the
2928
- // test's own list for every item and a case's `metrics` for its item. Scored
3006
+ // eval's own list for every item and a case's `metrics` for its item. Scored
2929
3007
  // all-must-pass -- every metric passes -- or weighted: points, each metric's
2930
3008
  // score times its weight, against a threshold (#150's points, a negative
2931
3009
  // weight taking them away).
@@ -2934,6 +3012,13 @@ LEGACY_TESTS.graded = {
2934
3012
  const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
2935
3013
  const SCORING_MODES = ["all", "weighted"];
2936
3014
 
3015
+ /** A metric's list option -- Contains all's values, Contains's exceptions --
3016
+ one entry a line, or a list of them. */
3017
+ function metricLines(v ) {
3018
+ const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
3019
+ return all.map(x => x.trim()).filter(Boolean);
3020
+ }
3021
+
2937
3022
  /** What is wrong with a list of metrics, as sentences naming [at]. */
2938
3023
  function metricsProblems(list , at , bad ) {
2939
3024
  if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
@@ -3003,14 +3088,16 @@ function readMetric(m , input , ctx )
3003
3088
  } catch (e) { return timed(failed(e)); }
3004
3089
  }
3005
3090
 
3006
- /** A test's score from its metrics' readings, the way [mode] says: all must
3091
+ /** An eval's score from its metrics' readings, the way [mode] says: all must
3007
3092
  pass, or weighted points against [threshold]. What a case metric found,
3008
3093
  missed and invented is carried up, for a run's totals. Null for none. */
3009
3094
  function scoreOf(metrics , mode = "all", threshold = null) {
3010
3095
  if (!metrics.length) return null;
3011
- const detail = metrics.some(r => r.found || r.missed || r.invented) ? {
3012
- found: metrics.flatMap(r => r.found ?? []), missed: metrics.flatMap(r => r.missed ?? []),
3013
- invented: metrics.flatMap(r => r.invented ?? []) } : {};
3096
+ // A watched reading (weight 0) is reported, and counts toward nothing.
3097
+ const scored = metrics.filter(r => r.weight !== 0);
3098
+ const detail = scored.some(r => r.found || r.missed || r.invented) ? {
3099
+ found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
3100
+ invented: scored.flatMap(r => r.invented ?? []) } : {};
3014
3101
  if (mode === "weighted") {
3015
3102
  const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
3016
3103
  return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
@@ -3053,7 +3140,7 @@ function readRun(list , run ) {
3053
3140
  });
3054
3141
  }
3055
3142
 
3056
- TEST_TYPES.metrics = {
3143
+ EVAL_TYPES.metrics = {
3057
3144
  label: "Metrics",
3058
3145
  description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3059
3146
  fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
@@ -3077,16 +3164,16 @@ TEST_TYPES.metrics = {
3077
3164
  if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
3078
3165
  if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
3079
3166
  if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
3080
- // The test's own grader, or the lab's.
3167
+ // The eval's own grader, or the lab's.
3081
3168
  const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
3082
- if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a Setup profile the lab's grader");
3169
+ if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
3083
3170
  // A grader is asked words, and needs a model to ask.
3084
3171
  const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3085
- if (grader && ctx.profiles && !p) bad.push(`Setup profile ${grader.name || grader.id} not found`);
3172
+ if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
3086
3173
  else if (p) {
3087
3174
  const type = CONNECTION_TYPES[typeOf(p )];
3088
3175
  if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
3089
- else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Setup profile");
3176
+ else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
3090
3177
  }
3091
3178
  },
3092
3179
  rules: t => (Array.isArray(t.metrics) ? t.metrics : []).map((m , x ) => ({
@@ -3094,7 +3181,7 @@ TEST_TYPES.metrics = {
3094
3181
  want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3095
3182
  .filter(Boolean).join(" ") })),
3096
3183
  profiles: t => (isRef(t.grader) ? [t.grader] : []),
3097
- // A run carries the lab's grader on a test that names none and may ask one:
3184
+ // A run carries the lab's grader on an eval that names none and may ask one:
3098
3185
  // a model-graded metric of its own, or a case's, over a dataset.
3099
3186
  resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
3100
3187
  ? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
@@ -3109,44 +3196,28 @@ TEST_TYPES.metrics = {
3109
3196
  ran: run.replies.length,
3110
3197
  checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3111
3198
  },
3112
- // Rule m{x} is the test's own metric x, read in that place on every item.
3199
+ // Rule m{x} is the eval's own metric x, read in that place on every item.
3113
3200
  // A score stored before Metrics -- a Graded set's, read as its conversion --
3114
3201
  // has no readings of its own: its one metric's reading is the score's.
3115
3202
  ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3116
3203
  : Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
3117
- // Every item, with its case's own metrics where it has a case.
3118
- read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((kase?.metrics ) || [])],
3204
+ // Every item, with its case's own metrics where the eval names the
3205
+ // dataset and the item has a case there.
3206
+ read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
3119
3207
  metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3120
3208
  };
3121
3209
 
3122
3210
  // ---- the lab's own scorers, as metrics ----------------------------------------
3123
- // What the Graded set and the Single Test check, one metric each, so a test of
3124
- // either converts to Metrics that read a run exactly as it did
3125
- // (metrics-parity-check.js). Here rather than in metrics/builtin.ts because
3126
- // each is the core's own matcher.
3127
-
3128
- // An item's case, as the Graded set scores it: its expectations, forbidden
3129
- // terms and count bounds, with what it found, missed and invented kept.
3130
- METRICS.case = {
3131
- label: "Matches its case",
3132
- description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
3133
- perItem: true,
3134
- needsTerms: true,
3135
- options: [],
3136
- defaults: () => ({}),
3137
- score(input) {
3138
- if (!input.kase) return null;
3139
- const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
3140
- return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
3141
- found: s.found, missed: s.missed, invented: s.invented };
3142
- },
3143
- };
3211
+ // What the Single Test checks, one metric each, so an eval of it converts to
3212
+ // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
3213
+ // rather than in metrics/builtin.ts because each is the core's own matcher.
3144
3214
 
3145
3215
  // Items the reply holds -- all of them, or any -- matched as the lab matches
3146
3216
  // a term; a kind that yields none is matched in the replies' text instead,
3147
3217
  // ignoring case, as the Single Test did.
3148
3218
  METRICS["has-items"] = {
3149
3219
  label: "Has items",
3220
+ family: "The reply's text",
3150
3221
  description: "Passes when the reply holds all, or any, of the items listed.",
3151
3222
  options: [{ key: "values", label: "Items", type: "textarea" },
3152
3223
  { key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
@@ -3166,6 +3237,7 @@ METRICS["has-items"] = {
3166
3237
  // A reply's length in characters, once trimmed.
3167
3238
  METRICS.length = {
3168
3239
  label: "Length",
3240
+ family: "The reply's text",
3169
3241
  description: "Compares the reply's length in characters.",
3170
3242
  options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
3171
3243
  { key: "n", label: "Characters", type: "number" }],
@@ -3181,6 +3253,7 @@ METRICS.length = {
3181
3253
  // as the count says.
3182
3254
  METRICS["parse-count"] = {
3183
3255
  label: "Parses",
3256
+ family: "The result",
3184
3257
  description: "Passes when the reply reads as the format chosen, holding the number of results set.",
3185
3258
  // Unformatted is one result a reply, as the Single Test counted it.
3186
3259
  options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
@@ -3199,12 +3272,13 @@ METRICS["parse-count"] = {
3199
3272
  },
3200
3273
  };
3201
3274
 
3202
- /** What every test carries, kept across a conversion. */
3275
+ /** What every eval carries, kept across a conversion. */
3203
3276
  const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
3204
3277
 
3205
- // A Graded set is a Metrics test over the same dataset, its one metric the case.
3278
+ // A Graded set is a Metrics eval over the same dataset, with none of its own:
3279
+ // each case's metrics are what it scored (dataset-parity-check.js).
3206
3280
  LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
3207
- dataset: t.dataset, over: "item", metrics: [{ type: "case" }] });
3281
+ dataset: t.dataset, over: "item", metrics: [] });
3208
3282
 
3209
3283
  // A Single Test is Metrics over the whole run: its lists over the run's items
3210
3284
  // together, and exact, length and parse over each reply alone.
@@ -3222,7 +3296,7 @@ LEGACY_TESTS.single .toMetrics = t => {
3222
3296
  return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
3223
3297
  };
3224
3298
 
3225
- /** The grader a Metrics test names, reached through what the runner hands it. */
3299
+ /** The grader a Metrics eval names, reached through what the runner hands it. */
3226
3300
  function graderCtx(t , more ) {
3227
3301
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3228
3302
  return ask ? { ask } : {};
@@ -3232,27 +3306,33 @@ function graderCtx(t , more ) {
3232
3306
  for one with none set: what History prints beside its name. */
3233
3307
  function metricSummary(m ) {
3234
3308
  const entry = METRICS[m.type];
3235
- // Each option as it reads in the editor: a choice's label, a box that is
3236
- // ticked by its own label, text by its first line.
3309
+ // Each option as it reads in the editor: a choice's label, a box by its
3310
+ // own label where it is not as a new metric has it (ticked, or "off"),
3311
+ // text by its first line.
3312
+ const fresh = entry?.defaults() ?? {};
3237
3313
  const said = (entry?.options || []).map(o => {
3238
3314
  const v = m[o.key];
3239
3315
  if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3240
- if (o.type === "checkbox") return v ? o.label.toLowerCase() : "";
3316
+ if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
3241
3317
  // Text that runs to lines (a schema) reads as its first words, run together.
3242
3318
  return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3243
3319
  }).filter(Boolean);
3244
3320
  return said.join(", ");
3245
3321
  }
3246
3322
 
3247
- // ---- a job's steps (pipeline-model §3) -------------------------------------
3323
+ // ---- a job's steps, and a target's (pipeline-model §16) ----------------------
3324
+
3325
+ // Content: what a job is given. Attach Content and the flow step are job
3326
+ // 1's; any job may send the item's image and map tokens.
3248
3327
 
3249
3328
  // Attach Content: the items a run goes over, job 1's first step.
3250
3329
  STEP_TYPES.attachContent = {
3251
- label: "Attach Content", slot: "content",
3252
- description: "The items the run goes through: each is sent to every scenario.",
3330
+ label: "Attach Content", slot: "content", rank: 0,
3331
+ description: "The items the run goes through: each is sent to every target.",
3253
3332
  in: null, out: "items",
3254
3333
  apply: "itemContent", // run-evals.js, through its file-type registry
3255
3334
  fields: ["type", "content"],
3335
+ firstJobOnly: "only job 1 attaches content -- a later job is handed the reply before it",
3256
3336
  validate(step, ctx, bad){
3257
3337
  const c = step?.content;
3258
3338
  if (!c) return void bad.push("set the content first");
@@ -3263,20 +3343,65 @@ STEP_TYPES.attachContent = {
3263
3343
  },
3264
3344
  };
3265
3345
 
3266
- // Prompt: a Call that sends each scenario's words to its model.
3346
+ // The flow step: the Power Automate action a job's records are calls of.
3347
+ STEP_TYPES.flowStep = {
3348
+ label: "Flow step", slot: "content", rank: 1,
3349
+ description: "The Power Automate step each record is a call of.",
3350
+ in: "item", out: "item",
3351
+ apply: "runPipeline",
3352
+ fields: ["type", "step", "loop", "api"],
3353
+ firstJobOnly: "a flow step's records are job 1's items",
3354
+ // Production's reply is the record's result, read as the flow reads it.
3355
+ production(flow, record){
3356
+ const result = isObj(record) && isObj(record.result) ? record.result : null;
3357
+ if (!result || typeof result.status !== "number") return null;
3358
+ try {
3359
+ return httpReplyOf(flow, { prompt: "" }, result.body, result.status,
3360
+ { scope: record.scope || {}, now: record.at ?? null }).raw;
3361
+ } catch { return null; }
3362
+ },
3363
+ validate(step, _ctx, bad, at){
3364
+ if (!isStr(step.step) || !step.step.trim()) bad.push(`${at}: a flow step names the action it is`);
3365
+ if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
3366
+ if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
3367
+ },
3368
+ };
3369
+
3370
+ // Attach Image: the item's image beside every target's words.
3371
+ STEP_TYPES.attachImage = {
3372
+ label: "Attach Image", slot: "content", rank: 2,
3373
+ description: "Sends each item's image beside the words.",
3374
+ in: "item", out: "item",
3375
+ apply: "runPipeline",
3376
+ fields: ["type"],
3377
+ validate(){},
3378
+ };
3379
+
3380
+ // Token mappings: what the words' {tokens} are resolved to.
3381
+ STEP_TYPES.tokenMappings = {
3382
+ label: "Token mappings", slot: "content", rank: 3,
3383
+ description: "What each {token} in the words is filled in with.",
3384
+ in: "item", out: "item",
3385
+ apply: "runPipeline",
3386
+ fields: ["type", "tokenMappings"],
3387
+ validate(step, _ctx, bad, at){ tokenMappingsProblems(step.tokenMappings, at, bad); },
3388
+ };
3389
+
3390
+ // Targets: what each target sends. A target's own steps, one per job.
3391
+
3392
+ // Prompt: a target's words, asked of its model.
3267
3393
  STEP_TYPES.prompt = {
3268
- label: "Prompt", slot: "call", asks: "prompt", modelFrom: "profile",
3269
- description: "Asks each scenario's model its prompt about the item.",
3394
+ label: "Prompt", slot: "target", asks: "prompt", modelFrom: "profile",
3395
+ description: "Asks each target's model its prompt about the item.",
3270
3396
  in: "item", out: "text",
3271
3397
  apply: "runPipeline",
3272
- fields: ["type", "withImage", "tokenMappings"],
3398
+ fields: ["type", "prompt", "profile", "from"],
3273
3399
  validate(step, _ctx, bad, at){
3274
- if (typeof step.withImage !== "boolean") bad.push(`${at}: withImage has to be true or false`);
3275
- tokenMappingsProblems(step.tokenMappings, at, bad);
3400
+ if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
3276
3401
  },
3277
3402
  };
3278
3403
 
3279
- // HTTP Request: a Call that sends a Power Automate step's own request,
3404
+ // HTTP Request: a target that sends a Power Automate step's own request,
3280
3405
  // rebuilt for each item from its record (docs/power-automate.md).
3281
3406
  const HTTP_METHODS = ["GET", "POST", "PUT", "PATCH", "DELETE"];
3282
3407
  /** The first field under [v] named like a key, as a path, or null. */
@@ -3293,63 +3418,85 @@ function keyField(v , at = "body") {
3293
3418
  return null;
3294
3419
  }
3295
3420
  STEP_TYPES.httpRequest = {
3296
- label: "HTTP Request", slot: "call", asks: "request", verbatim: true, modelFrom: "cell",
3297
- description: "Sends a Power Automate step's own request, rebuilt from each record, with each scenario's words in it.",
3298
- cellFields: ["system", "model", "settings", "prefill"],
3299
- firstJobOnly: "an HTTP Request is sent from its item's record, so it is job 1's call",
3300
- // Production's reply is the record's result, read as the flow reads it.
3301
- production(call, record){
3302
- const result = isObj(record) && isObj(record.result) ? record.result : null;
3303
- if (!result || typeof result.status !== "number") return null;
3304
- try {
3305
- return httpReplyOf(call, { prompt: "" }, result.body, result.status,
3306
- { scope: record.scope || {}, now: record.at ?? null }).raw;
3307
- } catch { return null; }
3308
- },
3309
- // A new scenario starts from the flow's own words and fields.
3310
- newCell: (call) => (HTTP_APIS[call.api] ?? HTTP_APIS.raw ).cellOf(call.body),
3311
- // A cell's fields, where its HTTP API has a place for them.
3312
- cellProblems(cell, call, at, bad){
3313
- const api = HTTP_APIS[call.api];
3421
+ label: "HTTP Request", slot: "target", asks: "request", verbatim: true, modelFrom: "step",
3422
+ description: "Sends a Power Automate step's own request, rebuilt from each record, with each target's words in it.",
3423
+ firstJobOnly: "an HTTP Request is sent from its item's record, so it is job 1's",
3424
+ // A new target starts from the flow's own words and fields.
3425
+ newStep: (step) => ({ type: "httpRequest", api: step.api, method: step.method, path: step.path,
3426
+ query: clone(step.query), body: clone(step.body),
3427
+ ...(HTTP_APIS[step.api] ?? HTTP_APIS.raw ).cellOf(step.body) }),
3428
+ in: "item", out: "text",
3429
+ apply: "runPipeline",
3430
+ fields: ["type", "api", "method", "path", "query", "body", "prompt", "profile", "from", "system", "model", "settings", "prefill"],
3431
+ validate(step, _ctx, bad, at){
3432
+ const api = HTTP_APIS[step.api];
3433
+ if (!api) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
3434
+ if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
3435
+ if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
3436
+ if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
3437
+ if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
3438
+ // A key is the Setup profile's, sent by its Key header; one in the
3439
+ // template would be kept in the pipeline and every run of it.
3440
+ const hit = keyField(step.body) ?? keyField(step.query, "query");
3441
+ if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Target profile sends it`);
3314
3442
  if (!api) return;
3443
+ // Its fields, where its HTTP API has a place for them.
3315
3444
  for (const f of ["system", "model", "prefill"] ) {
3316
- if (cell[f] == null) continue;
3317
- if (!isStr(cell[f])) bad.push(`${at}: ${f} has to be text`);
3445
+ if (step[f] == null) continue;
3446
+ if (!isStr(step[f])) bad.push(`${at}: ${f} has to be text`);
3318
3447
  else if (!api.fields.includes(f)) bad.push(`${at}: ${api.label} has no place for a ${f}`);
3319
3448
  }
3320
- if (cell.settings != null) {
3321
- if (!isObj(cell.settings) || !Object.values(cell.settings).every(isStr)) bad.push(`${at}: settings are names to text`);
3322
- else for (const k of Object.keys(cell.settings)) {
3449
+ if (step.settings != null) {
3450
+ if (!isObj(step.settings) || !Object.values(step.settings).every(isStr)) bad.push(`${at}: settings are names to text`);
3451
+ else for (const k of Object.keys(step.settings)) {
3323
3452
  if (!api.settings.includes(k)) bad.push(`${at}: ${api.label} has no setting ${k}`);
3324
3453
  }
3325
3454
  }
3326
- if (isStr(cell.prompt)) {
3327
- try { api.place(call.body, cell ); }
3455
+ if (isStr(step.prompt)) {
3456
+ try { api.place(step.body, step ); }
3328
3457
  catch (e) { bad.push(`${at}: ${(e ).message}`); }
3329
3458
  }
3330
3459
  },
3460
+ };
3461
+
3462
+ // Echo: a target that sends nothing and needs no profile -- each text item
3463
+ // is its own reply, to grade replies recorded earlier (#97). Job 1's, over
3464
+ // text items only. Its words, where it has any, are what the reply is read
3465
+ // against; over Prompt only they are the reply.
3466
+ STEP_TYPES.echo = {
3467
+ label: "Echo", slot: "target", asks: "nothing", local: true, replaces: "echo",
3468
+ description: "Sends nothing: each text item is its own reply, to grade replies recorded earlier.",
3469
+ firstJobOnly: "Echo answers only job 1, with the item's own text",
3331
3470
  in: "item", out: "text",
3332
3471
  apply: "runPipeline",
3333
- fields: ["type", "step", "api", "method", "path", "query", "body", "readAs", "loop"],
3472
+ fields: ["type", "prompt", "from"],
3334
3473
  validate(step, _ctx, bad, at){
3335
- if (!isStr(step.step) || !step.step.trim()) bad.push(`${at}: an HTTP Request names the flow step it is`);
3336
- if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
3337
- if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
3338
- if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
3339
- if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
3340
- if (step.readAs != null && !isStr(step.readAs)) bad.push(`${at}: readAs is an expression or null`);
3341
- if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
3342
- // A key is the Setup profile's, sent by its Key header; one in the
3343
- // template would be kept in the pipeline and every run of it.
3344
- const hit = keyField(step.body) ?? keyField(step.query, "query");
3345
- if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Setup profile sends it`);
3474
+ if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
3346
3475
  },
3347
3476
  };
3348
3477
 
3349
- // Read Reply: how a job's answer is read -- its kind and modifiers.
3478
+ /** What a local target step is answered as: Echo's own connection, with no
3479
+ address, key or model. */
3480
+ const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
3481
+
3482
+ // Responses: how a job's reply is read, in order -- possibly not at all.
3483
+
3484
+ // Read as: what the flow does to the reply before it reads it.
3485
+ STEP_TYPES.readAs = {
3486
+ label: "Read as", slot: "responses", rank: 0,
3487
+ description: "What the flow makes of the reply before it reads it.",
3488
+ in: "text", out: "text",
3489
+ apply: "runPipeline",
3490
+ fields: ["type", "readAs"],
3491
+ validate(step, _ctx, bad, at){
3492
+ if (!isStr(step.readAs) || !step.readAs.trim()) bad.push(`${at}: readAs is an expression`);
3493
+ },
3494
+ };
3495
+
3496
+ // Response Format Validation: the kind a reply is read as.
3350
3497
  STEP_TYPES.readReply = {
3351
- label: "Read Reply", slot: "reply",
3352
- description: "How the reply is read before tests and later jobs see it.",
3498
+ label: "Read Reply", slot: "responses", rank: 1,
3499
+ description: "How the reply is read before evals and later jobs see it.",
3353
3500
  in: "text", out: step => step.out?.kind,
3354
3501
  // Read inside the stage: runPipeline parses each reply as it comes back.
3355
3502
  apply: "runPipeline",
@@ -3360,30 +3507,42 @@ STEP_TYPES.readReply = {
3360
3507
  return void bad.push(`${at} answers as "${out?.kind}", which is not an output kind this lab has`);
3361
3508
  }
3362
3509
  const kindEntry = OUTPUT_KINDS[out.kind] ;
3363
- onlyFields(out, `${at}'s output`, ["kind", "modifiers", ...(kindEntry.settings || []).map(o => o.key)], bad);
3510
+ onlyFields(out, `${at}'s output`, ["kind", ...(kindEntry.settings || []).map(o => o.key)], bad);
3364
3511
  kindEntry.validateSettings?.(out, `${at}'s output`, bad);
3365
- if (!Array.isArray(out.modifiers)) return void bad.push(`${at}: modifiers has to be a list`);
3366
- for (const m of out.modifiers) {
3367
- const entry = isObj(m) && MODIFIERS[m.type];
3368
- if (!entry) { bad.push(`${at}: "${m?.type}" is not a modifier this lab has`); continue; }
3369
- if (entry.accepts && !entry.accepts.includes(out.kind)) {
3370
- bad.push(`${at}: ${entry.label || m.type} does not apply to ${OUTPUT_KINDS[out.kind] .noun || out.kind}`);
3371
- }
3372
- entry.validate?.(m , `${at}: ${entry.label || m.type}`, bad);
3512
+ },
3513
+ };
3514
+
3515
+ // A modifier: one change to the parsed reply, after the kind that takes it.
3516
+ STEP_TYPES.modifier = {
3517
+ label: "Modifier", slot: "responses", rank: 2, many: true,
3518
+ description: "Changes the reply after it is read.",
3519
+ in: "value", out: "value",
3520
+ apply: "runPipeline",
3521
+ fields: ["type", "modifier"],
3522
+ validate(step, _ctx, bad, at, kind){
3523
+ const m = step.modifier;
3524
+ const entry = isObj(m) && MODIFIERS[m.type];
3525
+ if (!entry) return void bad.push(`${at}: "${m?.type}" is not a modifier this lab has`);
3526
+ if (entry.accepts && !entry.accepts.includes(kind)) {
3527
+ bad.push(`${at}: ${entry.label || m.type} does not apply to ${OUTPUT_KINDS[kind]?.noun || kind}`);
3373
3528
  }
3529
+ entry.validate?.(m , `${at}: ${entry.label || m.type}`, bad);
3374
3530
  },
3375
3531
  };
3376
3532
 
3377
- // ---- reading a job's steps ---------------------------------------------------
3378
- // Every reader asks these, never a step's position: a job's slots are fixed,
3379
- // but which of them a job has (Attach Content is job 1's alone) is its own.
3533
+ // ---- reading a job's steps, and a target's -----------------------------------
3534
+ // Every reader asks these, never a step's position or the document's shape:
3535
+ // which steps a job holds is its own, and a version that moves a field
3536
+ // changes these alone (pipeline-model §16).
3537
+
3538
+
3380
3539
 
3381
3540
  const slotOf = (st ) =>
3382
3541
  (isObj(st) ? STEP_TYPES[st.type ]?.slot : undefined);
3383
3542
 
3384
- /** The step in [job]'s [slot], whatever type fills it. */
3385
- const stepIn = (job , slot ) =>
3386
- (job?.steps || []).find(st => slotOf(st) === slot) ;
3543
+ /** [job]'s step of [type], when it holds one. */
3544
+ const stepOf = (job , type ) =>
3545
+ (job?.steps || []).find(st => isObj(st) && st.type === type) ;
3387
3546
 
3388
3547
  /** The registry entry that fills [slot] by default: what a new step there is. */
3389
3548
  function slotEntry(slot ) {
@@ -3391,9 +3550,26 @@ function slotEntry(slot )
3391
3550
  return hit ? { type: hit[0], entry: hit[1] } : undefined;
3392
3551
  }
3393
3552
 
3553
+ /** Where a step sits: its stage, then its place in it. */
3554
+ const placeOf = (st ) => {
3555
+ const e = isObj(st) ? STEP_TYPES[st.type ] : undefined;
3556
+ return (e?.slot ? SLOTS.indexOf(e.slot) : SLOTS.length) * 100 + (e?.rank ?? 0);
3557
+ };
3558
+
3559
+ /** [steps] in stage order, each stage in its own order; steps that tie
3560
+ (modifiers) keep theirs. */
3561
+ const ordered = (steps ) =>
3562
+ steps.map((st, x) => ({ st, x })).sort((a, b) => placeOf(a.st) - placeOf(b.st) || a.x - b.x).map(o => o.st);
3563
+
3564
+ /** [job] with its step of [type] set to [step], or taken away for null. */
3565
+ function withStep(job , type , step ) {
3566
+ const steps = job.steps.filter(st => st.type !== type);
3567
+ return { ...job, steps: ordered(step ? [...steps, step] : steps) };
3568
+ }
3569
+
3394
3570
  /** A pipeline's content: job 1's content step, or none. */
3395
3571
  function contentOf(doc ) {
3396
- return stepIn (doc?.jobs?.[0], "content")?.content ?? null;
3572
+ return stepOf (doc?.jobs?.[0], "attachContent")?.content ?? null;
3397
3573
  }
3398
3574
 
3399
3575
  /** [doc] with its content set -- job 1's content step added, replaced, or
@@ -3401,26 +3577,94 @@ function contentOf(doc )
3401
3577
  function withContent (doc , content ) {
3402
3578
  const [first, ...rest] = doc.jobs;
3403
3579
  if (!first) return doc;
3404
- const steps = first.steps.filter(st => slotOf(st) !== "content");
3405
- const type = slotEntry("content") .type;
3406
- return { ...doc, jobs: [{ ...first, steps: content ? [{ type, content } , ...steps] : steps }, ...rest] };
3580
+ return { ...doc, jobs: [withStep(first, "attachContent", content ? { type: "attachContent", content } : null), ...rest] };
3581
+ }
3582
+
3583
+ /** A document's targets, in column order. */
3584
+ function targetsOf(doc ) {
3585
+ return Array.isArray(doc?.targets) ? doc .targets : [];
3586
+ }
3587
+
3588
+ /** Target [i]'s name, or the number it has always shown. */
3589
+ function targetLabel(doc , i ) {
3590
+ return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
3591
+ }
3592
+
3593
+ /** What target [i] sends in job [k]: its step there. */
3594
+ function targetStepOf(doc , i , k ) {
3595
+ const st = targetsOf(doc)[i]?.steps?.[k];
3596
+ return isObj(st) ? st : undefined;
3597
+ }
3598
+
3599
+ /** The registry entry of target [i]'s step in job [k]: what it asks, and of whom. */
3600
+ const targetEntryOf = (doc , i , k ) =>
3601
+ STEP_TYPES[targetStepOf(doc, i, k)?.type ?? ""];
3602
+
3603
+ /** What job [k]'s targets send, as the job's editor shows it: the first
3604
+ target's step type there, or Prompt's while there is none. */
3605
+ function jobTargetType(doc , k ) {
3606
+ return targetStepOf(doc, 0, k)?.type ?? "prompt";
3607
+ }
3608
+
3609
+ /** The Setup profile target [i] asks in job [k]: its step's own, or the target's. */
3610
+ function targetProfileOf(doc , i , k ) {
3611
+ return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
3612
+ }
3613
+
3614
+ /** [doc] with target [i]'s step in job [k] changed: [change] merged into
3615
+ it, or the step [change] makes of it. */
3616
+ function withTarget (doc , i , k ,
3617
+ change ) {
3618
+ return { ...doc, targets: doc.targets.map((t, x) => (x !== i ? t : {
3619
+ ...t, steps: t.steps.map((st, j) => (j !== k ? st : typeof change === "function" ? change(st) : { ...st, ...change } )),
3620
+ })) };
3621
+ }
3622
+
3623
+ /** The token mappings [job]'s words are resolved under. */
3624
+ const tokensOf = (job ) => stepOf (job, "tokenMappings")?.tokenMappings ?? [];
3625
+
3626
+ /** [job] resolving its words under [list]: none takes the step away. */
3627
+ const withTokens = (job , list ) =>
3628
+ withStep(job, "tokenMappings", list.length ? { type: "tokenMappings", tokenMappings: list } : null);
3629
+
3630
+ /** Whether [job] sends the item's image beside its words. */
3631
+ const sendsImage = (job ) => !!stepOf(job, "attachImage");
3632
+
3633
+ /** [job] sending the item's image, or not. */
3634
+ const withImage = (job , on ) => withStep(job, "attachImage", on ? { type: "attachImage" } : null);
3635
+
3636
+ /** The Power Automate step [job]'s records are calls of: its name, which the
3637
+ flow's expressions read it by, the loop item() means, and the HTTP API
3638
+ its replies speak. None for a job that is not one. */
3639
+ function flowStepOf(job ) {
3640
+ const f = stepOf (job, "flowStep");
3641
+ return f ? { step: f.step, loop: f.loop ?? null, api: f.api } : null;
3407
3642
  }
3408
3643
 
3409
- /** A job's call: the step that asks something and is answered. */
3410
- const callOf = (job ) =>
3411
- stepIn (job, "call") ;
3644
+ /** What the flow does to [job]'s reply before it reads it, as an expression
3645
+ over the step's body; null reads the reply's own text. */
3646
+ const readAsOf = (job ) => stepOf (job, "readAs")?.readAs ?? null;
3412
3647
 
3413
- /** The registry entry of a job's call: what it asks, and of whom. */
3414
- const callEntryOf = (job ) =>
3415
- STEP_TYPES[(stepIn(job, "call") )?.type ?? ""];
3648
+ /** What a transport builds target [i]'s request in job [k] from: the
3649
+ target's step, with the flow step and reading the job holds. */
3650
+ function requestStepOf(doc , i , k ) {
3651
+ const job = doc.jobs[k];
3652
+ return { ...flowStepOf(job), ...targetStepOf(doc, i, k), readAs: readAsOf(job) } ;
3653
+ }
3654
+
3655
+ /** What httpRequestOf builds a request from: a target's template, and the
3656
+ loop its flow step sits in. */
3657
+
3658
+ /** What httpReplyOf reads a reply with: the flow step, its API, its reading. */
3659
+
3416
3660
 
3417
3661
  /** The request an HTTP Request [step] sends for [cell] over an item whose
3418
3662
  record read [ctx]'s scope: the cell's words and fields placed in the
3419
3663
  flow's body by its HTTP API, then every expression evaluated. */
3420
- function httpRequestOf(step , cell , ctx )
3664
+ function httpRequestOf(step , cell , ctx )
3421
3665
  {
3422
3666
  const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
3423
- const at = { ...ctx, loop: step.loop };
3667
+ const at = { ...ctx, loop: step.loop ?? null };
3424
3668
  const path = String(evaluate(step.path, at));
3425
3669
  // A path an expression wrote whole -- an address -- keeps its path and query.
3426
3670
  const [p, q] = /^https?:\/\//i.test(path) ? (() => { const u = new URL(path); return [u.pathname, u.search]; })() : [path, ""];
@@ -3431,35 +3675,50 @@ function httpRequestOf(step , cell , ctx
3431
3675
  /** What the flow reads of a reply [j] to [step]: its readAs evaluated with the
3432
3676
  reply as the step's body -- the prefill put back in front, as the flow has
3433
3677
  to -- or the HTTP API's own text. */
3434
- function httpReplyOf(step , cell , j , status ,
3678
+ function httpReplyOf(step , cell , j , status ,
3435
3679
  ctx ) {
3436
3680
  const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
3437
3681
  const whole = cell.prefill && api.withPrefill ? api.withPrefill(j, cell.prefill) : j;
3438
3682
  const { raw: said, finishReason } = api.reply(whole);
3439
3683
  if (!step.readAs) return { raw: said, said, finishReason };
3440
3684
  const scope = { ...ctx.scope, actions: { ...(ctx.scope.actions || {}), [step.step]: { body: whole, statusCode: status } } };
3441
- return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop })), said, finishReason };
3685
+ return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop ?? null })), said, finishReason };
3442
3686
  }
3443
3687
 
3444
- /** A job's reply step. */
3445
- const replyOf = (job ) =>
3446
- stepIn (job, "reply") ;
3447
-
3448
- /** How a job's reply is read: its kind, settings and modifiers. */
3449
- const outOf = (job ) => replyOf(job)?.out;
3688
+ /** A reply [text] from a model asked in words, read as the flow reads its
3689
+ step's reply: put in the body the flow's API would have sent it in
3690
+ (`HttpApiEntry.wrap`), then read through the job's Read as -- so a
3691
+ model's words and production's are read alike. */
3692
+ function readFlowReply(flow , text , record ) {
3693
+ const api = HTTP_APIS[flow.api] ?? HTTP_APIS.raw ;
3694
+ const read = httpReplyOf({ ...flow, api: HTTP_APIS[flow.api] ? flow.api : "raw" }, { prompt: "" }, api.wrap(text), 200,
3695
+ { scope: record.scope || {}, now: record.at ?? null });
3696
+ return { raw: read.raw, said: text };
3697
+ }
3450
3698
 
3451
- /** [job] with the step in [slot] changed by [change]. */
3452
- function withSlot(job , slot , change ) {
3453
- return { ...job, steps: job.steps.map(st => (slotOf(st) === slot ? change(st) : st)) };
3699
+ /** A job's Response Format Validation, when it has one. */
3700
+ const replyOf = (job ) => stepOf (job, "readReply");
3701
+
3702
+ /** How a job's reply is read: its kind, settings and modifiers. A job with
3703
+ no Format Validation reads its reply as text -- the text kind, by name,
3704
+ so a stored document never changes meaning when a plugin registers a
3705
+ kind of its own. */
3706
+ function outOf(job ) {
3707
+ const read = replyOf(job)?.out ?? { kind: "text", ...(OUTPUT_KINDS.text?.settingDefaults?.() || {}) };
3708
+ const modifiers = (job?.steps || []).filter((st) => isObj(st) && st.type === "modifier").map(st => st.modifier);
3709
+ return { ...read, modifiers } ;
3454
3710
  }
3455
3711
 
3456
- /** [job] reading its reply as [out]. */
3457
- const withOut = (job , out ) => withSlot(job, "reply", st => ({ ...st, out } ));
3712
+ /** [job] reading its reply as [out]: its Format Validation and modifiers. */
3713
+ function withOut(job , out ) {
3714
+ const { modifiers, ...read } = out;
3715
+ const steps = job.steps.filter(st => st.type !== "readReply" && st.type !== "modifier");
3716
+ return { ...job, steps: ordered([...steps, { type: "readReply", out: read },
3717
+ ...(modifiers || []).map(modifier => ({ type: "modifier", modifier }) )]) };
3718
+ }
3458
3719
 
3459
- /** [job] with its call changed by [patch]. */
3460
- const withCall = (job , patch ) => withSlot(job, "call", st => ({ ...st, ...patch } ));
3461
3720
  /** A new id: random, so one minted in one browser never collides with one
3462
- minted in another. A pipeline, its jobs and its scenarios each carry one
3721
+ minted in another. A pipeline, its jobs and its targets each carry one
3463
3722
  (#93), minted once and never shown. */
3464
3723
  function newId() {
3465
3724
  const bytes = new Uint8Array(6);
@@ -3470,104 +3729,99 @@ function newId() {
3470
3729
  /** A new job: the kind a stage answers in, that kind's own modifiers, and
3471
3730
  a copy of the token set. */
3472
3731
  function jobDefaults(kind = defaultOutputKind, tokens = TOKEN_DEFAULTS) {
3473
- return {
3474
- type: "job", id: newId(), name: "",
3475
- steps: [
3476
- { type: "prompt", withImage: false, tokenMappings: clone(tokens) },
3477
- { type: "readReply",
3478
- out: { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] } },
3479
- ],
3480
- };
3732
+ const job = withTokens({ type: "job", id: newId(), name: "", steps: [] }, clone(tokens));
3733
+ return withOut(job, { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] });
3481
3734
  }
3482
3735
  STEP_TYPES.job = {
3483
3736
  in: "item", out: job => outOf(job )?.kind,
3484
3737
  apply: "runPipeline",
3485
3738
  fields: ["type", "id", "name", "steps"],
3486
3739
  defaults: jobDefaults,
3487
- // A job's steps sit in fixed slots: Attach Content first, on job 1 only;
3488
- // one Call; Read Reply last. Each step is its entry's to check.
3740
+ // A job's steps run in stage order: Content, then Responses, each in its
3741
+ // own order. A target's step is the target's, never a job's. Each step is
3742
+ // its entry's to check.
3489
3743
  validate(c, ctx, bad, k, doc){
3490
3744
  const at = jobLabel(doc, k);
3491
3745
  if (!isObj(c) || c.type !== "job") return void bad.push(`${at} has to be a job step`);
3492
3746
  onlyFields(c, at, STEP_TYPES.job .fields , bad);
3493
- // An id is a job's own, like a scenario's: stable across renames, so a
3747
+ // An id is a job's own, like a target's: stable across renames, so a
3494
3748
  // stored run keeps pointing at the job that actually ran (#93).
3495
3749
  if (!isStr(c.id) || !c.id.trim()) bad.push(`${at} has no id`);
3496
3750
  else if ((doc.jobs ).findIndex((o) => isObj(o) && o.id === c.id) !== k) bad.push(`${at} has the id of another job`);
3497
3751
  if (c.name != null && !isStr(c.name)) bad.push(`${at}: name has to be text`);
3498
3752
  if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
3499
3753
  const steps = c.steps;
3500
- // A job's steps are the entries with a slot; one without (a job, the
3501
- // tests) or of no type the lab has is not one.
3754
+ // A job's steps are the entries with a stage; one without (a job, the
3755
+ // evals) or of no type the lab has is not one.
3502
3756
  const unknown = steps.find(st => !slotOf(st));
3503
3757
  if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
3504
- const slots = steps.map(st => slotOf(st) );
3505
- const want = [...(k === 0 && slots[0] === "content" ? ["content" ] : []), "call", "reply"];
3506
- if (slots.join() !== want.join()) {
3507
- // One sentence for the one mistake, the likeliest first.
3508
- const label = (slot ) => slotEntry(slot)?.entry.label ?? slot;
3509
- bad.push(k > 0 && slots.includes("content")
3510
- ? `${at}: only job 1 attaches content -- a later job is handed the reply before it`
3511
- : slots.filter(x => x === "call").length !== 1 ? `${at} has to have one call step`
3512
- : `${at}: steps are ${label("content")} (job 1), one call, then ${label("reply")}`);
3513
- return;
3758
+ if (steps.some(st => slotOf(st) === "target")) {
3759
+ return void bad.push(`${at}: what a target sends is the target's own step, not the job's`);
3760
+ }
3761
+ const twice = steps.find((st, x) => !STEP_TYPES[st.type] .many && steps.findIndex(o => o.type === st.type) !== x);
3762
+ if (twice) return void bad.push(`${at} has two ${STEP_TYPES[twice.type] .label ?? twice.type} steps`);
3763
+ if (steps.some((st, x) => x > 0 && placeOf(st) < placeOf(steps[x - 1]))) {
3764
+ return void bad.push(`${at}: steps are its Content, then its Responses, each in order`);
3514
3765
  }
3766
+ const kind = outOf(c ).kind;
3515
3767
  for (const st of steps) {
3516
3768
  const entry = STEP_TYPES[st.type] ;
3769
+ const label = `${at}'s ${entry.label ?? st.type}`;
3517
3770
  if (k > 0 && entry.firstJobOnly) { bad.push(`${at}: ${entry.firstJobOnly}`); continue; }
3518
- onlyFields(st, `${at}'s ${entry.label ?? st.type}`, entry.fields ?? [], bad);
3519
- entry.validate(st, ctx, bad, at);
3771
+ onlyFields(st, label, entry.fields ?? [], bad);
3772
+ entry.validate(st, ctx, bad, label, kind);
3520
3773
  }
3521
3774
  },
3522
3775
  };
3523
- // What a test carries besides its type's own fields.
3524
- const TEST_FIELDS = ["id", "name", "continueOnFailure"];
3776
+ // What an eval carries besides its type's own fields.
3777
+ const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
3525
3778
 
3526
- /** Test [j]'s name, or the number it has always shown. */
3527
- const testLabel = (doc , j ) =>
3528
- (doc.tests?.[j]?.name || "").trim() || `Test ${j + 1}`;
3779
+ /** Eval [j]'s name, or the number it has always shown. */
3780
+ const evalLabel = (doc , j ) =>
3781
+ (doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
3529
3782
 
3530
- /** A test type that settles over the whole run rather than item by item. */
3783
+ /** An eval type that settles over the whole run rather than item by item. */
3531
3784
  const isWholeRun = (type , t ) =>
3532
3785
  type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
3533
3786
 
3534
3787
  /**
3535
- * A document's tests as a list, whatever version wrote it: a queue row keeps
3536
- * the run document it was submitted with, so a run from before version 6
3537
- * holds one test or null, read here as the upgrade reads it (testsList).
3788
+ * A document's evals as a list, whatever version wrote it: a queue row keeps
3789
+ * the run document it was submitted with, so a run from before version 11
3790
+ * spells them `tests`, and one from before version 6 holds one or null, read
3791
+ * here as the upgrade reads it (testsList, evalsKey).
3538
3792
  */
3539
- function testsOf(doc ) {
3540
- const t = doc?.tests;
3793
+ function evalsOf(doc ) {
3794
+ const t = doc?.evals ?? doc?.tests;
3541
3795
  if (Array.isArray(t)) return t ;
3542
3796
  return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3543
3797
  }
3544
3798
 
3545
- /** The dataset a document's tests grade against, where one does: a run
3799
+ /** The dataset a document's evals grade against, where one does: a run
3546
3800
  grades against one (validatePipeline says so), so the first names it. */
3547
- function testsDataset(doc ) {
3548
- const t = testsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3801
+ function evalsDataset(doc ) {
3802
+ const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3549
3803
  return t && "dataset" in t ? t.dataset : null;
3550
3804
  }
3551
3805
 
3552
- STEP_TYPES.tests = {
3806
+ STEP_TYPES.evals = {
3553
3807
  in: "results", out: "verdict",
3554
3808
  apply: "score",
3555
- validate(tests, ctx, bad, lastKind, doc){
3556
- if (!Array.isArray(tests)) return void bad.push("tests has to be a list, empty for an unscored run");
3809
+ validate(evals, ctx, bad, lastKind, doc){
3810
+ if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
3557
3811
  const seen = new Set ();
3558
- tests.forEach((t , j ) => {
3559
- const at = testLabel(doc, j);
3560
- const type = isObj(t) && TEST_TYPES[t.type];
3812
+ evals.forEach((t , j ) => {
3813
+ const at = evalLabel(doc, j);
3814
+ const type = isObj(t) && EVAL_TYPES[t.type];
3561
3815
  if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
3562
3816
  const before = bad.length;
3563
3817
  if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
3564
- else if (seen.has(t.id)) bad.push(`${at} has the id of another test`);
3818
+ else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
3565
3819
  else seen.add(t.id);
3566
3820
  if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
3567
3821
  if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
3568
- onlyFields(t, at, [...type.fields, ...TEST_FIELDS], bad);
3822
+ onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
3569
3823
  type.validate(t, ctx, bad);
3570
- // Which kinds a test scores is only worth saying of a test that is whole.
3824
+ // Which kinds an eval scores is only worth saying of an eval that is whole.
3571
3825
  if (bad.length > before) return;
3572
3826
  const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
3573
3827
  if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
@@ -3576,14 +3830,14 @@ STEP_TYPES.tests = {
3576
3830
  }
3577
3831
  });
3578
3832
  // A run is handed one dataset's body to grade against (server-side-runs §4).
3579
- const named = new Set(tests.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3580
- if (named.size > 1) bad.push("the tests grade against one dataset at a time");
3833
+ const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3834
+ if (named.size > 1) bad.push("the evals grade against one dataset at a time");
3581
3835
  },
3582
3836
  };
3583
3837
 
3584
3838
  // ---- the document -------------------------------------------------------------
3585
3839
 
3586
- const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "scenarios", "tests"];
3840
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
3587
3841
  // What resolving adds, and nothing else: the profiles it resolved to and the
3588
3842
  // run's own comment, which belongs to the run and never to the pipeline.
3589
3843
  const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
@@ -3600,9 +3854,13 @@ function versionProblem(doc ) {
3600
3854
  /** What an upgrade from version 2 needs from outside the document: the rules
3601
3855
  of the dataset a pipeline was graded against, which the retired
3602
3856
  version-2 list kind read every reply under. Without them a job is upgraded with no
3603
- rules, as a run with no graded test parsed. */
3857
+ rules, as a run with no graded eval parsed. */
3604
3858
 
3605
3859
 
3860
+
3861
+
3862
+
3863
+
3606
3864
 
3607
3865
 
3608
3866
  /**
@@ -3621,7 +3879,8 @@ function versionProblem(doc ) {
3621
3879
  * as it was, for versionProblem to name. A copy: the caller's document is not
3622
3880
  * touched. From version 5, its test -- or none -- becomes a list of one (or
3623
3881
  * none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
3624
- * renames and every job's `type` becomes `"job"`.
3882
+ * renames and every job's `type` becomes `"job"`. From version 10, its
3883
+ * `tests` are `evals` (evalsKey).
3625
3884
  */
3626
3885
  /** Every id a current document must carry, minted for the ones [doc] lacks.
3627
3886
  An id it already has is kept. */
@@ -3637,25 +3896,33 @@ function upgradeIds(doc ) {
3637
3896
  return doc;
3638
3897
  }
3639
3898
 
3899
+ /** A document's columns and their steps, wherever its version keeps them:
3900
+ version 10's targets, or an earlier one's scenarios and their cells. */
3901
+ function columnsOf(doc ) {
3902
+ const list = Array.isArray(doc?.targets) ? doc.targets : Array.isArray(doc?.scenarios) ? doc.scenarios : [];
3903
+ return list.filter(isObj).map((column ) => ({
3904
+ column, steps: (Array.isArray(column.steps) ? column.steps : Array.isArray(column.stages) ? column.stages : []).filter(isObj),
3905
+ }));
3906
+ }
3907
+
3640
3908
  /** Every profile reference cut to { id, name }. The Runs tab once kept the
3641
3909
  whole Setup profile a scenario picked -- key and all -- so a document saved
3642
3910
  then is cleaned as it is read, and validation refuses one that is not. */
3643
3911
  function profileRefs(doc ) {
3644
3912
  const cut = (r ) => (isObj(r) && isStr(r.id) ? { id: r.id, name: isStr(r.name) ? r.name : "" } : r);
3645
- for (const sc of Array.isArray(doc.scenarios) ? doc.scenarios : []) {
3646
- if (!isObj(sc)) continue;
3647
- if ("profile" in sc) sc.profile = cut(sc.profile);
3648
- for (const cell of Array.isArray(sc.stages) ? sc.stages : []) {
3649
- if (isObj(cell) && cell.profile != null) cell.profile = cut(cell.profile);
3650
- }
3913
+ for (const { column, steps } of columnsOf(doc)) {
3914
+ if ("profile" in column) column.profile = cut(column.profile);
3915
+ for (const st of steps) if (st.profile != null) st.profile = cut(st.profile);
3651
3916
  }
3652
3917
  }
3653
3918
 
3919
+ /** [doc] with its profile references cut (profileRefs), for chaining. */
3920
+ const cutRefs = (doc ) => { profileRefs(doc); return doc; };
3921
+
3654
3922
  /** Whether any profile reference in a pipeline holds more than { id, name }. */
3655
3923
  function fatProfileRef(doc ) {
3656
3924
  const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
3657
- return isObj(doc) && Array.isArray(doc.scenarios) && doc.scenarios.some((sc ) =>
3658
- isObj(sc) && (fat(sc.profile) || (Array.isArray(sc.stages) && sc.stages.some((c ) => isObj(c) && fat(c.profile)))));
3925
+ return isObj(doc) && columnsOf(doc).some(({ column, steps }) => fat(column.profile) || steps.some(st => fat(st.profile)));
3659
3926
  }
3660
3927
 
3661
3928
  /** Version 5 to 6: the one test, or none, as a list. Its id is `t1`, so
@@ -3670,9 +3937,9 @@ function testsList(doc ) {
3670
3937
  return doc;
3671
3938
  }
3672
3939
 
3673
- /** A new test of [type] at the lab's defaults, named [name], continuing on failure. */
3674
- function newTest(type , name = "", fields = {}) {
3675
- const own = TEST_TYPES[type]?.defaults?.() ?? { type };
3940
+ /** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
3941
+ function newEval(type , name = "", fields = {}) {
3942
+ const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
3676
3943
  return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
3677
3944
  }
3678
3945
 
@@ -3723,22 +3990,148 @@ function metricsOf(next ) {
3723
3990
  return next;
3724
3991
  }
3725
3992
 
3993
+ /** Version 9 to 10: a job's stages, and each scenario a target
3994
+ (pipeline-model §16). A job's call becomes what it held for the whole
3995
+ job -- a Prompt's image flag and token mappings, an HTTP Request's flow
3996
+ step and reading -- and each scenario's cell in it, with the call's
3997
+ template where it has one, that target's own step there. Read Reply's
3998
+ modifiers are steps of their own after it. What each held is moved as it
3999
+ was, so the run sends the same requests and reads them the same way
4000
+ (fixtures/stages-v9.json, held to it by pipeline-check.js). */
4001
+ function targetsFromScenarios(next ) {
4002
+ next.version = PIPELINE_VERSION;
4003
+ const jobs = Array.isArray(next.jobs) ? next.jobs : [];
4004
+ const calls = jobs.map(job => (isObj(job) && Array.isArray(job.steps)
4005
+ ? job.steps.find((st ) => isObj(st) && (st.type === "prompt" || st.type === "httpRequest")) : undefined));
4006
+ jobs.forEach((job, k) => {
4007
+ if (!isObj(job) || !Array.isArray(job.steps)) return;
4008
+ job.steps = job.steps.flatMap((st ) => {
4009
+ if (!isObj(st)) return [st];
4010
+ if (st === calls[k] && st.type === "prompt") {
4011
+ return [...(st.withImage ? [{ type: "attachImage" }] : []),
4012
+ ...(Array.isArray(st.tokenMappings) && st.tokenMappings.length ? [{ type: "tokenMappings", tokenMappings: st.tokenMappings }] : [])];
4013
+ }
4014
+ if (st === calls[k]) {
4015
+ return [{ type: "flowStep", step: st.step, loop: st.loop ?? null, api: st.api },
4016
+ ...(st.readAs != null ? [{ type: "readAs", readAs: st.readAs }] : [])];
4017
+ }
4018
+ if (st.type === "readReply" && isObj(st.out)) {
4019
+ const { modifiers, ...out } = st.out;
4020
+ return [{ type: "readReply", out },
4021
+ ...(Array.isArray(modifiers) ? modifiers : []).map((modifier ) => ({ type: "modifier", modifier }))];
4022
+ }
4023
+ return [st];
4024
+ });
4025
+ });
4026
+ const targets = (Array.isArray(next.scenarios) ? next.scenarios : []).map((sc ) => {
4027
+ if (!isObj(sc)) return sc;
4028
+ const { stages, ...rest } = sc;
4029
+ return { ...rest, steps: (Array.isArray(stages) ? stages : []).map((cell , k ) => {
4030
+ const call = calls[k];
4031
+ if (!isObj(cell) || !isObj(call)) return cell;
4032
+ const { step: _step, loop: _loop, readAs: _readAs, withImage: _image, tokenMappings: _tokens, ...shape } = call;
4033
+ return { ...clone(shape), ...cell };
4034
+ }) };
4035
+ });
4036
+ // In the scenarios' place, so the document reads in the same order.
4037
+ return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
4038
+ }
4039
+
4040
+ /** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
4041
+ upgraded document reads the same, key for key, and each eval is as it
4042
+ was. Every version before 11 passes through here. */
4043
+ function evalsKey(next ) {
4044
+ next.version = PIPELINE_VERSION;
4045
+ // `tests` is the spelling of every version before 11.
4046
+ if (!("tests" in next)) return next;
4047
+ return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
4048
+ .map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
4049
+ }
4050
+
4051
+ /** Version 11 to 12: a Contains metric's Ignore case holds where the reply
4052
+ is matched item by item, as it does where it is matched as text, and it
4053
+ is kept as written. Until version 11's last hours (#199) every Contains
4054
+ metric matched the reply's text and minded its Ignore case, so what a
4055
+ stored metric says is what its author meant; only the item-by-item
4056
+ matching #199 added, case-blind for a few hours, read it otherwise.
4057
+ Every version before 12 ends here. */
4058
+ function caseAsWritten(next ) {
4059
+ next.version = PIPELINE_VERSION;
4060
+ return next;
4061
+ }
4062
+
3726
4063
  function upgradePipeline (doc , ctx = {}) {
3727
4064
  // A current document is read as it is, but for a profile reference the Runs
3728
- // tab saved whole (see profileRefs), which is cut back.
4065
+ // tab saved whole (see profileRefs), which is cut back, and a step on a
4066
+ // profile a target step stands in for (localSteps).
3729
4067
  if (isObj(doc) && doc.version === PIPELINE_VERSION) {
3730
- if (!fatProfileRef(doc)) return doc;
3731
- const cleaned = clone(doc) ;
3732
- profileRefs(cleaned);
3733
- return cleaned ;
4068
+ let out = doc ;
4069
+ if (fatProfileRef(out)) {
4070
+ out = clone(out) ;
4071
+ profileRefs(out);
4072
+ }
4073
+ return localSteps(withoutCaseMetric(out), ctx) ;
3734
4074
  }
3735
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8].includes(doc.version )) return doc;
3736
- let next = clone(doc) ;
3737
- if (next.version === 8) return metricsOf(next) ;
3738
- if (next.version === 7) return metricsOf(stepsOf(testsList(next))) ;
3739
- if (next.version === 6) return metricsOf(stepsOf(jobsOf(next))) ;
3740
- if (next.version === 5) return metricsOf(stepsOf(jobsOf(testsList(next)))) ;
3741
- if (next.version === 3 || next.version === 4) return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next))))) ;
4075
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
4076
+ if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
4077
+ // Every version before 10 reads as version 9 first, then as 10, then as 11
4078
+ // and 12.
4079
+ // A version-10 document is cut as a current one was (profileRefs); the
4080
+ // earlier ones are cut on their way through nineOf.
4081
+ const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4082
+ return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
4083
+ }
4084
+
4085
+ /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
4086
+ are metrics now (dataset body version 5), which an eval naming the
4087
+ dataset adds for each item, so the case's own metrics carry what that
4088
+ metric scored. Read so at every version, the current one included, as a
4089
+ document saved before it went still holds it. The same document where
4090
+ none does. */
4091
+ function withoutCaseMetric(doc ) {
4092
+ const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
4093
+ if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
4094
+ return { ...doc, evals: doc.evals.map((t ) => (has(t)
4095
+ ? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
4096
+ }
4097
+
4098
+ /** [doc] with each step asked of a profile whose type a target step stands
4099
+ in for (`replaces`: Echo for an Echo profile) read as that step -- its
4100
+ words kept, and no profile, since it asks none -- once the profiles'
4101
+ types are known: [ctx]'s, or a run document's own. A target whose every
4102
+ step then asks none, or names its own, keeps no profile either. The same
4103
+ document, untouched, when there is nothing to read so. */
4104
+ function localSteps(doc , ctx ) {
4105
+ const typeOfProfile = ctx.profileType
4106
+ ?? (isObj(doc.profiles) ? (id ) => (isObj(doc.profiles[id]) ? doc.profiles[id].type : undefined) : null);
4107
+ if (!typeOfProfile) return doc;
4108
+ const stands = new Map(Object.entries(STEP_TYPES).filter(([, e]) => e.slot === "target" && e.replaces).map(([type, e]) => [e.replaces , type]));
4109
+ const as = (ref ) => (isObj(ref) && isStr(ref.id) ? stands.get(typeOfProfile(ref.id) ?? "") : undefined);
4110
+ const turns = (t , st ) => isObj(st) && STEP_TYPES[st.type]?.asks === "prompt" && as(st.profile ?? t.profile);
4111
+ const targets = Array.isArray(doc.targets) ? doc.targets : [];
4112
+ if (!targets.some(t => isObj(t) && Array.isArray(t.steps) && t.steps.some((st ) => turns(t, st)))) return doc;
4113
+ const next = clone(doc) ;
4114
+ for (const t of next.targets ) {
4115
+ if (!isObj(t) || !Array.isArray(t.steps)) continue;
4116
+ t.steps = t.steps.map((st ) => {
4117
+ const type = turns(t, st);
4118
+ return type ? { type, prompt: st.prompt, ...(st.from ? { from: st.from } : {}) } : st;
4119
+ });
4120
+ // A run keeps the profile it ran as, so its report names it as it did.
4121
+ if (!isObj(doc.profiles) && as(t.profile)
4122
+ && t.steps.every((st ) => isObj(st) && (STEP_TYPES[st.type]?.local || st.profile != null))) t.profile = null;
4123
+ }
4124
+ return next;
4125
+ }
4126
+
4127
+ /** [next] as version 9 held it, from any version before 10. */
4128
+ function nineOf(next , ctx ) {
4129
+ if (next.version === 9) { profileRefs(next); return next; }
4130
+ if (next.version === 8) return metricsOf(next);
4131
+ if (next.version === 7) return metricsOf(stepsOf(testsList(next)));
4132
+ if (next.version === 6) return metricsOf(stepsOf(jobsOf(next)));
4133
+ if (next.version === 5) return metricsOf(stepsOf(jobsOf(testsList(next))));
4134
+ if (next.version === 3 || next.version === 4) return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
3742
4135
  if (next.version === 1) {
3743
4136
  next.version = 2;
3744
4137
  if (Array.isArray(next.chains)) {
@@ -3764,7 +4157,7 @@ function upgradePipeline (doc , ctx = {}) {
3764
4157
  if (isObj(cell)) cell.prompt = renamed(cell.prompt);
3765
4158
  }
3766
4159
  }
3767
- return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next))))) ;
4160
+ return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
3768
4161
  }
3769
4162
 
3770
4163
  /** Version 1's two maps as a list of mappings. */
@@ -3781,9 +4174,9 @@ function tokenMappingsFromV1(set ) {
3781
4174
  the item's image, as a job over a Source would; one over text content
3782
4175
  has none to send, and the page turns it off with the content. */
3783
4176
  function blankPipeline(opts = {}) {
3784
- const job = withCall(jobDefaults(opts.kind, opts.tokens), { withImage: true });
4177
+ const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
3785
4178
  return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
3786
- jobs: [job], scenarios: [], tests: [] };
4179
+ jobs: [job], targets: [], evals: [] };
3787
4180
  }
3788
4181
 
3789
4182
  /**
@@ -3808,37 +4201,50 @@ function validatePipeline(input , ctx = {}) {
3808
4201
  if (!Array.isArray(doc.jobs) || !doc.jobs.length) {
3809
4202
  return [...bad, "jobs has to be a list of at least one job"];
3810
4203
  }
3811
- if (!Array.isArray(doc.scenarios) || !doc.scenarios.length) bad.push("add a scenario first");
4204
+ if (!targetsOf(doc).length) bad.push("add a target first");
3812
4205
  // Content is job 1's Attach Content step; a job 1 without one has none yet.
3813
- if (!stepIn(doc.jobs[0], "content")) bad.push("set the content first");
4206
+ if (!contentOf(doc)) bad.push("set the content first");
3814
4207
  if (run) profilesProblems(doc.profiles, bad);
3815
4208
  doc.jobs.forEach((ch , k ) => STEP_TYPES.job .validate(ch, c, bad, k, doc));
3816
4209
  if (bad.length) return bad;
3817
4210
  const content = contentOf(doc) ;
3818
4211
 
3819
- const scenarios = doc.scenarios;
4212
+ const targets = targetsOf(doc);
3820
4213
  const n = doc.jobs.length;
3821
- scenarios.forEach((sc, i) => {
3822
- const at = scenarioLabel(doc, i);
3823
- if (!isObj(sc)) return void bad.push(`${at} has to be an object`);
3824
- onlyFields(sc, at, ["id", "name", "profile", "stages"], bad);
3825
- if (!isStr(sc.id) || !sc.id.trim()) bad.push(`${at} has no id`);
3826
- else if (scenarios.findIndex((o) => isObj(o) && o.id === sc.id) !== i) bad.push(`${at} has the id of another scenario`);
3827
- if (sc.name != null && !isStr(sc.name)) bad.push(`${at}: name has to be text`);
3828
- if (!isRef(sc.profile)) bad.push(`${at} has to name its Setup profile as { id, name }`);
3829
- else onlyFields(sc.profile, `${at}'s profile`, ["id", "name"], bad);
3830
- if (!Array.isArray(sc.stages) || sc.stages.length !== n) {
3831
- bad.push(`${at} has ${Array.isArray(sc.stages) ? sc.stages.length : "no"} prompts for ${n} job${n === 1 ? "" : "s"}`);
4214
+ targets.forEach((t, i) => {
4215
+ const at = targetLabel(doc, i);
4216
+ if (!isObj(t)) return void bad.push(`${at} has to be an object`);
4217
+ onlyFields(t, at, ["id", "name", "profile", "steps"], bad);
4218
+ if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
4219
+ else if (targets.findIndex((o) => isObj(o) && o.id === t.id) !== i) bad.push(`${at} has the id of another target`);
4220
+ if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
4221
+ if (t.profile != null && !isRef(t.profile)) bad.push(`${at} has to name its Target profile as { id, name }`);
4222
+ else if (t.profile != null) onlyFields(t.profile, `${at}'s profile`, ["id", "name"], bad);
4223
+ if (!Array.isArray(t.steps) || t.steps.length !== n) {
4224
+ bad.push(`${at} has ${Array.isArray(t.steps) ? t.steps.length : "no"} prompts for ${n} job${n === 1 ? "" : "s"}`);
3832
4225
  return;
3833
4226
  }
3834
- sc.stages.forEach((cell , k ) => {
3835
- if (!isObj(cell)) return void bad.push(`${at}, ${jobLabel(doc, k)} has to be an object`);
3836
- const call = callEntryOf(doc.jobs[k]);
3837
- onlyFields(cell, `${at}, ${jobLabel(doc, k)}`, ["prompt", "profile", "from", ...(call?.cellFields || [])], bad);
3838
- call?.cellProblems?.(cell, callOf(doc.jobs[k]), `${at}, ${jobLabel(doc, k)}`, bad);
3839
- if (cell.profile != null && !isRef(cell.profile)) bad.push(`${at}, ${jobLabel(doc, k)} has to name its Setup profile as { id, name }`);
3840
- else if (cell.profile != null) onlyFields(cell.profile, `${at}, ${jobLabel(doc, k)}'s profile`, ["id", "name"], bad);
3841
- if (cell.from != null && !isRef(cell.from)) bad.push(`${at}, ${jobLabel(doc, k)} has to name the prompt it was picked from as { id, name }`);
4227
+ // A target with no profile of its own is one whose steps need none, or
4228
+ // each name theirs.
4229
+ if (t.profile == null && t.steps.some((st ) => !(isObj(st) && STEP_TYPES[st.type]?.local) && !(isObj(st) && st.profile != null))) {
4230
+ bad.push(`${at} has to name its Target profile as { id, name }`);
4231
+ }
4232
+ t.steps.forEach((st , k ) => {
4233
+ const where = `${at}, ${jobLabel(doc, k)}`;
4234
+ if (!isObj(st)) return void bad.push(`${where} has to be an object`);
4235
+ const entry = STEP_TYPES[st.type];
4236
+ if (entry?.slot !== "target") return void bad.push(`${where}: "${st.type}" is not something a target can send`);
4237
+ if (entry.local && st.profile != null) return void bad.push(`${where}: ${entry.label} asks no profile`);
4238
+ if (k > 0 && entry.firstJobOnly) return void bad.push(`${where}: ${entry.firstJobOnly}`);
4239
+ onlyFields(st, where, entry.fields ?? [], bad);
4240
+ entry.validate(st, c, bad, where);
4241
+ if (st.profile != null && !isRef(st.profile)) bad.push(`${where} has to name its Target profile as { id, name }`);
4242
+ else if (st.profile != null) onlyFields(st.profile, `${where}'s profile`, ["id", "name"], bad);
4243
+ if (st.from != null && !isRef(st.from)) bad.push(`${where} has to name the prompt it was picked from as { id, name }`);
4244
+ // A request is a flow step's, rebuilt from its records, and its
4245
+ // template has no place for an image.
4246
+ if (entry.asks === "request" && !flowStepOf(doc.jobs[k])) bad.push(`${where}: ${entry.label} needs the job's flow step`);
4247
+ if (entry.asks === "request" && sendsImage(doc.jobs[k])) bad.push(`${where}: ${entry.label} has no place for an image`);
3842
4248
  });
3843
4249
  });
3844
4250
  if (bad.length) return bad;
@@ -3849,68 +4255,76 @@ function validatePipeline(input , ctx = {}) {
3849
4255
  const p = c.profiles ? c.profiles(ref.id) : {};
3850
4256
  if (!p && !missing.has(ref.id)) {
3851
4257
  missing.add(ref.id);
3852
- bad.push(`Setup profile ${ref.name || ref.id} not found`);
4258
+ bad.push(`Target profile ${ref.name || ref.id} not found`);
3853
4259
  }
3854
4260
  return p;
3855
4261
  };
3856
4262
  // A type that answers from the item (Echo) has no model, answers only
3857
4263
  // job 1, and answers only a text item.
3858
4264
  const local = (p ) => !!p && typeof p === "object" && !!CONNECTION_TYPES[typeOf(p )]?.local;
3859
- // Every scenario needs a prompt in every job. A blank one is an error
4265
+ // Every target needs a prompt in every job. A blank one is an error
3860
4266
  // rather than a stage dropped, which would hand the job before it
3861
4267
  // straight to the job after it, under numbers nobody sees. The one it
3862
4268
  // may leave blank is job 1 answered from the item's own text (Echo over
3863
4269
  // a Source or Text): the reply is the item, and a prompt would only be
3864
4270
  // what it is read against. Over Prompt only the prompt is the reply.
4271
+ // What target [i] asks in job [k]: Echo's own connection for a step
4272
+ // answered here, else its profile as the lookups hold it.
4273
+ const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION
4274
+ : c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
3865
4275
  const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
3866
4276
  for (let k = 0; k < n; k++) {
3867
- const i = scenarios.findIndex(sc => (!isStr(sc.stages[k].prompt) || !sc.stages[k].prompt.trim())
3868
- && !(k === 0 && itemsHaveText && local(c.profiles ? c.profiles((sc.stages[0].profile || sc.profile).id) : null)));
3869
- if (i >= 0) bad.push(n > 1 ? `${jobLabel(doc, k)}, ${scenarioLabel(doc, i)} has no prompt`
3870
- : `${scenarioLabel(doc, i)} has no prompt`);
4277
+ const i = targets.findIndex((_, i) => {
4278
+ const words = targetStepOf(doc, i, k)?.prompt;
4279
+ return (!isStr(words) || !words.trim())
4280
+ && !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
4281
+ });
4282
+ if (i >= 0) bad.push(n > 1 ? `${jobLabel(doc, k)}, ${targetLabel(doc, i)} has no prompt`
4283
+ : `${targetLabel(doc, i)} has no prompt`);
3871
4284
  }
3872
4285
  const contentFiles = content?.type === "source"
3873
4286
  ? (c.sources?.(content.ref?.id)?.files ?? content.files ?? null) : null;
3874
4287
  const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
3875
4288
  .find(n => !/\.(?:txt|md|csv)$/i.test(n));
3876
- scenarios.forEach((sc, i) => {
3877
- const conns = sc.stages.map((cell ) => lookup(cell.profile || sc.profile));
4289
+ targets.forEach((_, i) => {
4290
+ const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION : lookup(targetProfileOf(doc, i, k) )));
3878
4291
  conns.forEach((p , k ) => {
3879
4292
  if (!local(p)) return;
3880
4293
  const label = CONNECTION_TYPES[typeOf(p )] .label;
3881
- if (k > 0) bad.push(`${jobLabel(doc, k)}, ${scenarioLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
3882
- else if (nonText) bad.push(`${scenarioLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
4294
+ if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
4295
+ else if (nonText) bad.push(`${targetLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
3883
4296
  });
3884
- // A connection answers what its job's call asks: words, or a whole request.
4297
+ // A connection answers what its target's step asks: words, or a whole request.
3885
4298
  conns.forEach((p , k ) => {
3886
- const call = callEntryOf(doc.jobs[k]);
3887
- // Only a profile that was looked up says what it answers.
3888
- if (!c.profiles || !p || !isObj(p) || !call) return;
4299
+ const step = targetEntryOf(doc, i, k);
4300
+ // Only a profile that was looked up says what it answers; a step
4301
+ // answered here asks none.
4302
+ if (!c.profiles || !p || !isObj(p) || !step || step.local) return;
3889
4303
  const type = CONNECTION_TYPES[typeOf(p )];
3890
- if (type && !(type.answers ?? ["prompt"]).includes(call.asks ?? "prompt")) {
3891
- bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${scenarioLabel(doc, i)}: ${call.label ?? "this call"} needs a profile that answers it, and ${type.label} does not`);
4304
+ if (type && !(type.answers ?? ["prompt"]).includes(step.asks ?? "prompt")) {
4305
+ bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${targetLabel(doc, i)}: ${step.label ?? "this step"} needs a profile that answers it, and ${type.label} does not`);
3892
4306
  }
3893
4307
  });
3894
- // The model is the profile's unless the call's cell carries it.
4308
+ // The model is the profile's unless the target's step carries it.
3895
4309
  const bare = conns.findIndex((p , k ) => p && c.profiles && !local(p)
3896
- && callEntryOf(doc.jobs[k])?.modelFrom !== "cell" && !String(p.model || "").trim());
4310
+ && targetEntryOf(doc, i, k)?.modelFrom !== "step" && !String(p.model || "").trim());
3897
4311
  if (bare >= 0) {
3898
4312
  bad.push(n > 1
3899
- ? `no model on the profile ${jobLabel(doc, bare)} of ${scenarioLabel(doc, i)} uses — manage profiles on the Setup tab`
3900
- : `no model on ${scenarioLabel(doc, i)}'s Setup profile — manage profiles on the Setup tab`);
4313
+ ? `no model on the profile ${jobLabel(doc, bare)} of ${targetLabel(doc, i)} uses — manage profiles on the Setup tab`
4314
+ : `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
3901
4315
  }
3902
4316
  });
3903
- STEP_TYPES.tests .validate(doc.tests, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4317
+ STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
3904
4318
  if (bad.length) return bad;
3905
4319
 
3906
4320
  // Asked before anything is sent, so a misspelt token costs nothing and
3907
4321
  // says which job, instead of failing every item the same way.
3908
4322
  const text = CONTENT_TYPES[content?.type]?.text?.(content, c) ?? null;
3909
- scenarios.forEach((sc, i) => {
3910
- const stages = doc.jobs.map((ch , k ) => ({ text: sc.stages[k].prompt, kind: outOf(ch).kind,
3911
- verbatim: !!callEntryOf(ch)?.verbatim }));
3912
- const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) => callOf(ch).tokenMappings), text);
3913
- if (problem) bad.push(`${scenarioLabel(doc, i)}: ${jobWords(problem)}`);
4323
+ targets.forEach((_, i) => {
4324
+ const stages = doc.jobs.map((ch , k ) => ({ text: targetStepOf(doc, i, k) .prompt, kind: outOf(ch).kind,
4325
+ verbatim: !!targetEntryOf(doc, i, k)?.verbatim }));
4326
+ const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) => tokensOf(ch)), text);
4327
+ if (problem) bad.push(`${targetLabel(doc, i)}: ${jobWords(problem)}`);
3914
4328
  });
3915
4329
  return bad;
3916
4330
  }
@@ -3990,16 +4404,16 @@ function connectionProblems(conn , at , from
3990
4404
  }
3991
4405
 
3992
4406
  /** The profile ids a pipeline references, scenarios first, in order. */
3993
- function profileIds(doc ) {
4407
+ function profileIds(doc ) {
3994
4408
  const ids = [];
3995
- for (const sc of doc.scenarios || []) {
3996
- for (const ref of [sc.profile, ...(sc.stages || []).map((c ) => c.profile)]) {
4409
+ for (const t of targetsOf(doc)) {
4410
+ for (const ref of [t.profile, ...(Array.isArray(t.steps) ? t.steps : []).map(st => st?.profile)] ) {
3997
4411
  if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
3998
4412
  }
3999
4413
  }
4000
- // Then the ones a test asks (a grader), so a run carries them too.
4001
- for (const t of testsOf(doc)) {
4002
- for (const ref of TEST_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4414
+ // Then the ones an eval asks (a grader), so a run carries them too.
4415
+ for (const t of evalsOf(doc)) {
4416
+ for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4003
4417
  }
4004
4418
  return ids;
4005
4419
  }
@@ -4012,9 +4426,9 @@ function profileIds(doc )
4012
4426
  function resolvePipeline(doc ,
4013
4427
  ctx = {}) {
4014
4428
  const run = clone(doc) ;
4015
- // What the lab supplies a test -- its grader -- before the profiles it
4429
+ // What the lab supplies an eval -- its grader -- before the profiles it
4016
4430
  // asks are carried.
4017
- run.tests = (Array.isArray(run.tests) ? run.tests : []).map((t ) => TEST_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4431
+ run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4018
4432
  run.profiles = {};
4019
4433
  for (const id of profileIds(run)) {
4020
4434
  const p = ctx.profiles?.(id);
@@ -4023,7 +4437,7 @@ function resolvePipeline(doc ,
4023
4437
  const content = contentOf(run) ;
4024
4438
  if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4025
4439
  if (ctx.datasetVersion) {
4026
- for (const t of run.tests || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4440
+ for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4027
4441
  }
4028
4442
  if (isStr(ctx.comment)) run.comment = ctx.comment;
4029
4443
  return run;
@@ -4037,7 +4451,7 @@ function pipelineOfRun(run , ctx = {}) {
4037
4451
  delete doc.plugins;
4038
4452
  const content = contentOf(doc) ;
4039
4453
  if (content) { delete content.files; delete content.revs; }
4040
- for (const t of doc.tests || []) if (isObj(t.dataset)) delete t.dataset.version;
4454
+ for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
4041
4455
  return doc;
4042
4456
  }
4043
4457
 
@@ -4076,7 +4490,7 @@ function importPipeline(input , ctx = {}) {
4076
4490
  const missing = [], seen = new Set ();
4077
4491
  const lists = { profile: profiles, source: sources, dataset: datasets };
4078
4492
  const words = {
4079
- profile: (ref ) => `Setup profile ${ref.name || ref.id} not found`,
4493
+ profile: (ref ) => `Target profile ${ref.name || ref.id} not found`,
4080
4494
  source: (ref ) => `Source ${ref.name || ref.id} not found`,
4081
4495
  dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
4082
4496
  };
@@ -4089,15 +4503,20 @@ function importPipeline(input , ctx = {}) {
4089
4503
  };
4090
4504
  const content = contentOf(next) ;
4091
4505
  if (content?.type === "source") content.ref = remap(content.ref, "source");
4092
- for (const sc of next.scenarios || []) {
4093
- sc.profile = remap(sc.profile, "profile");
4094
- for (const cell of sc.stages || []) {
4095
- if (cell.profile) cell.profile = remap(cell.profile, "profile");
4506
+ for (const t of targetsOf(next)) {
4507
+ if (t.profile) t.profile = remap(t.profile, "profile");
4508
+ for (const st of Array.isArray(t.steps) ? t.steps : []) {
4509
+ if (st?.profile) st.profile = remap(st.profile, "profile");
4096
4510
  }
4097
4511
  }
4098
- for (const t of Array.isArray(next.tests) ? next.tests : []) {
4512
+ for (const t of Array.isArray(next.evals) ? next.evals : []) {
4099
4513
  if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4100
4514
  }
4515
+ // Its references are this lab's now, so a step asking this lab's Echo
4516
+ // profile is read as an Echo step (upgradePipeline did it for ids that
4517
+ // already matched).
4518
+ const local = localSteps(next, ctx);
4519
+ if (local !== next) Object.assign(next, local);
4101
4520
  // A hand-written file needs no id of its own: mint the ones it lacks, and
4102
4521
  // keep the ones it carries, so export → import → export is the same
4103
4522
  // document and a re-import can recognise the same pipeline (#93).
@@ -4119,10 +4538,10 @@ function mintIds(doc ) {
4119
4538
  for (const ch of Array.isArray(doc.jobs) ? doc.jobs : []) {
4120
4539
  if (isObj(ch) && (!isStr(ch.id) || !ch.id)) ch.id = newId();
4121
4540
  }
4122
- for (const sc of Array.isArray(doc.scenarios) ? doc.scenarios : []) {
4123
- if (isObj(sc) && (!isStr(sc.id) || !sc.id)) sc.id = newId();
4541
+ for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
4542
+ if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4124
4543
  }
4125
- for (const t of Array.isArray(doc.tests) ? doc.tests : []) {
4544
+ for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4126
4545
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4127
4546
  }
4128
4547
  return doc;
@@ -4146,26 +4565,34 @@ function yamlToPipeline(text , ctx ) {
4146
4565
  */
4147
4566
  function stagesFor(run , i )
4148
4567
 
4149
- {
4150
- const sc = run.scenarios[i] ;
4568
+ {
4151
4569
  const stages = run.jobs.map((ch, k) => {
4152
4570
  const { kind, modifiers, ...settings } = outOf(ch);
4153
- return { text: sc.stages[k] .prompt, kind, withImage: !!callOf(ch).withImage,
4154
- verbatim: !!callEntryOf(ch)?.verbatim, modifiers: modifiers || [], settings };
4571
+ return { text: targetStepOf(run, i, k) .prompt, kind, withImage: sendsImage(ch),
4572
+ verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: modifiers || [], settings };
4155
4573
  });
4156
- const connections = run.jobs.map((ch, k) => {
4157
- const id = sc.stages[k] .profile?.id ?? sc.profile.id;
4574
+ const connections = run.jobs.map((_, k) => {
4575
+ // A step answered here asks nothing: the profile a run made before it
4576
+ // kept is what it ran as, else Echo itself.
4577
+ if (targetEntryOf(run, i, k)?.local) {
4578
+ const id = targetProfileOf(run, i, k)?.id;
4579
+ const held = id != null ? run.profiles?.[id] : undefined;
4580
+ return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...ECHO_CONNECTION };
4581
+ }
4582
+ const id = targetProfileOf(run, i, k) .id;
4158
4583
  return { id, ...(run.profiles?.[id] || {}) };
4159
4584
  });
4160
- // Each job's call step and this scenario's cell in it, for a transport
4161
- // that builds its own request from them (an HTTP Request).
4162
- return { stages, tokens: run.jobs.map(ch => callOf(ch).tokenMappings), connections,
4163
- calls: run.jobs.map(ch => stepIn(ch, "call") ), cells: sc.stages };
4585
+ // What a transport that builds its own request (an HTTP Request) builds
4586
+ // it from in each job, and this target's step there.
4587
+ return { stages, tokens: run.jobs.map(tokensOf), connections,
4588
+ calls: run.jobs.map((_, k) => requestStepOf(run, i, k)), cells: run.jobs.map((_, k) => targetStepOf(run, i, k) ) };
4164
4589
  }
4165
4590
 
4166
4591
  /** The Setup profile scenario [i] of a run ran under, as History shows it: keyless. */
4167
4592
  function scenarioProfile(run , i ) {
4168
- const id = run.scenarios?.[i]?.profile?.id;
4593
+ const id = targetsOf(run)[i]?.profile?.id;
4594
+ // A target that asks nothing ran as Echo.
4595
+ if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name: ECHO_CONNECTION.name, settings: { ...ECHO_CONNECTION } };
4169
4596
  const conn = id != null ? run.profiles?.[id] : null;
4170
4597
  return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
4171
4598
  }
@@ -4192,14 +4619,14 @@ function modifierSummary(m ) {
4192
4619
  return said.length ? said.join(", ") : "on";
4193
4620
  }
4194
4621
 
4195
- // ---- the tests, in order (docs/pipeline-model.md §3) ---------------------------
4196
- // Tests only read a run: none changes what a later one sees, and none stops
4622
+ // ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
4623
+ // Evals only read a run: none changes what a later one sees, and none stops
4197
4624
  // the model being sent the next item. What order changes is Continue on
4198
- // failure. A per-item test that fails and does not continue stops the tests
4625
+ // failure. A per-item eval that fails and does not continue stops the evals
4199
4626
  // after it for that item alone -- they read Skipped there, and a whole-run
4200
- // test after it pools the items it did not stop. A whole-run test settles
4627
+ // eval after it pools the items it did not stop. A whole-run eval settles
4201
4628
  // once every item is in, and one that fails then and does not continue
4202
- // leaves every test after it Skipped.
4629
+ // leaves every eval after it Skipped.
4203
4630
 
4204
4631
  const SKIPPED = Object.freeze({ skipped: true });
4205
4632
  const isSkipped = (s ) => isObj(s) && s.skipped === true;
@@ -4207,16 +4634,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
4207
4634
  const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
4208
4635
 
4209
4636
  /**
4210
- * One reply's scores under a run's per-item tests, by test id, in order: a
4211
- * test with no case to score leaves no entry, and one after a failure that
4637
+ * One reply's scores under a run's per-item evals, by eval id, in order: a
4638
+ * eval with no case to score leaves no entry, and one after a failure that
4212
4639
  * does not continue reads Skipped. Null where nothing was scored.
4213
4640
  */
4214
- function itemScores(run , kase , res ) {
4641
+ function itemScores(run , kase , res ) {
4215
4642
  if (!kase) return null;
4216
- const out = {};
4643
+ const out = {};
4217
4644
  let stopped = false;
4218
- for (const t of testsOf(run)) {
4219
- const type = TEST_TYPES[t.type];
4645
+ for (const t of evalsOf(run)) {
4646
+ const type = EVAL_TYPES[t.type];
4220
4647
  if (!type?.score) continue;
4221
4648
  if (stopped) { out[t.id] = SKIPPED; continue; }
4222
4649
  const s = type.score(t, kase, res);
@@ -4229,25 +4656,24 @@ function itemScores(run , kase , res
4229
4656
  /** What production replied to an item, for a run's metrics: its last job's
4230
4657
  call says, from the item's record, or nobody does. */
4231
4658
  function productionOf(run , record ) {
4232
- const last = run.jobs.at(-1);
4233
- const call = stepIn(last, "call");
4234
- return record ? STEP_TYPES[(call )?.type ?? ""]?.production?.(call, record) ?? null : null;
4659
+ const last = run.jobs.at(-1), flow = flowStepOf(last);
4660
+ return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
4235
4661
  }
4236
4662
 
4237
4663
  /**
4238
- * itemScores for the runner, which can wait: a test that reads every item
4664
+ * itemScores for the runner, which can wait: an eval that reads every item
4239
4665
  * (`read`: the Metrics, which may ask a grader) scores one with no case too.
4240
4666
  */
4241
4667
  async function itemScoresAsync(run , kase , res ,
4242
4668
  more = {}) {
4243
4669
  const out = {};
4244
4670
  let stopped = false;
4245
- for (const t of testsOf(run)) {
4246
- const type = TEST_TYPES[t.type];
4671
+ for (const t of evalsOf(run)) {
4672
+ const type = EVAL_TYPES[t.type];
4247
4673
  if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
4248
4674
  if (stopped) { out[t.id] = SKIPPED; continue; }
4249
- // A test with nothing to read on this item leaves no entry, as a graded
4250
- // test does on an item with no case.
4675
+ // An eval with nothing to read on this item leaves no entry, as a graded
4676
+ // eval does on an item with no case.
4251
4677
  const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
4252
4678
  : type.score (t, kase , res);
4253
4679
  if (!s) continue;
@@ -4258,22 +4684,22 @@ async function itemScoresAsync(run , kase ,
4258
4684
  }
4259
4685
 
4260
4686
  /**
4261
- * Every test's reading of scenario [i] of a run, in the run's order, from
4687
+ * Every eval's reading of scenario [i] of a run, in the run's order, from
4262
4688
  * the items so far. [settled] says every item is in: only then has a
4263
- * whole-run test settled, so only then does its failure skip the tests
4689
+ * whole-run eval settled, so only then does its failure skip the evals
4264
4690
  * after it.
4265
4691
  */
4266
- function scenarioTests(run , i , items ,
4692
+ function scenarioEvals(run , i , items ,
4267
4693
  settled = true) {
4268
4694
  const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
4269
- // Which items a per-item test has stopped so far, for the tests after it.
4695
+ // Which items a per-item eval has stopped so far, for the evals after it.
4270
4696
  const stopped = cells.map(() => false);
4271
4697
  let skipRest = false;
4272
- const tests = testsOf(run);
4273
- return tests.map((t, j) => {
4274
- const type = TEST_TYPES[t.type];
4698
+ const evals = evalsOf(run);
4699
+ return evals.map((t, j) => {
4700
+ const type = EVAL_TYPES[t.type];
4275
4701
  const whole = isWholeRun(type, t);
4276
- const base = { id: t.id, label: testLabel({ tests }, j), whole, skipped: skipRest,
4702
+ const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
4277
4703
  verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
4278
4704
  ran: 0, passed: 0, skippedItems: 0 };
4279
4705
  if (skipRest) {
@@ -4300,7 +4726,7 @@ function scenarioTests(run , i , items
4300
4726
  });
4301
4727
  }
4302
4728
 
4303
- /** A scenario's pass or fail over every test that read it: null where none
4729
+ /** A scenario's pass or fail over every eval that read it: null where none
4304
4730
  has anything to say yet. */
4305
4731
  function scenarioPasses(outcomes ) {
4306
4732
  let said = false;
@@ -4341,95 +4767,77 @@ function validateEvals(ev , files
4341
4767
  return ["the graded set has to be a JSON object with a `cases` list"];
4342
4768
  }
4343
4769
  if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
4770
+ if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
4344
4771
  if (bad.length) return bad;
4345
4772
 
4346
4773
  const graded = ev.cases || [];
4347
4774
  // Identity first: every rule below reports which case is at fault, so a
4348
4775
  // case with no usable id makes the rest of the report unreadable.
4349
- const seen = new Map ();
4776
+ const seen = new Set ();
4350
4777
  for (const c of graded) {
4351
- {
4352
- if (!c || typeof c !== "object" || Array.isArray(c)) {
4353
- bad.push("cases holds something that is not a case");
4354
- continue;
4355
- }
4356
- if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
4357
- else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4358
- else seen.set(c.id, "cases");
4359
- if (!caseFile(c).trim()) {
4360
- bad.push(`${c.id || "a case"} names no file`);
4361
- }
4362
- for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
4363
- if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
4364
- }
4365
- for (const key of ["minCount", "maxCount"]) {
4366
- if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
4367
- }
4368
- if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
4369
- // A case's own metrics, which a Metrics test adds to its own for this item.
4370
- if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4778
+ if (!isObj(c)) {
4779
+ bad.push("cases holds something that is not a case");
4780
+ continue;
4371
4781
  }
4782
+ if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
4783
+ else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4784
+ else seen.add(c.id);
4785
+ if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
4786
+ if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
4787
+ if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
4788
+ if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4372
4789
  }
4373
4790
  if (bad.length) return bad;
4374
4791
 
4375
- // A term the scorer cannot match is a term that can never be produced, so
4376
- // the case is unpassable however well the model answers.
4792
+ // What a case says, read from its metrics: a term the scorer cannot match
4793
+ // can never be produced, so the case is unpassable however well the model
4794
+ // answers -- and the same goes for a bound no reply can meet.
4377
4795
  const matchable = (term ) => termIn([String(term).trim()], term);
4378
-
4796
+ const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
4379
4797
  for (const c of graded) {
4380
4798
  if (c.todo) continue;
4381
4799
  const say = (m ) => bad.push(`${c.id}: ${m}`);
4382
- const expect = c.expect || [], forbid = c.forbid || [], anyOf = c.anyOf || [];
4383
-
4384
- if (c.discarded === true) {
4800
+ const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
4801
+ if (!scored.length) { say("graded but states nothing to expect"); continue; }
4802
+ if (scored.some(m => m.type === "discarded" && !m.not)) {
4385
4803
  // What is discarded holds nothing, so there is nothing else to expect.
4386
- if (expect.length || anyOf.length || forbid.length || c.minCount != null || c.maxCount != null) {
4387
- say("expects its answer discarded, and states something the answer should hold too");
4388
- }
4804
+ if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
4389
4805
  continue;
4390
4806
  }
4391
- const metrics = Array.isArray(c.metrics) ? c.metrics : [];
4392
- if (!expect.length && !anyOf.length && !metrics.length) say("graded but states nothing to expect");
4393
- for (const t of expect) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4394
- for (const g of anyOf) {
4395
- if (!Array.isArray(g)) { say("an anyOf group has to be a list of terms"); continue; }
4396
- if (!g.length) say("an anyOf group is empty, so nothing can satisfy it");
4397
- for (const t of g) if (!matchable(t)) say(`offers ${t}, which its own scorer cannot match`);
4398
- }
4399
- // An exception excuses only the forbidden terms inside it, so one that
4400
- // holds none of them changes nothing and reads as if it did.
4401
- for (const a of c.allow || []) {
4402
- if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
4807
+ const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
4808
+ const forbids = scored.filter(m => m.not && m.type === "contains");
4809
+ for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4810
+ // An exception excuses only the forbidden term inside it, so one that
4811
+ // holds it not changes nothing and reads as if it did.
4812
+ for (const m of forbids) {
4813
+ for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
4403
4814
  }
4404
4815
  // A term on both lists cannot be produced and cannot be withheld.
4405
- for (const t of [...expect, ...anyOf.flat()]) {
4406
- if (forbid.includes(t)) say(`${t} is both expected and forbidden`);
4407
- }
4408
- const { minCount: lo, maxCount: hi } = c;
4409
- if (lo != null && hi != null && lo > hi) say(`minCount ${lo} is above maxCount ${hi}`);
4410
- if (lo != null && lo < 1) say(`minCount ${lo} is not a bound`);
4411
- // The mirror of the floor: a ceiling below one says no answer is
4412
- // acceptable, which is an entry that can never pass rather than a
4413
- // strict one.
4414
- if (hi != null && hi < 1) say(`maxCount ${hi} leaves no answer that could pass`);
4415
- // A case that names more distinct things than the reply may carry, in
4416
- // the scorer's own counting of a thing (#486: an expect term is one, an
4417
- // anyOf group is one), can never pass however well the model answers.
4418
- const things = expect.length + anyOf.length;
4419
- if (hi != null && things > hi) {
4420
- say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
4816
+ const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
4817
+ for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
4818
+ for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
4819
+ const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
4820
+ if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
4821
+ // A ceiling below one says no answer is acceptable, which is a case
4822
+ // that can never pass rather than a strict one.
4823
+ if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
4824
+ // A case that names more distinct things than the reply may carry, in
4825
+ // the scorer's own counting of a thing (#486: each term Contains all
4826
+ // wants is one, each Contains any is one), can never pass.
4827
+ const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
4828
+ if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
4421
4829
  }
4422
4830
  }
4423
4831
 
4424
- // One file, one set of expectations. A file here twice is two sets of
4425
- // them, graded separately, and both would be listed.
4832
+ // One item, one case. An item here twice is two cases of it, graded
4833
+ // separately, and both would be listed.
4426
4834
  const where = new Map ();
4427
4835
  for (const c of graded) {
4428
- const name = caseFile(c);
4836
+ const name = caseItem(c);
4429
4837
  const counted = where.get(name);
4430
4838
  if (counted) {
4431
- bad.push(`${name} is graded twice — one file, `
4432
- + `two sets of expectations. Grade it once.`);
4839
+ bad.push(`${name} is graded twice — one item, `
4840
+ + `two cases. Grade it once.`);
4433
4841
  } else where.set(name, true);
4434
4842
  }
4435
4843
  const gradedAt = (name ) => where.has(name);
@@ -4450,7 +4858,7 @@ function validateEvals(ev , files
4450
4858
  for (const f of shown) {
4451
4859
  if (!gradedAt(f)) {
4452
4860
  bad.push(`${f} is in the Source and this dataset does not grade it, `
4453
- + `so the Evals tab does not list it`);
4861
+ + `so the Cases view does not list it`);
4454
4862
  }
4455
4863
  }
4456
4864
  return bad;
@@ -4472,7 +4880,9 @@ function evalsWarnings(ev , prompt ) {
4472
4880
  const want = Number(n), warn = [];
4473
4881
  for (const c of ev.cases || []) {
4474
4882
  if (!c || c.todo) continue;
4475
- const things = (c.expect || []).length + (c.anyOf || []).length;
4883
+ const things = (Array.isArray(c.metrics) ? c.metrics : [])
4884
+ .filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
4885
+ .reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
4476
4886
  if (things > want) {
4477
4887
  warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
4478
4888
  }
@@ -4491,7 +4901,7 @@ function evalsWarnings(ev , prompt ) {
4491
4901
  * `evals-check.js` asserts the round trip against the real file.
4492
4902
  */
4493
4903
  function evalsJson(ev ) {
4494
- return JSON.stringify(canonicalCases(ev), null, 2) + "\n";
4904
+ return JSON.stringify(ev, null, 2) + "\n";
4495
4905
  }
4496
4906
 
4497
4907
 
@@ -4505,22 +4915,24 @@ registerMetrics({ registerKinds });
4505
4915
  export {
4506
4916
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
4507
4917
  loopReplyError, preparedSize,
4508
- termIn, forbiddenIn, scoreCase, gradedSetFrom, caseFile, canonicalCases, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4918
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4509
4919
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
4510
- CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, callEntryOf, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4920
+ CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4511
4921
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
4512
4922
  tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
4513
4923
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
4514
4924
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
4515
- PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, TEST_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4925
+ PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4516
4926
  SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4517
4927
  registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
4518
4928
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
4519
4929
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
4520
- contentOf, withContent, callOf, replyOf, outOf, withOut, withCall, slotEntry, SLOTS,
4521
- blankPipeline, upgradePipeline, fatProfileRef, newId, newTest, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4522
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioTests, scenarioPasses, isSkipped, failedScore,
4523
- testLabel, testsOf, testsDataset, isWholeRun, TEST_FIELDS,
4930
+ contentOf, withContent, replyOf,
4931
+ targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
4932
+ tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
4933
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4934
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
4935
+ evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
4524
4936
  pipelineToYaml, importPipeline, yamlToPipeline,
4525
4937
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
4526
4938
  };