evals-lab 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +43 -0
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +36 -30
- package/lab/demo/pipelines/demo-2.json +36 -30
- package/lab/evals-core.mjs +1122 -710
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +77 -49
- package/lab/server.py +544 -154
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-61NS6C4m.js +0 -18
- package/lab/web/dist/assets/main-DjQQums6.css +0 -1
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
// WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
|
|
27
27
|
// script has not, so `prepare()` stays in the page and the script shells out
|
|
28
28
|
// to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
|
|
29
|
-
// the script writes JSON. Both read the same verdict out of
|
|
29
|
+
// the script writes JSON. Both read the same verdict out of the same Metrics.
|
|
30
30
|
//
|
|
31
31
|
// `runner-check.js` runs one set through two of the callers and asserts the
|
|
32
32
|
// verdicts and the totals are identical, because sharing a file is a claim
|
|
@@ -94,49 +94,76 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
94
94
|
|
|
95
95
|
|
|
96
96
|
|
|
97
|
-
/**
|
|
98
|
-
|
|
99
|
-
|
|
97
|
+
/** The Power Automate step a job's records are calls of (job 1, Content):
|
|
98
|
+
the action's name its expressions read it by, the loop item() means,
|
|
99
|
+
and the HTTP API its replies speak -- how production's reply is read. */
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
/** The item's image goes beside every target's words in this job. */
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
/** The token mappings a job's words are resolved under. */
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
/** What the flow does to the reply before it reads it -- its Parse JSON's
|
|
119
|
+
content -- as an expression over body('<step>'). First of Responses. */
|
|
100
120
|
|
|
101
121
|
|
|
102
|
-
|
|
103
|
-
|
|
122
|
+
|
|
104
123
|
|
|
105
124
|
|
|
106
|
-
/**
|
|
125
|
+
/** Response Format Validation: the output kind a reply is read as, and that
|
|
126
|
+
kind's own settings. A job with none reads its reply as text. */
|
|
107
127
|
|
|
108
128
|
|
|
109
|
-
|
|
129
|
+
|
|
110
130
|
|
|
111
131
|
|
|
112
|
-
/**
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
132
|
+
/** One change applied to the parsed reply, after the Format Validation
|
|
133
|
+
whose kind accepts it. */
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
/** A job's steps: its Content stage, then its Responses (§16). */
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
/** A target step that prompts a model: its words, resolved under the job's
|
|
144
|
+
token mappings, and a profile of its own where it names one. */
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
/** A target step that sends a Power Automate step's request
|
|
150
|
+
(docs/power-automate.md § The HTTP Request step): the flow's own template
|
|
151
|
+
-- method, path, query and body, Workflow Definition Language expressions
|
|
152
|
+
and all -- with the target's words and fields placed in the body by its
|
|
153
|
+
HTTP API, evaluated against each item's record. The profile gives the
|
|
154
|
+
address and the key. */
|
|
155
|
+
|
|
119
156
|
|
|
120
|
-
|
|
121
|
-
|
|
122
157
|
|
|
123
158
|
|
|
124
159
|
|
|
125
160
|
|
|
126
161
|
|
|
127
162
|
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
163
|
|
|
135
164
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
/** A job is its steps, in fixed slots (pipeline-model §3): an Attach Content
|
|
139
|
-
on job 1 only, one Call, then Read Reply. */
|
|
165
|
+
/** A job is its steps, staged (pipeline-model §16): Content, then
|
|
166
|
+
Responses. What each target sends in it is the target's own. */
|
|
140
167
|
|
|
141
168
|
|
|
142
169
|
|
|
@@ -170,13 +197,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
170
197
|
|
|
171
198
|
|
|
172
199
|
|
|
173
|
-
/**
|
|
200
|
+
/** What a target's step says beside its type: what to ask in that job,
|
|
201
|
+
and whom, where it names a profile of its own. */
|
|
174
202
|
|
|
175
203
|
|
|
176
204
|
|
|
177
|
-
|
|
178
205
|
|
|
179
|
-
|
|
206
|
+
|
|
207
|
+
|
|
180
208
|
|
|
181
209
|
|
|
182
210
|
|
|
@@ -187,17 +215,25 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
187
215
|
|
|
188
216
|
|
|
189
217
|
|
|
190
|
-
|
|
191
|
-
|
|
218
|
+
/** What a target sends in one job: one of the target step types. */
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
/** A target (pipeline-model §16): one column through every job, compared
|
|
222
|
+
side by side in Results -- what version 9 called a scenario. Its step in
|
|
223
|
+
each job says what it sends there. */
|
|
224
|
+
|
|
225
|
+
|
|
192
226
|
|
|
193
227
|
|
|
194
228
|
|
|
195
|
-
|
|
196
|
-
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
|
|
197
233
|
|
|
198
234
|
|
|
199
|
-
/** What every
|
|
200
|
-
shown, an optional name ("
|
|
235
|
+
/** What every eval carries whatever its type: an id minted once and never
|
|
236
|
+
shown, an optional name ("Eval 2" when blank), and whether the evals after
|
|
201
237
|
it still read what it failed on (docs/pipeline-model.md §3). */
|
|
202
238
|
|
|
203
239
|
|
|
@@ -224,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
224
260
|
|
|
225
261
|
|
|
226
262
|
|
|
227
|
-
/** Checks of every reply (
|
|
263
|
+
/** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
|
|
228
264
|
a case's own for its item; all must pass, or weighted points reach the
|
|
229
265
|
threshold. A model-graded one asks the grader. */
|
|
230
266
|
|
|
@@ -239,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
239
275
|
|
|
240
276
|
|
|
241
277
|
|
|
242
|
-
/**
|
|
278
|
+
/** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
|
|
243
279
|
older document held (upgradePipeline reads it converted). */
|
|
244
280
|
|
|
245
281
|
|
|
246
|
-
/** A pipeline's
|
|
282
|
+
/** A pipeline's evals, in the order they read a run. */
|
|
247
283
|
|
|
248
284
|
|
|
249
285
|
|
|
@@ -255,7 +291,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
255
291
|
|
|
256
292
|
|
|
257
293
|
|
|
258
|
-
|
|
294
|
+
|
|
259
295
|
|
|
260
296
|
|
|
261
297
|
|
|
@@ -323,7 +359,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
323
359
|
/** A pipeline read from YAML, references remapped to this lab. */
|
|
324
360
|
|
|
325
361
|
|
|
326
|
-
|
|
362
|
+
|
|
327
363
|
|
|
328
364
|
|
|
329
365
|
|
|
@@ -335,6 +371,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
335
371
|
|
|
336
372
|
|
|
337
373
|
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
|
|
338
377
|
|
|
339
378
|
|
|
340
379
|
/** An import's outcome: the remapped pipeline, or one sentence refusing it. */
|
|
@@ -417,12 +456,23 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
417
456
|
|
|
418
457
|
|
|
419
458
|
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
/** What an expression token reads: a record's scope, its time, the loop. */
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
|
|
420
470
|
|
|
421
471
|
|
|
422
472
|
/** A stage of a run as a transport asks it: stagesFor's answer. */
|
|
423
473
|
|
|
424
474
|
|
|
425
|
-
|
|
475
|
+
|
|
426
476
|
|
|
427
477
|
|
|
428
478
|
|
|
@@ -432,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
432
482
|
|
|
433
483
|
// ---- grading ----
|
|
434
484
|
|
|
435
|
-
/** A graded case
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
485
|
+
/** A graded case (dataset body version 5): an Item and the Metrics its
|
|
486
|
+
reply is held to.
|
|
487
|
+
|
|
488
|
+
The item is named by `item`: a dataset joins on an item's name in the
|
|
489
|
+
Source it grades, exactly, and nothing about that is an image -- the
|
|
490
|
+
next dataset addresses a row of a spreadsheet or a line of a log by the
|
|
491
|
+
same key. Version 4 called it `filename`, and the lab's first app had a
|
|
492
|
+
name of its own for it; `upgradeDatasetBody` reads both as `item`.
|
|
493
|
+
|
|
494
|
+
A case's metrics are what a good answer is: each a metric as a
|
|
495
|
+
pipeline's Metrics eval holds one, added to the eval's own for this item
|
|
496
|
+
when an eval names the dataset. One with weight 0 is watched -- reported,
|
|
497
|
+
never scored. */
|
|
446
498
|
|
|
447
499
|
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
500
|
+
|
|
501
|
+
|
|
451
502
|
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
503
|
+
|
|
504
|
+
|
|
462
505
|
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
506
|
|
|
468
507
|
|
|
469
508
|
|
|
470
|
-
/** A graded set: cases.
|
|
509
|
+
/** A graded set: a dataset's body, or anything holding its cases. */
|
|
471
510
|
|
|
472
511
|
|
|
473
512
|
|
|
474
513
|
|
|
475
514
|
|
|
476
|
-
/** One row the
|
|
477
|
-
a case of a set validateEvals accepts, so it has its id and
|
|
515
|
+
/** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
|
|
516
|
+
a case of a set validateEvals accepts, so it has its id and item. */
|
|
478
517
|
|
|
479
518
|
|
|
480
|
-
|
|
519
|
+
|
|
481
520
|
|
|
482
521
|
|
|
483
522
|
|
|
484
523
|
/** A requirement as a score reports it: a term, or the group it was. */
|
|
485
524
|
|
|
486
525
|
|
|
487
|
-
/** How a graded case read a reply: the same shape both engines agree on. */
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
526
|
/** A run's totals, summed across cases. */
|
|
505
527
|
|
|
506
528
|
|
|
@@ -509,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
509
531
|
|
|
510
532
|
|
|
511
533
|
|
|
512
|
-
/** The whole-run verdict of
|
|
534
|
+
/** The whole-run verdict of an eval type, where it has one of its own. */
|
|
513
535
|
|
|
514
536
|
|
|
515
537
|
|
|
@@ -520,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
520
542
|
|
|
521
543
|
|
|
522
544
|
|
|
523
|
-
/** One thing
|
|
545
|
+
/** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
|
|
524
546
|
|
|
525
547
|
|
|
526
548
|
|
|
@@ -549,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
549
571
|
|
|
550
572
|
|
|
551
573
|
|
|
552
|
-
/** A per-item
|
|
574
|
+
/** A per-item eval's score as a stored row carries it. */
|
|
553
575
|
|
|
554
576
|
|
|
555
577
|
|
|
@@ -628,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
628
650
|
more entry, and no reader changes. */
|
|
629
651
|
|
|
630
652
|
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
|
|
631
656
|
|
|
632
657
|
|
|
633
658
|
|
|
@@ -648,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
648
673
|
|
|
649
674
|
|
|
650
675
|
|
|
651
|
-
/** What an earlier
|
|
676
|
+
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
652
677
|
|
|
653
678
|
|
|
654
679
|
|
|
655
680
|
|
|
656
|
-
/** One
|
|
681
|
+
/** One eval's reading of one scenario of a run (scenarioEvals). */
|
|
657
682
|
|
|
658
683
|
|
|
659
684
|
|
|
@@ -710,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
710
735
|
|
|
711
736
|
// ---- the registries ----
|
|
712
737
|
|
|
713
|
-
/** An option a modifier or
|
|
738
|
+
/** An option a modifier or an eval type exposes for editing. */
|
|
714
739
|
|
|
715
740
|
|
|
716
741
|
|
|
@@ -770,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
770
795
|
|
|
771
796
|
|
|
772
797
|
|
|
773
|
-
|
|
798
|
+
|
|
774
799
|
|
|
775
800
|
|
|
776
801
|
|
|
@@ -846,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
846
871
|
|
|
847
872
|
|
|
848
873
|
|
|
849
|
-
/**
|
|
874
|
+
/** An eval type: what it checks of a document, and how it scores a run. */
|
|
850
875
|
|
|
851
876
|
|
|
852
|
-
|
|
877
|
+
|
|
853
878
|
|
|
854
879
|
|
|
855
880
|
|
|
@@ -859,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
859
884
|
|
|
860
885
|
|
|
861
886
|
|
|
862
|
-
|
|
887
|
+
|
|
863
888
|
|
|
864
889
|
|
|
865
890
|
|
|
866
|
-
|
|
891
|
+
|
|
867
892
|
|
|
868
893
|
|
|
869
894
|
|
|
@@ -884,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
884
909
|
|
|
885
910
|
|
|
886
911
|
|
|
887
|
-
/** What
|
|
912
|
+
/** What an eval's `read` is handed besides the reply: production's reply to
|
|
888
913
|
the item, and a grader, where the run has them. */
|
|
889
914
|
|
|
890
915
|
|
|
@@ -894,45 +919,54 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
894
919
|
|
|
895
920
|
|
|
896
921
|
|
|
897
|
-
/** A job's
|
|
898
|
-
|
|
899
|
-
|
|
922
|
+
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
923
|
+
step is in no job: it is the target's, in that job's place. */
|
|
924
|
+
|
|
925
|
+
const SLOTS = ["content", "target", "responses"];
|
|
900
926
|
|
|
901
927
|
|
|
902
928
|
|
|
903
929
|
|
|
904
930
|
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
931
|
+
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
|
|
908
935
|
|
|
909
|
-
|
|
910
936
|
|
|
911
937
|
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
|
|
912
943
|
|
|
913
944
|
|
|
914
945
|
|
|
915
946
|
|
|
916
|
-
|
|
947
|
+
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
|
|
917
951
|
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
918
955
|
|
|
919
956
|
|
|
920
957
|
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
958
|
|
|
959
|
+
|
|
925
960
|
|
|
926
|
-
|
|
961
|
+
|
|
927
962
|
|
|
963
|
+
|
|
928
964
|
|
|
929
|
-
|
|
930
|
-
|
|
965
|
+
|
|
966
|
+
|
|
931
967
|
|
|
932
|
-
|
|
968
|
+
|
|
933
969
|
|
|
934
|
-
|
|
935
|
-
|
|
936
970
|
|
|
937
971
|
|
|
938
972
|
|
|
@@ -945,42 +979,122 @@ const SLOTS = ["content", "call", "reply"];
|
|
|
945
979
|
* submitted against, so nothing grades from a file.
|
|
946
980
|
*/
|
|
947
981
|
|
|
982
|
+
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
|
|
948
988
|
|
|
949
989
|
|
|
950
990
|
|
|
951
991
|
|
|
992
|
+
/** The dataset body's version: 6 marks itself; 5 named its Source; 4 and
|
|
993
|
+
earlier, neither. */
|
|
994
|
+
const DATASET_BODY_VERSION = 6 ;
|
|
995
|
+
|
|
996
|
+
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
997
|
+
reply read as a list, as well as one read as text. */
|
|
998
|
+
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
999
|
+
|
|
952
1000
|
/**
|
|
953
|
-
* [body] as this version of a dataset (
|
|
1001
|
+
* [body] as this version of a dataset (6), from any earlier one. Version 1
|
|
954
1002
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
955
1003
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
956
1004
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
957
1005
|
* job (docs/pipeline-model.md §13): they leave, and the terms a case
|
|
958
1006
|
* watches for are its `watch`. Version 3 held the `prompt` a new scenario
|
|
959
|
-
* started from, which the Prompt library holds now: it leaves.
|
|
960
|
-
*
|
|
961
|
-
*
|
|
962
|
-
*
|
|
1007
|
+
* started from, which the Prompt library holds now: it leaves. Version 4's
|
|
1008
|
+
* case named its item `filename` and said what a good answer is in
|
|
1009
|
+
* expectations; each becomes the metric it is (`caseMetrics`), `why` is the
|
|
1010
|
+
* `note`, `traits` go, and the body names no Source yet. Version 5 matched
|
|
1011
|
+
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1012
|
+
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1013
|
+
* the body says its version. Every reader of a
|
|
1014
|
+
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1015
|
+
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1016
|
+
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
1017
|
+
* same answer, and server.py's upgrade_body is its twin. Anything else
|
|
1018
|
+
* comes back as it was.
|
|
963
1019
|
*/
|
|
964
1020
|
function upgradeDatasetBody (body ) {
|
|
965
1021
|
if (!isObj(body)) return body;
|
|
966
|
-
if (
|
|
967
|
-
if (!("prompt" in body)) return body;
|
|
968
|
-
const { prompt: _library, ...rest } = body ;
|
|
969
|
-
return rest ;
|
|
970
|
-
}
|
|
1022
|
+
if (body.version === DATASET_BODY_VERSION) return body;
|
|
971
1023
|
const b = body ;
|
|
972
|
-
//
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
const out = {};
|
|
977
|
-
for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
|
|
978
|
-
return out;
|
|
979
|
-
};
|
|
1024
|
+
// A body that names its Source, even as null, is version 5.
|
|
1025
|
+
if ("source" in b) {
|
|
1026
|
+
return { version: DATASET_BODY_VERSION, ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) };
|
|
1027
|
+
}
|
|
980
1028
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
981
1029
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
982
1030
|
if (!raw) return body;
|
|
983
|
-
return {
|
|
1031
|
+
return { version: DATASET_BODY_VERSION, source: null, cases: raw.map(caseOfV4).map(caseOfV5) };
|
|
1032
|
+
}
|
|
1033
|
+
|
|
1034
|
+
/** A version-5 case as a version-6 one: each Contains metric says Ignore
|
|
1035
|
+
case, as version 5 matched a list's items whatever it said. A reply read
|
|
1036
|
+
as text did mind its setting, but a case's metrics were written for a
|
|
1037
|
+
list -- each one converted from version 4, and each the case form made. */
|
|
1038
|
+
function caseOfV5(c ) {
|
|
1039
|
+
if (!isObj(c) || !Array.isArray(c.metrics)) return c;
|
|
1040
|
+
return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
|
|
1041
|
+
? { ...m, ignoreCase: true } : m)) };
|
|
1042
|
+
}
|
|
1043
|
+
|
|
1044
|
+
// vocab: the names older versions gave these fields
|
|
1045
|
+
const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
|
|
1046
|
+
/** What a version-4 case said, which its metrics say now. */
|
|
1047
|
+
const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
|
|
1048
|
+
|
|
1049
|
+
/** One case of any earlier version as a version-5 one: its id, item, todo
|
|
1050
|
+
and note, its expectations as metrics ahead of any metrics it held, and
|
|
1051
|
+
anything else it carried kept as it was. */
|
|
1052
|
+
function caseOfV4(c ) {
|
|
1053
|
+
if (!isObj(c)) return c;
|
|
1054
|
+
const was = {};
|
|
1055
|
+
for (const [k, v] of Object.entries(c)) {
|
|
1056
|
+
const key = CASE_RENAMED[k] ?? k;
|
|
1057
|
+
// A case naming its item both ways keeps the newer key's.
|
|
1058
|
+
if (!(key in was) || key === k) was[key] = v;
|
|
1059
|
+
}
|
|
1060
|
+
if (isStr(was.item)) delete was.filename;
|
|
1061
|
+
const out = {};
|
|
1062
|
+
if ("id" in was) out.id = was.id;
|
|
1063
|
+
out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
|
|
1064
|
+
out.todo = was.todo === true;
|
|
1065
|
+
out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
|
|
1066
|
+
out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
|
|
1067
|
+
for (const [k, v] of Object.entries(was)) {
|
|
1068
|
+
if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
|
|
1069
|
+
}
|
|
1070
|
+
return out;
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
/** A version-4 case's expectations as the metrics that say the same:
|
|
1074
|
+
`expect` is Contains all, each `anyOf` group a Contains any, each
|
|
1075
|
+
forbidden term a Contains turned round with the `allow` phrases that
|
|
1076
|
+
excuse it, the count bounds an Item count, `discarded` the Discarded
|
|
1077
|
+
metric, and each watched term a Contains any of weight 0 -- reported,
|
|
1078
|
+
never scored. The order is the one scoreCase read them in, so a reason
|
|
1079
|
+
reads in the order it did. */
|
|
1080
|
+
function caseMetrics(c ) {
|
|
1081
|
+
const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
|
|
1082
|
+
if (c.discarded === true) return [{ type: "discarded" }];
|
|
1083
|
+
const out = [];
|
|
1084
|
+
const expect = list(c.expect), allow = list(c.allow);
|
|
1085
|
+
if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
|
|
1086
|
+
for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
|
|
1087
|
+
if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
|
|
1088
|
+
}
|
|
1089
|
+
for (const t of list(c.forbid)) {
|
|
1090
|
+
// An exception excuses only the forbidden term inside it.
|
|
1091
|
+
const except = allow.filter(a => termIn([a], t));
|
|
1092
|
+
out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
|
|
1093
|
+
}
|
|
1094
|
+
const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
|
|
1095
|
+
if (min != null || max != null) out.push({ type: "item-count", min, max });
|
|
1096
|
+
for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
|
|
1097
|
+
return out;
|
|
984
1098
|
}
|
|
985
1099
|
|
|
986
1100
|
/** The rules a version-1 or version-2 dataset body held, or null: what a
|
|
@@ -996,6 +1110,9 @@ function datasetRules(body ) {
|
|
|
996
1110
|
|
|
997
1111
|
|
|
998
1112
|
|
|
1113
|
+
|
|
1114
|
+
|
|
1115
|
+
|
|
999
1116
|
|
|
1000
1117
|
|
|
1001
1118
|
|
|
@@ -1023,6 +1140,10 @@ function datasetRules(body ) {
|
|
|
1023
1140
|
|
|
1024
1141
|
|
|
1025
1142
|
|
|
1143
|
+
|
|
1144
|
+
|
|
1145
|
+
|
|
1146
|
+
|
|
1026
1147
|
|
|
1027
1148
|
|
|
1028
1149
|
|
|
@@ -1067,7 +1188,7 @@ function datasetRules(body ) {
|
|
|
1067
1188
|
|
|
1068
1189
|
|
|
1069
1190
|
/** What a Call hands its connection. */
|
|
1070
|
-
|
|
1191
|
+
|
|
1071
1192
|
|
|
1072
1193
|
/** What the core hands a registry module (kinds/list.ts) to register with. */
|
|
1073
1194
|
|
|
@@ -1120,7 +1241,9 @@ function mappingsFor(prompt ) {
|
|
|
1120
1241
|
|
|
1121
1242
|
|
|
1122
1243
|
|
|
1123
|
-
|
|
1244
|
+
|
|
1245
|
+
|
|
1246
|
+
|
|
1124
1247
|
|
|
1125
1248
|
|
|
1126
1249
|
|
|
@@ -1157,6 +1280,32 @@ TOKEN_TYPES.block = {
|
|
|
1157
1280
|
validate(m, at, bad){ if (typeof m.enabled !== "boolean") bad.push(`${at}: a block's enabled has to be true or false`); },
|
|
1158
1281
|
};
|
|
1159
1282
|
|
|
1283
|
+
// Expression: a Power Automate expression, evaluated against the item's
|
|
1284
|
+
// record -- `@triggerBody()?['note']`, or text with `@{…}` in it -- so a
|
|
1285
|
+
// prompt can say what the flow's own request says (docs/pipeline-model.md
|
|
1286
|
+
// §16 › Targets).
|
|
1287
|
+
TOKEN_TYPES.expression = {
|
|
1288
|
+
label: "Expression",
|
|
1289
|
+
fields: ["value"],
|
|
1290
|
+
defaults: () => ({ value: "" }),
|
|
1291
|
+
notation: name => `{${name}}`,
|
|
1292
|
+
names: name => [name],
|
|
1293
|
+
order: 1,
|
|
1294
|
+
apply(prompt, m, record){
|
|
1295
|
+
const token = `{${m.name}}`;
|
|
1296
|
+
if (!prompt.includes(token)) return prompt;
|
|
1297
|
+
if (!record) throw new WdlError(`{${m.name}} reads the item's record, and this item is not one`);
|
|
1298
|
+
try {
|
|
1299
|
+
return prompt.replaceAll(token, asText(evaluate(m.value ?? "", { scope: record.scope || {}, now: record.at ?? null, loop: record.loop ?? null })));
|
|
1300
|
+
} catch (e) {
|
|
1301
|
+
throw new WdlError(`{${m.name}}: ${(e ).message}`);
|
|
1302
|
+
}
|
|
1303
|
+
},
|
|
1304
|
+
validate(m, at, bad){
|
|
1305
|
+
if (!isStr(m.value)) bad.push(`${at}: an expression's value has to be text`);
|
|
1306
|
+
},
|
|
1307
|
+
};
|
|
1308
|
+
|
|
1160
1309
|
/** A mapping of [type] named [name], holding its type's defaults. */
|
|
1161
1310
|
function tokenMapping(name , type ) {
|
|
1162
1311
|
return { name, type, ...(TOKEN_TYPES[type]?.defaults() ?? {}) };
|
|
@@ -1186,12 +1335,12 @@ function tokenMappingsProblems(list , at , bad ) {
|
|
|
1186
1335
|
* The tokens are a parameter and not a global, because only one of the three
|
|
1187
1336
|
* callers has a localStorage to have loaded a set from.
|
|
1188
1337
|
*/
|
|
1189
|
-
function resolvePrompt(tpl , tokens ) {
|
|
1338
|
+
function resolvePrompt(tpl , tokens , record = null) {
|
|
1190
1339
|
const order = (m ) => TOKEN_TYPES[m.type]?.order ?? Infinity;
|
|
1191
1340
|
let out = tpl;
|
|
1192
1341
|
for (const m of [...tokens].sort((a, b) => order(a) - order(b))) {
|
|
1193
1342
|
const type = TOKEN_TYPES[m.type];
|
|
1194
|
-
if (type) out = type.apply(out, m);
|
|
1343
|
+
if (type) out = type.apply(out, m, record);
|
|
1195
1344
|
}
|
|
1196
1345
|
// Dropping a block leaves the spaces that surrounded it.
|
|
1197
1346
|
return out.replace(/[ \t]{2,}/g, " ").trim();
|
|
@@ -1231,8 +1380,10 @@ function textPrompt(instruction , text ) {
|
|
|
1231
1380
|
}
|
|
1232
1381
|
|
|
1233
1382
|
// Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
|
|
1234
|
-
// work in, so that "New Zealand." and "new zealand" are the same two words
|
|
1235
|
-
|
|
1383
|
+
// work in, so that "New Zealand." and "new zealand" are the same two words --
|
|
1384
|
+
// unless [keepCase], for a metric whose Ignore case is off.
|
|
1385
|
+
const words = (s , keepCase = false) =>
|
|
1386
|
+
(keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
|
|
1236
1387
|
|
|
1237
1388
|
// The request Tagger builds, field for field -- when the target reads those
|
|
1238
1389
|
// fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
|
|
@@ -1376,8 +1527,10 @@ CONNECTION_TYPES["ollama-cloud"] = ollamaType("ollama-cloud", "Ollama (Cloud)",
|
|
|
1376
1527
|
// each scenario holds one. Only job 1 can use it: a later job's item is the
|
|
1377
1528
|
// job before's answer, not a file.
|
|
1378
1529
|
const sendsNothing = () => { throw new Error("Echo answers from the item and sends no request"); };
|
|
1530
|
+
// Echo is a target step now (STEP_TYPES.echo), with no profile; the type
|
|
1531
|
+
// stays so a stored run, or a profile made before, still reads.
|
|
1379
1532
|
CONNECTION_TYPES.echo = {
|
|
1380
|
-
id: "echo", label: "Echo", settings: [], keyless: true,
|
|
1533
|
+
id: "echo", label: "Echo", settings: [], keyless: true, picker: false,
|
|
1381
1534
|
description: "No model: answers with each text item itself, to grade replies recorded earlier.",
|
|
1382
1535
|
local: (item, sent) => item.text ?? sent,
|
|
1383
1536
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
@@ -1547,16 +1700,19 @@ CONNECTION_TYPES["llama.cpp"] = {
|
|
|
1547
1700
|
|
|
1548
1701
|
|
|
1549
1702
|
|
|
1550
|
-
|
|
1703
|
+
|
|
1551
1704
|
|
|
1552
|
-
|
|
1705
|
+
|
|
1553
1706
|
|
|
1554
|
-
|
|
1707
|
+
|
|
1555
1708
|
|
|
1556
1709
|
|
|
1557
1710
|
|
|
1558
1711
|
|
|
1559
1712
|
|
|
1713
|
+
|
|
1714
|
+
|
|
1715
|
+
|
|
1560
1716
|
|
|
1561
1717
|
|
|
1562
1718
|
|
|
@@ -1622,6 +1778,7 @@ HTTP_APIS.anthropic = {
|
|
|
1622
1778
|
const r = j ;
|
|
1623
1779
|
return { raw: (r?.content || []).filter(b => b?.type === "text").map(b => b.text).join(""), finishReason: r?.stop_reason ?? null };
|
|
1624
1780
|
},
|
|
1781
|
+
wrap: text => ({ type: "message", role: "assistant", content: [{ type: "text", text }], stop_reason: "end_turn" }),
|
|
1625
1782
|
withPrefill(j, prefill) {
|
|
1626
1783
|
const r = j ;
|
|
1627
1784
|
if (!isObj(r) || !Array.isArray(r.content)) return j;
|
|
@@ -1667,6 +1824,7 @@ const openAiApi = (label , matches ) =>
|
|
|
1667
1824
|
return { ...b, messages: msgs };
|
|
1668
1825
|
},
|
|
1669
1826
|
reply: chatReply,
|
|
1827
|
+
wrap: text => ({ object: "chat.completion", choices: [{ index: 0, message: { role: "assistant", content: text }, finish_reason: "stop" }] }),
|
|
1670
1828
|
});
|
|
1671
1829
|
HTTP_APIS.openai = openAiApi("OpenAI", u => u.hostname === "api.openai.com");
|
|
1672
1830
|
HTTP_APIS["azure-openai"] = openAiApi("Azure OpenAI", u => u.hostname.endsWith(".openai.azure.com"));
|
|
@@ -1684,6 +1842,9 @@ HTTP_APIS.raw = {
|
|
|
1684
1842
|
try { return JSON.parse(cell.prompt); } catch { throw new Error("the body is not JSON"); }
|
|
1685
1843
|
},
|
|
1686
1844
|
reply: j => ({ raw: isStr(j) ? j : JSON.stringify(j), finishReason: null }),
|
|
1845
|
+
// A body the flow reads is JSON where it parses as JSON, as the HTTP
|
|
1846
|
+
// action hands it on, and text otherwise.
|
|
1847
|
+
wrap: text => { try { return JSON.parse(text); } catch { return text; } },
|
|
1687
1848
|
};
|
|
1688
1849
|
|
|
1689
1850
|
/** The HTTP API a flow's request to [uri] speaks: the first entry that
|
|
@@ -1938,11 +2099,12 @@ function budgetLabel(target ) {
|
|
|
1938
2099
|
// A term is present when its words appear in some item, in order and adjacent:
|
|
1939
2100
|
// "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
|
|
1940
2101
|
// looser scores "new zealand" against "zealandia" and flatters every run.
|
|
1941
|
-
|
|
1942
|
-
|
|
2102
|
+
// Letter case counts only where a metric's Ignore case is off.
|
|
2103
|
+
function termIn(items , term , ignoreCase = true) {
|
|
2104
|
+
const t = words(term, !ignoreCase);
|
|
1943
2105
|
if (!t.length) return false;
|
|
1944
2106
|
return items.some(item => {
|
|
1945
|
-
const w = words(item);
|
|
2107
|
+
const w = words(item, !ignoreCase);
|
|
1946
2108
|
for (let i = 0; i + t.length <= w.length; i++) {
|
|
1947
2109
|
if (t.every((x, j) => w[i + j] === x)) return true;
|
|
1948
2110
|
}
|
|
@@ -1969,153 +2131,55 @@ function forbiddenIn(items , term , allow
|
|
|
1969
2131
|
* check green. `evals-check.js` calls this directly.
|
|
1970
2132
|
*/
|
|
1971
2133
|
function gradedSetFrom(ev ) {
|
|
1972
|
-
return (ev.cases || []).map(c => ({ ...c,
|
|
2134
|
+
return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
|
|
1973
2135
|
}
|
|
1974
2136
|
|
|
1975
|
-
/**
|
|
1976
|
-
|
|
1977
|
-
|
|
1978
|
-
|
|
1979
|
-
|
|
1980
|
-
* case to an item -- the runner, the Datasets tab, a mapping against a Source
|
|
1981
|
-
* -- goes through here.
|
|
1982
|
-
*/
|
|
1983
|
-
function caseFile(kase ) { // vocab: the older spelling
|
|
1984
|
-
const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
|
|
1985
|
-
return typeof name === "string" ? name : "";
|
|
2137
|
+
/** The item a case grades: its `item`, or "" for none. The one reader, so
|
|
2138
|
+
everything that joins a case to an item -- the runner, the Library, a
|
|
2139
|
+
run's Results -- joins on the same key, exactly. */
|
|
2140
|
+
function caseItem(kase ) {
|
|
2141
|
+
return isStr(kase?.item) ? kase .item : "";
|
|
1986
2142
|
}
|
|
1987
2143
|
|
|
1988
|
-
/**
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
|
|
1993
|
-
|
|
1994
|
-
|
|
1995
|
-
|
|
1996
|
-
|
|
1997
|
-
|
|
1998
|
-
|
|
1999
|
-
|
|
2000
|
-
|
|
2001
|
-
|
|
2002
|
-
|
|
2003
|
-
|
|
2004
|
-
|
|
2005
|
-
|
|
2006
|
-
|
|
2007
|
-
|
|
2008
|
-
if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
|
|
2009
|
-
else out[k] = v;
|
|
2010
|
-
}
|
|
2011
|
-
return out;
|
|
2012
|
-
}),
|
|
2013
|
-
};
|
|
2014
|
-
}
|
|
2144
|
+
/** One graded case's reading of a reply, now: its own metrics, scored as a
|
|
2145
|
+
Metrics eval scores them (all must pass; weight 0 is watched, never
|
|
2146
|
+
scored), with what they found, missed and invented carried up, and
|
|
2147
|
+
`watch` the readings of weight 0. What the page draws and the runner's
|
|
2148
|
+
case-by-case report writes. A model-graded metric needs a grader and the
|
|
2149
|
+
wait for one, so here it reads as not met: the runner's `--run` asks it. */
|
|
2150
|
+
|
|
2151
|
+
|
|
2152
|
+
|
|
2153
|
+
|
|
2154
|
+
|
|
2155
|
+
|
|
2156
|
+
|
|
2157
|
+
|
|
2158
|
+
|
|
2159
|
+
|
|
2160
|
+
|
|
2161
|
+
|
|
2162
|
+
|
|
2163
|
+
|
|
2015
2164
|
|
|
2016
|
-
|
|
2017
|
-
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
* were wanted has not done what was asked.
|
|
2024
|
-
*
|
|
2025
|
-
* Pure on purpose, and returning `reasons` as plain sentences rather than
|
|
2026
|
-
* markup. The dashboard was the first consumer; `run-evals.js` is the second
|
|
2027
|
-
* and CI the third, and neither can reach into a page for a rendered cell.
|
|
2028
|
-
* Issue #248 wants a failure written out as a task an agent can act on.
|
|
2029
|
-
*/
|
|
2030
|
-
function scoreCase(kase , res ) {
|
|
2031
|
-
// A case that expects its answer discarded passes on a discard and on
|
|
2032
|
-
// nothing else: the answer the job threw away is the finding.
|
|
2033
|
-
if (kase.discarded === true) {
|
|
2034
|
-
const thrown = !!res.error && res.error.startsWith("discarded: ");
|
|
2035
|
-
const n = (res.terms || []).length;
|
|
2036
|
-
return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
|
|
2037
|
-
unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
|
|
2038
|
-
reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
|
|
2039
|
-
}
|
|
2040
|
-
// A discarded answer is a failed item: whatever the pipeline produced
|
|
2041
|
-
// along the way does not count.
|
|
2042
|
-
const discarded = !!res.error;
|
|
2043
|
-
const terms = discarded ? [] : (res.terms || []);
|
|
2044
|
-
const expect = kase.expect || [], forbid = kase.forbid || [];
|
|
2045
|
-
|
|
2046
|
-
// `missed` stays populated for a discarded reply even though it reads as
|
|
2047
|
-
// vacuous, because the score depends on it: docs/datasets.md has such a reply
|
|
2048
|
-
// scoring zero against everything the entry asked for, groups included, and
|
|
2049
|
-
// `addToTally` gets there through `found + missed + invented`. Clear it and
|
|
2050
|
-
// a run that discarded every item would contribute nothing to the
|
|
2051
|
-
// total instead of contributing a nought, which flatters it.
|
|
2052
|
-
//
|
|
2053
|
-
// One member of a group is enough. Without this rule the set manufactures
|
|
2054
|
-
// failures out of synonyms.
|
|
2055
|
-
//
|
|
2056
|
-
// A requirement reads back as a term when it had one member and as the
|
|
2057
|
-
// group itself where a synonym list was allowed, so `missed` carries just
|
|
2058
|
-
// enough to state the reason: "FAIL: Missed dog" for a plain expect term,
|
|
2059
|
-
// "none of dog / puppy" for a group.
|
|
2060
|
-
//
|
|
2061
|
-
// Nothing satisfies a group when nothing was stored, so a discarded reply
|
|
2062
|
-
// misses every requirement rather than none -- the same cast that makes its
|
|
2063
|
-
// count bounds not breached by one. `unmet` is a finding only now that
|
|
2064
|
-
// `missed` names the unsatisfied groups itself, but the run report has
|
|
2065
|
-
// always carried it, so it stays.
|
|
2066
|
-
const met = (g ) => g.some(t => termIn(terms, t));
|
|
2067
|
-
const spoken = (g ) => g.length === 1 ? g[0] : g;
|
|
2068
|
-
const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
|
|
2069
|
-
const found = requirements.filter(met).map(spoken);
|
|
2070
|
-
const missed = requirements.filter(g => !met(g)).map(spoken);
|
|
2071
|
-
const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
|
|
2072
|
-
const unmet = discarded ? []
|
|
2073
|
-
: (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
|
|
2074
|
-
|
|
2075
|
-
const n = terms.length;
|
|
2076
|
-
// A null bound is not checked -- and neither is a bound on a reply that was
|
|
2077
|
-
// discarded. Nothing was counted out and found wanting there; the reply was
|
|
2078
|
-
// thrown away before it had a count, and saying "0 items, wanted at least 4"
|
|
2079
|
-
// states a second failure that never happened.
|
|
2080
|
-
const under = !discarded && kase.minCount != null && n < kase.minCount;
|
|
2081
|
-
const over = !discarded && kase.maxCount != null && n > kase.maxCount;
|
|
2082
|
-
|
|
2083
|
-
const denom = requirements.length + invented.length;
|
|
2084
|
-
// A case that passes is one whose whole expectation was met, not one that
|
|
2085
|
-
// scored well: an unsatisfied group is in `missed` alongside any term
|
|
2086
|
-
// missed, so pass needs nothing more than the terms already covered.
|
|
2087
|
-
const pass = !discarded && !missed.length && !invented.length && !under && !over;
|
|
2088
|
-
|
|
2089
|
-
// `found` and `missed` are groups, not terms, so the reasons word them one
|
|
2090
|
-
// by one: a bare term under one heading, a group on its own line.
|
|
2091
|
-
const reasons = [];
|
|
2092
|
-
if (discarded) {
|
|
2093
|
-
// Everything else would be derived from this one fact -- there are no
|
|
2094
|
-
// terms -- and would bury it. `scoreHtml` above suppresses `missed` for
|
|
2095
|
-
// the same reason: the reason that matters is already on the row.
|
|
2096
|
-
reasons.push(`Error - ${res.error}`);
|
|
2097
|
-
} else {
|
|
2098
|
-
const plain = missed.filter(t => typeof t === "string");
|
|
2099
|
-
if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
|
|
2100
|
-
for (const g of missed) {
|
|
2101
|
-
if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
|
|
2165
|
+
function readCase(kase , res , plain = false) {
|
|
2166
|
+
const input = metricInput(res , kase, { plain });
|
|
2167
|
+
const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
|
|
2168
|
+
const r = readMetric(m, input, {});
|
|
2169
|
+
if (r && typeof (r ).then === "function") {
|
|
2170
|
+
return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
|
|
2171
|
+
pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
|
|
2102
2172
|
}
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2110
|
-
|
|
2111
|
-
|
|
2112
|
-
|
|
2113
|
-
// beside an entry that asked for nothing -- a shape `evals-check.js` allows
|
|
2114
|
-
// even though every graded case now names an expectation.
|
|
2115
|
-
return { pass, score: denom ? found.length / denom : null, discarded,
|
|
2116
|
-
found, missed, invented, unmet, under, over, count: n,
|
|
2117
|
-
watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
|
|
2118
|
-
reasons };
|
|
2173
|
+
return r ? [r ] : [];
|
|
2174
|
+
});
|
|
2175
|
+
const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
|
|
2176
|
+
const counted = metrics.filter(r => r.weight !== 0);
|
|
2177
|
+
return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
|
|
2178
|
+
discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
|
|
2179
|
+
// A reply the job threw away fails on that one fact, which would
|
|
2180
|
+
// be buried under everything that derives from it.
|
|
2181
|
+
reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
|
|
2182
|
+
watch: metrics.filter(r => r.weight === 0) };
|
|
2119
2183
|
}
|
|
2120
2184
|
|
|
2121
2185
|
/**
|
|
@@ -2156,11 +2220,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
|
|
|
2156
2220
|
// docs/datasets.md: an item naming more terms should weigh more than
|
|
2157
2221
|
// one naming fewer, and averaging percentages lets the easiest case carry the
|
|
2158
2222
|
// score.
|
|
2159
|
-
function addToTally(t , s
|
|
2223
|
+
function addToTally(t , s ) {
|
|
2224
|
+
const found = s.found?.length ?? 0;
|
|
2160
2225
|
t.ran++;
|
|
2161
2226
|
if (s.pass) t.passed++;
|
|
2162
|
-
t.found +=
|
|
2163
|
-
t.of +=
|
|
2227
|
+
t.found += found;
|
|
2228
|
+
t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
|
|
2164
2229
|
}
|
|
2165
2230
|
|
|
2166
2231
|
// The headline percentage, in one place because it is a number and this file
|
|
@@ -2175,7 +2240,7 @@ function tallyPercent(t ) {
|
|
|
2175
2240
|
// A whole-run assertion with no graded set: All of / Any of / None of, a
|
|
2176
2241
|
// parse, a count, an exact reply and a length bound. It lives in the shared
|
|
2177
2242
|
// core because it is a pass/fail the tab, the worker and History all have to
|
|
2178
|
-
// agree on -- the same one-definition rule that keeps
|
|
2243
|
+
// agree on -- the same one-definition rule that keeps the case reader here.
|
|
2179
2244
|
function parseCount(raw , parse ) {
|
|
2180
2245
|
// An unparsed reply is one result, not no result. It used to return null,
|
|
2181
2246
|
// which made a Count of 1 fail as "null results" against a reply that
|
|
@@ -2432,7 +2497,7 @@ function jobProblem(stages , call
|
|
|
2432
2497
|
// 1 the way `textPrompt` says, and any later stage that places {text}.
|
|
2433
2498
|
async function runPipeline(stages , dataUrl , call ,
|
|
2434
2499
|
opts = {}) {
|
|
2435
|
-
const { tokens = TOKEN_DEFAULTS, mode, text = null } = opts;
|
|
2500
|
+
const { tokens = TOKEN_DEFAULTS, mode, text = null, record = null } = opts;
|
|
2436
2501
|
const transcript = [];
|
|
2437
2502
|
const problem = jobProblem(stages, call, tokens, text);
|
|
2438
2503
|
if (problem) return { error: problem, terms: [], transcript, ms: 0 };
|
|
@@ -2466,8 +2531,15 @@ async function runPipeline(stages , dataUrl , cal
|
|
|
2466
2531
|
forwarded.set("reply", prev.reply);
|
|
2467
2532
|
for (const [name, v] of Object.entries(prev.rendered)) forwarded.set(name, v);
|
|
2468
2533
|
}
|
|
2469
|
-
// A verbatim stage's words are
|
|
2470
|
-
|
|
2534
|
+
// A verbatim stage's words are its target step's own template, sent as written.
|
|
2535
|
+
// An expression token that cannot be read is the stage's error, said
|
|
2536
|
+
// before anything is sent.
|
|
2537
|
+
let instruction ;
|
|
2538
|
+
try { instruction = st.verbatim ? st.text : resolvePrompt(st.text, tokenSet(tokens, i), record); }
|
|
2539
|
+
catch (e) {
|
|
2540
|
+
if (!(e instanceof WdlError)) throw e;
|
|
2541
|
+
return result({ error: named(e.message) });
|
|
2542
|
+
}
|
|
2471
2543
|
const fill = (to ) => instruction.replace(TOKEN, (m, name ) =>
|
|
2472
2544
|
forwarded.has(name) ? to(forwarded.get(name) ) : m);
|
|
2473
2545
|
const sent = st.verbatim ? instruction : i === 0 && text != null ? textPrompt(instruction, text) : fill(v => v);
|
|
@@ -2558,10 +2630,10 @@ function applyModifiers (list , kind ,
|
|
|
2558
2630
|
// queue and run-evals.js alike, and every one of them reads it through the
|
|
2559
2631
|
// functions below rather than through a translation of its own.
|
|
2560
2632
|
//
|
|
2561
|
-
// The lab is generic, so what a reply is, what
|
|
2633
|
+
// The lab is generic, so what a reply is, what an eval scores and how a value
|
|
2562
2634
|
// is changed are registry entries. The ones here are the lab's own, and so
|
|
2563
2635
|
// are kinds/list.ts's -- the List kind and its modifiers. A dataset
|
|
2564
|
-
// registers nothing: it is data a graded
|
|
2636
|
+
// registers nothing: it is data a graded eval names by id, and a run carries.
|
|
2565
2637
|
|
|
2566
2638
|
// 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
|
|
2567
2639
|
// version 1 kept them as two maps (tokens: { values, blocks }).
|
|
@@ -2577,10 +2649,16 @@ function applyModifiers (list , kind ,
|
|
|
2577
2649
|
// 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
|
|
2578
2650
|
// 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
|
|
2579
2651
|
// Call (Prompt: the image flag and token mappings), Read Reply (the output).
|
|
2580
|
-
// 9:
|
|
2652
|
+
// 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
|
|
2581
2653
|
// they convert to (LEGACY_TESTS' toMetrics, proven equal by
|
|
2582
2654
|
// metrics-parity-check.js).
|
|
2583
|
-
|
|
2655
|
+
// 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
|
|
2656
|
+
// Content, the flow step, Attach Image, Token mappings) and Responses (Read
|
|
2657
|
+
// as, Response Format Validation, modifiers) -- and its call is gone: each
|
|
2658
|
+
// scenario is a target, whose own step in each job is what it sends there.
|
|
2659
|
+
// 11: `tests` are `evals`: the key renames and nothing in an eval changes,
|
|
2660
|
+
// so a stored result's scores, keyed by eval id, read as they did.
|
|
2661
|
+
const PIPELINE_VERSION = 12 ;
|
|
2584
2662
|
|
|
2585
2663
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2586
2664
|
// to know which registered it.
|
|
@@ -2588,7 +2666,7 @@ const STEP_TYPES = Object.create(null);
|
|
|
2588
2666
|
const CONTENT_TYPES = Object.create(null);
|
|
2589
2667
|
const OUTPUT_KINDS = Object.create(null);
|
|
2590
2668
|
const MODIFIERS = Object.create(null);
|
|
2591
|
-
const
|
|
2669
|
+
const EVAL_TYPES = Object.create(null);
|
|
2592
2670
|
const METRICS = Object.create(null);
|
|
2593
2671
|
const SOURCE_TYPES = Object.create(null);
|
|
2594
2672
|
|
|
@@ -2602,11 +2680,15 @@ let defaultOutputKind = "text";
|
|
|
2602
2680
|
const defaultKind = () => defaultOutputKind;
|
|
2603
2681
|
const kindOf = (st ) => st.kind ?? defaultOutputKind;
|
|
2604
2682
|
|
|
2683
|
+
/** A module's eval types, under either spelling (Kinds.testTypes). */
|
|
2684
|
+
const evalTypesOf = (k ) =>
|
|
2685
|
+
({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
|
|
2686
|
+
|
|
2605
2687
|
/** A module's entries into the registries, in one call. */
|
|
2606
2688
|
function registerKinds(k ) {
|
|
2607
2689
|
Object.assign(OUTPUT_KINDS, k.outputKinds || {});
|
|
2608
2690
|
Object.assign(MODIFIERS, k.modifiers || {});
|
|
2609
|
-
Object.assign(
|
|
2691
|
+
Object.assign(EVAL_TYPES, evalTypesOf(k));
|
|
2610
2692
|
Object.assign(SOURCE_TYPES, k.sourceTypes || {});
|
|
2611
2693
|
Object.assign(METRICS, k.metrics || {});
|
|
2612
2694
|
if (k.defaultKind) defaultOutputKind = k.defaultKind;
|
|
@@ -2637,7 +2719,7 @@ function pluginHost(pluginId ) {
|
|
|
2637
2719
|
registerKinds(k) {
|
|
2638
2720
|
taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
|
|
2639
2721
|
taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
|
|
2640
|
-
taken(
|
|
2722
|
+
taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
|
|
2641
2723
|
taken(METRICS, "metric", Object.keys(k.metrics || {}));
|
|
2642
2724
|
// The server keeps its own copy of the Source types, to refuse a row of
|
|
2643
2725
|
// one it does not know, and a plugin never reaches the server's code.
|
|
@@ -2741,8 +2823,7 @@ function encoderQuality(quality ) {
|
|
|
2741
2823
|
// always shown, so a sentence names what the reader sees.
|
|
2742
2824
|
const jobLabel = (doc , k ) =>
|
|
2743
2825
|
(doc.jobs?.[k]?.name || "").trim() || `Job ${k + 1}`;
|
|
2744
|
-
const scenarioLabel = (doc
|
|
2745
|
-
(doc.scenarios?.[i]?.name || "").trim() || `Scenario ${i + 1}`;
|
|
2826
|
+
const scenarioLabel = (doc , i ) => targetLabel(doc, i);
|
|
2746
2827
|
// The runner calls a job's link a stage; the model calls it a job.
|
|
2747
2828
|
const jobWords = (s ) => s && String(s).replace(/\bstage[ ](\d+)/g, "job $1");
|
|
2748
2829
|
|
|
@@ -2850,7 +2931,7 @@ CONTENT_TYPES.text = {
|
|
|
2850
2931
|
// being the recorded reply.
|
|
2851
2932
|
CONTENT_TYPES.prompt = {
|
|
2852
2933
|
label: "Prompt only",
|
|
2853
|
-
description: "No item: each
|
|
2934
|
+
description: "No item: each target's prompt is sent on its own.",
|
|
2854
2935
|
inline: true,
|
|
2855
2936
|
bare: true,
|
|
2856
2937
|
fields: ["type"],
|
|
@@ -2904,11 +2985,9 @@ LEGACY_TESTS.single = {
|
|
|
2904
2985
|
},
|
|
2905
2986
|
};
|
|
2906
2987
|
|
|
2907
|
-
// A dataset's cases, scored item by item
|
|
2908
|
-
//
|
|
2909
|
-
//
|
|
2910
|
-
// yields, so it accepts every kind that yields any: plain text has none, and
|
|
2911
|
-
// would fail every case.
|
|
2988
|
+
// A dataset's cases, scored item by item. The eval references the dataset
|
|
2989
|
+
// the way a pipeline references a Source, and the run carries the body it
|
|
2990
|
+
// was submitted against.
|
|
2912
2991
|
LEGACY_TESTS.graded = {
|
|
2913
2992
|
label: "Graded set",
|
|
2914
2993
|
fields: ["type", "dataset"],
|
|
@@ -2916,16 +2995,15 @@ LEGACY_TESTS.graded = {
|
|
|
2916
2995
|
validate(t, ctx, bad){
|
|
2917
2996
|
const d = t.dataset;
|
|
2918
2997
|
if (!isRef(d) || (d.version != null && !isStr(d.version))) {
|
|
2919
|
-
return void bad.push("a graded
|
|
2998
|
+
return void bad.push("a graded eval has to name its dataset as { id, name }");
|
|
2920
2999
|
}
|
|
2921
3000
|
if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
|
|
2922
3001
|
},
|
|
2923
|
-
score: (t, kase, res) => scoreCase(kase, res),
|
|
2924
3002
|
};
|
|
2925
3003
|
|
|
2926
3004
|
// Metrics: checks of each reply, deterministic or model-graded, from the
|
|
2927
3005
|
// METRICS registry (metrics/builtin.ts registers the lab's own), with the
|
|
2928
|
-
//
|
|
3006
|
+
// eval's own list for every item and a case's `metrics` for its item. Scored
|
|
2929
3007
|
// all-must-pass -- every metric passes -- or weighted: points, each metric's
|
|
2930
3008
|
// score times its weight, against a threshold (#150's points, a negative
|
|
2931
3009
|
// weight taking them away).
|
|
@@ -2934,6 +3012,13 @@ LEGACY_TESTS.graded = {
|
|
|
2934
3012
|
const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
|
|
2935
3013
|
const SCORING_MODES = ["all", "weighted"];
|
|
2936
3014
|
|
|
3015
|
+
/** A metric's list option -- Contains all's values, Contains's exceptions --
|
|
3016
|
+
one entry a line, or a list of them. */
|
|
3017
|
+
function metricLines(v ) {
|
|
3018
|
+
const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
|
|
3019
|
+
return all.map(x => x.trim()).filter(Boolean);
|
|
3020
|
+
}
|
|
3021
|
+
|
|
2937
3022
|
/** What is wrong with a list of metrics, as sentences naming [at]. */
|
|
2938
3023
|
function metricsProblems(list , at , bad ) {
|
|
2939
3024
|
if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
|
|
@@ -3003,14 +3088,16 @@ function readMetric(m , input , ctx )
|
|
|
3003
3088
|
} catch (e) { return timed(failed(e)); }
|
|
3004
3089
|
}
|
|
3005
3090
|
|
|
3006
|
-
/**
|
|
3091
|
+
/** An eval's score from its metrics' readings, the way [mode] says: all must
|
|
3007
3092
|
pass, or weighted points against [threshold]. What a case metric found,
|
|
3008
3093
|
missed and invented is carried up, for a run's totals. Null for none. */
|
|
3009
3094
|
function scoreOf(metrics , mode = "all", threshold = null) {
|
|
3010
3095
|
if (!metrics.length) return null;
|
|
3011
|
-
|
|
3012
|
-
|
|
3013
|
-
|
|
3096
|
+
// A watched reading (weight 0) is reported, and counts toward nothing.
|
|
3097
|
+
const scored = metrics.filter(r => r.weight !== 0);
|
|
3098
|
+
const detail = scored.some(r => r.found || r.missed || r.invented) ? {
|
|
3099
|
+
found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
|
|
3100
|
+
invented: scored.flatMap(r => r.invented ?? []) } : {};
|
|
3014
3101
|
if (mode === "weighted") {
|
|
3015
3102
|
const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
|
|
3016
3103
|
return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
|
|
@@ -3053,7 +3140,7 @@ function readRun(list , run ) {
|
|
|
3053
3140
|
});
|
|
3054
3141
|
}
|
|
3055
3142
|
|
|
3056
|
-
|
|
3143
|
+
EVAL_TYPES.metrics = {
|
|
3057
3144
|
label: "Metrics",
|
|
3058
3145
|
description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
|
|
3059
3146
|
fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
|
|
@@ -3077,16 +3164,16 @@ TEST_TYPES.metrics = {
|
|
|
3077
3164
|
if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
|
|
3078
3165
|
if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
|
|
3079
3166
|
if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
|
|
3080
|
-
// The
|
|
3167
|
+
// The eval's own grader, or the lab's.
|
|
3081
3168
|
const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3082
|
-
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the
|
|
3169
|
+
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
|
|
3083
3170
|
// A grader is asked words, and needs a model to ask.
|
|
3084
3171
|
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3085
|
-
if (grader && ctx.profiles && !p) bad.push(`
|
|
3172
|
+
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
3086
3173
|
else if (p) {
|
|
3087
3174
|
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3088
3175
|
if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
|
|
3089
|
-
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's
|
|
3176
|
+
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
|
|
3090
3177
|
}
|
|
3091
3178
|
},
|
|
3092
3179
|
rules: t => (Array.isArray(t.metrics) ? t.metrics : []).map((m , x ) => ({
|
|
@@ -3094,7 +3181,7 @@ TEST_TYPES.metrics = {
|
|
|
3094
3181
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3095
3182
|
.filter(Boolean).join(" ") })),
|
|
3096
3183
|
profiles: t => (isRef(t.grader) ? [t.grader] : []),
|
|
3097
|
-
// A run carries the lab's grader on
|
|
3184
|
+
// A run carries the lab's grader on an eval that names none and may ask one:
|
|
3098
3185
|
// a model-graded metric of its own, or a case's, over a dataset.
|
|
3099
3186
|
resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
|
|
3100
3187
|
? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
|
|
@@ -3109,44 +3196,28 @@ TEST_TYPES.metrics = {
|
|
|
3109
3196
|
ran: run.replies.length,
|
|
3110
3197
|
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3111
3198
|
},
|
|
3112
|
-
// Rule m{x} is the
|
|
3199
|
+
// Rule m{x} is the eval's own metric x, read in that place on every item.
|
|
3113
3200
|
// A score stored before Metrics -- a Graded set's, read as its conversion --
|
|
3114
3201
|
// has no readings of its own: its one metric's reading is the score's.
|
|
3115
3202
|
ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
|
|
3116
3203
|
: Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
|
|
3117
|
-
// Every item, with its case's own metrics where
|
|
3118
|
-
|
|
3204
|
+
// Every item, with its case's own metrics where the eval names the
|
|
3205
|
+
// dataset and the item has a case there.
|
|
3206
|
+
read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
|
|
3119
3207
|
metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
|
|
3120
3208
|
};
|
|
3121
3209
|
|
|
3122
3210
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3123
|
-
// What the
|
|
3124
|
-
//
|
|
3125
|
-
//
|
|
3126
|
-
// each is the core's own matcher.
|
|
3127
|
-
|
|
3128
|
-
// An item's case, as the Graded set scores it: its expectations, forbidden
|
|
3129
|
-
// terms and count bounds, with what it found, missed and invented kept.
|
|
3130
|
-
METRICS.case = {
|
|
3131
|
-
label: "Matches its case",
|
|
3132
|
-
description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
|
|
3133
|
-
perItem: true,
|
|
3134
|
-
needsTerms: true,
|
|
3135
|
-
options: [],
|
|
3136
|
-
defaults: () => ({}),
|
|
3137
|
-
score(input) {
|
|
3138
|
-
if (!input.kase) return null;
|
|
3139
|
-
const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
|
|
3140
|
-
return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
|
|
3141
|
-
found: s.found, missed: s.missed, invented: s.invented };
|
|
3142
|
-
},
|
|
3143
|
-
};
|
|
3211
|
+
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3212
|
+
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
3213
|
+
// rather than in metrics/builtin.ts because each is the core's own matcher.
|
|
3144
3214
|
|
|
3145
3215
|
// Items the reply holds -- all of them, or any -- matched as the lab matches
|
|
3146
3216
|
// a term; a kind that yields none is matched in the replies' text instead,
|
|
3147
3217
|
// ignoring case, as the Single Test did.
|
|
3148
3218
|
METRICS["has-items"] = {
|
|
3149
3219
|
label: "Has items",
|
|
3220
|
+
family: "The reply's text",
|
|
3150
3221
|
description: "Passes when the reply holds all, or any, of the items listed.",
|
|
3151
3222
|
options: [{ key: "values", label: "Items", type: "textarea" },
|
|
3152
3223
|
{ key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
|
|
@@ -3166,6 +3237,7 @@ METRICS["has-items"] = {
|
|
|
3166
3237
|
// A reply's length in characters, once trimmed.
|
|
3167
3238
|
METRICS.length = {
|
|
3168
3239
|
label: "Length",
|
|
3240
|
+
family: "The reply's text",
|
|
3169
3241
|
description: "Compares the reply's length in characters.",
|
|
3170
3242
|
options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
|
|
3171
3243
|
{ key: "n", label: "Characters", type: "number" }],
|
|
@@ -3181,6 +3253,7 @@ METRICS.length = {
|
|
|
3181
3253
|
// as the count says.
|
|
3182
3254
|
METRICS["parse-count"] = {
|
|
3183
3255
|
label: "Parses",
|
|
3256
|
+
family: "The result",
|
|
3184
3257
|
description: "Passes when the reply reads as the format chosen, holding the number of results set.",
|
|
3185
3258
|
// Unformatted is one result a reply, as the Single Test counted it.
|
|
3186
3259
|
options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
|
|
@@ -3199,12 +3272,13 @@ METRICS["parse-count"] = {
|
|
|
3199
3272
|
},
|
|
3200
3273
|
};
|
|
3201
3274
|
|
|
3202
|
-
/** What every
|
|
3275
|
+
/** What every eval carries, kept across a conversion. */
|
|
3203
3276
|
const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
|
|
3204
3277
|
|
|
3205
|
-
// A Graded set is a Metrics
|
|
3278
|
+
// A Graded set is a Metrics eval over the same dataset, with none of its own:
|
|
3279
|
+
// each case's metrics are what it scored (dataset-parity-check.js).
|
|
3206
3280
|
LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
|
|
3207
|
-
dataset: t.dataset, over: "item", metrics: [
|
|
3281
|
+
dataset: t.dataset, over: "item", metrics: [] });
|
|
3208
3282
|
|
|
3209
3283
|
// A Single Test is Metrics over the whole run: its lists over the run's items
|
|
3210
3284
|
// together, and exact, length and parse over each reply alone.
|
|
@@ -3222,7 +3296,7 @@ LEGACY_TESTS.single .toMetrics = t => {
|
|
|
3222
3296
|
return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
|
|
3223
3297
|
};
|
|
3224
3298
|
|
|
3225
|
-
/** The grader a Metrics
|
|
3299
|
+
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3226
3300
|
function graderCtx(t , more ) {
|
|
3227
3301
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3228
3302
|
return ask ? { ask } : {};
|
|
@@ -3232,27 +3306,33 @@ function graderCtx(t , more ) {
|
|
|
3232
3306
|
for one with none set: what History prints beside its name. */
|
|
3233
3307
|
function metricSummary(m ) {
|
|
3234
3308
|
const entry = METRICS[m.type];
|
|
3235
|
-
// Each option as it reads in the editor: a choice's label, a box
|
|
3236
|
-
//
|
|
3309
|
+
// Each option as it reads in the editor: a choice's label, a box by its
|
|
3310
|
+
// own label where it is not as a new metric has it (ticked, or "off"),
|
|
3311
|
+
// text by its first line.
|
|
3312
|
+
const fresh = entry?.defaults() ?? {};
|
|
3237
3313
|
const said = (entry?.options || []).map(o => {
|
|
3238
3314
|
const v = m[o.key];
|
|
3239
3315
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3240
|
-
if (o.type === "checkbox") return v ? o.label.toLowerCase() :
|
|
3316
|
+
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
3241
3317
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3242
3318
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3243
3319
|
}).filter(Boolean);
|
|
3244
3320
|
return said.join(", ");
|
|
3245
3321
|
}
|
|
3246
3322
|
|
|
3247
|
-
// ---- a job's steps (pipeline-model §
|
|
3323
|
+
// ---- a job's steps, and a target's (pipeline-model §16) ----------------------
|
|
3324
|
+
|
|
3325
|
+
// Content: what a job is given. Attach Content and the flow step are job
|
|
3326
|
+
// 1's; any job may send the item's image and map tokens.
|
|
3248
3327
|
|
|
3249
3328
|
// Attach Content: the items a run goes over, job 1's first step.
|
|
3250
3329
|
STEP_TYPES.attachContent = {
|
|
3251
|
-
label: "Attach Content", slot: "content",
|
|
3252
|
-
description: "The items the run goes through: each is sent to every
|
|
3330
|
+
label: "Attach Content", slot: "content", rank: 0,
|
|
3331
|
+
description: "The items the run goes through: each is sent to every target.",
|
|
3253
3332
|
in: null, out: "items",
|
|
3254
3333
|
apply: "itemContent", // run-evals.js, through its file-type registry
|
|
3255
3334
|
fields: ["type", "content"],
|
|
3335
|
+
firstJobOnly: "only job 1 attaches content -- a later job is handed the reply before it",
|
|
3256
3336
|
validate(step, ctx, bad){
|
|
3257
3337
|
const c = step?.content;
|
|
3258
3338
|
if (!c) return void bad.push("set the content first");
|
|
@@ -3263,20 +3343,65 @@ STEP_TYPES.attachContent = {
|
|
|
3263
3343
|
},
|
|
3264
3344
|
};
|
|
3265
3345
|
|
|
3266
|
-
//
|
|
3346
|
+
// The flow step: the Power Automate action a job's records are calls of.
|
|
3347
|
+
STEP_TYPES.flowStep = {
|
|
3348
|
+
label: "Flow step", slot: "content", rank: 1,
|
|
3349
|
+
description: "The Power Automate step each record is a call of.",
|
|
3350
|
+
in: "item", out: "item",
|
|
3351
|
+
apply: "runPipeline",
|
|
3352
|
+
fields: ["type", "step", "loop", "api"],
|
|
3353
|
+
firstJobOnly: "a flow step's records are job 1's items",
|
|
3354
|
+
// Production's reply is the record's result, read as the flow reads it.
|
|
3355
|
+
production(flow, record){
|
|
3356
|
+
const result = isObj(record) && isObj(record.result) ? record.result : null;
|
|
3357
|
+
if (!result || typeof result.status !== "number") return null;
|
|
3358
|
+
try {
|
|
3359
|
+
return httpReplyOf(flow, { prompt: "" }, result.body, result.status,
|
|
3360
|
+
{ scope: record.scope || {}, now: record.at ?? null }).raw;
|
|
3361
|
+
} catch { return null; }
|
|
3362
|
+
},
|
|
3363
|
+
validate(step, _ctx, bad, at){
|
|
3364
|
+
if (!isStr(step.step) || !step.step.trim()) bad.push(`${at}: a flow step names the action it is`);
|
|
3365
|
+
if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
|
|
3366
|
+
if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3367
|
+
},
|
|
3368
|
+
};
|
|
3369
|
+
|
|
3370
|
+
// Attach Image: the item's image beside every target's words.
|
|
3371
|
+
STEP_TYPES.attachImage = {
|
|
3372
|
+
label: "Attach Image", slot: "content", rank: 2,
|
|
3373
|
+
description: "Sends each item's image beside the words.",
|
|
3374
|
+
in: "item", out: "item",
|
|
3375
|
+
apply: "runPipeline",
|
|
3376
|
+
fields: ["type"],
|
|
3377
|
+
validate(){},
|
|
3378
|
+
};
|
|
3379
|
+
|
|
3380
|
+
// Token mappings: what the words' {tokens} are resolved to.
|
|
3381
|
+
STEP_TYPES.tokenMappings = {
|
|
3382
|
+
label: "Token mappings", slot: "content", rank: 3,
|
|
3383
|
+
description: "What each {token} in the words is filled in with.",
|
|
3384
|
+
in: "item", out: "item",
|
|
3385
|
+
apply: "runPipeline",
|
|
3386
|
+
fields: ["type", "tokenMappings"],
|
|
3387
|
+
validate(step, _ctx, bad, at){ tokenMappingsProblems(step.tokenMappings, at, bad); },
|
|
3388
|
+
};
|
|
3389
|
+
|
|
3390
|
+
// Targets: what each target sends. A target's own steps, one per job.
|
|
3391
|
+
|
|
3392
|
+
// Prompt: a target's words, asked of its model.
|
|
3267
3393
|
STEP_TYPES.prompt = {
|
|
3268
|
-
label: "Prompt", slot: "
|
|
3269
|
-
description: "Asks each
|
|
3394
|
+
label: "Prompt", slot: "target", asks: "prompt", modelFrom: "profile",
|
|
3395
|
+
description: "Asks each target's model its prompt about the item.",
|
|
3270
3396
|
in: "item", out: "text",
|
|
3271
3397
|
apply: "runPipeline",
|
|
3272
|
-
fields: ["type", "
|
|
3398
|
+
fields: ["type", "prompt", "profile", "from"],
|
|
3273
3399
|
validate(step, _ctx, bad, at){
|
|
3274
|
-
if (
|
|
3275
|
-
tokenMappingsProblems(step.tokenMappings, at, bad);
|
|
3400
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3276
3401
|
},
|
|
3277
3402
|
};
|
|
3278
3403
|
|
|
3279
|
-
// HTTP Request: a
|
|
3404
|
+
// HTTP Request: a target that sends a Power Automate step's own request,
|
|
3280
3405
|
// rebuilt for each item from its record (docs/power-automate.md).
|
|
3281
3406
|
const HTTP_METHODS = ["GET", "POST", "PUT", "PATCH", "DELETE"];
|
|
3282
3407
|
/** The first field under [v] named like a key, as a path, or null. */
|
|
@@ -3293,63 +3418,85 @@ function keyField(v , at = "body") {
|
|
|
3293
3418
|
return null;
|
|
3294
3419
|
}
|
|
3295
3420
|
STEP_TYPES.httpRequest = {
|
|
3296
|
-
label: "HTTP Request", slot: "
|
|
3297
|
-
description: "Sends a Power Automate step's own request, rebuilt from each record, with each
|
|
3298
|
-
|
|
3299
|
-
|
|
3300
|
-
|
|
3301
|
-
|
|
3302
|
-
|
|
3303
|
-
|
|
3304
|
-
|
|
3305
|
-
|
|
3306
|
-
|
|
3307
|
-
|
|
3308
|
-
|
|
3309
|
-
|
|
3310
|
-
|
|
3311
|
-
|
|
3312
|
-
|
|
3313
|
-
|
|
3421
|
+
label: "HTTP Request", slot: "target", asks: "request", verbatim: true, modelFrom: "step",
|
|
3422
|
+
description: "Sends a Power Automate step's own request, rebuilt from each record, with each target's words in it.",
|
|
3423
|
+
firstJobOnly: "an HTTP Request is sent from its item's record, so it is job 1's",
|
|
3424
|
+
// A new target starts from the flow's own words and fields.
|
|
3425
|
+
newStep: (step) => ({ type: "httpRequest", api: step.api, method: step.method, path: step.path,
|
|
3426
|
+
query: clone(step.query), body: clone(step.body),
|
|
3427
|
+
...(HTTP_APIS[step.api] ?? HTTP_APIS.raw ).cellOf(step.body) }),
|
|
3428
|
+
in: "item", out: "text",
|
|
3429
|
+
apply: "runPipeline",
|
|
3430
|
+
fields: ["type", "api", "method", "path", "query", "body", "prompt", "profile", "from", "system", "model", "settings", "prefill"],
|
|
3431
|
+
validate(step, _ctx, bad, at){
|
|
3432
|
+
const api = HTTP_APIS[step.api];
|
|
3433
|
+
if (!api) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3434
|
+
if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
|
|
3435
|
+
if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
|
|
3436
|
+
if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
|
|
3437
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3438
|
+
// A key is the Setup profile's, sent by its Key header; one in the
|
|
3439
|
+
// template would be kept in the pipeline and every run of it.
|
|
3440
|
+
const hit = keyField(step.body) ?? keyField(step.query, "query");
|
|
3441
|
+
if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Target profile sends it`);
|
|
3314
3442
|
if (!api) return;
|
|
3443
|
+
// Its fields, where its HTTP API has a place for them.
|
|
3315
3444
|
for (const f of ["system", "model", "prefill"] ) {
|
|
3316
|
-
if (
|
|
3317
|
-
if (!isStr(
|
|
3445
|
+
if (step[f] == null) continue;
|
|
3446
|
+
if (!isStr(step[f])) bad.push(`${at}: ${f} has to be text`);
|
|
3318
3447
|
else if (!api.fields.includes(f)) bad.push(`${at}: ${api.label} has no place for a ${f}`);
|
|
3319
3448
|
}
|
|
3320
|
-
if (
|
|
3321
|
-
if (!isObj(
|
|
3322
|
-
else for (const k of Object.keys(
|
|
3449
|
+
if (step.settings != null) {
|
|
3450
|
+
if (!isObj(step.settings) || !Object.values(step.settings).every(isStr)) bad.push(`${at}: settings are names to text`);
|
|
3451
|
+
else for (const k of Object.keys(step.settings)) {
|
|
3323
3452
|
if (!api.settings.includes(k)) bad.push(`${at}: ${api.label} has no setting ${k}`);
|
|
3324
3453
|
}
|
|
3325
3454
|
}
|
|
3326
|
-
if (isStr(
|
|
3327
|
-
try { api.place(
|
|
3455
|
+
if (isStr(step.prompt)) {
|
|
3456
|
+
try { api.place(step.body, step ); }
|
|
3328
3457
|
catch (e) { bad.push(`${at}: ${(e ).message}`); }
|
|
3329
3458
|
}
|
|
3330
3459
|
},
|
|
3460
|
+
};
|
|
3461
|
+
|
|
3462
|
+
// Echo: a target that sends nothing and needs no profile -- each text item
|
|
3463
|
+
// is its own reply, to grade replies recorded earlier (#97). Job 1's, over
|
|
3464
|
+
// text items only. Its words, where it has any, are what the reply is read
|
|
3465
|
+
// against; over Prompt only they are the reply.
|
|
3466
|
+
STEP_TYPES.echo = {
|
|
3467
|
+
label: "Echo", slot: "target", asks: "nothing", local: true, replaces: "echo",
|
|
3468
|
+
description: "Sends nothing: each text item is its own reply, to grade replies recorded earlier.",
|
|
3469
|
+
firstJobOnly: "Echo answers only job 1, with the item's own text",
|
|
3331
3470
|
in: "item", out: "text",
|
|
3332
3471
|
apply: "runPipeline",
|
|
3333
|
-
fields: ["type", "
|
|
3472
|
+
fields: ["type", "prompt", "from"],
|
|
3334
3473
|
validate(step, _ctx, bad, at){
|
|
3335
|
-
if (!isStr(step.
|
|
3336
|
-
if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3337
|
-
if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
|
|
3338
|
-
if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
|
|
3339
|
-
if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
|
|
3340
|
-
if (step.readAs != null && !isStr(step.readAs)) bad.push(`${at}: readAs is an expression or null`);
|
|
3341
|
-
if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
|
|
3342
|
-
// A key is the Setup profile's, sent by its Key header; one in the
|
|
3343
|
-
// template would be kept in the pipeline and every run of it.
|
|
3344
|
-
const hit = keyField(step.body) ?? keyField(step.query, "query");
|
|
3345
|
-
if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Setup profile sends it`);
|
|
3474
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3346
3475
|
},
|
|
3347
3476
|
};
|
|
3348
3477
|
|
|
3349
|
-
|
|
3478
|
+
/** What a local target step is answered as: Echo's own connection, with no
|
|
3479
|
+
address, key or model. */
|
|
3480
|
+
const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
|
|
3481
|
+
|
|
3482
|
+
// Responses: how a job's reply is read, in order -- possibly not at all.
|
|
3483
|
+
|
|
3484
|
+
// Read as: what the flow does to the reply before it reads it.
|
|
3485
|
+
STEP_TYPES.readAs = {
|
|
3486
|
+
label: "Read as", slot: "responses", rank: 0,
|
|
3487
|
+
description: "What the flow makes of the reply before it reads it.",
|
|
3488
|
+
in: "text", out: "text",
|
|
3489
|
+
apply: "runPipeline",
|
|
3490
|
+
fields: ["type", "readAs"],
|
|
3491
|
+
validate(step, _ctx, bad, at){
|
|
3492
|
+
if (!isStr(step.readAs) || !step.readAs.trim()) bad.push(`${at}: readAs is an expression`);
|
|
3493
|
+
},
|
|
3494
|
+
};
|
|
3495
|
+
|
|
3496
|
+
// Response Format Validation: the kind a reply is read as.
|
|
3350
3497
|
STEP_TYPES.readReply = {
|
|
3351
|
-
label: "Read Reply", slot: "
|
|
3352
|
-
description: "How the reply is read before
|
|
3498
|
+
label: "Read Reply", slot: "responses", rank: 1,
|
|
3499
|
+
description: "How the reply is read before evals and later jobs see it.",
|
|
3353
3500
|
in: "text", out: step => step.out?.kind,
|
|
3354
3501
|
// Read inside the stage: runPipeline parses each reply as it comes back.
|
|
3355
3502
|
apply: "runPipeline",
|
|
@@ -3360,30 +3507,42 @@ STEP_TYPES.readReply = {
|
|
|
3360
3507
|
return void bad.push(`${at} answers as "${out?.kind}", which is not an output kind this lab has`);
|
|
3361
3508
|
}
|
|
3362
3509
|
const kindEntry = OUTPUT_KINDS[out.kind] ;
|
|
3363
|
-
onlyFields(out, `${at}'s output`, ["kind",
|
|
3510
|
+
onlyFields(out, `${at}'s output`, ["kind", ...(kindEntry.settings || []).map(o => o.key)], bad);
|
|
3364
3511
|
kindEntry.validateSettings?.(out, `${at}'s output`, bad);
|
|
3365
|
-
|
|
3366
|
-
|
|
3367
|
-
|
|
3368
|
-
|
|
3369
|
-
|
|
3370
|
-
|
|
3371
|
-
|
|
3372
|
-
|
|
3512
|
+
},
|
|
3513
|
+
};
|
|
3514
|
+
|
|
3515
|
+
// A modifier: one change to the parsed reply, after the kind that takes it.
|
|
3516
|
+
STEP_TYPES.modifier = {
|
|
3517
|
+
label: "Modifier", slot: "responses", rank: 2, many: true,
|
|
3518
|
+
description: "Changes the reply after it is read.",
|
|
3519
|
+
in: "value", out: "value",
|
|
3520
|
+
apply: "runPipeline",
|
|
3521
|
+
fields: ["type", "modifier"],
|
|
3522
|
+
validate(step, _ctx, bad, at, kind){
|
|
3523
|
+
const m = step.modifier;
|
|
3524
|
+
const entry = isObj(m) && MODIFIERS[m.type];
|
|
3525
|
+
if (!entry) return void bad.push(`${at}: "${m?.type}" is not a modifier this lab has`);
|
|
3526
|
+
if (entry.accepts && !entry.accepts.includes(kind)) {
|
|
3527
|
+
bad.push(`${at}: ${entry.label || m.type} does not apply to ${OUTPUT_KINDS[kind]?.noun || kind}`);
|
|
3373
3528
|
}
|
|
3529
|
+
entry.validate?.(m , `${at}: ${entry.label || m.type}`, bad);
|
|
3374
3530
|
},
|
|
3375
3531
|
};
|
|
3376
3532
|
|
|
3377
|
-
// ---- reading a job's steps
|
|
3378
|
-
// Every reader asks these, never a step's position
|
|
3379
|
-
//
|
|
3533
|
+
// ---- reading a job's steps, and a target's -----------------------------------
|
|
3534
|
+
// Every reader asks these, never a step's position or the document's shape:
|
|
3535
|
+
// which steps a job holds is its own, and a version that moves a field
|
|
3536
|
+
// changes these alone (pipeline-model §16).
|
|
3537
|
+
|
|
3538
|
+
|
|
3380
3539
|
|
|
3381
3540
|
const slotOf = (st ) =>
|
|
3382
3541
|
(isObj(st) ? STEP_TYPES[st.type ]?.slot : undefined);
|
|
3383
3542
|
|
|
3384
|
-
/**
|
|
3385
|
-
const
|
|
3386
|
-
(job?.steps || []).find(st =>
|
|
3543
|
+
/** [job]'s step of [type], when it holds one. */
|
|
3544
|
+
const stepOf = (job , type ) =>
|
|
3545
|
+
(job?.steps || []).find(st => isObj(st) && st.type === type) ;
|
|
3387
3546
|
|
|
3388
3547
|
/** The registry entry that fills [slot] by default: what a new step there is. */
|
|
3389
3548
|
function slotEntry(slot ) {
|
|
@@ -3391,9 +3550,26 @@ function slotEntry(slot )
|
|
|
3391
3550
|
return hit ? { type: hit[0], entry: hit[1] } : undefined;
|
|
3392
3551
|
}
|
|
3393
3552
|
|
|
3553
|
+
/** Where a step sits: its stage, then its place in it. */
|
|
3554
|
+
const placeOf = (st ) => {
|
|
3555
|
+
const e = isObj(st) ? STEP_TYPES[st.type ] : undefined;
|
|
3556
|
+
return (e?.slot ? SLOTS.indexOf(e.slot) : SLOTS.length) * 100 + (e?.rank ?? 0);
|
|
3557
|
+
};
|
|
3558
|
+
|
|
3559
|
+
/** [steps] in stage order, each stage in its own order; steps that tie
|
|
3560
|
+
(modifiers) keep theirs. */
|
|
3561
|
+
const ordered = (steps ) =>
|
|
3562
|
+
steps.map((st, x) => ({ st, x })).sort((a, b) => placeOf(a.st) - placeOf(b.st) || a.x - b.x).map(o => o.st);
|
|
3563
|
+
|
|
3564
|
+
/** [job] with its step of [type] set to [step], or taken away for null. */
|
|
3565
|
+
function withStep(job , type , step ) {
|
|
3566
|
+
const steps = job.steps.filter(st => st.type !== type);
|
|
3567
|
+
return { ...job, steps: ordered(step ? [...steps, step] : steps) };
|
|
3568
|
+
}
|
|
3569
|
+
|
|
3394
3570
|
/** A pipeline's content: job 1's content step, or none. */
|
|
3395
3571
|
function contentOf(doc ) {
|
|
3396
|
-
return
|
|
3572
|
+
return stepOf (doc?.jobs?.[0], "attachContent")?.content ?? null;
|
|
3397
3573
|
}
|
|
3398
3574
|
|
|
3399
3575
|
/** [doc] with its content set -- job 1's content step added, replaced, or
|
|
@@ -3401,26 +3577,94 @@ function contentOf(doc )
|
|
|
3401
3577
|
function withContent (doc , content ) {
|
|
3402
3578
|
const [first, ...rest] = doc.jobs;
|
|
3403
3579
|
if (!first) return doc;
|
|
3404
|
-
|
|
3405
|
-
|
|
3406
|
-
|
|
3580
|
+
return { ...doc, jobs: [withStep(first, "attachContent", content ? { type: "attachContent", content } : null), ...rest] };
|
|
3581
|
+
}
|
|
3582
|
+
|
|
3583
|
+
/** A document's targets, in column order. */
|
|
3584
|
+
function targetsOf(doc ) {
|
|
3585
|
+
return Array.isArray(doc?.targets) ? doc .targets : [];
|
|
3586
|
+
}
|
|
3587
|
+
|
|
3588
|
+
/** Target [i]'s name, or the number it has always shown. */
|
|
3589
|
+
function targetLabel(doc , i ) {
|
|
3590
|
+
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
|
|
3591
|
+
}
|
|
3592
|
+
|
|
3593
|
+
/** What target [i] sends in job [k]: its step there. */
|
|
3594
|
+
function targetStepOf(doc , i , k ) {
|
|
3595
|
+
const st = targetsOf(doc)[i]?.steps?.[k];
|
|
3596
|
+
return isObj(st) ? st : undefined;
|
|
3597
|
+
}
|
|
3598
|
+
|
|
3599
|
+
/** The registry entry of target [i]'s step in job [k]: what it asks, and of whom. */
|
|
3600
|
+
const targetEntryOf = (doc , i , k ) =>
|
|
3601
|
+
STEP_TYPES[targetStepOf(doc, i, k)?.type ?? ""];
|
|
3602
|
+
|
|
3603
|
+
/** What job [k]'s targets send, as the job's editor shows it: the first
|
|
3604
|
+
target's step type there, or Prompt's while there is none. */
|
|
3605
|
+
function jobTargetType(doc , k ) {
|
|
3606
|
+
return targetStepOf(doc, 0, k)?.type ?? "prompt";
|
|
3607
|
+
}
|
|
3608
|
+
|
|
3609
|
+
/** The Setup profile target [i] asks in job [k]: its step's own, or the target's. */
|
|
3610
|
+
function targetProfileOf(doc , i , k ) {
|
|
3611
|
+
return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
|
|
3612
|
+
}
|
|
3613
|
+
|
|
3614
|
+
/** [doc] with target [i]'s step in job [k] changed: [change] merged into
|
|
3615
|
+
it, or the step [change] makes of it. */
|
|
3616
|
+
function withTarget (doc , i , k ,
|
|
3617
|
+
change ) {
|
|
3618
|
+
return { ...doc, targets: doc.targets.map((t, x) => (x !== i ? t : {
|
|
3619
|
+
...t, steps: t.steps.map((st, j) => (j !== k ? st : typeof change === "function" ? change(st) : { ...st, ...change } )),
|
|
3620
|
+
})) };
|
|
3621
|
+
}
|
|
3622
|
+
|
|
3623
|
+
/** The token mappings [job]'s words are resolved under. */
|
|
3624
|
+
const tokensOf = (job ) => stepOf (job, "tokenMappings")?.tokenMappings ?? [];
|
|
3625
|
+
|
|
3626
|
+
/** [job] resolving its words under [list]: none takes the step away. */
|
|
3627
|
+
const withTokens = (job , list ) =>
|
|
3628
|
+
withStep(job, "tokenMappings", list.length ? { type: "tokenMappings", tokenMappings: list } : null);
|
|
3629
|
+
|
|
3630
|
+
/** Whether [job] sends the item's image beside its words. */
|
|
3631
|
+
const sendsImage = (job ) => !!stepOf(job, "attachImage");
|
|
3632
|
+
|
|
3633
|
+
/** [job] sending the item's image, or not. */
|
|
3634
|
+
const withImage = (job , on ) => withStep(job, "attachImage", on ? { type: "attachImage" } : null);
|
|
3635
|
+
|
|
3636
|
+
/** The Power Automate step [job]'s records are calls of: its name, which the
|
|
3637
|
+
flow's expressions read it by, the loop item() means, and the HTTP API
|
|
3638
|
+
its replies speak. None for a job that is not one. */
|
|
3639
|
+
function flowStepOf(job ) {
|
|
3640
|
+
const f = stepOf (job, "flowStep");
|
|
3641
|
+
return f ? { step: f.step, loop: f.loop ?? null, api: f.api } : null;
|
|
3407
3642
|
}
|
|
3408
3643
|
|
|
3409
|
-
/**
|
|
3410
|
-
|
|
3411
|
-
|
|
3644
|
+
/** What the flow does to [job]'s reply before it reads it, as an expression
|
|
3645
|
+
over the step's body; null reads the reply's own text. */
|
|
3646
|
+
const readAsOf = (job ) => stepOf (job, "readAs")?.readAs ?? null;
|
|
3412
3647
|
|
|
3413
|
-
/**
|
|
3414
|
-
|
|
3415
|
-
|
|
3648
|
+
/** What a transport builds target [i]'s request in job [k] from: the
|
|
3649
|
+
target's step, with the flow step and reading the job holds. */
|
|
3650
|
+
function requestStepOf(doc , i , k ) {
|
|
3651
|
+
const job = doc.jobs[k];
|
|
3652
|
+
return { ...flowStepOf(job), ...targetStepOf(doc, i, k), readAs: readAsOf(job) } ;
|
|
3653
|
+
}
|
|
3654
|
+
|
|
3655
|
+
/** What httpRequestOf builds a request from: a target's template, and the
|
|
3656
|
+
loop its flow step sits in. */
|
|
3657
|
+
|
|
3658
|
+
/** What httpReplyOf reads a reply with: the flow step, its API, its reading. */
|
|
3659
|
+
|
|
3416
3660
|
|
|
3417
3661
|
/** The request an HTTP Request [step] sends for [cell] over an item whose
|
|
3418
3662
|
record read [ctx]'s scope: the cell's words and fields placed in the
|
|
3419
3663
|
flow's body by its HTTP API, then every expression evaluated. */
|
|
3420
|
-
function httpRequestOf(step
|
|
3664
|
+
function httpRequestOf(step , cell , ctx )
|
|
3421
3665
|
{
|
|
3422
3666
|
const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
|
|
3423
|
-
const at = { ...ctx, loop: step.loop };
|
|
3667
|
+
const at = { ...ctx, loop: step.loop ?? null };
|
|
3424
3668
|
const path = String(evaluate(step.path, at));
|
|
3425
3669
|
// A path an expression wrote whole -- an address -- keeps its path and query.
|
|
3426
3670
|
const [p, q] = /^https?:\/\//i.test(path) ? (() => { const u = new URL(path); return [u.pathname, u.search]; })() : [path, ""];
|
|
@@ -3431,35 +3675,50 @@ function httpRequestOf(step , cell , ctx
|
|
|
3431
3675
|
/** What the flow reads of a reply [j] to [step]: its readAs evaluated with the
|
|
3432
3676
|
reply as the step's body -- the prefill put back in front, as the flow has
|
|
3433
3677
|
to -- or the HTTP API's own text. */
|
|
3434
|
-
function httpReplyOf(step
|
|
3678
|
+
function httpReplyOf(step , cell , j , status ,
|
|
3435
3679
|
ctx ) {
|
|
3436
3680
|
const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
|
|
3437
3681
|
const whole = cell.prefill && api.withPrefill ? api.withPrefill(j, cell.prefill) : j;
|
|
3438
3682
|
const { raw: said, finishReason } = api.reply(whole);
|
|
3439
3683
|
if (!step.readAs) return { raw: said, said, finishReason };
|
|
3440
3684
|
const scope = { ...ctx.scope, actions: { ...(ctx.scope.actions || {}), [step.step]: { body: whole, statusCode: status } } };
|
|
3441
|
-
return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop })), said, finishReason };
|
|
3685
|
+
return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop ?? null })), said, finishReason };
|
|
3442
3686
|
}
|
|
3443
3687
|
|
|
3444
|
-
/** A
|
|
3445
|
-
|
|
3446
|
-
|
|
3447
|
-
|
|
3448
|
-
|
|
3449
|
-
const
|
|
3688
|
+
/** A reply [text] from a model asked in words, read as the flow reads its
|
|
3689
|
+
step's reply: put in the body the flow's API would have sent it in
|
|
3690
|
+
(`HttpApiEntry.wrap`), then read through the job's Read as -- so a
|
|
3691
|
+
model's words and production's are read alike. */
|
|
3692
|
+
function readFlowReply(flow , text , record ) {
|
|
3693
|
+
const api = HTTP_APIS[flow.api] ?? HTTP_APIS.raw ;
|
|
3694
|
+
const read = httpReplyOf({ ...flow, api: HTTP_APIS[flow.api] ? flow.api : "raw" }, { prompt: "" }, api.wrap(text), 200,
|
|
3695
|
+
{ scope: record.scope || {}, now: record.at ?? null });
|
|
3696
|
+
return { raw: read.raw, said: text };
|
|
3697
|
+
}
|
|
3450
3698
|
|
|
3451
|
-
/**
|
|
3452
|
-
|
|
3453
|
-
|
|
3699
|
+
/** A job's Response Format Validation, when it has one. */
|
|
3700
|
+
const replyOf = (job ) => stepOf (job, "readReply");
|
|
3701
|
+
|
|
3702
|
+
/** How a job's reply is read: its kind, settings and modifiers. A job with
|
|
3703
|
+
no Format Validation reads its reply as text -- the text kind, by name,
|
|
3704
|
+
so a stored document never changes meaning when a plugin registers a
|
|
3705
|
+
kind of its own. */
|
|
3706
|
+
function outOf(job ) {
|
|
3707
|
+
const read = replyOf(job)?.out ?? { kind: "text", ...(OUTPUT_KINDS.text?.settingDefaults?.() || {}) };
|
|
3708
|
+
const modifiers = (job?.steps || []).filter((st) => isObj(st) && st.type === "modifier").map(st => st.modifier);
|
|
3709
|
+
return { ...read, modifiers } ;
|
|
3454
3710
|
}
|
|
3455
3711
|
|
|
3456
|
-
/** [job] reading its reply as [out]. */
|
|
3457
|
-
|
|
3712
|
+
/** [job] reading its reply as [out]: its Format Validation and modifiers. */
|
|
3713
|
+
function withOut(job , out ) {
|
|
3714
|
+
const { modifiers, ...read } = out;
|
|
3715
|
+
const steps = job.steps.filter(st => st.type !== "readReply" && st.type !== "modifier");
|
|
3716
|
+
return { ...job, steps: ordered([...steps, { type: "readReply", out: read },
|
|
3717
|
+
...(modifiers || []).map(modifier => ({ type: "modifier", modifier }) )]) };
|
|
3718
|
+
}
|
|
3458
3719
|
|
|
3459
|
-
/** [job] with its call changed by [patch]. */
|
|
3460
|
-
const withCall = (job , patch ) => withSlot(job, "call", st => ({ ...st, ...patch } ));
|
|
3461
3720
|
/** A new id: random, so one minted in one browser never collides with one
|
|
3462
|
-
minted in another. A pipeline, its jobs and its
|
|
3721
|
+
minted in another. A pipeline, its jobs and its targets each carry one
|
|
3463
3722
|
(#93), minted once and never shown. */
|
|
3464
3723
|
function newId() {
|
|
3465
3724
|
const bytes = new Uint8Array(6);
|
|
@@ -3470,104 +3729,99 @@ function newId() {
|
|
|
3470
3729
|
/** A new job: the kind a stage answers in, that kind's own modifiers, and
|
|
3471
3730
|
a copy of the token set. */
|
|
3472
3731
|
function jobDefaults(kind = defaultOutputKind, tokens = TOKEN_DEFAULTS) {
|
|
3473
|
-
|
|
3474
|
-
|
|
3475
|
-
steps: [
|
|
3476
|
-
{ type: "prompt", withImage: false, tokenMappings: clone(tokens) },
|
|
3477
|
-
{ type: "readReply",
|
|
3478
|
-
out: { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] } },
|
|
3479
|
-
],
|
|
3480
|
-
};
|
|
3732
|
+
const job = withTokens({ type: "job", id: newId(), name: "", steps: [] }, clone(tokens));
|
|
3733
|
+
return withOut(job, { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] });
|
|
3481
3734
|
}
|
|
3482
3735
|
STEP_TYPES.job = {
|
|
3483
3736
|
in: "item", out: job => outOf(job )?.kind,
|
|
3484
3737
|
apply: "runPipeline",
|
|
3485
3738
|
fields: ["type", "id", "name", "steps"],
|
|
3486
3739
|
defaults: jobDefaults,
|
|
3487
|
-
// A job's steps
|
|
3488
|
-
//
|
|
3740
|
+
// A job's steps run in stage order: Content, then Responses, each in its
|
|
3741
|
+
// own order. A target's step is the target's, never a job's. Each step is
|
|
3742
|
+
// its entry's to check.
|
|
3489
3743
|
validate(c, ctx, bad, k, doc){
|
|
3490
3744
|
const at = jobLabel(doc, k);
|
|
3491
3745
|
if (!isObj(c) || c.type !== "job") return void bad.push(`${at} has to be a job step`);
|
|
3492
3746
|
onlyFields(c, at, STEP_TYPES.job .fields , bad);
|
|
3493
|
-
// An id is a job's own, like a
|
|
3747
|
+
// An id is a job's own, like a target's: stable across renames, so a
|
|
3494
3748
|
// stored run keeps pointing at the job that actually ran (#93).
|
|
3495
3749
|
if (!isStr(c.id) || !c.id.trim()) bad.push(`${at} has no id`);
|
|
3496
3750
|
else if ((doc.jobs ).findIndex((o) => isObj(o) && o.id === c.id) !== k) bad.push(`${at} has the id of another job`);
|
|
3497
3751
|
if (c.name != null && !isStr(c.name)) bad.push(`${at}: name has to be text`);
|
|
3498
3752
|
if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
|
|
3499
3753
|
const steps = c.steps;
|
|
3500
|
-
// A job's steps are the entries with a
|
|
3501
|
-
//
|
|
3754
|
+
// A job's steps are the entries with a stage; one without (a job, the
|
|
3755
|
+
// evals) or of no type the lab has is not one.
|
|
3502
3756
|
const unknown = steps.find(st => !slotOf(st));
|
|
3503
3757
|
if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
|
|
3504
|
-
|
|
3505
|
-
|
|
3506
|
-
|
|
3507
|
-
|
|
3508
|
-
|
|
3509
|
-
|
|
3510
|
-
|
|
3511
|
-
: slots.filter(x => x === "call").length !== 1 ? `${at} has to have one call step`
|
|
3512
|
-
: `${at}: steps are ${label("content")} (job 1), one call, then ${label("reply")}`);
|
|
3513
|
-
return;
|
|
3758
|
+
if (steps.some(st => slotOf(st) === "target")) {
|
|
3759
|
+
return void bad.push(`${at}: what a target sends is the target's own step, not the job's`);
|
|
3760
|
+
}
|
|
3761
|
+
const twice = steps.find((st, x) => !STEP_TYPES[st.type] .many && steps.findIndex(o => o.type === st.type) !== x);
|
|
3762
|
+
if (twice) return void bad.push(`${at} has two ${STEP_TYPES[twice.type] .label ?? twice.type} steps`);
|
|
3763
|
+
if (steps.some((st, x) => x > 0 && placeOf(st) < placeOf(steps[x - 1]))) {
|
|
3764
|
+
return void bad.push(`${at}: steps are its Content, then its Responses, each in order`);
|
|
3514
3765
|
}
|
|
3766
|
+
const kind = outOf(c ).kind;
|
|
3515
3767
|
for (const st of steps) {
|
|
3516
3768
|
const entry = STEP_TYPES[st.type] ;
|
|
3769
|
+
const label = `${at}'s ${entry.label ?? st.type}`;
|
|
3517
3770
|
if (k > 0 && entry.firstJobOnly) { bad.push(`${at}: ${entry.firstJobOnly}`); continue; }
|
|
3518
|
-
onlyFields(st,
|
|
3519
|
-
entry.validate(st, ctx, bad,
|
|
3771
|
+
onlyFields(st, label, entry.fields ?? [], bad);
|
|
3772
|
+
entry.validate(st, ctx, bad, label, kind);
|
|
3520
3773
|
}
|
|
3521
3774
|
},
|
|
3522
3775
|
};
|
|
3523
|
-
// What
|
|
3524
|
-
const
|
|
3776
|
+
// What an eval carries besides its type's own fields.
|
|
3777
|
+
const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
|
|
3525
3778
|
|
|
3526
|
-
/**
|
|
3527
|
-
const
|
|
3528
|
-
(doc.
|
|
3779
|
+
/** Eval [j]'s name, or the number it has always shown. */
|
|
3780
|
+
const evalLabel = (doc , j ) =>
|
|
3781
|
+
(doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
|
|
3529
3782
|
|
|
3530
|
-
/**
|
|
3783
|
+
/** An eval type that settles over the whole run rather than item by item. */
|
|
3531
3784
|
const isWholeRun = (type , t ) =>
|
|
3532
3785
|
type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
|
|
3533
3786
|
|
|
3534
3787
|
/**
|
|
3535
|
-
* A document's
|
|
3536
|
-
* the run document it was submitted with, so a run from before version
|
|
3537
|
-
*
|
|
3788
|
+
* A document's evals as a list, whatever version wrote it: a queue row keeps
|
|
3789
|
+
* the run document it was submitted with, so a run from before version 11
|
|
3790
|
+
* spells them `tests`, and one from before version 6 holds one or null, read
|
|
3791
|
+
* here as the upgrade reads it (testsList, evalsKey).
|
|
3538
3792
|
*/
|
|
3539
|
-
function
|
|
3540
|
-
const t = doc?.tests;
|
|
3793
|
+
function evalsOf(doc ) {
|
|
3794
|
+
const t = doc?.evals ?? doc?.tests;
|
|
3541
3795
|
if (Array.isArray(t)) return t ;
|
|
3542
3796
|
return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
|
|
3543
3797
|
}
|
|
3544
3798
|
|
|
3545
|
-
/** The dataset a document's
|
|
3799
|
+
/** The dataset a document's evals grade against, where one does: a run
|
|
3546
3800
|
grades against one (validatePipeline says so), so the first names it. */
|
|
3547
|
-
function
|
|
3548
|
-
const t =
|
|
3801
|
+
function evalsDataset(doc ) {
|
|
3802
|
+
const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
|
|
3549
3803
|
return t && "dataset" in t ? t.dataset : null;
|
|
3550
3804
|
}
|
|
3551
3805
|
|
|
3552
|
-
STEP_TYPES.
|
|
3806
|
+
STEP_TYPES.evals = {
|
|
3553
3807
|
in: "results", out: "verdict",
|
|
3554
3808
|
apply: "score",
|
|
3555
|
-
validate(
|
|
3556
|
-
if (!Array.isArray(
|
|
3809
|
+
validate(evals, ctx, bad, lastKind, doc){
|
|
3810
|
+
if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
|
|
3557
3811
|
const seen = new Set ();
|
|
3558
|
-
|
|
3559
|
-
const at =
|
|
3560
|
-
const type = isObj(t) &&
|
|
3812
|
+
evals.forEach((t , j ) => {
|
|
3813
|
+
const at = evalLabel(doc, j);
|
|
3814
|
+
const type = isObj(t) && EVAL_TYPES[t.type];
|
|
3561
3815
|
if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
|
|
3562
3816
|
const before = bad.length;
|
|
3563
3817
|
if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
|
|
3564
|
-
else if (seen.has(t.id)) bad.push(`${at} has the id of another
|
|
3818
|
+
else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
|
|
3565
3819
|
else seen.add(t.id);
|
|
3566
3820
|
if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
|
|
3567
3821
|
if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
|
|
3568
|
-
onlyFields(t, at, [...type.fields, ...
|
|
3822
|
+
onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
|
|
3569
3823
|
type.validate(t, ctx, bad);
|
|
3570
|
-
// Which kinds
|
|
3824
|
+
// Which kinds an eval scores is only worth saying of an eval that is whole.
|
|
3571
3825
|
if (bad.length > before) return;
|
|
3572
3826
|
const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
|
|
3573
3827
|
if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
|
|
@@ -3576,14 +3830,14 @@ STEP_TYPES.tests = {
|
|
|
3576
3830
|
}
|
|
3577
3831
|
});
|
|
3578
3832
|
// A run is handed one dataset's body to grade against (server-side-runs §4).
|
|
3579
|
-
const named = new Set(
|
|
3580
|
-
if (named.size > 1) bad.push("the
|
|
3833
|
+
const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
|
|
3834
|
+
if (named.size > 1) bad.push("the evals grade against one dataset at a time");
|
|
3581
3835
|
},
|
|
3582
3836
|
};
|
|
3583
3837
|
|
|
3584
3838
|
// ---- the document -------------------------------------------------------------
|
|
3585
3839
|
|
|
3586
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "
|
|
3840
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
|
|
3587
3841
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3588
3842
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3589
3843
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3600,9 +3854,13 @@ function versionProblem(doc ) {
|
|
|
3600
3854
|
/** What an upgrade from version 2 needs from outside the document: the rules
|
|
3601
3855
|
of the dataset a pipeline was graded against, which the retired
|
|
3602
3856
|
version-2 list kind read every reply under. Without them a job is upgraded with no
|
|
3603
|
-
rules, as a run with no graded
|
|
3857
|
+
rules, as a run with no graded eval parsed. */
|
|
3604
3858
|
|
|
3605
3859
|
|
|
3860
|
+
|
|
3861
|
+
|
|
3862
|
+
|
|
3863
|
+
|
|
3606
3864
|
|
|
3607
3865
|
|
|
3608
3866
|
/**
|
|
@@ -3621,7 +3879,8 @@ function versionProblem(doc ) {
|
|
|
3621
3879
|
* as it was, for versionProblem to name. A copy: the caller's document is not
|
|
3622
3880
|
* touched. From version 5, its test -- or none -- becomes a list of one (or
|
|
3623
3881
|
* none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
|
|
3624
|
-
* renames and every job's `type` becomes `"job"`.
|
|
3882
|
+
* renames and every job's `type` becomes `"job"`. From version 10, its
|
|
3883
|
+
* `tests` are `evals` (evalsKey).
|
|
3625
3884
|
*/
|
|
3626
3885
|
/** Every id a current document must carry, minted for the ones [doc] lacks.
|
|
3627
3886
|
An id it already has is kept. */
|
|
@@ -3637,25 +3896,33 @@ function upgradeIds(doc ) {
|
|
|
3637
3896
|
return doc;
|
|
3638
3897
|
}
|
|
3639
3898
|
|
|
3899
|
+
/** A document's columns and their steps, wherever its version keeps them:
|
|
3900
|
+
version 10's targets, or an earlier one's scenarios and their cells. */
|
|
3901
|
+
function columnsOf(doc ) {
|
|
3902
|
+
const list = Array.isArray(doc?.targets) ? doc.targets : Array.isArray(doc?.scenarios) ? doc.scenarios : [];
|
|
3903
|
+
return list.filter(isObj).map((column ) => ({
|
|
3904
|
+
column, steps: (Array.isArray(column.steps) ? column.steps : Array.isArray(column.stages) ? column.stages : []).filter(isObj),
|
|
3905
|
+
}));
|
|
3906
|
+
}
|
|
3907
|
+
|
|
3640
3908
|
/** Every profile reference cut to { id, name }. The Runs tab once kept the
|
|
3641
3909
|
whole Setup profile a scenario picked -- key and all -- so a document saved
|
|
3642
3910
|
then is cleaned as it is read, and validation refuses one that is not. */
|
|
3643
3911
|
function profileRefs(doc ) {
|
|
3644
3912
|
const cut = (r ) => (isObj(r) && isStr(r.id) ? { id: r.id, name: isStr(r.name) ? r.name : "" } : r);
|
|
3645
|
-
for (const
|
|
3646
|
-
if (
|
|
3647
|
-
if (
|
|
3648
|
-
for (const cell of Array.isArray(sc.stages) ? sc.stages : []) {
|
|
3649
|
-
if (isObj(cell) && cell.profile != null) cell.profile = cut(cell.profile);
|
|
3650
|
-
}
|
|
3913
|
+
for (const { column, steps } of columnsOf(doc)) {
|
|
3914
|
+
if ("profile" in column) column.profile = cut(column.profile);
|
|
3915
|
+
for (const st of steps) if (st.profile != null) st.profile = cut(st.profile);
|
|
3651
3916
|
}
|
|
3652
3917
|
}
|
|
3653
3918
|
|
|
3919
|
+
/** [doc] with its profile references cut (profileRefs), for chaining. */
|
|
3920
|
+
const cutRefs = (doc ) => { profileRefs(doc); return doc; };
|
|
3921
|
+
|
|
3654
3922
|
/** Whether any profile reference in a pipeline holds more than { id, name }. */
|
|
3655
3923
|
function fatProfileRef(doc ) {
|
|
3656
3924
|
const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
|
|
3657
|
-
return isObj(doc) &&
|
|
3658
|
-
isObj(sc) && (fat(sc.profile) || (Array.isArray(sc.stages) && sc.stages.some((c ) => isObj(c) && fat(c.profile)))));
|
|
3925
|
+
return isObj(doc) && columnsOf(doc).some(({ column, steps }) => fat(column.profile) || steps.some(st => fat(st.profile)));
|
|
3659
3926
|
}
|
|
3660
3927
|
|
|
3661
3928
|
/** Version 5 to 6: the one test, or none, as a list. Its id is `t1`, so
|
|
@@ -3670,9 +3937,9 @@ function testsList(doc ) {
|
|
|
3670
3937
|
return doc;
|
|
3671
3938
|
}
|
|
3672
3939
|
|
|
3673
|
-
/** A new
|
|
3674
|
-
function
|
|
3675
|
-
const own =
|
|
3940
|
+
/** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
|
|
3941
|
+
function newEval(type , name = "", fields = {}) {
|
|
3942
|
+
const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
|
|
3676
3943
|
return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
|
|
3677
3944
|
}
|
|
3678
3945
|
|
|
@@ -3723,22 +3990,148 @@ function metricsOf(next ) {
|
|
|
3723
3990
|
return next;
|
|
3724
3991
|
}
|
|
3725
3992
|
|
|
3993
|
+
/** Version 9 to 10: a job's stages, and each scenario a target
|
|
3994
|
+
(pipeline-model §16). A job's call becomes what it held for the whole
|
|
3995
|
+
job -- a Prompt's image flag and token mappings, an HTTP Request's flow
|
|
3996
|
+
step and reading -- and each scenario's cell in it, with the call's
|
|
3997
|
+
template where it has one, that target's own step there. Read Reply's
|
|
3998
|
+
modifiers are steps of their own after it. What each held is moved as it
|
|
3999
|
+
was, so the run sends the same requests and reads them the same way
|
|
4000
|
+
(fixtures/stages-v9.json, held to it by pipeline-check.js). */
|
|
4001
|
+
function targetsFromScenarios(next ) {
|
|
4002
|
+
next.version = PIPELINE_VERSION;
|
|
4003
|
+
const jobs = Array.isArray(next.jobs) ? next.jobs : [];
|
|
4004
|
+
const calls = jobs.map(job => (isObj(job) && Array.isArray(job.steps)
|
|
4005
|
+
? job.steps.find((st ) => isObj(st) && (st.type === "prompt" || st.type === "httpRequest")) : undefined));
|
|
4006
|
+
jobs.forEach((job, k) => {
|
|
4007
|
+
if (!isObj(job) || !Array.isArray(job.steps)) return;
|
|
4008
|
+
job.steps = job.steps.flatMap((st ) => {
|
|
4009
|
+
if (!isObj(st)) return [st];
|
|
4010
|
+
if (st === calls[k] && st.type === "prompt") {
|
|
4011
|
+
return [...(st.withImage ? [{ type: "attachImage" }] : []),
|
|
4012
|
+
...(Array.isArray(st.tokenMappings) && st.tokenMappings.length ? [{ type: "tokenMappings", tokenMappings: st.tokenMappings }] : [])];
|
|
4013
|
+
}
|
|
4014
|
+
if (st === calls[k]) {
|
|
4015
|
+
return [{ type: "flowStep", step: st.step, loop: st.loop ?? null, api: st.api },
|
|
4016
|
+
...(st.readAs != null ? [{ type: "readAs", readAs: st.readAs }] : [])];
|
|
4017
|
+
}
|
|
4018
|
+
if (st.type === "readReply" && isObj(st.out)) {
|
|
4019
|
+
const { modifiers, ...out } = st.out;
|
|
4020
|
+
return [{ type: "readReply", out },
|
|
4021
|
+
...(Array.isArray(modifiers) ? modifiers : []).map((modifier ) => ({ type: "modifier", modifier }))];
|
|
4022
|
+
}
|
|
4023
|
+
return [st];
|
|
4024
|
+
});
|
|
4025
|
+
});
|
|
4026
|
+
const targets = (Array.isArray(next.scenarios) ? next.scenarios : []).map((sc ) => {
|
|
4027
|
+
if (!isObj(sc)) return sc;
|
|
4028
|
+
const { stages, ...rest } = sc;
|
|
4029
|
+
return { ...rest, steps: (Array.isArray(stages) ? stages : []).map((cell , k ) => {
|
|
4030
|
+
const call = calls[k];
|
|
4031
|
+
if (!isObj(cell) || !isObj(call)) return cell;
|
|
4032
|
+
const { step: _step, loop: _loop, readAs: _readAs, withImage: _image, tokenMappings: _tokens, ...shape } = call;
|
|
4033
|
+
return { ...clone(shape), ...cell };
|
|
4034
|
+
}) };
|
|
4035
|
+
});
|
|
4036
|
+
// In the scenarios' place, so the document reads in the same order.
|
|
4037
|
+
return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
|
|
4038
|
+
}
|
|
4039
|
+
|
|
4040
|
+
/** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
|
|
4041
|
+
upgraded document reads the same, key for key, and each eval is as it
|
|
4042
|
+
was. Every version before 11 passes through here. */
|
|
4043
|
+
function evalsKey(next ) {
|
|
4044
|
+
next.version = PIPELINE_VERSION;
|
|
4045
|
+
// `tests` is the spelling of every version before 11.
|
|
4046
|
+
if (!("tests" in next)) return next;
|
|
4047
|
+
return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
|
|
4048
|
+
.map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
|
|
4049
|
+
}
|
|
4050
|
+
|
|
4051
|
+
/** Version 11 to 12: a Contains metric's Ignore case holds where the reply
|
|
4052
|
+
is matched item by item, as it does where it is matched as text, and it
|
|
4053
|
+
is kept as written. Until version 11's last hours (#199) every Contains
|
|
4054
|
+
metric matched the reply's text and minded its Ignore case, so what a
|
|
4055
|
+
stored metric says is what its author meant; only the item-by-item
|
|
4056
|
+
matching #199 added, case-blind for a few hours, read it otherwise.
|
|
4057
|
+
Every version before 12 ends here. */
|
|
4058
|
+
function caseAsWritten(next ) {
|
|
4059
|
+
next.version = PIPELINE_VERSION;
|
|
4060
|
+
return next;
|
|
4061
|
+
}
|
|
4062
|
+
|
|
3726
4063
|
function upgradePipeline (doc , ctx = {}) {
|
|
3727
4064
|
// A current document is read as it is, but for a profile reference the Runs
|
|
3728
|
-
// tab saved whole (see profileRefs), which is cut back
|
|
4065
|
+
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
4066
|
+
// profile a target step stands in for (localSteps).
|
|
3729
4067
|
if (isObj(doc) && doc.version === PIPELINE_VERSION) {
|
|
3730
|
-
|
|
3731
|
-
|
|
3732
|
-
|
|
3733
|
-
|
|
4068
|
+
let out = doc ;
|
|
4069
|
+
if (fatProfileRef(out)) {
|
|
4070
|
+
out = clone(out) ;
|
|
4071
|
+
profileRefs(out);
|
|
4072
|
+
}
|
|
4073
|
+
return localSteps(withoutCaseMetric(out), ctx) ;
|
|
3734
4074
|
}
|
|
3735
|
-
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8].includes(doc.version )) return doc;
|
|
3736
|
-
|
|
3737
|
-
|
|
3738
|
-
|
|
3739
|
-
|
|
3740
|
-
|
|
3741
|
-
|
|
4075
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
|
|
4076
|
+
if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
|
|
4077
|
+
// Every version before 10 reads as version 9 first, then as 10, then as 11
|
|
4078
|
+
// and 12.
|
|
4079
|
+
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4080
|
+
// earlier ones are cut on their way through nineOf.
|
|
4081
|
+
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4082
|
+
return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
|
|
4083
|
+
}
|
|
4084
|
+
|
|
4085
|
+
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
4086
|
+
are metrics now (dataset body version 5), which an eval naming the
|
|
4087
|
+
dataset adds for each item, so the case's own metrics carry what that
|
|
4088
|
+
metric scored. Read so at every version, the current one included, as a
|
|
4089
|
+
document saved before it went still holds it. The same document where
|
|
4090
|
+
none does. */
|
|
4091
|
+
function withoutCaseMetric(doc ) {
|
|
4092
|
+
const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
|
|
4093
|
+
if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
|
|
4094
|
+
return { ...doc, evals: doc.evals.map((t ) => (has(t)
|
|
4095
|
+
? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
|
|
4096
|
+
}
|
|
4097
|
+
|
|
4098
|
+
/** [doc] with each step asked of a profile whose type a target step stands
|
|
4099
|
+
in for (`replaces`: Echo for an Echo profile) read as that step -- its
|
|
4100
|
+
words kept, and no profile, since it asks none -- once the profiles'
|
|
4101
|
+
types are known: [ctx]'s, or a run document's own. A target whose every
|
|
4102
|
+
step then asks none, or names its own, keeps no profile either. The same
|
|
4103
|
+
document, untouched, when there is nothing to read so. */
|
|
4104
|
+
function localSteps(doc , ctx ) {
|
|
4105
|
+
const typeOfProfile = ctx.profileType
|
|
4106
|
+
?? (isObj(doc.profiles) ? (id ) => (isObj(doc.profiles[id]) ? doc.profiles[id].type : undefined) : null);
|
|
4107
|
+
if (!typeOfProfile) return doc;
|
|
4108
|
+
const stands = new Map(Object.entries(STEP_TYPES).filter(([, e]) => e.slot === "target" && e.replaces).map(([type, e]) => [e.replaces , type]));
|
|
4109
|
+
const as = (ref ) => (isObj(ref) && isStr(ref.id) ? stands.get(typeOfProfile(ref.id) ?? "") : undefined);
|
|
4110
|
+
const turns = (t , st ) => isObj(st) && STEP_TYPES[st.type]?.asks === "prompt" && as(st.profile ?? t.profile);
|
|
4111
|
+
const targets = Array.isArray(doc.targets) ? doc.targets : [];
|
|
4112
|
+
if (!targets.some(t => isObj(t) && Array.isArray(t.steps) && t.steps.some((st ) => turns(t, st)))) return doc;
|
|
4113
|
+
const next = clone(doc) ;
|
|
4114
|
+
for (const t of next.targets ) {
|
|
4115
|
+
if (!isObj(t) || !Array.isArray(t.steps)) continue;
|
|
4116
|
+
t.steps = t.steps.map((st ) => {
|
|
4117
|
+
const type = turns(t, st);
|
|
4118
|
+
return type ? { type, prompt: st.prompt, ...(st.from ? { from: st.from } : {}) } : st;
|
|
4119
|
+
});
|
|
4120
|
+
// A run keeps the profile it ran as, so its report names it as it did.
|
|
4121
|
+
if (!isObj(doc.profiles) && as(t.profile)
|
|
4122
|
+
&& t.steps.every((st ) => isObj(st) && (STEP_TYPES[st.type]?.local || st.profile != null))) t.profile = null;
|
|
4123
|
+
}
|
|
4124
|
+
return next;
|
|
4125
|
+
}
|
|
4126
|
+
|
|
4127
|
+
/** [next] as version 9 held it, from any version before 10. */
|
|
4128
|
+
function nineOf(next , ctx ) {
|
|
4129
|
+
if (next.version === 9) { profileRefs(next); return next; }
|
|
4130
|
+
if (next.version === 8) return metricsOf(next);
|
|
4131
|
+
if (next.version === 7) return metricsOf(stepsOf(testsList(next)));
|
|
4132
|
+
if (next.version === 6) return metricsOf(stepsOf(jobsOf(next)));
|
|
4133
|
+
if (next.version === 5) return metricsOf(stepsOf(jobsOf(testsList(next))));
|
|
4134
|
+
if (next.version === 3 || next.version === 4) return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
|
|
3742
4135
|
if (next.version === 1) {
|
|
3743
4136
|
next.version = 2;
|
|
3744
4137
|
if (Array.isArray(next.chains)) {
|
|
@@ -3764,7 +4157,7 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
3764
4157
|
if (isObj(cell)) cell.prompt = renamed(cell.prompt);
|
|
3765
4158
|
}
|
|
3766
4159
|
}
|
|
3767
|
-
return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))))
|
|
4160
|
+
return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
|
|
3768
4161
|
}
|
|
3769
4162
|
|
|
3770
4163
|
/** Version 1's two maps as a list of mappings. */
|
|
@@ -3781,9 +4174,9 @@ function tokenMappingsFromV1(set ) {
|
|
|
3781
4174
|
the item's image, as a job over a Source would; one over text content
|
|
3782
4175
|
has none to send, and the page turns it off with the content. */
|
|
3783
4176
|
function blankPipeline(opts = {}) {
|
|
3784
|
-
const job =
|
|
4177
|
+
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
3785
4178
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
3786
|
-
jobs: [job],
|
|
4179
|
+
jobs: [job], targets: [], evals: [] };
|
|
3787
4180
|
}
|
|
3788
4181
|
|
|
3789
4182
|
/**
|
|
@@ -3808,37 +4201,50 @@ function validatePipeline(input , ctx = {}) {
|
|
|
3808
4201
|
if (!Array.isArray(doc.jobs) || !doc.jobs.length) {
|
|
3809
4202
|
return [...bad, "jobs has to be a list of at least one job"];
|
|
3810
4203
|
}
|
|
3811
|
-
if (!
|
|
4204
|
+
if (!targetsOf(doc).length) bad.push("add a target first");
|
|
3812
4205
|
// Content is job 1's Attach Content step; a job 1 without one has none yet.
|
|
3813
|
-
if (!
|
|
4206
|
+
if (!contentOf(doc)) bad.push("set the content first");
|
|
3814
4207
|
if (run) profilesProblems(doc.profiles, bad);
|
|
3815
4208
|
doc.jobs.forEach((ch , k ) => STEP_TYPES.job .validate(ch, c, bad, k, doc));
|
|
3816
4209
|
if (bad.length) return bad;
|
|
3817
4210
|
const content = contentOf(doc) ;
|
|
3818
4211
|
|
|
3819
|
-
const
|
|
4212
|
+
const targets = targetsOf(doc);
|
|
3820
4213
|
const n = doc.jobs.length;
|
|
3821
|
-
|
|
3822
|
-
const at =
|
|
3823
|
-
if (!isObj(
|
|
3824
|
-
onlyFields(
|
|
3825
|
-
if (!isStr(
|
|
3826
|
-
else if (
|
|
3827
|
-
if (
|
|
3828
|
-
if (!isRef(
|
|
3829
|
-
else onlyFields(
|
|
3830
|
-
if (!Array.isArray(
|
|
3831
|
-
bad.push(`${at} has ${Array.isArray(
|
|
4214
|
+
targets.forEach((t, i) => {
|
|
4215
|
+
const at = targetLabel(doc, i);
|
|
4216
|
+
if (!isObj(t)) return void bad.push(`${at} has to be an object`);
|
|
4217
|
+
onlyFields(t, at, ["id", "name", "profile", "steps"], bad);
|
|
4218
|
+
if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
|
|
4219
|
+
else if (targets.findIndex((o) => isObj(o) && o.id === t.id) !== i) bad.push(`${at} has the id of another target`);
|
|
4220
|
+
if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
|
|
4221
|
+
if (t.profile != null && !isRef(t.profile)) bad.push(`${at} has to name its Target profile as { id, name }`);
|
|
4222
|
+
else if (t.profile != null) onlyFields(t.profile, `${at}'s profile`, ["id", "name"], bad);
|
|
4223
|
+
if (!Array.isArray(t.steps) || t.steps.length !== n) {
|
|
4224
|
+
bad.push(`${at} has ${Array.isArray(t.steps) ? t.steps.length : "no"} prompts for ${n} job${n === 1 ? "" : "s"}`);
|
|
3832
4225
|
return;
|
|
3833
4226
|
}
|
|
3834
|
-
|
|
3835
|
-
|
|
3836
|
-
|
|
3837
|
-
|
|
3838
|
-
|
|
3839
|
-
|
|
3840
|
-
|
|
3841
|
-
if (
|
|
4227
|
+
// A target with no profile of its own is one whose steps need none, or
|
|
4228
|
+
// each name theirs.
|
|
4229
|
+
if (t.profile == null && t.steps.some((st ) => !(isObj(st) && STEP_TYPES[st.type]?.local) && !(isObj(st) && st.profile != null))) {
|
|
4230
|
+
bad.push(`${at} has to name its Target profile as { id, name }`);
|
|
4231
|
+
}
|
|
4232
|
+
t.steps.forEach((st , k ) => {
|
|
4233
|
+
const where = `${at}, ${jobLabel(doc, k)}`;
|
|
4234
|
+
if (!isObj(st)) return void bad.push(`${where} has to be an object`);
|
|
4235
|
+
const entry = STEP_TYPES[st.type];
|
|
4236
|
+
if (entry?.slot !== "target") return void bad.push(`${where}: "${st.type}" is not something a target can send`);
|
|
4237
|
+
if (entry.local && st.profile != null) return void bad.push(`${where}: ${entry.label} asks no profile`);
|
|
4238
|
+
if (k > 0 && entry.firstJobOnly) return void bad.push(`${where}: ${entry.firstJobOnly}`);
|
|
4239
|
+
onlyFields(st, where, entry.fields ?? [], bad);
|
|
4240
|
+
entry.validate(st, c, bad, where);
|
|
4241
|
+
if (st.profile != null && !isRef(st.profile)) bad.push(`${where} has to name its Target profile as { id, name }`);
|
|
4242
|
+
else if (st.profile != null) onlyFields(st.profile, `${where}'s profile`, ["id", "name"], bad);
|
|
4243
|
+
if (st.from != null && !isRef(st.from)) bad.push(`${where} has to name the prompt it was picked from as { id, name }`);
|
|
4244
|
+
// A request is a flow step's, rebuilt from its records, and its
|
|
4245
|
+
// template has no place for an image.
|
|
4246
|
+
if (entry.asks === "request" && !flowStepOf(doc.jobs[k])) bad.push(`${where}: ${entry.label} needs the job's flow step`);
|
|
4247
|
+
if (entry.asks === "request" && sendsImage(doc.jobs[k])) bad.push(`${where}: ${entry.label} has no place for an image`);
|
|
3842
4248
|
});
|
|
3843
4249
|
});
|
|
3844
4250
|
if (bad.length) return bad;
|
|
@@ -3849,68 +4255,76 @@ function validatePipeline(input , ctx = {}) {
|
|
|
3849
4255
|
const p = c.profiles ? c.profiles(ref.id) : {};
|
|
3850
4256
|
if (!p && !missing.has(ref.id)) {
|
|
3851
4257
|
missing.add(ref.id);
|
|
3852
|
-
bad.push(`
|
|
4258
|
+
bad.push(`Target profile ${ref.name || ref.id} not found`);
|
|
3853
4259
|
}
|
|
3854
4260
|
return p;
|
|
3855
4261
|
};
|
|
3856
4262
|
// A type that answers from the item (Echo) has no model, answers only
|
|
3857
4263
|
// job 1, and answers only a text item.
|
|
3858
4264
|
const local = (p ) => !!p && typeof p === "object" && !!CONNECTION_TYPES[typeOf(p )]?.local;
|
|
3859
|
-
// Every
|
|
4265
|
+
// Every target needs a prompt in every job. A blank one is an error
|
|
3860
4266
|
// rather than a stage dropped, which would hand the job before it
|
|
3861
4267
|
// straight to the job after it, under numbers nobody sees. The one it
|
|
3862
4268
|
// may leave blank is job 1 answered from the item's own text (Echo over
|
|
3863
4269
|
// a Source or Text): the reply is the item, and a prompt would only be
|
|
3864
4270
|
// what it is read against. Over Prompt only the prompt is the reply.
|
|
4271
|
+
// What target [i] asks in job [k]: Echo's own connection for a step
|
|
4272
|
+
// answered here, else its profile as the lookups hold it.
|
|
4273
|
+
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION
|
|
4274
|
+
: c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
|
|
3865
4275
|
const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
|
|
3866
4276
|
for (let k = 0; k < n; k++) {
|
|
3867
|
-
const i =
|
|
3868
|
-
|
|
3869
|
-
|
|
3870
|
-
|
|
4277
|
+
const i = targets.findIndex((_, i) => {
|
|
4278
|
+
const words = targetStepOf(doc, i, k)?.prompt;
|
|
4279
|
+
return (!isStr(words) || !words.trim())
|
|
4280
|
+
&& !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
|
|
4281
|
+
});
|
|
4282
|
+
if (i >= 0) bad.push(n > 1 ? `${jobLabel(doc, k)}, ${targetLabel(doc, i)} has no prompt`
|
|
4283
|
+
: `${targetLabel(doc, i)} has no prompt`);
|
|
3871
4284
|
}
|
|
3872
4285
|
const contentFiles = content?.type === "source"
|
|
3873
4286
|
? (c.sources?.(content.ref?.id)?.files ?? content.files ?? null) : null;
|
|
3874
4287
|
const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
|
|
3875
4288
|
.find(n => !/\.(?:txt|md|csv)$/i.test(n));
|
|
3876
|
-
|
|
3877
|
-
const conns =
|
|
4289
|
+
targets.forEach((_, i) => {
|
|
4290
|
+
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION : lookup(targetProfileOf(doc, i, k) )));
|
|
3878
4291
|
conns.forEach((p , k ) => {
|
|
3879
4292
|
if (!local(p)) return;
|
|
3880
4293
|
const label = CONNECTION_TYPES[typeOf(p )] .label;
|
|
3881
|
-
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${
|
|
3882
|
-
else if (nonText) bad.push(`${
|
|
4294
|
+
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
|
|
4295
|
+
else if (nonText) bad.push(`${targetLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
|
|
3883
4296
|
});
|
|
3884
|
-
// A connection answers what its
|
|
4297
|
+
// A connection answers what its target's step asks: words, or a whole request.
|
|
3885
4298
|
conns.forEach((p , k ) => {
|
|
3886
|
-
const
|
|
3887
|
-
// Only a profile that was looked up says what it answers
|
|
3888
|
-
|
|
4299
|
+
const step = targetEntryOf(doc, i, k);
|
|
4300
|
+
// Only a profile that was looked up says what it answers; a step
|
|
4301
|
+
// answered here asks none.
|
|
4302
|
+
if (!c.profiles || !p || !isObj(p) || !step || step.local) return;
|
|
3889
4303
|
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3890
|
-
if (type && !(type.answers ?? ["prompt"]).includes(
|
|
3891
|
-
bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${
|
|
4304
|
+
if (type && !(type.answers ?? ["prompt"]).includes(step.asks ?? "prompt")) {
|
|
4305
|
+
bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${targetLabel(doc, i)}: ${step.label ?? "this step"} needs a profile that answers it, and ${type.label} does not`);
|
|
3892
4306
|
}
|
|
3893
4307
|
});
|
|
3894
|
-
// The model is the profile's unless the
|
|
4308
|
+
// The model is the profile's unless the target's step carries it.
|
|
3895
4309
|
const bare = conns.findIndex((p , k ) => p && c.profiles && !local(p)
|
|
3896
|
-
&&
|
|
4310
|
+
&& targetEntryOf(doc, i, k)?.modelFrom !== "step" && !String(p.model || "").trim());
|
|
3897
4311
|
if (bare >= 0) {
|
|
3898
4312
|
bad.push(n > 1
|
|
3899
|
-
? `no model on the profile ${jobLabel(doc, bare)} of ${
|
|
3900
|
-
: `no model on ${
|
|
4313
|
+
? `no model on the profile ${jobLabel(doc, bare)} of ${targetLabel(doc, i)} uses — manage profiles on the Setup tab`
|
|
4314
|
+
: `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
|
|
3901
4315
|
}
|
|
3902
4316
|
});
|
|
3903
|
-
STEP_TYPES.
|
|
4317
|
+
STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
3904
4318
|
if (bad.length) return bad;
|
|
3905
4319
|
|
|
3906
4320
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
3907
4321
|
// says which job, instead of failing every item the same way.
|
|
3908
4322
|
const text = CONTENT_TYPES[content?.type]?.text?.(content, c) ?? null;
|
|
3909
|
-
|
|
3910
|
-
const stages = doc.jobs.map((ch , k ) => ({ text:
|
|
3911
|
-
verbatim: !!
|
|
3912
|
-
const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) =>
|
|
3913
|
-
if (problem) bad.push(`${
|
|
4323
|
+
targets.forEach((_, i) => {
|
|
4324
|
+
const stages = doc.jobs.map((ch , k ) => ({ text: targetStepOf(doc, i, k) .prompt, kind: outOf(ch).kind,
|
|
4325
|
+
verbatim: !!targetEntryOf(doc, i, k)?.verbatim }));
|
|
4326
|
+
const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) => tokensOf(ch)), text);
|
|
4327
|
+
if (problem) bad.push(`${targetLabel(doc, i)}: ${jobWords(problem)}`);
|
|
3914
4328
|
});
|
|
3915
4329
|
return bad;
|
|
3916
4330
|
}
|
|
@@ -3990,16 +4404,16 @@ function connectionProblems(conn , at , from
|
|
|
3990
4404
|
}
|
|
3991
4405
|
|
|
3992
4406
|
/** The profile ids a pipeline references, scenarios first, in order. */
|
|
3993
|
-
function profileIds(doc
|
|
4407
|
+
function profileIds(doc ) {
|
|
3994
4408
|
const ids = [];
|
|
3995
|
-
for (const
|
|
3996
|
-
for (const ref of [
|
|
4409
|
+
for (const t of targetsOf(doc)) {
|
|
4410
|
+
for (const ref of [t.profile, ...(Array.isArray(t.steps) ? t.steps : []).map(st => st?.profile)] ) {
|
|
3997
4411
|
if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
3998
4412
|
}
|
|
3999
4413
|
}
|
|
4000
|
-
// Then the ones
|
|
4001
|
-
for (const t of
|
|
4002
|
-
for (const ref of
|
|
4414
|
+
// Then the ones an eval asks (a grader), so a run carries them too.
|
|
4415
|
+
for (const t of evalsOf(doc)) {
|
|
4416
|
+
for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
4003
4417
|
}
|
|
4004
4418
|
return ids;
|
|
4005
4419
|
}
|
|
@@ -4012,9 +4426,9 @@ function profileIds(doc )
|
|
|
4012
4426
|
function resolvePipeline(doc ,
|
|
4013
4427
|
ctx = {}) {
|
|
4014
4428
|
const run = clone(doc) ;
|
|
4015
|
-
// What the lab supplies
|
|
4429
|
+
// What the lab supplies an eval -- its grader -- before the profiles it
|
|
4016
4430
|
// asks are carried.
|
|
4017
|
-
run.
|
|
4431
|
+
run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
|
|
4018
4432
|
run.profiles = {};
|
|
4019
4433
|
for (const id of profileIds(run)) {
|
|
4020
4434
|
const p = ctx.profiles?.(id);
|
|
@@ -4023,7 +4437,7 @@ function resolvePipeline(doc ,
|
|
|
4023
4437
|
const content = contentOf(run) ;
|
|
4024
4438
|
if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
|
|
4025
4439
|
if (ctx.datasetVersion) {
|
|
4026
|
-
for (const t of run.
|
|
4440
|
+
for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
|
|
4027
4441
|
}
|
|
4028
4442
|
if (isStr(ctx.comment)) run.comment = ctx.comment;
|
|
4029
4443
|
return run;
|
|
@@ -4037,7 +4451,7 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4037
4451
|
delete doc.plugins;
|
|
4038
4452
|
const content = contentOf(doc) ;
|
|
4039
4453
|
if (content) { delete content.files; delete content.revs; }
|
|
4040
|
-
for (const t of doc.
|
|
4454
|
+
for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
|
|
4041
4455
|
return doc;
|
|
4042
4456
|
}
|
|
4043
4457
|
|
|
@@ -4076,7 +4490,7 @@ function importPipeline(input , ctx = {}) {
|
|
|
4076
4490
|
const missing = [], seen = new Set ();
|
|
4077
4491
|
const lists = { profile: profiles, source: sources, dataset: datasets };
|
|
4078
4492
|
const words = {
|
|
4079
|
-
profile: (ref ) => `
|
|
4493
|
+
profile: (ref ) => `Target profile ${ref.name || ref.id} not found`,
|
|
4080
4494
|
source: (ref ) => `Source ${ref.name || ref.id} not found`,
|
|
4081
4495
|
dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
|
|
4082
4496
|
};
|
|
@@ -4089,15 +4503,20 @@ function importPipeline(input , ctx = {}) {
|
|
|
4089
4503
|
};
|
|
4090
4504
|
const content = contentOf(next) ;
|
|
4091
4505
|
if (content?.type === "source") content.ref = remap(content.ref, "source");
|
|
4092
|
-
for (const
|
|
4093
|
-
|
|
4094
|
-
for (const
|
|
4095
|
-
if (
|
|
4506
|
+
for (const t of targetsOf(next)) {
|
|
4507
|
+
if (t.profile) t.profile = remap(t.profile, "profile");
|
|
4508
|
+
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4509
|
+
if (st?.profile) st.profile = remap(st.profile, "profile");
|
|
4096
4510
|
}
|
|
4097
4511
|
}
|
|
4098
|
-
for (const t of Array.isArray(next.
|
|
4512
|
+
for (const t of Array.isArray(next.evals) ? next.evals : []) {
|
|
4099
4513
|
if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
|
|
4100
4514
|
}
|
|
4515
|
+
// Its references are this lab's now, so a step asking this lab's Echo
|
|
4516
|
+
// profile is read as an Echo step (upgradePipeline did it for ids that
|
|
4517
|
+
// already matched).
|
|
4518
|
+
const local = localSteps(next, ctx);
|
|
4519
|
+
if (local !== next) Object.assign(next, local);
|
|
4101
4520
|
// A hand-written file needs no id of its own: mint the ones it lacks, and
|
|
4102
4521
|
// keep the ones it carries, so export → import → export is the same
|
|
4103
4522
|
// document and a re-import can recognise the same pipeline (#93).
|
|
@@ -4119,10 +4538,10 @@ function mintIds(doc ) {
|
|
|
4119
4538
|
for (const ch of Array.isArray(doc.jobs) ? doc.jobs : []) {
|
|
4120
4539
|
if (isObj(ch) && (!isStr(ch.id) || !ch.id)) ch.id = newId();
|
|
4121
4540
|
}
|
|
4122
|
-
for (const
|
|
4123
|
-
if (isObj(
|
|
4541
|
+
for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
|
|
4542
|
+
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4124
4543
|
}
|
|
4125
|
-
for (const t of Array.isArray(doc.
|
|
4544
|
+
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4126
4545
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4127
4546
|
}
|
|
4128
4547
|
return doc;
|
|
@@ -4146,26 +4565,34 @@ function yamlToPipeline(text , ctx ) {
|
|
|
4146
4565
|
*/
|
|
4147
4566
|
function stagesFor(run , i )
|
|
4148
4567
|
|
|
4149
|
-
|
|
4150
|
-
const sc = run.scenarios[i] ;
|
|
4568
|
+
{
|
|
4151
4569
|
const stages = run.jobs.map((ch, k) => {
|
|
4152
4570
|
const { kind, modifiers, ...settings } = outOf(ch);
|
|
4153
|
-
return { text:
|
|
4154
|
-
verbatim: !!
|
|
4571
|
+
return { text: targetStepOf(run, i, k) .prompt, kind, withImage: sendsImage(ch),
|
|
4572
|
+
verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: modifiers || [], settings };
|
|
4155
4573
|
});
|
|
4156
|
-
const connections = run.jobs.map((
|
|
4157
|
-
|
|
4574
|
+
const connections = run.jobs.map((_, k) => {
|
|
4575
|
+
// A step answered here asks nothing: the profile a run made before it
|
|
4576
|
+
// kept is what it ran as, else Echo itself.
|
|
4577
|
+
if (targetEntryOf(run, i, k)?.local) {
|
|
4578
|
+
const id = targetProfileOf(run, i, k)?.id;
|
|
4579
|
+
const held = id != null ? run.profiles?.[id] : undefined;
|
|
4580
|
+
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...ECHO_CONNECTION };
|
|
4581
|
+
}
|
|
4582
|
+
const id = targetProfileOf(run, i, k) .id;
|
|
4158
4583
|
return { id, ...(run.profiles?.[id] || {}) };
|
|
4159
4584
|
});
|
|
4160
|
-
//
|
|
4161
|
-
//
|
|
4162
|
-
return { stages, tokens: run.jobs.map(
|
|
4163
|
-
calls: run.jobs.map(
|
|
4585
|
+
// What a transport that builds its own request (an HTTP Request) builds
|
|
4586
|
+
// it from in each job, and this target's step there.
|
|
4587
|
+
return { stages, tokens: run.jobs.map(tokensOf), connections,
|
|
4588
|
+
calls: run.jobs.map((_, k) => requestStepOf(run, i, k)), cells: run.jobs.map((_, k) => targetStepOf(run, i, k) ) };
|
|
4164
4589
|
}
|
|
4165
4590
|
|
|
4166
4591
|
/** The Setup profile scenario [i] of a run ran under, as History shows it: keyless. */
|
|
4167
4592
|
function scenarioProfile(run , i ) {
|
|
4168
|
-
const id = run
|
|
4593
|
+
const id = targetsOf(run)[i]?.profile?.id;
|
|
4594
|
+
// A target that asks nothing ran as Echo.
|
|
4595
|
+
if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name: ECHO_CONNECTION.name, settings: { ...ECHO_CONNECTION } };
|
|
4169
4596
|
const conn = id != null ? run.profiles?.[id] : null;
|
|
4170
4597
|
return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
|
|
4171
4598
|
}
|
|
@@ -4192,14 +4619,14 @@ function modifierSummary(m ) {
|
|
|
4192
4619
|
return said.length ? said.join(", ") : "on";
|
|
4193
4620
|
}
|
|
4194
4621
|
|
|
4195
|
-
// ---- the
|
|
4196
|
-
//
|
|
4622
|
+
// ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
|
|
4623
|
+
// Evals only read a run: none changes what a later one sees, and none stops
|
|
4197
4624
|
// the model being sent the next item. What order changes is Continue on
|
|
4198
|
-
// failure. A per-item
|
|
4625
|
+
// failure. A per-item eval that fails and does not continue stops the evals
|
|
4199
4626
|
// after it for that item alone -- they read Skipped there, and a whole-run
|
|
4200
|
-
//
|
|
4627
|
+
// eval after it pools the items it did not stop. A whole-run eval settles
|
|
4201
4628
|
// once every item is in, and one that fails then and does not continue
|
|
4202
|
-
// leaves every
|
|
4629
|
+
// leaves every eval after it Skipped.
|
|
4203
4630
|
|
|
4204
4631
|
const SKIPPED = Object.freeze({ skipped: true });
|
|
4205
4632
|
const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
@@ -4207,16 +4634,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
|
4207
4634
|
const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
|
|
4208
4635
|
|
|
4209
4636
|
/**
|
|
4210
|
-
* One reply's scores under a run's per-item
|
|
4211
|
-
*
|
|
4637
|
+
* One reply's scores under a run's per-item evals, by eval id, in order: a
|
|
4638
|
+
* eval with no case to score leaves no entry, and one after a failure that
|
|
4212
4639
|
* does not continue reads Skipped. Null where nothing was scored.
|
|
4213
4640
|
*/
|
|
4214
|
-
function itemScores(run , kase , res )
|
|
4641
|
+
function itemScores(run , kase , res ) {
|
|
4215
4642
|
if (!kase) return null;
|
|
4216
|
-
const out
|
|
4643
|
+
const out = {};
|
|
4217
4644
|
let stopped = false;
|
|
4218
|
-
for (const t of
|
|
4219
|
-
const type =
|
|
4645
|
+
for (const t of evalsOf(run)) {
|
|
4646
|
+
const type = EVAL_TYPES[t.type];
|
|
4220
4647
|
if (!type?.score) continue;
|
|
4221
4648
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4222
4649
|
const s = type.score(t, kase, res);
|
|
@@ -4229,25 +4656,24 @@ function itemScores(run , kase , res
|
|
|
4229
4656
|
/** What production replied to an item, for a run's metrics: its last job's
|
|
4230
4657
|
call says, from the item's record, or nobody does. */
|
|
4231
4658
|
function productionOf(run , record ) {
|
|
4232
|
-
const last = run.jobs.at(-1);
|
|
4233
|
-
|
|
4234
|
-
return record ? STEP_TYPES[(call )?.type ?? ""]?.production?.(call, record) ?? null : null;
|
|
4659
|
+
const last = run.jobs.at(-1), flow = flowStepOf(last);
|
|
4660
|
+
return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
|
|
4235
4661
|
}
|
|
4236
4662
|
|
|
4237
4663
|
/**
|
|
4238
|
-
* itemScores for the runner, which can wait:
|
|
4664
|
+
* itemScores for the runner, which can wait: an eval that reads every item
|
|
4239
4665
|
* (`read`: the Metrics, which may ask a grader) scores one with no case too.
|
|
4240
4666
|
*/
|
|
4241
4667
|
async function itemScoresAsync(run , kase , res ,
|
|
4242
4668
|
more = {}) {
|
|
4243
4669
|
const out = {};
|
|
4244
4670
|
let stopped = false;
|
|
4245
|
-
for (const t of
|
|
4246
|
-
const type =
|
|
4671
|
+
for (const t of evalsOf(run)) {
|
|
4672
|
+
const type = EVAL_TYPES[t.type];
|
|
4247
4673
|
if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
|
|
4248
4674
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4249
|
-
//
|
|
4250
|
-
//
|
|
4675
|
+
// An eval with nothing to read on this item leaves no entry, as a graded
|
|
4676
|
+
// eval does on an item with no case.
|
|
4251
4677
|
const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
|
|
4252
4678
|
: type.score (t, kase , res);
|
|
4253
4679
|
if (!s) continue;
|
|
@@ -4258,22 +4684,22 @@ async function itemScoresAsync(run , kase ,
|
|
|
4258
4684
|
}
|
|
4259
4685
|
|
|
4260
4686
|
/**
|
|
4261
|
-
* Every
|
|
4687
|
+
* Every eval's reading of scenario [i] of a run, in the run's order, from
|
|
4262
4688
|
* the items so far. [settled] says every item is in: only then has a
|
|
4263
|
-
* whole-run
|
|
4689
|
+
* whole-run eval settled, so only then does its failure skip the evals
|
|
4264
4690
|
* after it.
|
|
4265
4691
|
*/
|
|
4266
|
-
function
|
|
4692
|
+
function scenarioEvals(run , i , items ,
|
|
4267
4693
|
settled = true) {
|
|
4268
4694
|
const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
|
|
4269
|
-
// Which items a per-item
|
|
4695
|
+
// Which items a per-item eval has stopped so far, for the evals after it.
|
|
4270
4696
|
const stopped = cells.map(() => false);
|
|
4271
4697
|
let skipRest = false;
|
|
4272
|
-
const
|
|
4273
|
-
return
|
|
4274
|
-
const type =
|
|
4698
|
+
const evals = evalsOf(run);
|
|
4699
|
+
return evals.map((t, j) => {
|
|
4700
|
+
const type = EVAL_TYPES[t.type];
|
|
4275
4701
|
const whole = isWholeRun(type, t);
|
|
4276
|
-
const base = { id: t.id, label:
|
|
4702
|
+
const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
|
|
4277
4703
|
verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
|
|
4278
4704
|
ran: 0, passed: 0, skippedItems: 0 };
|
|
4279
4705
|
if (skipRest) {
|
|
@@ -4300,7 +4726,7 @@ function scenarioTests(run , i , items
|
|
|
4300
4726
|
});
|
|
4301
4727
|
}
|
|
4302
4728
|
|
|
4303
|
-
/** A scenario's pass or fail over every
|
|
4729
|
+
/** A scenario's pass or fail over every eval that read it: null where none
|
|
4304
4730
|
has anything to say yet. */
|
|
4305
4731
|
function scenarioPasses(outcomes ) {
|
|
4306
4732
|
let said = false;
|
|
@@ -4341,95 +4767,77 @@ function validateEvals(ev , files
|
|
|
4341
4767
|
return ["the graded set has to be a JSON object with a `cases` list"];
|
|
4342
4768
|
}
|
|
4343
4769
|
if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
|
|
4770
|
+
if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
|
|
4344
4771
|
if (bad.length) return bad;
|
|
4345
4772
|
|
|
4346
4773
|
const graded = ev.cases || [];
|
|
4347
4774
|
// Identity first: every rule below reports which case is at fault, so a
|
|
4348
4775
|
// case with no usable id makes the rest of the report unreadable.
|
|
4349
|
-
const seen = new
|
|
4776
|
+
const seen = new Set ();
|
|
4350
4777
|
for (const c of graded) {
|
|
4351
|
-
{
|
|
4352
|
-
|
|
4353
|
-
|
|
4354
|
-
continue;
|
|
4355
|
-
}
|
|
4356
|
-
if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
|
|
4357
|
-
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4358
|
-
else seen.set(c.id, "cases");
|
|
4359
|
-
if (!caseFile(c).trim()) {
|
|
4360
|
-
bad.push(`${c.id || "a case"} names no file`);
|
|
4361
|
-
}
|
|
4362
|
-
for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
|
|
4363
|
-
if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
|
|
4364
|
-
}
|
|
4365
|
-
for (const key of ["minCount", "maxCount"]) {
|
|
4366
|
-
if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
|
|
4367
|
-
}
|
|
4368
|
-
if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
|
|
4369
|
-
// A case's own metrics, which a Metrics test adds to its own for this item.
|
|
4370
|
-
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4778
|
+
if (!isObj(c)) {
|
|
4779
|
+
bad.push("cases holds something that is not a case");
|
|
4780
|
+
continue;
|
|
4371
4781
|
}
|
|
4782
|
+
if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
|
|
4783
|
+
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4784
|
+
else seen.add(c.id);
|
|
4785
|
+
if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
|
|
4786
|
+
if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
|
|
4787
|
+
if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
|
|
4788
|
+
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4372
4789
|
}
|
|
4373
4790
|
if (bad.length) return bad;
|
|
4374
4791
|
|
|
4375
|
-
//
|
|
4376
|
-
// the case is unpassable however well the model
|
|
4792
|
+
// What a case says, read from its metrics: a term the scorer cannot match
|
|
4793
|
+
// can never be produced, so the case is unpassable however well the model
|
|
4794
|
+
// answers -- and the same goes for a bound no reply can meet.
|
|
4377
4795
|
const matchable = (term ) => termIn([String(term).trim()], term);
|
|
4378
|
-
|
|
4796
|
+
const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
|
|
4379
4797
|
for (const c of graded) {
|
|
4380
4798
|
if (c.todo) continue;
|
|
4381
4799
|
const say = (m ) => bad.push(`${c.id}: ${m}`);
|
|
4382
|
-
const
|
|
4383
|
-
|
|
4384
|
-
if (
|
|
4800
|
+
const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
|
|
4801
|
+
if (!scored.length) { say("graded but states nothing to expect"); continue; }
|
|
4802
|
+
if (scored.some(m => m.type === "discarded" && !m.not)) {
|
|
4385
4803
|
// What is discarded holds nothing, so there is nothing else to expect.
|
|
4386
|
-
if (
|
|
4387
|
-
say("expects its answer discarded, and states something the answer should hold too");
|
|
4388
|
-
}
|
|
4804
|
+
if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
|
|
4389
4805
|
continue;
|
|
4390
4806
|
}
|
|
4391
|
-
const
|
|
4392
|
-
|
|
4393
|
-
for (const t of
|
|
4394
|
-
|
|
4395
|
-
|
|
4396
|
-
|
|
4397
|
-
for (const
|
|
4398
|
-
}
|
|
4399
|
-
// An exception excuses only the forbidden terms inside it, so one that
|
|
4400
|
-
// holds none of them changes nothing and reads as if it did.
|
|
4401
|
-
for (const a of c.allow || []) {
|
|
4402
|
-
if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
|
|
4807
|
+
const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
|
|
4808
|
+
const forbids = scored.filter(m => m.not && m.type === "contains");
|
|
4809
|
+
for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
|
|
4810
|
+
// An exception excuses only the forbidden term inside it, so one that
|
|
4811
|
+
// holds it not changes nothing and reads as if it did.
|
|
4812
|
+
for (const m of forbids) {
|
|
4813
|
+
for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
|
|
4403
4814
|
}
|
|
4404
4815
|
// A term on both lists cannot be produced and cannot be withheld.
|
|
4405
|
-
|
|
4406
|
-
|
|
4407
|
-
|
|
4408
|
-
|
|
4409
|
-
|
|
4410
|
-
|
|
4411
|
-
|
|
4412
|
-
|
|
4413
|
-
|
|
4414
|
-
|
|
4415
|
-
|
|
4416
|
-
|
|
4417
|
-
|
|
4418
|
-
const things = expect.length + anyOf.length;
|
|
4419
|
-
if (hi != null && things > hi) {
|
|
4420
|
-
say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
|
|
4816
|
+
const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
|
|
4817
|
+
for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
|
|
4818
|
+
for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
|
|
4819
|
+
const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
|
|
4820
|
+
if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
|
|
4821
|
+
// A ceiling below one says no answer is acceptable, which is a case
|
|
4822
|
+
// that can never pass rather than a strict one.
|
|
4823
|
+
if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
|
|
4824
|
+
// A case that names more distinct things than the reply may carry, in
|
|
4825
|
+
// the scorer's own counting of a thing (#486: each term Contains all
|
|
4826
|
+
// wants is one, each Contains any is one), can never pass.
|
|
4827
|
+
const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
|
|
4828
|
+
if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
|
|
4421
4829
|
}
|
|
4422
4830
|
}
|
|
4423
4831
|
|
|
4424
|
-
// One
|
|
4425
|
-
//
|
|
4832
|
+
// One item, one case. An item here twice is two cases of it, graded
|
|
4833
|
+
// separately, and both would be listed.
|
|
4426
4834
|
const where = new Map ();
|
|
4427
4835
|
for (const c of graded) {
|
|
4428
|
-
const name =
|
|
4836
|
+
const name = caseItem(c);
|
|
4429
4837
|
const counted = where.get(name);
|
|
4430
4838
|
if (counted) {
|
|
4431
|
-
bad.push(`${name} is graded twice — one
|
|
4432
|
-
+ `two
|
|
4839
|
+
bad.push(`${name} is graded twice — one item, `
|
|
4840
|
+
+ `two cases. Grade it once.`);
|
|
4433
4841
|
} else where.set(name, true);
|
|
4434
4842
|
}
|
|
4435
4843
|
const gradedAt = (name ) => where.has(name);
|
|
@@ -4450,7 +4858,7 @@ function validateEvals(ev , files
|
|
|
4450
4858
|
for (const f of shown) {
|
|
4451
4859
|
if (!gradedAt(f)) {
|
|
4452
4860
|
bad.push(`${f} is in the Source and this dataset does not grade it, `
|
|
4453
|
-
+ `so the
|
|
4861
|
+
+ `so the Cases view does not list it`);
|
|
4454
4862
|
}
|
|
4455
4863
|
}
|
|
4456
4864
|
return bad;
|
|
@@ -4472,7 +4880,9 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4472
4880
|
const want = Number(n), warn = [];
|
|
4473
4881
|
for (const c of ev.cases || []) {
|
|
4474
4882
|
if (!c || c.todo) continue;
|
|
4475
|
-
const things = (c.
|
|
4883
|
+
const things = (Array.isArray(c.metrics) ? c.metrics : [])
|
|
4884
|
+
.filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
|
|
4885
|
+
.reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
|
|
4476
4886
|
if (things > want) {
|
|
4477
4887
|
warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
|
|
4478
4888
|
}
|
|
@@ -4491,7 +4901,7 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4491
4901
|
* `evals-check.js` asserts the round trip against the real file.
|
|
4492
4902
|
*/
|
|
4493
4903
|
function evalsJson(ev ) {
|
|
4494
|
-
return JSON.stringify(
|
|
4904
|
+
return JSON.stringify(ev, null, 2) + "\n";
|
|
4495
4905
|
}
|
|
4496
4906
|
|
|
4497
4907
|
|
|
@@ -4505,22 +4915,24 @@ registerMetrics({ registerKinds });
|
|
|
4505
4915
|
export {
|
|
4506
4916
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
4507
4917
|
loopReplyError, preparedSize,
|
|
4508
|
-
termIn, forbiddenIn,
|
|
4918
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4509
4919
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
4510
|
-
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf,
|
|
4920
|
+
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
4511
4921
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
4512
4922
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
4513
4923
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
4514
4924
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
4515
|
-
PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS,
|
|
4925
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
|
|
4516
4926
|
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
4517
4927
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
4518
4928
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
4519
4929
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
4520
|
-
contentOf, withContent,
|
|
4521
|
-
|
|
4522
|
-
|
|
4523
|
-
|
|
4930
|
+
contentOf, withContent, replyOf,
|
|
4931
|
+
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
4932
|
+
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
4933
|
+
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
4934
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
4935
|
+
evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
|
|
4524
4936
|
pipelineToYaml, importPipeline, yamlToPipeline,
|
|
4525
4937
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
4526
4938
|
};
|