evals-lab 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/LICENSE.md +57 -0
  2. package/README.md +98 -0
  3. package/bin/evals-lab.js +196 -0
  4. package/lab/demo/CREDITS.md +92 -0
  5. package/lab/demo/datasets/demo-1.json +1839 -0
  6. package/lab/demo/datasets/demo-2.json +1683 -0
  7. package/lab/demo/manifest.json +12 -0
  8. package/lab/demo/pipelines/demo-1.json +104 -0
  9. package/lab/demo/pipelines/demo-2.json +104 -0
  10. package/lab/demo/sources/Demo 1/basketball-hoop.jpg +0 -0
  11. package/lab/demo/sources/Demo 1/blue-boardwalk.jpg +0 -0
  12. package/lab/demo/sources/Demo 1/busy-beach.jpg +0 -0
  13. package/lab/demo/sources/Demo 1/butcher-sign.jpg +0 -0
  14. package/lab/demo/sources/Demo 1/cactus-flower.jpg +0 -0
  15. package/lab/demo/sources/Demo 1/corner-shop.jpg +0 -0
  16. package/lab/demo/sources/Demo 1/crosswalk-cyclist-panorama.jpg +0 -0
  17. package/lab/demo/sources/Demo 1/cyanotype-room.jpg +0 -0
  18. package/lab/demo/sources/Demo 1/dead-end-sign.jpg +0 -0
  19. package/lab/demo/sources/Demo 1/dog-in-snow.jpg +0 -0
  20. package/lab/demo/sources/Demo 1/empty-bedroom.jpg +0 -0
  21. package/lab/demo/sources/Demo 1/farmers-market-stall.jpg +0 -0
  22. package/lab/demo/sources/Demo 1/four-jets.jpg +0 -0
  23. package/lab/demo/sources/Demo 1/german-shepherd.jpg +0 -0
  24. package/lab/demo/sources/Demo 1/harbour-bridge-dusk.jpg +0 -0
  25. package/lab/demo/sources/Demo 1/harrow-on-the-hill-sign.jpg +0 -0
  26. package/lab/demo/sources/Demo 1/hong-kong-street.jpg +0 -0
  27. package/lab/demo/sources/Demo 1/house-salisbury-street.jpg +0 -0
  28. package/lab/demo/sources/Demo 1/infrared-orchard.jpg +0 -0
  29. package/lab/demo/sources/Demo 1/keyboard-desk.jpg +0 -0
  30. package/lab/demo/sources/Demo 1/lighthouse-dunes.jpg +0 -0
  31. package/lab/demo/sources/Demo 1/mountain-lake.jpg +0 -0
  32. package/lab/demo/sources/Demo 1/museum-skeleton.jpg +0 -0
  33. package/lab/demo/sources/Demo 1/music-room.jpg +0 -0
  34. package/lab/demo/sources/Demo 1/old-red-car.jpg +0 -0
  35. package/lab/demo/sources/Demo 1/parked-car-plate.jpg +0 -0
  36. package/lab/demo/sources/Demo 1/phone-and-wallet.png +0 -0
  37. package/lab/demo/sources/Demo 1/phone-box.jpg +0 -0
  38. package/lab/demo/sources/Demo 1/pink-flamingo.jpg +0 -0
  39. package/lab/demo/sources/Demo 1/postcards.jpg +0 -0
  40. package/lab/demo/sources/Demo 1/red-roof-church.jpg +0 -0
  41. package/lab/demo/sources/Demo 1/roadside-mailboxes.jpg +0 -0
  42. package/lab/demo/sources/Demo 1/rodeo.jpg +0 -0
  43. package/lab/demo/sources/Demo 1/running-tap.tif +0 -0
  44. package/lab/demo/sources/Demo 1/snail-on-stem.jpg +0 -0
  45. package/lab/demo/sources/Demo 1/two-horses-field.jpg +0 -0
  46. package/lab/demo/sources/Demo 1/university-sign.jpg +0 -0
  47. package/lab/demo/sources/Demo 1/vegetable-crates.jpg +0 -0
  48. package/lab/demo/sources/Demo 1/watermarked-car.jpg +0 -0
  49. package/lab/demo/sources/Demo 1/watermarked-pills.jpg +0 -0
  50. package/lab/demo/sources/Demo 1/whiteboard-delegate.jpg +0 -0
  51. package/lab/demo/sources/Demo 1/whiteboard-germ-layers.jpg +0 -0
  52. package/lab/demo/sources/Demo 2/aerial-city.jpg +0 -0
  53. package/lab/demo/sources/Demo 2/alligator-pen.jpg +0 -0
  54. package/lab/demo/sources/Demo 2/arm-tattoo.jpg +0 -0
  55. package/lab/demo/sources/Demo 2/bed-and-plant.jpg +0 -0
  56. package/lab/demo/sources/Demo 2/bright-bedroom.jpg +0 -0
  57. package/lab/demo/sources/Demo 2/car-headlight.jpg +0 -0
  58. package/lab/demo/sources/Demo 2/child-in-surf.jpg +0 -0
  59. package/lab/demo/sources/Demo 2/city-highway.jpg +0 -0
  60. package/lab/demo/sources/Demo 2/corner-bakery.jpg +0 -0
  61. package/lab/demo/sources/Demo 2/cyclist-yellow-jacket.jpg +0 -0
  62. package/lab/demo/sources/Demo 2/excavator-street-signs.jpg +0 -0
  63. package/lab/demo/sources/Demo 2/globe-closeup.jpg +0 -0
  64. package/lab/demo/sources/Demo 2/harbour-village.jpg +0 -0
  65. package/lab/demo/sources/Demo 2/high-street-walkers.jpg +0 -0
  66. package/lab/demo/sources/Demo 2/hillside-rooftops.jpg +0 -0
  67. package/lab/demo/sources/Demo 2/house-at-night.jpg +0 -0
  68. package/lab/demo/sources/Demo 2/ivy-leaves.jpg +0 -0
  69. package/lab/demo/sources/Demo 2/man-with-alligator.jpg +0 -0
  70. package/lab/demo/sources/Demo 2/orange-mailbox-hedge.jpg +0 -0
  71. package/lab/demo/sources/Demo 2/pedal-boats.jpg +0 -0
  72. package/lab/demo/sources/Demo 2/pink-trees-vignette.jpg +0 -0
  73. package/lab/demo/sources/Demo 2/please-leave-quietly-sign.jpg +0 -0
  74. package/lab/demo/sources/Demo 2/railroad-crossing-sign.jpg +0 -0
  75. package/lab/demo/sources/Demo 2/roadworks-sign.jpg +0 -0
  76. package/lab/demo/sources/Demo 2/roundabout-sign.jpg +0 -0
  77. package/lab/demo/sources/Demo 2/sea-cave.jpg +0 -0
  78. package/lab/demo/sources/Demo 2/shadow-on-sand.jpg +0 -0
  79. package/lab/demo/sources/Demo 2/signpost-a404.jpg +0 -0
  80. package/lab/demo/sources/Demo 2/street-fruit-cart.jpg +0 -0
  81. package/lab/demo/sources/Demo 2/street-sign-parnassusweg.jpg +0 -0
  82. package/lab/demo/sources/Demo 2/two-horses-close.jpg +0 -0
  83. package/lab/demo/sources/Demo 2/victorian-house.jpg +0 -0
  84. package/lab/demo/sources/Demo 2/vintage-dashboard.jpg +0 -0
  85. package/lab/demo/sources/Demo 2/volcano-at-dusk.jpg +0 -0
  86. package/lab/demo/sources/Demo 2/wall-camera.jpg +0 -0
  87. package/lab/demo/sources/Demo 2/watch-for-rocks-sign.jpg +0 -0
  88. package/lab/demo/sources/Demo 2/waterfall.jpg +0 -0
  89. package/lab/demo/sources/Demo 2/white-domes.jpg +0 -0
  90. package/lab/demo/sources/Demo 2/whiteboard-meeting-notes.jpg +0 -0
  91. package/lab/demo/sources/Demo 2/whiteboard-messages.jpg +0 -0
  92. package/lab/demo/sources/Demo 2/wooden-house-fence.jpg +0 -0
  93. package/lab/evals-core.mjs +4526 -0
  94. package/lab/flows/wdl.mjs +373 -0
  95. package/lab/js-yaml.mjs +3851 -0
  96. package/lab/kinds/list.mjs +411 -0
  97. package/lab/metrics/builtin.mjs +353 -0
  98. package/lab/presets.json +62 -0
  99. package/lab/run-evals.js +1262 -0
  100. package/lab/server.py +5044 -0
  101. package/lab/web/dist/assets/dist-DI3ewZYj.js +1 -0
  102. package/lab/web/dist/assets/gallery-DipkRvqJ.js +3 -0
  103. package/lab/web/dist/assets/gallery-o7c4lfpn.css +1 -0
  104. package/lab/web/dist/assets/main-CpVssvrb.js +18 -0
  105. package/lab/web/dist/assets/main-DjQQums6.css +1 -0
  106. package/lab/web/dist/assets/tokens-B9intIuT.js +51 -0
  107. package/lab/web/dist/assets/tokens-s6I-RMVq.css +1 -0
  108. package/lab/web/dist/gallery.html +18 -0
  109. package/lab/web/dist/index.html +23 -0
  110. package/package.json +20 -0
@@ -0,0 +1,4526 @@
1
+ // The eval runner. One definition of what a pass is, with three callers: the
2
+ // Prompt Lab tab, `run-evals.js`, and CI.
3
+ //
4
+ // WHY THIS FILE EXISTS. The scoring used to live in the page, and a CLI and a
5
+ // CI job would each have grown a copy of it. Three scorers agree on the day
6
+ // they are written and not after: this repository has already produced a ~6x
7
+ // unit error, a note written to prevent a magnitude error that carried the
8
+ // wrong magnitude, and a scorer that reported 100% for a set it had not run.
9
+ // A number stated in more than one place drifts. Issue #281.
10
+ //
11
+ // WHY JAVASCRIPT, AND NOT PYTHON. The scorer was already JavaScript and was
12
+ // already being run outside a browser -- `evals-check.js` loaded it into a
13
+ // Node `vm` from the day it was written. A Python core would mean rewriting it,
14
+ // which is the second implementation this file exists to prevent, or having
15
+ // the tab post every row to the relay to be scored, which puts the deployed
16
+ // container in the path of a number the page can work out for itself.
17
+ //
18
+ // ONE MODULE, TYPED AT THE SOURCE (#45). The page imports this file and
19
+ // Vite bundles it; run-evals.js and the checks import the same file, and
20
+ // Node runs it by stripping its types, so there is no build step between a
21
+ // change here and any of its callers. That rules out the TypeScript that
22
+ // has to be compiled rather than erased -- enums, namespaces, parameter
23
+ // properties -- and `tsc` (web/'s typecheck, `erasableSyntaxOnly`) says so.
24
+ // The pipeline document's types are declared here and nowhere else.
25
+ //
26
+ // WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
27
+ // script has not, so `prepare()` stays in the page and the script shells out
28
+ // to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
29
+ // the script writes JSON. Both read the same verdict out of `scoreCase`.
30
+ //
31
+ // `runner-check.js` runs one set through two of the callers and asserts the
32
+ // verdicts and the totals are identical, because sharing a file is a claim
33
+ // and that is a measurement.
34
+
35
+ import yaml from "./js-yaml.mjs";
36
+
37
+ import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
38
+ import registerMetrics from "./metrics/builtin.mjs";
39
+ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
40
+
41
+ // ---- The types -------------------------------------------------------------
42
+ // docs/pipeline-model.md as types, and the shapes of the registries every
43
+ // caller reads. They are declared here rather than beside any one caller so
44
+ // the page, the worker and the checks cannot disagree about a document: web/
45
+ // imports them from this file and declares none of its own.
46
+
47
+ /** A reference to something the lab holds: the id is looked up, the name is
48
+ the label and what an import falls back to when the id is unknown. */
49
+
50
+
51
+
52
+
53
+
54
+ /**
55
+ * One of a job's token mappings: a name the prompt writes in braces, and a
56
+ * TOKEN_TYPES entry that says what the prompt does with it. What else a
57
+ * mapping holds is its type's: a value's text, a block's on/off.
58
+ */
59
+
60
+
61
+
62
+
63
+
64
+
65
+
66
+
67
+
68
+ /** A job's mappings, in the order the page lists them. */
69
+
70
+
71
+ /** The version-1 shape of a job's mappings, read only to upgrade it. */
72
+
73
+
74
+
75
+
76
+
77
+ /** One change applied to a job's parsed value, in order: a MODIFIERS
78
+ entry's type and that entry's options. */
79
+
80
+
81
+
82
+
83
+
84
+
85
+
86
+
87
+
88
+
89
+
90
+
91
+ /** Job 1's first step, when it has one: the items a run goes over. */
92
+
93
+
94
+
95
+
96
+
97
+ /** A Call that prompts a model: each scenario's cell says with what words
98
+ and whom to ask; the step says whether the item's image goes too, and the
99
+ token mappings the words are resolved under. */
100
+
101
+
102
+
103
+
104
+
105
+
106
+ /** How a job's reply is read: its output kind, and the modifiers applied. */
107
+
108
+
109
+
110
+
111
+
112
+ /** A Call that sends a Power Automate step's request (docs/power-automate.md
113
+ § The HTTP Request step): the flow's own template -- method, path, query
114
+ and body, Workflow Definition Language expressions and all -- evaluated
115
+ against each item's record. The Setup profile gives the address and the
116
+ key; each scenario's cell the words, model, settings and prefill the
117
+ HTTP API puts in the body. */
118
+
119
+
120
+
121
+
122
+
123
+
124
+
125
+
126
+
127
+
128
+
129
+
130
+
131
+
132
+
133
+
134
+
135
+
136
+
137
+
138
+ /** A job is its steps, in fixed slots (pipeline-model §3): an Attach Content
139
+ on job 1 only, one Call, then Read Reply. */
140
+
141
+
142
+
143
+
144
+
145
+
146
+
147
+
148
+
149
+
150
+
151
+
152
+
153
+
154
+
155
+  
156
+
157
+
158
+
159
+
160
+
161
+
162
+
163
+
164
+
165
+
166
+ /** No item: a stage is sent its prompt alone (CONTENT_TYPES.prompt). */
167
+
168
+
169
+
170
+
171
+
172
+
173
+ /** One cell of a scenario: what to ask in that job, and whom. */
174
+
175
+
176
+
177
+
178
+
179
+
180
+
181
+
182
+
183
+
184
+
185
+
186
+
187
+
188
+
189
+
190
+
191
+
192
+
193
+
194
+
195
+
196
+
197
+
198
+
199
+ /** What every test carries whatever its type: an id minted once and never
200
+ shown, an optional name ("Test 2" when blank), and whether the tests after
201
+ it still read what it failed on (docs/pipeline-model.md §3). */
202
+
203
+
204
+
205
+
206
+
207
+
208
+
209
+
210
+
211
+
212
+
213
+
214
+
215
+
216
+
217
+
218
+
219
+
220
+
221
+
222
+
223
+
224
+
225
+
226
+
227
+ /** Checks of every reply (TEST_TYPES.metrics): its own for every item, and
228
+ a case's own for its item; all must pass, or weighted points reach the
229
+ threshold. A model-graded one asks the grader. */
230
+
231
+
232
+
233
+
234
+
235
+
236
+
237
+
238
+
239
+
240
+
241
+
242
+ /** A test, from version 9: Metrics. A Single Test or a Graded set is what an
243
+ older document held (upgradePipeline reads it converted). */
244
+
245
+
246
+ /** A pipeline's tests, in the order they read a run. */
247
+
248
+
249
+
250
+
251
+
252
+
253
+
254
+
255
+
256
+
257
+
258
+
259
+
260
+
261
+
262
+ /** A Setup profile as a run carries it: request settings, never the key. */
263
+
264
+
265
+
266
+
267
+
268
+
269
+
270
+
271
+
272
+
273
+
274
+
275
+
276
+ /** A pipeline resolved at submit (§5): what the queue runs and History
277
+ reads. The additions, and nothing else. */
278
+
279
+
280
+
281
+
282
+
283
+
284
+
285
+
286
+
287
+ /** A plugin as a run records it. */
288
+
289
+
290
+
291
+
292
+
293
+
294
+ /** Either document: what a reader that only looks at jobs takes. */
295
+
296
+
297
+ /**
298
+ * A Setup profile as the Setup tab stores it: a connection type's settings,
299
+ * the key, and -- until it is converted -- the old `profile` it was made
300
+ * with (#46). Its fields are whatever its type's settings are.
301
+ */
302
+
303
+
304
+
305
+
306
+
307
+
308
+
309
+
310
+
311
+
312
+
313
+ /** A connection's settings in one flat set, the shape a type reads. */
314
+
315
+
316
+ /** A value read before it is checked -- a document off the wire, out of a
317
+ file or out of a store. The functions that take one are the ones that
318
+ establish its shape, so they read it field by field, as it came. */
319
+
320
+
321
+ // ---- the pipeline's file form (#29) ----
322
+
323
+ /** A pipeline read from YAML, references remapped to this lab. */
324
+
325
+
326
+
327
+
328
+
329
+
330
+
331
+ /** What an import matches references against: { id, name } each. */
332
+
333
+
334
+
335
+
336
+
337
+
338
+
339
+
340
+ /** An import's outcome: the remapped pipeline, or one sentence refusing it. */
341
+
342
+
343
+ // ---- running a job ----
344
+
345
+ /** A stage as runPipeline takes it. */
346
+
347
+
348
+
349
+
350
+
351
+
352
+
353
+
354
+
355
+
356
+
357
+ /** An item a modifier dropped, and what dropped it. */
358
+ /** An item a modifier took out of a list, and what took it: a rule's id, or a modifier's own reason. */
359
+
360
+
361
+
362
+
363
+
364
+ /** What a transport answered for one stage. */
365
+
366
+
367
+
368
+
369
+
370
+
371
+
372
+
373
+
374
+
375
+
376
+
377
+
378
+
379
+
380
+
381
+
382
+ /** The transport: a stage's prompt and image in, its reply out. */
383
+
384
+
385
+ /** One stage of a run as it happened: what was sent and what came back. */
386
+
387
+
388
+
389
+
390
+
391
+
392
+
393
+
394
+
395
+
396
+
397
+
398
+
399
+
400
+ /** A job's result for one item: the terms its last stage holds, and how. */
401
+
402
+
403
+
404
+
405
+
406
+
407
+
408
+
409
+
410
+
411
+
412
+
413
+ /** What runPipeline is told besides the stages. */
414
+
415
+
416
+
417
+
418
+
419
+
420
+
421
+
422
+ /** A stage of a run as a transport asks it: stagesFor's answer. */
423
+
424
+
425
+
426
+
427
+
428
+
429
+
430
+
431
+
432
+
433
+ // ---- grading ----
434
+
435
+ /** A graded case, as cases.json writes one.
436
+
437
+ The item a case grades is named by `filename`: a dataset joins on a file
438
+ name, and nothing about that is an image -- the next dataset addresses
439
+ a row of a spreadsheet or a line of a log by the same key. The field had
440
+ an older name while the lab was one app's bench, and it is still *read*:
441
+ a set re-synced from that app's repository arrives spelled that way, and
442
+ a case that silently stopped matching would be worse than one that reads
443
+ both.
444
+ `caseFile` is the one reader; `evalsJson` is the one writer, and it writes
445
+ `filename`. */
446
+
447
+
448
+
449
+
450
+
451
+
452
+
453
+
454
+
455
+
456
+
457
+
458
+
459
+
460
+
461
+
462
+
463
+
464
+
465
+
466
+
467
+
468
+
469
+
470
+ /** A graded set: cases.json. */
471
+
472
+
473
+
474
+
475
+
476
+ /** One row the Datasets tab's Cases group draws, as gradedSetFrom builds it:
477
+ a case of a set validateEvals accepts, so it has its id and file. */
478
+
479
+
480
+
481
+
482
+
483
+
484
+ /** A requirement as a score reports it: a term, or the group it was. */
485
+
486
+
487
+ /** How a graded case read a reply: the same shape both engines agree on. */
488
+
489
+
490
+
491
+
492
+
493
+
494
+
495
+
496
+
497
+
498
+
499
+
500
+
501
+
502
+
503
+
504
+ /** A run's totals, summed across cases. */
505
+
506
+
507
+
508
+
509
+
510
+
511
+
512
+ /** The whole-run verdict of a test type, where it has one of its own. */
513
+
514
+
515
+
516
+
517
+
518
+
519
+
520
+
521
+
522
+
523
+ /** One thing a test asserts, as its type names it: "Exact" "outdoor". */
524
+
525
+
526
+
527
+
528
+
529
+
530
+ /** A rule read against a scenario's replies, and what broke it if it failed. */
531
+
532
+
533
+
534
+
535
+
536
+ /** A transcript stage as a stored row carries it: an older row may lack
537
+ anything but its number. */
538
+
539
+
540
+ /** A job's result as a stored row carries it. */
541
+
542
+
543
+ /** One scenario's cell of a queue row's result item. */
544
+
545
+
546
+
547
+
548
+
549
+
550
+
551
+
552
+ /** A per-item test's score as a stored row carries it. */
553
+
554
+
555
+
556
+
557
+
558
+ // ---- Metrics (docs/pipeline-model.md § Metrics) ----
559
+
560
+ /** One check of a reply: a METRICS entry's type and its options, and how it
561
+ counts -- `not` turns it round, `weight` is what it is worth (negative
562
+ takes points away), `metric` names the group it is reported under. */
563
+
564
+
565
+
566
+
567
+
568
+
569
+
570
+
571
+
572
+
573
+
574
+
575
+
576
+ /** What a metric reads of one reply. */
577
+
578
+
579
+
580
+
581
+
582
+
583
+
584
+
585
+
586
+
587
+
588
+
589
+
590
+
591
+
592
+
593
+
594
+
595
+
596
+
597
+
598
+ /** What a metric needs besides the reply: a model to grade with. */
599
+
600
+
601
+
602
+
603
+
604
+
605
+ /** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
606
+
607
+
608
+
609
+
610
+
611
+
612
+
613
+
614
+
615
+
616
+
617
+ /** One metric's reading of one reply, as a stored row keeps it. */
618
+
619
+
620
+
621
+
622
+
623
+
624
+
625
+
626
+
627
+ /** A kind of check. A registry: a check a flow or a dataset needs is one
628
+ more entry, and no reader changes. */
629
+
630
+
631
+
632
+
633
+
634
+
635
+
636
+
637
+
638
+
639
+
640
+
641
+
642
+
643
+
644
+
645
+
646
+
647
+
648
+
649
+
650
+
651
+ /** What an earlier test that failed and does not continue leaves a later one. */
652
+
653
+
654
+
655
+
656
+ /** One test's reading of one scenario of a run (scenarioTests). */
657
+
658
+
659
+
660
+
661
+
662
+
663
+
664
+
665
+
666
+
667
+
668
+
669
+
670
+
671
+
672
+
673
+
674
+
675
+
676
+
677
+ /** One item of a queue row's results: a file or the pasted text. */
678
+
679
+
680
+
681
+
682
+
683
+
684
+
685
+
686
+ // ---- the rules file ----
687
+
688
+ /** A bound on a number: either end may be left open. */
689
+
690
+
691
+
692
+
693
+
694
+ /** A rule as a version-2 dataset held it, scoped to one item or to the whole reply. Read
695
+ only to upgrade a pipeline graded against such a dataset: `legacyTagsOut`
696
+ turns these into a List's Drop items and Reject rules. */
697
+
698
+
699
+
700
+
701
+
702
+
703
+
704
+
705
+ /** A version-2 dataset's `rules`: `{ rules: DudRule[] }`. */
706
+
707
+
708
+
709
+
710
+
711
+ // ---- the registries ----
712
+
713
+ /** An option a modifier or a test type exposes for editing. */
714
+
715
+
716
+
717
+
718
+
719
+
720
+
721
+
722
+
723
+
724
+ /** What an output kind made of a reply. */
725
+
726
+
727
+
728
+
729
+
730
+
731
+
732
+
733
+
734
+ /** What a modifier is told besides its value: the stage's own instruction,
735
+ and the text the value was read from. */
736
+
737
+
738
+
739
+
740
+
741
+ /** A modifier's answer when it did more than change the value: what it
742
+ removed, or that the whole answer is rejected and why. */
743
+
744
+
745
+
746
+
747
+
748
+
749
+ /**
750
+ * How a job's reply is read, and what its value can be tested for. [V] is
751
+ * the value a reply reads as -- text, or a shape of its own, such as a list
752
+ * of items. Methods, so a kind with a value of its own is
753
+ * assignable to the registry that holds every kind.
754
+ */
755
+
756
+
757
+
758
+
759
+
760
+
761
+
762
+
763
+
764
+
765
+
766
+
767
+
768
+
769
+
770
+
771
+
772
+
773
+
774
+
775
+
776
+
777
+
778
+
779
+ /** A change applied to a job's parsed value. */
780
+
781
+
782
+
783
+
784
+
785
+
786
+
787
+
788
+
789
+
790
+
791
+
792
+
793
+
794
+ /** What validation looks references up in: `profiles(id)` and `sources(id)`
795
+ answer with what the id names, or nothing; `datasets` is the datasets the
796
+ lab holds, as `GET /api/datasets` lists them. Each is optional, and a
797
+ reference nothing is given to look up is not checked. */
798
+
799
+
800
+
801
+
802
+
803
+
804
+
805
+
806
+
807
+ /** Lookups, and whether it is a run document being validated. */
808
+
809
+
810
+
811
+
812
+ /** A kind of Source: what one of its items is, and what the page and the
813
+ runner do with it. A reader asks the entry; it never branches on an id. */
814
+
815
+
816
+
817
+
818
+
819
+
820
+
821
+
822
+
823
+
824
+
825
+
826
+
827
+
828
+
829
+
830
+
831
+
832
+
833
+
834
+
835
+
836
+
837
+
838
+
839
+
840
+
841
+
842
+
843
+
844
+
845
+
846
+
847
+
848
+
849
+ /** A test type: what it checks of a document, and how it scores a run. */
850
+
851
+
852
+
853
+
854
+
855
+
856
+
857
+
858
+
859
+
860
+
861
+
862
+
863
+
864
+
865
+
866
+
867
+
868
+
869
+
870
+
871
+
872
+
873
+
874
+
875
+
876
+
877
+
878
+
879
+
880
+
881
+
882
+
883
+
884
+
885
+
886
+
887
+ /** What a test's `read` is handed besides the reply: production's reply to
888
+ the item, and a grader, where the run has them. */
889
+
890
+
891
+
892
+
893
+
894
+
895
+
896
+
897
+ /** A job's slots, in the order they run. */
898
+
899
+ const SLOTS = ["content", "call", "reply"];
900
+
901
+
902
+
903
+
904
+
905
+  
906
+
907
+
908
+
909
+
910
+
911
+
912
+
913
+
914
+
915
+
916
+
917
+
918
+
919
+
920
+
921
+
922
+
923
+
924
+
925
+
926
+
927
+
928
+
929
+
930
+
931
+
932
+
933
+
934
+
935
+
936
+
937
+
938
+
939
+
940
+
941
+
942
+ /**
943
+ * A dataset: data, never code. The lab keeps each one as a row in SQLite
944
+ * (server.py's `datasets` table), and a graded run carries the body it was
945
+ * submitted against, so nothing grades from a file.
946
+ */
947
+
948
+
949
+
950
+
951
+
952
+ /**
953
+ * [body] as this version of a dataset (4), from any earlier one. Version 1
954
+ * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
955
+ * `replays` and `conformance` (now fixtures/replays.json beside the checks).
956
+ * Version 2 held `rules`, which clean a job's answer and so belong to the
957
+ * job (docs/pipeline-model.md §13): they leave, and the terms a case
958
+ * watches for are its `watch`. Version 3 held the `prompt` a new scenario
959
+ * started from, which the Prompt library holds now: it leaves. Every reader of a body calls this: the runner, the page, a
960
+ * run's kept copy. A reader that needs a version-2 body's rules -- to upgrade
961
+ * a pipeline graded against it -- takes them first (`datasetRules`).
962
+ * Anything else comes back as it was.
963
+ */
964
+ function upgradeDatasetBody (body ) {
965
+ if (!isObj(body)) return body;
966
+ if (!("imageCases" in body || "rules" in body)) {
967
+ if (!("prompt" in body)) return body;
968
+ const { prompt: _library, ...rest } = body ;
969
+ return rest ;
970
+ }
971
+ const b = body ;
972
+ // vocab: the names older versions gave these fields
973
+ const RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch" }; // vocab: as above
974
+ const renamed = (c ) => {
975
+ if (!isObj(c)) return c;
976
+ const out = {};
977
+ for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
978
+ return out;
979
+ };
980
+ const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
981
+ // Not a body of any version -- a copy kept while a dataset was an overlay.
982
+ if (!raw) return body;
983
+ return { cases: (canonicalCases({ cases: raw.map(renamed) }) ).cases };
984
+ }
985
+
986
+ /** The rules a version-1 or version-2 dataset body held, or null: what a
987
+ pipeline graded against it was read under, before rules were the job's. */
988
+ function datasetRules(body ) {
989
+ return isObj(body) && isObj(body.rules) && Array.isArray(body.rules.rules) ? body.rules : null;
990
+ }
991
+
992
+ /** Registry entries a module adds in one call: see kinds/list.ts. */
993
+
994
+
995
+
996
+
997
+
998
+
999
+
1000
+
1001
+
1002
+
1003
+ /** A connection type's setting, as the form draws it. */
1004
+
1005
+
1006
+
1007
+
1008
+
1009
+
1010
+
1011
+
1012
+
1013
+ /** An HTTP request as a connection type builds it, for the relay to send. */
1014
+
1015
+
1016
+
1017
+
1018
+
1019
+
1020
+
1021
+
1022
+ /** A registered connection type: its label, its settings, how it transports. */
1023
+
1024
+
1025
+
1026
+
1027
+
1028
+
1029
+
1030
+
1031
+
1032
+
1033
+
1034
+
1035
+
1036
+
1037
+
1038
+
1039
+
1040
+
1041
+
1042
+
1043
+
1044
+
1045
+
1046
+
1047
+
1048
+
1049
+
1050
+
1051
+
1052
+
1053
+
1054
+
1055
+
1056
+
1057
+
1058
+
1059
+
1060
+
1061
+
1062
+
1063
+
1064
+
1065
+
1066
+
1067
+
1068
+
1069
+ /** What a Call hands its connection. */
1070
+
1071
+
1072
+ /** What the core hands a registry module (kinds/list.ts) to register with. */
1073
+
1074
+
1075
+
1076
+
1077
+ // The core reads no dataset as it loads. A dataset is data the lab keeps in
1078
+ // SQLite -- its cases and its default prompt -- and whoever
1079
+ // holds one hands it over: the worker the body a run was submitted with, the
1080
+ // page the one it fetched. So the page, the worker and the checks all load
1081
+ // the same module with nothing carried beside it.
1082
+
1083
+ // A new job's token mappings: none. A mapping is what a prompt of the
1084
+ // pipeline's own asks for, so the lab names none of its own; a prompt that
1085
+ // asks for a token no mapping fills is refused, naming it (`jobProblem`),
1086
+ // and `mappingsFor` maps the ones a given prompt names, blank.
1087
+ const TOKEN_DEFAULTS = [];
1088
+
1089
+ /** A blank mapping for every token [prompt] names that only a mapping can
1090
+ fill: a {name}…{/name} pair as a block, off; any other {name} as a value,
1091
+ empty. What a pipeline built around a dataset's prompt starts from, so it
1092
+ names what to fill in rather than refusing to run. */
1093
+ function mappingsFor(prompt ) {
1094
+ const offered = new Set(Object.values(OUTPUT_KINDS).flatMap(k => k.offers || []));
1095
+ const names = [...String(prompt).matchAll(TOKEN)].map(m => m[1] );
1096
+ const closed = new Set(names.filter(n => n.startsWith("/")).map(n => n.slice(1)));
1097
+ const out = [];
1098
+ for (const name of names) {
1099
+ if (name.startsWith("/") || name === "text" || name === "reply" || offered.has(name)
1100
+ || /^stage[1-9]\d*(?:\..+)?$/.test(name) || out.some(m => m.name === name)) continue;
1101
+ out.push(closed.has(name) ? { name, type: "block", enabled: false } : { name, type: "value", value: "" });
1102
+ }
1103
+ return out;
1104
+ }
1105
+
1106
+ /** A token type: what a mapping of it holds, how a prompt writes it, and
1107
+ what it does to a prompt. The page and the core read these; neither
1108
+ names a type. */
1109
+
1110
+
1111
+
1112
+
1113
+
1114
+
1115
+  
1116
+
1117
+
1118
+
1119
+
1120
+
1121
+
1122
+
1123
+
1124
+
1125
+
1126
+
1127
+ const TOKEN_TYPES = Object.create(null);
1128
+ TOKEN_TYPES.value = {
1129
+ label: "Value",
1130
+ fields: ["value"],
1131
+ defaults: () => ({ value: "" }),
1132
+ notation: name => `{${name}}`,
1133
+ names: name => [name],
1134
+ order: 1,
1135
+ apply: (prompt, m) => prompt.replaceAll(`{${m.name}}`, m.value ?? ""),
1136
+ validate(m, at, bad){ if (!isStr(m.value)) bad.push(`${at}: a value's value has to be text`); },
1137
+ };
1138
+ TOKEN_TYPES.block = {
1139
+ label: "Conditional block",
1140
+ fields: ["enabled"],
1141
+ defaults: () => ({ enabled: true }),
1142
+ notation: name => `{${name}}…{/${name}}`,
1143
+ names: name => [name, `/${name}`],
1144
+ order: 0,
1145
+ apply(prompt, m){
1146
+ const open = `{${m.name}}`, close = `{/${m.name}}`;
1147
+ let out = prompt;
1148
+ for (;;) {
1149
+ const s = out.indexOf(open);
1150
+ const e = out.indexOf(close, s + open.length);
1151
+ if (s < 0 || e < 0) break;
1152
+ const inner = out.slice(s + open.length, e);
1153
+ out = out.slice(0, s) + (m.enabled ? inner : "") + out.slice(e + close.length);
1154
+ }
1155
+ return out;
1156
+ },
1157
+ validate(m, at, bad){ if (typeof m.enabled !== "boolean") bad.push(`${at}: a block's enabled has to be true or false`); },
1158
+ };
1159
+
1160
+ /** A mapping of [type] named [name], holding its type's defaults. */
1161
+ function tokenMapping(name , type ) {
1162
+ return { name, type, ...(TOKEN_TYPES[type]?.defaults() ?? {}) };
1163
+ }
1164
+
1165
+ /** Every reason [list] is not a job's token mappings; [at] names the job. */
1166
+ function tokenMappingsProblems(list , at , bad ) {
1167
+ if (!Array.isArray(list)) return void bad.push(`${at}: tokenMappings has to be a list`);
1168
+ const seen = new Set ();
1169
+ list.forEach((m , i ) => {
1170
+ const where = `${at}'s token mapping ${i + 1}`;
1171
+ if (!isObj(m)) return void bad.push(`${where} has to be an object`);
1172
+ if (!isStr(m.name) || !m.name.trim()) return void bad.push(`${where} has no name`);
1173
+ const type = TOKEN_TYPES[m.type];
1174
+ if (!type) return void bad.push(`${where} ({${m.name}}) is of type "${m.type}", which is not a token type this lab has`);
1175
+ onlyFields(m, where, ["name", "type", ...type.fields], bad);
1176
+ type.validate(m, `${where} ({${m.name}})`, bad);
1177
+ if (seen.has(m.name)) bad.push(`${at} maps {${m.name}} twice`);
1178
+ seen.add(m.name);
1179
+ });
1180
+ }
1181
+
1182
+ /**
1183
+ * Mirrors PromptStore.resolve, generalised: values substitute, paired blocks
1184
+ * survive or vanish. The app applies exactly this to {language} and {geo}.
1185
+ *
1186
+ * The tokens are a parameter and not a global, because only one of the three
1187
+ * callers has a localStorage to have loaded a set from.
1188
+ */
1189
+ function resolvePrompt(tpl , tokens ) {
1190
+ const order = (m ) => TOKEN_TYPES[m.type]?.order ?? Infinity;
1191
+ let out = tpl;
1192
+ for (const m of [...tokens].sort((a, b) => order(a) - order(b))) {
1193
+ const type = TOKEN_TYPES[m.type];
1194
+ if (type) out = type.apply(out, m);
1195
+ }
1196
+ // Dropping a block leaves the spaces that surrounded it.
1197
+ return out.replace(/[ \t]{2,}/g, " ").trim();
1198
+ }
1199
+
1200
+ /**
1201
+ * The token set a stage reads. A scenario can carry one set for every stage,
1202
+ * or one per stage -- a job owns its set, so a per-job setup hands each
1203
+ * stage the set its job was given. A missing entry falls back to the
1204
+ * defaults either way, so an older scenario without them still runs.
1205
+ */
1206
+ function tokenSet(tokens , k ) {
1207
+ // One job's list is a list of mappings; one list per stage is a list of
1208
+ // lists. An empty list is one job's, mapping nothing.
1209
+ const perStage = tokens.length > 0 && tokens.every(t => t === undefined || Array.isArray(t));
1210
+ return perStage ? ((tokens )[k] || TOKEN_DEFAULTS) : tokens ;
1211
+ }
1212
+
1213
+ /** The names a list of mappings answers to inside braces. */
1214
+ const tokenNames = (tokens ) =>
1215
+ tokens.flatMap(m => TOKEN_TYPES[m.type]?.names(m.name) ?? []);
1216
+
1217
+ /**
1218
+ * Where a common piece of text goes inside an instruction: at `{text}` if the
1219
+ * wording places it, otherwise on a line of its own at the end.
1220
+ *
1221
+ * Shared rather than written twice because both pages do this and they have
1222
+ * to agree: text content and image content are the same question
1223
+ * asked on two screens, and a wording carried between them that meant
1224
+ * different things would be the drift this file exists to stop. The line
1225
+ * break matters -- gluing the text onto the end of the instruction produced
1226
+ * "of this text.one two three" in issue #310.
1227
+ */
1228
+ function textPrompt(instruction , text ) {
1229
+ return instruction.includes("{text}") ? instruction.replaceAll("{text}", text)
1230
+ : `${instruction}\n${text}`;
1231
+ }
1232
+
1233
+ // Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
1234
+ // work in, so that "New Zealand." and "new zealand" are the same two words.
1235
+ const words = (s ) => String(s||"").toLowerCase().match(/[\p{L}\p{N}]+/gu) || [];
1236
+
1237
+ // The request Tagger builds, field for field -- when the target reads those
1238
+ // fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
1239
+ // not, and reject unknown arguments outright ("Unrecognized request
1240
+ // arguments supplied"), which is what a run against one of them died on.
1241
+ // For those the request carries only the portable OpenAI shape, with
1242
+ // max_completion_tokens standing in for n_predict (max_tokens is rejected
1243
+ // by gpt-5 and other reasoning models: "Use 'max_completion_tokens'
1244
+ // instead"). The hosts named here are the same two the relay allow-lists
1245
+ // as providers, so what is named is what can be reached at all.
1246
+ const HOSTED = new Set(["api.openai.com", "openrouter.ai"]);
1247
+
1248
+ /**
1249
+ * Whether an endpoint takes the hosted request shape rather than llama.cpp's.
1250
+ *
1251
+ * Shared with the connection editor so the UI and the request agree on what a
1252
+ * connection is: the editor was offering the Speed profile on hosted
1253
+ * endpoints, where building the body then threw.
1254
+ */
1255
+ function isHostedUrl(url = "") {
1256
+ try {
1257
+ return HOSTED.has(new URL(/:\/\//.test(url) ? url : "http://" + url).hostname.toLowerCase());
1258
+ } catch { return false; } // unreadable means unrecognised; llama.cpp is the default
1259
+ }
1260
+
1261
+ // Kept in step with Tagger.SEED, so the lab and the handset decode alike.
1262
+ const SEED = 1234;
1263
+
1264
+ // The reply-token cap every OpenAI-compatible, llama.cpp and Anthropic request
1265
+ // carried before a profile had to set one (generic-lab step 4). Now a request
1266
+ // carries none unless its profile sets it, so a long answer is not cut to one
1267
+ // app's length; `pinReplyTokens` writes this into a profile saved before, so
1268
+ // what it measures does not move.
1269
+ const REPLY_TOKENS_BEFORE = 160;
1270
+ // The Messages API will not answer without a max_tokens, so an Anthropic
1271
+ // profile that sets none still sends one: a ceiling, not a length.
1272
+ const ANTHROPIC_MAX_TOKENS = 4096;
1273
+
1274
+ // The decoding fields a llama.cpp profile carries, as ordinary settings. The
1275
+ // measured presets of these are data, Setup's presets, so nothing here is one
1276
+ // app's candidate list.
1277
+ const DECODING_KEYS = ["repeatPenalty", "frequencyPenalty", "presencePenalty",
1278
+ "dryMultiplier", "dryBase", "dryAllowedLength", "grammar"];
1279
+
1280
+ /** A field the user may have left blank. Blank means "use the default". */
1281
+ function asNumber(v ) {
1282
+ if (v == null) return null;
1283
+ const t = String(v).trim();
1284
+ if (t === "" || !/^-?\d+(\.\d+)?$/.test(t)) return null;
1285
+ const n = parseFloat(t);
1286
+ return isFinite(n) ? n : null;
1287
+ }
1288
+
1289
+ // ---- the connection-type registry (#46) -----------------------------------
1290
+ // A Setup profile is one registered connection type plus that type's own
1291
+ // settings; nothing is one app's shape. The registry is where the transport
1292
+ // lives -- how a request is built, how a reply is read, how models are listed
1293
+ // -- and what the form offers comes from it too, so a new type is a new entry
1294
+ // and nothing that reads a connection changes. The ones that ship are generic;
1295
+ // a tuned local engine is a llama.cpp profile with its settings.
1296
+ const CONNECTION_TYPES = Object.create(null);
1297
+
1298
+ /** The key [key] as the header most endpoints expect, when there is one. */
1299
+ const bearer = (key ) => (key ? { Authorization: "Bearer " + key } : {});
1300
+
1301
+ /**
1302
+ * An entered address normalised to the /v1 base the chat paths hang off: a
1303
+ * bare host and port, a base URL, a /v1, or the full chat path. The runner,
1304
+ * the relay and the page's Test connection all build URLs from one copy.
1305
+ */
1306
+ function apiBase(raw ) {
1307
+ let s = String(raw || "").trim().replace(/\/+$/, "");
1308
+ if (!s) return "";
1309
+ if (!/^https?:\/\//.test(s)) s = "http://" + s;
1310
+ if (s.endsWith("/chat/completions")) s = s.slice(0, -"/chat/completions".length);
1311
+ return s.endsWith("/v1") ? s : s + "/v1";
1312
+ }
1313
+
1314
+ /** The model ids a /v1/models reply lists: every type serves that shape. */
1315
+ const idsFrom = (j ) => ((j )?.data || []).map(m => m.id);
1316
+
1317
+ // The models list every type reaches at its base + /models, for the page's
1318
+ // Test connection through the relay; only the auth differs (Anthropic's).
1319
+ const listModels = (base , key , auth = bearer(key)) => ({ url: base + "/models", headers: auth });
1320
+
1321
+ /** A /v1/chat/completions reply, as the OpenAI-shaped types read it. */
1322
+ function chatReply(j ) {
1323
+ const r = j ;
1324
+ return { raw: r?.choices?.[0]?.message?.content ?? "", finishReason: r?.choices?.[0]?.finish_reason ?? null };
1325
+ }
1326
+
1327
+ function ollamaType(id , label , cloud ) {
1328
+ return {
1329
+ id, label,
1330
+ description: cloud ? "Ollama's cloud at ollama.com, with a key." : "Ollama on a machine of your own; it takes no key.",
1331
+ settings: [
1332
+ { key: "seed", label: "Seed", control: "text", default: "", hint: "" },
1333
+ { key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
1334
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1335
+ { key: "repeatPenalty", label: "Repeat penalty", control: "text", default: "", hint: "" },
1336
+ ],
1337
+ // Ollama's native /api/chat, at the server root rather than under /v1.
1338
+ request(conn, prompt, dataUrl, base, key) {
1339
+ // Seeded by default, like every profile: an eval graded against a
1340
+ // resample is grading noise, and the lab exists to tell wordings apart.
1341
+ const options = { seed: asNumber(conn.seed) ?? SEED };
1342
+ const num_predict = asNumber(conn.nPredict); if (num_predict != null) options.num_predict = num_predict;
1343
+ const temperature = asNumber(conn.temperature); if (temperature != null) options.temperature = temperature;
1344
+ const repeat_penalty = asNumber(conn.repeatPenalty); if (repeat_penalty != null) options.repeat_penalty = repeat_penalty;
1345
+ const message = { role: "user", content: prompt };
1346
+ // Ollama's own endpoint takes the bytes as bare base64, where every /v1
1347
+ // type takes a data URL. Handing it the whole URL puts "data:image/..."
1348
+ // through its decoder as payload, and it refuses the image outright --
1349
+ // "illegal base64 data at input byte 4", once per item, for the whole run.
1350
+ if (dataUrl) message.images = [String(dataUrl).replace(/^data:[^,]*,/, "")];
1351
+ return {
1352
+ url: base.replace(/\/v1$/, "") + "/api/chat",
1353
+ headers: cloud ? bearer(key) : {},
1354
+ body: { model: conn.model, messages: [message], options, stream: false },
1355
+ };
1356
+ },
1357
+ parseReply(j) {
1358
+ const r = j ;
1359
+ return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
1360
+ },
1361
+ listModels: (base, key) => listModels(base, cloud ? key : null),
1362
+ parseModels: idsFrom,
1363
+ ...(cloud ? { url: "https://ollama.com" } : { keyless: true }),
1364
+ };
1365
+ }
1366
+ // Two types, one API: Ollama on a machine of your own takes no key -- there
1367
+ // is nowhere to set one -- and Ollama's cloud (ollama.com) takes one as a
1368
+ // bearer token and has an address of its own.
1369
+ CONNECTION_TYPES.ollama = ollamaType("ollama", "Ollama (Local)", false);
1370
+ CONNECTION_TYPES["ollama-cloud"] = ollamaType("ollama-cloud", "Ollama (Cloud)", true);
1371
+
1372
+ // Echo answers each text item with the item's own text, so a pipeline can be
1373
+ // graded over replies recorded earlier -- a Source of them, one a file --
1374
+ // with no model at all (#97). Over Prompt only it answers with the
1375
+ // scenario's prompt, so a reply can be written straight into the prompt and
1376
+ // each scenario holds one. Only job 1 can use it: a later job's item is the
1377
+ // job before's answer, not a file.
1378
+ const sendsNothing = () => { throw new Error("Echo answers from the item and sends no request"); };
1379
+ CONNECTION_TYPES.echo = {
1380
+ id: "echo", label: "Echo", settings: [], keyless: true,
1381
+ description: "No model: answers with each text item itself, to grade replies recorded earlier.",
1382
+ local: (item, sent) => item.text ?? sent,
1383
+ request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1384
+ };
1385
+
1386
+ /** A local type's answer to one stage, as a model's would come back. An item
1387
+ with no text of its own (Prompt only) is answered from the prompt, and
1388
+ that reply is read against nothing -- the prompt is the reply, not an
1389
+ instruction it could repeat. */
1390
+ function localAnswer(conn , text , sent ) {
1391
+ const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent);
1392
+ return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
1393
+ }
1394
+
1395
+ CONNECTION_TYPES["openai-compatible"] = {
1396
+ id: "openai-compatible", label: "OpenAI-compatible",
1397
+ description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
1398
+ settings: [
1399
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1400
+ { key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
1401
+ ],
1402
+ // /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio. The
1403
+ // reasoning-model temperature rule and max_completion_tokens stay with this
1404
+ // type, as the hosted providers reject the fields a llama.cpp server takes.
1405
+ request(conn, prompt, dataUrl, base, key) {
1406
+ const hosted = isHostedUrl(base);
1407
+ const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
1408
+ const temp = asNumber(conn.temperature);
1409
+ const predict = asNumber(conn.nPredict);
1410
+ return {
1411
+ url: base + "/chat/completions",
1412
+ headers: bearer(key),
1413
+ body: {
1414
+ model: conn.model,
1415
+ ...(isReasoning || temp == null ? {} : { temperature: temp }),
1416
+ ...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
1417
+ messages: [{ role: "user", content: [
1418
+ { type: "text", text: prompt },
1419
+ ...(dataUrl ? [{ type: "image_url", image_url: { url: dataUrl } }] : []),
1420
+ ] }],
1421
+ },
1422
+ };
1423
+ },
1424
+ parseReply: chatReply,
1425
+ listModels, parseModels: idsFrom,
1426
+ };
1427
+
1428
+ CONNECTION_TYPES.anthropic = {
1429
+ id: "anthropic", label: "Anthropic",
1430
+ description: "Claude, through Anthropic's Messages API.",
1431
+ settings: [
1432
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1433
+ { key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
1434
+ ],
1435
+ // The Messages API: the x-api-key and anthropic-version headers, base64
1436
+ // image content blocks, and /v1/models for the list.
1437
+ request(conn, prompt, dataUrl, base, key) {
1438
+ const temp = asNumber(conn.temperature);
1439
+ const predict = asNumber(conn.maxTokens) ?? ANTHROPIC_MAX_TOKENS;
1440
+ const content = [];
1441
+ if (dataUrl) {
1442
+ const m = /^data:(image\/[^;]+);base64,(.*)$/s.exec(dataUrl);
1443
+ if (m) content.push({ type: "image", source: { type: "base64", media_type: m[1], data: m[2] } });
1444
+ }
1445
+ content.push({ type: "text", text: prompt });
1446
+ return {
1447
+ url: base + "/messages",
1448
+ headers: { ...(key ? { "x-api-key": key } : {}), "anthropic-version": "2023-06-01" },
1449
+ body: {
1450
+ model: conn.model,
1451
+ max_tokens: predict,
1452
+ ...(temp == null ? {} : { temperature: temp }),
1453
+ messages: [{ role: "user", content }],
1454
+ },
1455
+ };
1456
+ },
1457
+ parseReply(j) {
1458
+ const r = j ;
1459
+ const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
1460
+ return { raw, finishReason: r?.stop_reason ?? null };
1461
+ },
1462
+ listModels: (base, key) => listModels(base, key,
1463
+ { ...(key ? { "x-api-key": key } : {}), "anthropic-version": "2023-06-01" }),
1464
+ parseModels: idsFrom,
1465
+ };
1466
+
1467
+ // The values Speed sent today, kept as a converted Speed profile's settings
1468
+ // so a llama.cpp profile reproduces the Speed request byte for byte.
1469
+ const SPEED_IMAGE = "before", SPEED_CACHE = "off", SPEED_THINKING = "off";
1470
+
1471
+ CONNECTION_TYPES["llama.cpp"] = {
1472
+ id: "llama.cpp", label: "llama.cpp",
1473
+ description: "A llama-server you run, with its sampler settings.",
1474
+ settings: [
1475
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1476
+ { key: "seed", label: "Seed", control: "text", default: "", hint: "" },
1477
+ { key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
1478
+ { key: "repeatPenalty", label: "Repeat penalty", control: "text", default: "", hint: "" },
1479
+ { key: "frequencyPenalty", label: "Frequency penalty", control: "text", default: "", hint: "" },
1480
+ { key: "presencePenalty", label: "Presence penalty", control: "text", default: "", hint: "" },
1481
+ { key: "dryMultiplier", label: "DRY multiplier", control: "text", default: "", hint: "" },
1482
+ { key: "dryBase", label: "DRY base", control: "text", default: "", hint: "" },
1483
+ { key: "dryAllowedLength", label: "DRY allowed length", control: "text", default: "", hint: "" },
1484
+ { key: "grammar", label: "Grammar", control: "text", default: "", hint: "" },
1485
+ { key: "imagePosition", label: "Image position", control: "select", default: "after",
1486
+ choices: [{ value: "before", label: "before the prompt" }, { value: "after", label: "after the prompt" }], hint: "" },
1487
+ { key: "cachePrompt", label: "Cache prompt", control: "select", default: "off",
1488
+ choices: [{ value: "on", label: "on" }, { value: "off", label: "off" }], hint: "" },
1489
+ { key: "thinking", label: "Thinking", control: "select", default: "off",
1490
+ choices: [{ value: "on", label: "on" }, { value: "off", label: "off" }], hint: "" },
1491
+ { key: "rejectLoops", label: "Reject looping replies", control: "select", default: "off",
1492
+ choices: [{ value: "on", label: "on" }, { value: "off", label: "off" }], hint: "" },
1493
+ ],
1494
+ request(conn, prompt, dataUrl, base, key) {
1495
+ if (isHostedUrl(base)) throw new Error("llama.cpp requires its llama-server, not a hosted model");
1496
+ const temp = asNumber(conn.temperature);
1497
+ const seed = asNumber(conn.seed) ?? SEED;
1498
+ const predict = asNumber(conn.nPredict);
1499
+ const penalty = asNumber(conn.repeatPenalty);
1500
+ const sampler = {
1501
+ frequency_penalty: asNumber(conn.frequencyPenalty),
1502
+ presence_penalty: asNumber(conn.presencePenalty),
1503
+ dry_multiplier: asNumber(conn.dryMultiplier),
1504
+ dry_base: asNumber(conn.dryBase),
1505
+ dry_allowed_length: asNumber(conn.dryAllowedLength),
1506
+ grammar: String(conn.grammar ?? "").trim() || null,
1507
+ };
1508
+ const content = [
1509
+ { type: "text", text: prompt },
1510
+ ...(dataUrl ? [{ type: "image_url", image_url: { url: dataUrl } }] : []),
1511
+ ];
1512
+ if ((conn.imagePosition || "after") === "before") content.reverse();
1513
+ return {
1514
+ url: base + "/chat/completions",
1515
+ headers: bearer(key),
1516
+ body: {
1517
+ model: conn.model,
1518
+ seed,
1519
+ ...(temp == null ? {} : { temperature: temp }),
1520
+ ...(penalty == null ? {} : { repeat_penalty: penalty }),
1521
+ ...Object.fromEntries(Object.entries(sampler).filter(([, v]) => v != null)),
1522
+ cache_prompt: (conn.cachePrompt || "off") === "on",
1523
+ ...(predict == null ? {} : { n_predict: predict }),
1524
+ chat_template_kwargs: { enable_thinking: (conn.thinking || "off") === "on" },
1525
+ messages: [{ role: "user", content }],
1526
+ },
1527
+ };
1528
+ },
1529
+ parseReply: chatReply,
1530
+ listModels, parseModels: idsFrom,
1531
+ // A reply that arrived is still a failure when this profile asked for the
1532
+ // Speed looping check (today's speedReplyError, now an opt-in setting).
1533
+ replyProblem(conn, raw, finishReason) {
1534
+ return (conn.rejectLoops || "off") === "on" ? loopReplyError(raw, finishReason) : null;
1535
+ },
1536
+ };
1537
+
1538
+ // ---- HTTP APIs: how a request body is shaped --------------------------------
1539
+ // An HTTP Request step sends a flow's own body; its HTTP API says where in
1540
+ // that body a scenario's words, system prompt, model, settings and prefill
1541
+ // go, and how the reply's text is read. Same API in and out: a request is
1542
+ // never translated to another provider's shape. "Any endpoint" holds the
1543
+ // whole body as its words, for an API the lab has no entry for.
1544
+
1545
+ /** A body shape an HTTP Request step can send. */
1546
+
1547
+
1548
+
1549
+
1550
+
1551
+
1552
+
1553
+
1554
+
1555
+
1556
+
1557
+
1558
+
1559
+
1560
+
1561
+
1562
+
1563
+
1564
+
1565
+ const HTTP_APIS = Object.create(null);
1566
+
1567
+ /** A setting's text as the body wants it: a number when it reads as one. */
1568
+ const settingValue = (v ) => {
1569
+ const t = v.trim();
1570
+ if (t === "true" || t === "false") return t === "true";
1571
+ return t !== "" && !Number.isNaN(Number(t)) ? Number(t) : v;
1572
+ };
1573
+ const pick = (o , keys ) =>
1574
+ Object.fromEntries(keys.filter(k => isObj(o) && o[k] != null && !isObj(o[k])).map(k => [k, isStr(o[k]) ? o[k] : JSON.stringify(o[k])]));
1575
+ /** A message's text: the string, or its first text block's. */
1576
+ const messageText = (m ) => isStr(m?.content) ? m.content
1577
+ : Array.isArray(m?.content) ? String(m.content.find((b ) => b?.type === "text")?.text ?? "") : "";
1578
+ const withMessageText = (m , text ) => !Array.isArray(m?.content) ? { ...m, content: text }
1579
+ : { ...m, content: m.content.map((b ) => (b?.type === "text" ? { ...b, text } : b)) };
1580
+ /** [body] with the cell's settings set: a blank one leaves the key out. */
1581
+ const withSettings = (body , cell , keys ) => {
1582
+ const out = { ...body };
1583
+ for (const [k, v] of Object.entries(cell.settings || {})) {
1584
+ if (!keys.includes(k)) continue;
1585
+ if (String(v).trim() === "") delete out[k];
1586
+ else out[k] = settingValue(v);
1587
+ }
1588
+ return out;
1589
+ };
1590
+
1591
+ HTTP_APIS.anthropic = {
1592
+ label: "Anthropic",
1593
+ matches: u => u.hostname === "api.anthropic.com",
1594
+ fields: ["system", "model", "settings", "prefill"],
1595
+ settings: ["max_tokens", "temperature", "top_p", "top_k"],
1596
+ cellOf(body) {
1597
+ const b = isObj(body) ? body : {};
1598
+ const msgs = Array.isArray(b.messages) ? b.messages : [];
1599
+ const user = [...msgs].reverse().find(m => m?.role === "user");
1600
+ const last = msgs.at(-1);
1601
+ return {
1602
+ prompt: messageText(user),
1603
+ ...(isStr(b.system) ? { system: b.system } : {}),
1604
+ ...(isStr(b.model) ? { model: b.model } : {}),
1605
+ settings: pick(b, this.settings),
1606
+ ...(last?.role === "assistant" ? { prefill: messageText(last) } : {}),
1607
+ };
1608
+ },
1609
+ place(body, cell) {
1610
+ const b = withSettings(isObj(body) ? body : {}, cell, this.settings);
1611
+ if (cell.model) b.model = cell.model;
1612
+ if (cell.system != null) { if (cell.system === "") delete b.system; else b.system = cell.system; }
1613
+ let msgs = Array.isArray(b.messages) ? [...b.messages] : [];
1614
+ if (msgs.at(-1)?.role === "assistant") msgs = msgs.slice(0, -1);
1615
+ const at = msgs.map(m => m?.role).lastIndexOf("user");
1616
+ if (at >= 0) msgs[at] = withMessageText(msgs[at], cell.prompt);
1617
+ else msgs.push({ role: "user", content: cell.prompt });
1618
+ if (cell.prefill) msgs.push({ role: "assistant", content: cell.prefill });
1619
+ return { ...b, messages: msgs };
1620
+ },
1621
+ reply(j) {
1622
+ const r = j ;
1623
+ return { raw: (r?.content || []).filter(b => b?.type === "text").map(b => b.text).join(""), finishReason: r?.stop_reason ?? null };
1624
+ },
1625
+ withPrefill(j, prefill) {
1626
+ const r = j ;
1627
+ if (!isObj(r) || !Array.isArray(r.content)) return j;
1628
+ let done = false;
1629
+ return { ...r, content: r.content.map((b ) => {
1630
+ if (done || b?.type !== "text") return b;
1631
+ done = true;
1632
+ return { ...b, text: prefill + (b.text ?? "") };
1633
+ }) };
1634
+ },
1635
+ };
1636
+
1637
+ // OpenAI's chat completions, and Azure OpenAI's, which is the same body at a
1638
+ // deployment's address.
1639
+ const openAiApi = (label , matches ) => ({
1640
+ label, matches,
1641
+ fields: ["system", "model", "settings"],
1642
+ settings: ["max_tokens", "max_completion_tokens", "temperature", "top_p"],
1643
+ cellOf(body) {
1644
+ const b = isObj(body) ? body : {};
1645
+ const msgs = Array.isArray(b.messages) ? b.messages : [];
1646
+ const system = msgs.find(m => m?.role === "system" || m?.role === "developer");
1647
+ return {
1648
+ prompt: messageText([...msgs].reverse().find(m => m?.role === "user")),
1649
+ ...(system ? { system: messageText(system) } : {}),
1650
+ ...(isStr(b.model) ? { model: b.model } : {}),
1651
+ settings: pick(b, this.settings),
1652
+ };
1653
+ },
1654
+ place(body, cell) {
1655
+ const b = withSettings(isObj(body) ? body : {}, cell, this.settings);
1656
+ if (cell.model) b.model = cell.model;
1657
+ let msgs = Array.isArray(b.messages) ? [...b.messages] : [];
1658
+ if (cell.system != null) {
1659
+ const at = msgs.findIndex(m => m?.role === "system" || m?.role === "developer");
1660
+ if (cell.system === "") msgs = msgs.filter((_, x) => x !== at);
1661
+ else if (at >= 0) msgs[at] = withMessageText(msgs[at], cell.system);
1662
+ else msgs.unshift({ role: "system", content: cell.system });
1663
+ }
1664
+ const at = msgs.map(m => m?.role).lastIndexOf("user");
1665
+ if (at >= 0) msgs[at] = withMessageText(msgs[at], cell.prompt);
1666
+ else msgs.push({ role: "user", content: cell.prompt });
1667
+ return { ...b, messages: msgs };
1668
+ },
1669
+ reply: chatReply,
1670
+ });
1671
+ HTTP_APIS.openai = openAiApi("OpenAI", u => u.hostname === "api.openai.com");
1672
+ HTTP_APIS["azure-openai"] = openAiApi("Azure OpenAI", u => u.hostname.endsWith(".openai.azure.com"));
1673
+
1674
+ // Any endpoint: the words are the whole body, as JSON with the flow's
1675
+ // expressions in it, and the reply is its body as text.
1676
+ HTTP_APIS.raw = {
1677
+ label: "Any endpoint",
1678
+ matches: () => true,
1679
+ fields: [],
1680
+ settings: [],
1681
+ cellOf: body => ({ prompt: isStr(body) ? body : JSON.stringify(body, null, 2) }),
1682
+ place(body, cell) {
1683
+ if (!isObj(body) && !Array.isArray(body)) return cell.prompt;
1684
+ try { return JSON.parse(cell.prompt); } catch { throw new Error("the body is not JSON"); }
1685
+ },
1686
+ reply: j => ({ raw: isStr(j) ? j : JSON.stringify(j), finishReason: null }),
1687
+ };
1688
+
1689
+ /** The HTTP API a flow's request to [uri] speaks: the first entry that
1690
+ matches, Any endpoint otherwise. */
1691
+ function httpApiFor(uri ) {
1692
+ let u = null;
1693
+ try { u = new URL(uri); } catch { return "raw"; }
1694
+ return Object.entries(HTTP_APIS).find(([id, e]) => id !== "raw" && e.matches(u ))?.[0] ?? "raw";
1695
+ }
1696
+
1697
+ // An HTTP endpoint: an address, and how its key travels. It sends whole
1698
+ // requests an HTTP Request step builds -- the flow's method, path and body
1699
+ // -- so it has no model and no chat path of its own. Its templates fill in
1700
+ // the common providers.
1701
+ const KEY_HEADERS = {
1702
+ bearer: key => ({ Authorization: "Bearer " + key }),
1703
+ "x-api-key": key => ({ "x-api-key": key }),
1704
+ "api-key": key => ({ "api-key": key }),
1705
+ none: () => ({}),
1706
+ };
1707
+ /** A profile's extra headers: `name: value` pairs, separated by `;`. */
1708
+ function extraHeaders(raw ) {
1709
+ const out = {};
1710
+ for (const part of String(raw ?? "").split(";")) {
1711
+ const at = part.indexOf(":");
1712
+ if (at > 0) out[part.slice(0, at).trim()] = part.slice(at + 1).trim();
1713
+ }
1714
+ return out;
1715
+ }
1716
+ /** An address as a whole-request type hangs paths off it: its origin, no /v1. */
1717
+ function originBase(raw ) {
1718
+ let s = String(raw || "").trim().replace(/\/+$/, "");
1719
+ if (!s) return "";
1720
+ if (!/^https?:\/\//.test(s)) s = "https://" + s;
1721
+ return s;
1722
+ }
1723
+ const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
1724
+ CONNECTION_TYPES.http = {
1725
+ id: "http", label: "HTTP endpoint",
1726
+ description: "Any API a Power Automate step calls: an address, a key and how the key is sent.",
1727
+ settings: [
1728
+ { key: "auth", label: "Key header", control: "select", default: "bearer", hint: "",
1729
+ choices: [{ value: "bearer", label: "Authorization: Bearer" }, { value: "x-api-key", label: "x-api-key" },
1730
+ { value: "api-key", label: "api-key" }, { value: "none", label: "none" }] },
1731
+ { key: "headers", label: "Headers", control: "text", default: "", hint: "" },
1732
+ // What Test connection asks of it: a GET, keyed as a run is.
1733
+ { key: "testPath", label: "Test path", control: "text", default: "", hint: "" },
1734
+ ],
1735
+ answers: ["request"],
1736
+ lists: false,
1737
+ baseOf: originBase,
1738
+ send(conn, req, base, key) {
1739
+ const query = new URLSearchParams(req.query).toString();
1740
+ return {
1741
+ method: req.method,
1742
+ url: base + req.path + (query ? (req.path.includes("?") ? "&" : "?") + query : ""),
1743
+ headers: { ...extraHeaders(conn.headers), ...(key ? KEY_HEADERS[String(conn.auth || "bearer")]?.(key) ?? {} : {}) },
1744
+ body: req.body,
1745
+ };
1746
+ },
1747
+ request: wholeRequestsOnly,
1748
+ parseReply: j => ({ raw: isStr(j) ? j : JSON.stringify(j), finishReason: null }),
1749
+ listModels: wholeRequestsOnly,
1750
+ parseModels: () => [],
1751
+ templates: [
1752
+ { id: "anthropic", label: "Anthropic",
1753
+ settings: { url: "https://api.anthropic.com", auth: "x-api-key", headers: "anthropic-version: 2023-06-01", testPath: "/v1/models" } },
1754
+ { id: "openai", label: "OpenAI", settings: { url: "https://api.openai.com", auth: "bearer", headers: "", testPath: "/v1/models" } },
1755
+ { id: "azure-openai", label: "Azure OpenAI",
1756
+ settings: { url: "https://RESOURCE.openai.azure.com", auth: "api-key", headers: "", testPath: "/openai/models?api-version=2024-10-21" } },
1757
+ ],
1758
+ // A key goes in the key box, where it is sent by the Key header and never
1759
+ // kept in a pipeline; a header that would carry one is refused.
1760
+ settingsProblems(conn) {
1761
+ const bad = [];
1762
+ if (!String(conn.url ?? "").trim()) bad.push("needs its address");
1763
+ for (const name of Object.keys(extraHeaders(conn.headers))) {
1764
+ if (/^(authorization|x-api-key|api-key|cookie)$/i.test(name) || /key|token|secret/i.test(name)) {
1765
+ bad.push(`sends a key in its ${name} header -- put the key in the key box`);
1766
+ }
1767
+ }
1768
+ return bad;
1769
+ },
1770
+ };
1771
+
1772
+ /** The registry entry for [conn]'s type. An unregistered one throws where
1773
+ it is used, as it always has: validation is what refuses one politely. */
1774
+ const typeEntry = (conn ) => CONNECTION_TYPES[typeOf(conn)] ;
1775
+
1776
+ /** The registered type a connection is, converting a legacy profile. */
1777
+ function typeOf(conn ) {
1778
+ return (conn?.type ) || profileType(conn);
1779
+ }
1780
+
1781
+ /**
1782
+ * The type an old `profile: "quality" | "speed"` connection converts to
1783
+ * (#46): quality is openai-compatible on a hosted or /v1 address, ollama on
1784
+ * a bare host; speed is llama.cpp.
1785
+ */
1786
+ function profileType(p ) {
1787
+ if (p?.profile === "speed") return "llama.cpp";
1788
+ if (p?.profile === "quality") {
1789
+ const url = String(p.url || "");
1790
+ return (isHostedUrl(url) || /\/v1(\/|$)/.test(url)) ? "openai-compatible" : "ollama";
1791
+ }
1792
+ return "ollama";
1793
+ }
1794
+
1795
+ /**
1796
+ * A stored profile converted once to the registry shape: keeps every field
1797
+ * and sets `type`, with the settings a converted Speed profile carries so its
1798
+ * request is byte-identical to today's. Runs once per profile on load.
1799
+ */
1800
+ function convertProfile(p ) {
1801
+ const out = { ...p };
1802
+ if (p.profile === "speed") {
1803
+ out.type = "llama.cpp";
1804
+ out.temperature = "0.1";
1805
+ out.repeatPenalty = "1.15";
1806
+ out.seed = "1234";
1807
+ out.imagePosition = SPEED_IMAGE;
1808
+ out.cachePrompt = SPEED_CACHE;
1809
+ out.thinking = SPEED_THINKING;
1810
+ out.rejectLoops = "on";
1811
+ out.nPredict = String(REPLY_TOKENS_BEFORE);
1812
+ } else {
1813
+ out.type = profileType(p);
1814
+ }
1815
+ delete out.profile;
1816
+ return out;
1817
+ }
1818
+
1819
+ /**
1820
+ * An Ollama profile from before Ollama was two types: one holding a key was
1821
+ * reaching a server that takes one, so it is the cloud type; one without is
1822
+ * a machine of the owner's own. Anything else comes back as it was.
1823
+ */
1824
+ /**
1825
+ * A profile saved before requests stopped carrying a reply-token cap unless
1826
+ * one was set: its type's Reply tokens setting, absent, is the cap it has
1827
+ * been sending, written in, so its runs go on measuring what they did. A
1828
+ * profile made since has the setting, blank or not, and comes back as it
1829
+ * was, as does one of a type that never sent the cap.
1830
+ */
1831
+ function pinReplyTokens(p ) {
1832
+ const key = ({ "openai-compatible": "nPredict", "llama.cpp": "nPredict", anthropic: "maxTokens" } )[String(p.type)];
1833
+ return !key || key in p ? p : { ...p, [key]: String(REPLY_TOKENS_BEFORE) };
1834
+ }
1835
+
1836
+ function splitOllama(p ) {
1837
+ if (p.type !== "ollama" || !String(p.key || "").trim()) return p;
1838
+ return { ...p, type: "ollama-cloud", url: String(p.url || "").trim() || CONNECTION_TYPES["ollama-cloud"] .url };
1839
+ }
1840
+
1841
+ /**
1842
+ * The request [conn] sends for [prompt] (and an image [dataUrl], if any),
1843
+ * as its type builds it: the URL, the headers, and the body. [base] is the
1844
+ * /v1-normalised base, [key] the auth key if there is one. The resolved
1845
+ * connection carries its type's settings under `options`, so they are merged
1846
+ * up before the type reads them.
1847
+ */
1848
+ function connectionRequest(conn , prompt , dataUrl , base , key ) {
1849
+ const flat = { ...conn, ...(conn?.options || {}) };
1850
+ return typeEntry(flat).request(flat, prompt, dataUrl, base, key);
1851
+ }
1852
+
1853
+ /** Where [conn]'s requests hang off: its type's own reading of its address,
1854
+ or the /v1 base a chat API's paths hang off. */
1855
+ function connectionBase(conn ) {
1856
+ const type = CONNECTION_TYPES[typeOf(conn)];
1857
+ return type?.baseOf ? type.baseOf(conn.url) : apiBase(conn.url);
1858
+ }
1859
+
1860
+ /** The HTTP request [conn] sends for an HTTP Request step's [req]. */
1861
+ function connectionSend(conn , req ,
1862
+ base , key ) {
1863
+ const flat = { ...conn, ...(conn?.options || {}) };
1864
+ const type = typeEntry(flat);
1865
+ if (!type.send) throw new Error(`a ${type.label} profile cannot send a whole request`);
1866
+ return type.send(flat, req, base, key);
1867
+ }
1868
+
1869
+ /** A reply [conn] received, read as its raw text and finish reason. */
1870
+ function connectionReply(conn , j ) {
1871
+ return typeEntry(conn).parseReply(j);
1872
+ }
1873
+
1874
+ /** Why a reply [conn] received is a failure though it arrived, or null. */
1875
+ function connectionReplyProblem(conn , raw , finishReason ) {
1876
+ return typeEntry(conn).replyProblem?.(conn, raw, finishReason) ?? null;
1877
+ }
1878
+
1879
+ /**
1880
+ * Whether a reply the Speed looping check guards against looped or truncated:
1881
+ * today's speedReplyError, now the llama.cpp type's rejectLoops setting.
1882
+ */
1883
+ function loopReplyError(reply , finishReason ) {
1884
+ if (finishReason === "length") return "the model repeated or truncated its answer";
1885
+ const items = String(reply).toLowerCase().split(/[,;\n]/).map(s => s.trim().replace(/^\.+|\.+$/g, "")).filter(Boolean);
1886
+ const frequencies = new Map ();
1887
+ for (const item of items) frequencies.set(item, (frequencies.get(item) || 0) + 1);
1888
+ if (items.length >= 4 && [...frequencies.values()].some(n => n >= 3)) return "the model repeated or truncated its answer";
1889
+ const ws = words(reply);
1890
+ for (let width = 1; width <= 6; width++) for (let start = 0; start + width * 3 <= ws.length; start++) {
1891
+ if (Array.from({length: width}, (_, n) => n).every(n => ws[start+n] === ws[start+n+width] && ws[start+n] === ws[start+n+width*2]))
1892
+ return "the model repeated or truncated its answer";
1893
+ }
1894
+ return null;
1895
+ }
1896
+
1897
+ // ---- Scoring -------------------------------------------------------------
1898
+ /**
1899
+ * The dimensions an image is sent at, for a budget.
1900
+ *
1901
+ * [target] is an area in pixels, or the string "edge448" for the one mode an
1902
+ * area cannot express -- see below.
1903
+ */
1904
+ function preparedSize(width , height , target ) {
1905
+ const ratio = width / height;
1906
+ // 448 on the longest side, aspect kept. Not an area budget and not
1907
+ // expressible as one: it is the single-tile case WITHOUT the squash that
1908
+ // `Tagger.prepare` applies, kept here so the two can be graded against each
1909
+ // other. A 4:3 image goes 448x336 rather than 448x448, so the tile
1910
+ // carries a letterboxed image instead of a distorted one -- which of those
1911
+ // the model reads better is a question, not a known.
1912
+ //
1913
+ // Lab-only: the app has no longest-edge mode, so a result measured here is
1914
+ // not something the handset can currently reproduce.
1915
+ if (target === EDGE_448) {
1916
+ const scale = 448 / Math.max(width, height);
1917
+ const snap = (v ) => Math.min(448, Math.max(56, Math.round(v * scale / 56) * 56));
1918
+ const tw = snap(width), th = snap(height);
1919
+ return width <= tw && height <= th ? [width, height] : [tw, th];
1920
+ }
1921
+ const square = Number(target) <= 448 * 448;
1922
+ const h = Math.sqrt(Number(target) / ratio);
1923
+ const tw = square ? 448 : Math.max(56, Math.round(h * ratio / 56) * 56);
1924
+ const th = square ? 448 : Math.max(56, Math.round(h / 56) * 56);
1925
+ return width <= tw && height <= th ? [width, height] : [tw, th];
1926
+ }
1927
+
1928
+ /** The one resolution setting that is a mode rather than a number. */
1929
+ const EDGE_448 = "edge448";
1930
+
1931
+ /** How a budget reads in a sentence, for the run record and the log. */
1932
+ function budgetLabel(target ) {
1933
+ return target === EDGE_448 ? "448 longest edge" : `${(Number(target) / 1e6).toFixed(1)}MP`;
1934
+ }
1935
+
1936
+ // What `docs/datasets.md` says the graded set means, in code.
1937
+
1938
+ // A term is present when its words appear in some item, in order and adjacent:
1939
+ // "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
1940
+ // looser scores "new zealand" against "zealandia" and flatters every run.
1941
+ function termIn(items , term ) {
1942
+ const t = words(term);
1943
+ if (!t.length) return false;
1944
+ return items.some(item => {
1945
+ const w = words(item);
1946
+ for (let i = 0; i + t.length <= w.length; i++) {
1947
+ if (t.every((x, j) => w[i + j] === x)) return true;
1948
+ }
1949
+ return false;
1950
+ });
1951
+ }
1952
+
1953
+ // A forbidden term counts wherever termIn finds it, except inside a item that
1954
+ // holds one of the case's `allow` phrases and that phrase holds the term:
1955
+ // "train car" is a fair item for a locomotive whose case forbids "car", and a
1956
+ // bare "car" beside it still counts.
1957
+ function forbiddenIn(items , term , allow ) {
1958
+ return items.some(item => termIn([item], term)
1959
+ && !(allow || []).some(a => termIn([a], term) && termIn([item], a)));
1960
+ }
1961
+
1962
+ /**
1963
+ * The graded half as one list, each entry carrying where it is graded.
1964
+ *
1965
+ * A named function rather than a bare array, so that something can check it.
1966
+ * Issue #270 was this list being built from one half of the old file alone, and an inline
1967
+ * assembly can quietly go back to that: nothing in CI can reach into an async
1968
+ * boot for the array it produced, so the fault would ship again with every
1969
+ * check green. `evals-check.js` calls this directly.
1970
+ */
1971
+ function gradedSetFrom(ev ) {
1972
+ return (ev.cases || []).map(c => ({ ...c, filename: caseFile(c), half: "cases" }) );
1973
+ }
1974
+
1975
+ /**
1976
+ * The file a case grades, whichever of the two keys names it.
1977
+ *
1978
+ * One reader, so that accepting the older spelling is a fact about this
1979
+ * function rather than a branch every caller carries. Everything that joins a
1980
+ * case to an item -- the runner, the Datasets tab, a mapping against a Source
1981
+ * -- goes through here.
1982
+ */
1983
+ function caseFile(kase ) { // vocab: the older spelling
1984
+ const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
1985
+ return typeof name === "string" ? name : "";
1986
+ }
1987
+
1988
+ /**
1989
+ * A graded set with every case naming its file as `filename`.
1990
+ *
1991
+ * What `evalsJson` writes, so the committed bytes carry one key and a set
1992
+ * re-synced from an app's own repository is normalised the first time it is
1993
+ * exported. The key takes the place the older one held, so normalising a set
1994
+ * changes the spelling of
1995
+ * one key and not the order of any.
1996
+ */
1997
+ function canonicalCases(ev ) {
1998
+ const OLD = "photo"; // vocab: the older spelling of filename
1999
+ const set = ev ;
2000
+ const cases = set && typeof set === "object" && Array.isArray(set.cases) ? set.cases : null;
2001
+ if (!cases || !cases.some(c => c && typeof c === "object" && OLD in (c ))) return ev;
2002
+ return {
2003
+ ...set,
2004
+ cases: cases.map(c => {
2005
+ if (!c || typeof c !== "object" || !(OLD in (c ))) return c;
2006
+ const out = {};
2007
+ for (const [k, v] of Object.entries(c )) {
2008
+ if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
2009
+ else out[k] = v;
2010
+ }
2011
+ return out;
2012
+ }),
2013
+ };
2014
+ }
2015
+
2016
+ /**
2017
+ * One graded case, one result.
2018
+ *
2019
+ * Every expectation is a group, and a plain `expect` term is a group of one:
2020
+ * the requirements are the `expect` terms and the `anyOf` groups alike, and
2021
+ * the score is found requirements over all of them plus `forbid`. A count
2022
+ * bound stays pass/fail: an item that produced two perfect terms when five
2023
+ * were wanted has not done what was asked.
2024
+ *
2025
+ * Pure on purpose, and returning `reasons` as plain sentences rather than
2026
+ * markup. The dashboard was the first consumer; `run-evals.js` is the second
2027
+ * and CI the third, and neither can reach into a page for a rendered cell.
2028
+ * Issue #248 wants a failure written out as a task an agent can act on.
2029
+ */
2030
+ function scoreCase(kase , res ) {
2031
+ // A case that expects its answer discarded passes on a discard and on
2032
+ // nothing else: the answer the job threw away is the finding.
2033
+ if (kase.discarded === true) {
2034
+ const thrown = !!res.error && res.error.startsWith("discarded: ");
2035
+ const n = (res.terms || []).length;
2036
+ return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
2037
+ unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
2038
+ reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
2039
+ }
2040
+ // A discarded answer is a failed item: whatever the pipeline produced
2041
+ // along the way does not count.
2042
+ const discarded = !!res.error;
2043
+ const terms = discarded ? [] : (res.terms || []);
2044
+ const expect = kase.expect || [], forbid = kase.forbid || [];
2045
+
2046
+ // `missed` stays populated for a discarded reply even though it reads as
2047
+ // vacuous, because the score depends on it: docs/datasets.md has such a reply
2048
+ // scoring zero against everything the entry asked for, groups included, and
2049
+ // `addToTally` gets there through `found + missed + invented`. Clear it and
2050
+ // a run that discarded every item would contribute nothing to the
2051
+ // total instead of contributing a nought, which flatters it.
2052
+ //
2053
+ // One member of a group is enough. Without this rule the set manufactures
2054
+ // failures out of synonyms.
2055
+ //
2056
+ // A requirement reads back as a term when it had one member and as the
2057
+ // group itself where a synonym list was allowed, so `missed` carries just
2058
+ // enough to state the reason: "FAIL: Missed dog" for a plain expect term,
2059
+ // "none of dog / puppy" for a group.
2060
+ //
2061
+ // Nothing satisfies a group when nothing was stored, so a discarded reply
2062
+ // misses every requirement rather than none -- the same cast that makes its
2063
+ // count bounds not breached by one. `unmet` is a finding only now that
2064
+ // `missed` names the unsatisfied groups itself, but the run report has
2065
+ // always carried it, so it stays.
2066
+ const met = (g ) => g.some(t => termIn(terms, t));
2067
+ const spoken = (g ) => g.length === 1 ? g[0] : g;
2068
+ const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
2069
+ const found = requirements.filter(met).map(spoken);
2070
+ const missed = requirements.filter(g => !met(g)).map(spoken);
2071
+ const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
2072
+ const unmet = discarded ? []
2073
+ : (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
2074
+
2075
+ const n = terms.length;
2076
+ // A null bound is not checked -- and neither is a bound on a reply that was
2077
+ // discarded. Nothing was counted out and found wanting there; the reply was
2078
+ // thrown away before it had a count, and saying "0 items, wanted at least 4"
2079
+ // states a second failure that never happened.
2080
+ const under = !discarded && kase.minCount != null && n < kase.minCount;
2081
+ const over = !discarded && kase.maxCount != null && n > kase.maxCount;
2082
+
2083
+ const denom = requirements.length + invented.length;
2084
+ // A case that passes is one whose whole expectation was met, not one that
2085
+ // scored well: an unsatisfied group is in `missed` alongside any term
2086
+ // missed, so pass needs nothing more than the terms already covered.
2087
+ const pass = !discarded && !missed.length && !invented.length && !under && !over;
2088
+
2089
+ // `found` and `missed` are groups, not terms, so the reasons word them one
2090
+ // by one: a bare term under one heading, a group on its own line.
2091
+ const reasons = [];
2092
+ if (discarded) {
2093
+ // Everything else would be derived from this one fact -- there are no
2094
+ // terms -- and would bury it. `scoreHtml` above suppresses `missed` for
2095
+ // the same reason: the reason that matters is already on the row.
2096
+ reasons.push(`Error - ${res.error}`);
2097
+ } else {
2098
+ const plain = missed.filter(t => typeof t === "string");
2099
+ if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
2100
+ for (const g of missed) {
2101
+ if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
2102
+ }
2103
+ if (invented.length) reasons.push(`invented ${invented.join(", ")}`);
2104
+ if (under) reasons.push(`${n} items, wanted at least ${kase.minCount}`);
2105
+ if (over) reasons.push(`${n} items, wanted at most ${kase.maxCount}`);
2106
+ }
2107
+
2108
+ // Observed rather than scored: a dataset that wants to know whether some
2109
+ // terms turn up, without grading on them, lists them as `watch`.
2110
+ const watch = kase.watch || [];
2111
+ // `null`, not 1, when there are no requirements and nothing forbidden turned
2112
+ // up: there is no score to report. Returning 1 there printed "fail 100%"
2113
+ // beside an entry that asked for nothing -- a shape `evals-check.js` allows
2114
+ // even though every graded case now names an expectation.
2115
+ return { pass, score: denom ? found.length / denom : null, discarded,
2116
+ found, missed, invented, unmet, under, over, count: n,
2117
+ watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
2118
+ reasons };
2119
+ }
2120
+
2121
+ /**
2122
+ * A queue row's stored results as today's: before generic-lab step 3 a
2123
+ * result's terms and a dropped item were spelled as the retired kind spelled
2124
+ * them (`upgradeResults` names them). Rows are
2125
+ * kept as they were written -- a run's replies cost a model's time to make
2126
+ * again -- so every reader of one (the page's API client, the runner's
2127
+ * resume and re-score) reads it through this. Anything else comes back as it
2128
+ * was.
2129
+ */
2130
+ function upgradeResults (items ) {
2131
+ if (!Array.isArray(items)) return items;
2132
+ const res = (r ) => {
2133
+ if (!isObj(r) || !("tags" in r) || "terms" in r) return r; // vocab: a pre-v3 row's spelling
2134
+ const { tags: was, dropped, ...rest } = r ; // vocab: as above
2135
+ return {
2136
+ ...rest, terms: Array.isArray(was) ? was : [],
2137
+ ...(Array.isArray(dropped) ? { dropped: dropped.map((d ) =>
2138
+ isObj(d) && "tag" in d ? { item: d.tag, by: d.rule } : d) } : {}), // vocab: as above
2139
+ };
2140
+ };
2141
+ // Before version 6 a run had one test, whose score a cell kept as `score`:
2142
+ // the upgrade names that test t1 (testsList).
2143
+ const cell = (sc ) => {
2144
+ const next = sc.res ? { ...sc, res: res(sc.res) } : sc;
2145
+ if (!("score" in next) || "scores" in next) return next;
2146
+ const { score, ...rest } = next;
2147
+ return { ...rest, scores: isObj(score) ? { t1: score } : null };
2148
+ };
2149
+ return items.map(it => !isObj(it) || !Array.isArray(it.scenarios) ? it
2150
+ : { ...it, scenarios: (it.scenarios ).map(sc => isObj(sc) ? cell(sc) : sc) }) ;
2151
+ }
2152
+
2153
+ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
2154
+
2155
+ // Summed across cases rather than averaged, for the reason in
2156
+ // docs/datasets.md: an item naming more terms should weigh more than
2157
+ // one naming fewer, and averaging percentages lets the easiest case carry the
2158
+ // score.
2159
+ function addToTally(t , s ) {
2160
+ t.ran++;
2161
+ if (s.pass) t.passed++;
2162
+ t.found += s.found.length;
2163
+ t.of += s.found.length + s.missed.length + s.invented.length;
2164
+ }
2165
+
2166
+ // The headline percentage, in one place because it is a number and this file
2167
+ // is about numbers stated twice. `null` where there is nothing to take a
2168
+ // percentage of -- a set whose entries name no terms -- so that a caller
2169
+ // prints a dash rather than a confident nought.
2170
+ function tallyPercent(t ) {
2171
+ return t.of ? Math.round(100 * t.found / t.of) : null;
2172
+ }
2173
+
2174
+ // ---- the Single Test -----------------------------------------------------
2175
+ // A whole-run assertion with no graded set: All of / Any of / None of, a
2176
+ // parse, a count, an exact reply and a length bound. It lives in the shared
2177
+ // core because it is a pass/fail the tab, the worker and History all have to
2178
+ // agree on -- the same one-definition rule that keeps `scoreCase` here.
2179
+ function parseCount(raw , parse ) {
2180
+ // An unparsed reply is one result, not no result. It used to return null,
2181
+ // which made a Count of 1 fail as "null results" against a reply that
2182
+ // plainly was one -- the answer was simply not in a format we had split up.
2183
+ if (!parse) return {valid:true, count: 1};
2184
+ if (parse === "csv") {
2185
+ const parts = raw.split(",").map(s=>s.trim()).filter(Boolean);
2186
+ return {valid:true, count: parts.length};
2187
+ }
2188
+ if (parse === "json") {
2189
+ try { const j = JSON.parse(raw); return {valid:true, count: Array.isArray(j) ? j.length : 1}; }
2190
+ catch { return {valid:false, count:null}; }
2191
+ }
2192
+ if (parse === "xml") {
2193
+ const t = raw.trim();
2194
+ if (!t.startsWith("<") || !t.endsWith(">")) return {valid:false, count:null};
2195
+ const marks = t.match(/<[^>]+>/g) || [];
2196
+ return {valid: marks.length > 0, count: marks.length ? Math.floor(marks.length/2) || 1 : 0};
2197
+ }
2198
+ return {valid:true, count:null};
2199
+ }
2200
+
2201
+ // How a wanted count is compared with the one that came back.
2202
+ // "Exactly" was the only comparison there was, which made the field useless
2203
+ // for the thing people actually want to say: that a reply should have *at
2204
+ // least* a few items, or *no more than* a handful. The op is stored beside
2205
+ // the number so an older setup, which has a number and no op, still means
2206
+ // exactly what it always meant.
2207
+ const COUNT_OPS = [
2208
+ ["atleast", "At least"],
2209
+ ["exactly", "Exactly"],
2210
+ ["atmost", "No more than"],
2211
+ ["not", "Not"],
2212
+ ];
2213
+ const COUNT_OP_LABEL = new Map(COUNT_OPS);
2214
+
2215
+ // How a reply's length is compared. Characters of the reply once trimmed,
2216
+ // because trailing whitespace is the model's punctuation rather than content.
2217
+ const LENGTH_OPS = [
2218
+ ["longer", "Longer than"],
2219
+ ["exactly", "Exactly"],
2220
+ ["shorter", "No longer than"],
2221
+ ];
2222
+ const LENGTH_OP_LABEL = new Map(LENGTH_OPS);
2223
+
2224
+ function lengthHolds(op , got , want ) {
2225
+ switch (op) {
2226
+ case "longer": return got > want;
2227
+ case "shorter": return got <= want;
2228
+ default: return got === want;
2229
+ }
2230
+ }
2231
+
2232
+ function countHolds(op , got , want ) {
2233
+ switch (op) {
2234
+ case "atleast": return Number(got) >= want;
2235
+ case "atmost": return Number(got) <= want;
2236
+ case "not": return got !== want;
2237
+ default: return got === want;
2238
+ }
2239
+ }
2240
+
2241
+ /**
2242
+ * The rules a Single Test asserts, in the order it checks them: one per field
2243
+ * it was given, none for a field left empty. Read from the test alone, so a
2244
+ * test skipped before it ran still names what it would have checked.
2245
+ *
2246
+ * A parse only has a rule where it can fail: csv and an unformatted reply are
2247
+ * always read, and a Parse row on every test would say nothing.
2248
+ */
2249
+ function singleRules(test ) {
2250
+ const rules = [];
2251
+ const list = (key , label ) => {
2252
+ if (test[key]?.length) rules.push({ key, label, want: test[key] .join(", ") });
2253
+ };
2254
+ list("all", "All of");
2255
+ list("any", "Any of");
2256
+ list("none", "None of");
2257
+ if (test.exact != null && String(test.exact).trim() !== "") {
2258
+ rules.push({ key: "exact", label: "Exact", want: String(test.exact).trim() });
2259
+ }
2260
+ if (test.length != null) {
2261
+ rules.push({ key: "length", label: `Length ${(LENGTH_OP_LABEL.get(test.lengthOp || "exactly") || "Exactly").toLowerCase()}`,
2262
+ want: String(test.length) });
2263
+ }
2264
+ if (test.parse === "json" || test.parse === "xml") rules.push({ key: "parse", label: "Parses as", want: test.parse });
2265
+ if (test.count != null) {
2266
+ rules.push({ key: "count", label: `Count ${(COUNT_OP_LABEL.get(test.countOp || "exactly") || "Exactly").toLowerCase()}`,
2267
+ want: String(test.count) });
2268
+ }
2269
+ return rules;
2270
+ }
2271
+
2272
+ /** "3 of 12 replies", or "the reply" where there was only one. */
2273
+ const ofReplies = (k , n ) => (n === 1 ? "the reply" : `${k} of ${n} replies`);
2274
+ const quoted = (s ) => `"${s.trim().slice(0, 60)}"`;
2275
+
2276
+ /**
2277
+ * The Single Test's verdict over a whole run: [terms] and [raws] pool every
2278
+ * item's reply, [plain] is whether the last stage answered as text, which
2279
+ * decides whether the All/Any/None terms are matched in the terms or the raw
2280
+ * replies. Pass is the absence of any reason. [checks] is each of its rules
2281
+ * (singleRules) met or not, with what broke it, so a reader can say which
2282
+ * rule failed without parsing the reasons.
2283
+ */
2284
+ function scoreSingle(test , terms , raws , plain = false)
2285
+ {
2286
+ const reasons = [];
2287
+ const broke = new Map ();
2288
+ const all = test.all || [], any = test.any || [], none = test.none || [];
2289
+ const checkIn = plain
2290
+ ? (raws , term ) => raws.some(r => r.toLowerCase().includes(term.toLowerCase()))
2291
+ : (items , term ) => items.some(item => termIn([item], term));
2292
+ const poolTerms = terms;
2293
+ const poolRaws = raws;
2294
+ const missed = all.filter(s => plain ? !checkIn(poolRaws, s) : !checkIn(poolTerms, s));
2295
+ if (missed.length) {
2296
+ reasons.push(`FAIL: Missed ${missed.join(", ")}`);
2297
+ broke.set("all", `missed ${missed.join(", ")}`);
2298
+ }
2299
+ if (any.length && !any.some(s => plain ? checkIn(poolRaws, s) : checkIn(poolTerms, s))) {
2300
+ reasons.push(`FAIL: Missed any of ${any.join(", ")}`);
2301
+ broke.set("any", "found none");
2302
+ }
2303
+ const invented = none.filter(s => plain ? checkIn(poolRaws, s) : checkIn(poolTerms, s));
2304
+ if (invented.length) {
2305
+ reasons.push(`invented ${invented.join(", ")}`);
2306
+ broke.set("none", `found ${invented.join(", ")}`);
2307
+ }
2308
+ // Exact and Length judge the reply as it came back, not the items read out
2309
+ // of it: "did it answer with precisely this" is a question about the string.
2310
+ if (test.exact != null && String(test.exact).trim() !== "") {
2311
+ const want = String(test.exact).trim();
2312
+ const off = raws.filter(r => String(r).trim() !== want);
2313
+ if (off.length) {
2314
+ reasons.push(`not an exact match: got "${String(off[0]).trim().slice(0, 60)}"`);
2315
+ broke.set("exact", `${ofReplies(off.length, raws.length)} ${raws.length === 1 ? "was" : "differ, first"} ${quoted(String(off[0]))}`);
2316
+ }
2317
+ }
2318
+ if (test.length != null) {
2319
+ const op = test.lengthOp || "exactly", want = test.length;
2320
+ const bad = raws.filter(r => !lengthHolds(op, String(r).trim().length, want));
2321
+ if (bad.length) {
2322
+ const got = String(bad[0]).trim().length;
2323
+ reasons.push(`${got} characters, wanted ` +
2324
+ `${(LENGTH_OP_LABEL.get(op) || "Exactly").toLowerCase()} ${test.length}`);
2325
+ broke.set("length", `${ofReplies(bad.length, raws.length)}, ${raws.length === 1 ? "" : "first "}${got} characters`);
2326
+ }
2327
+ }
2328
+ for (const raw of raws) {
2329
+ const pc = parseCount(raw, test.parse);
2330
+ if (!pc.valid) { reasons.push(`invalid ${test.parse}`); break; }
2331
+ if (test.count != null && !countHolds(test.countOp || "exactly", pc.count, test.count)) {
2332
+ reasons.push(`${pc.count} results, wanted ${(COUNT_OP_LABEL.get(test.countOp || "exactly") || "Exactly").toLowerCase()} ${test.count}`);
2333
+ break;
2334
+ }
2335
+ }
2336
+ // The reasons stop at the first reply that fails a parse or a count; each
2337
+ // rule's check counts every reply, and a reply that did not parse has no
2338
+ // count to hold against.
2339
+ const counted = raws.map(r => parseCount(r, test.parse));
2340
+ const invalid = counted.filter(pc => !pc.valid);
2341
+ if (invalid.length) broke.set("parse", `${ofReplies(invalid.length, raws.length)} invalid`);
2342
+ if (test.count != null) {
2343
+ const off = counted.filter(pc => pc.valid && !countHolds(test.countOp || "exactly", pc.count, test.count ));
2344
+ if (off.length) broke.set("count", `${ofReplies(off.length, raws.length)}, ${raws.length === 1 ? "" : "first "}${off[0] .count} results`);
2345
+ }
2346
+ const pass = !reasons.length;
2347
+ const checks = singleRules(test).map(r => (broke.has(r.key) ? { ...r, pass: false, note: broke.get(r.key) } : { ...r, pass: true }));
2348
+ return {pass, reasons, score: pass ? 1 : 0, checks};
2349
+ }
2350
+
2351
+ // A `{name}` in a stage's wording. Narrow on purpose -- a letter first, then
2352
+ // word characters, dots and hyphens -- so a JSON example in a prompt is not
2353
+ // read as a token and refused.
2354
+ const TOKEN = /\{(\/?[A-Za-z][\w.-]*)\}/g;
2355
+
2356
+ /**
2357
+ * The first thing in a job that cannot run, as a sentence naming its stage,
2358
+ * or null. Asked before anything is sent, so a misspelt token costs nothing
2359
+ * rather than a vision pass per image -- and so it is an error at all:
2360
+ * left in place, `{stage2.itme}` reaches the model as those characters
2361
+ * and the run looks like a finding about the wording.
2362
+ */
2363
+ function jobProblem(stages , call ,
2364
+ tokens , text ) {
2365
+ const kinds = stages.map(kindOf);
2366
+ // The names an output kind renders into tokens -- {items} is the List's
2367
+ // -- so a token is checked against the kinds the lab has, not a list here.
2368
+ const offered = new Set(Object.values(OUTPUT_KINDS).flatMap(k => k.offers || []));
2369
+ for (let i = 0; i < stages.length; i++){
2370
+ const at = `stage ${i + 1}`;
2371
+ const base = new Set(tokenNames(tokenSet(tokens, i)));
2372
+ if (!OUTPUT_KINDS[kinds[i] ]) {
2373
+ return `${at} answers as "${kinds[i]}", which is not an output kind this lab has`;
2374
+ }
2375
+ if (typeof (Array.isArray(call) ? call[i] : call) !== "function") {
2376
+ return `${at} has no connection to send it to`;
2377
+ }
2378
+ const why = (name ) => {
2379
+ if (base.has(name)) return null;
2380
+ if (name === "text") return text != null ? null : "and there is no text content to put there";
2381
+ const m = /^stage([1-9]\d*)(?:\.(.+))?$/.exec(name);
2382
+ let from , sub ;
2383
+ if (m) {
2384
+ from = +m[1] - 1; sub = m[2] ?? null;
2385
+ if (sub != null && !offered.has(sub)) return "which is not a token";
2386
+ } else if (name === "reply") { from = i - 1; sub = null; }
2387
+ else if (offered.has(name)) { from = i - 1; sub = name; }
2388
+ else return "which is not a token";
2389
+ if (from < 0 || from >= i) return i === 0 ? "and nothing comes before stage 1"
2390
+ : "which is not an earlier stage";
2391
+ const kind = OUTPUT_KINDS[kinds[from] ] ;
2392
+ if (sub && !(kind.offers || []).includes(sub)) {
2393
+ return `and stage ${from + 1} answers with ${kind.noun || kinds[from]}, which has no ${sub} — `
2394
+ + `{${m ? `stage${from + 1}` : "reply"}} is its reply`;
2395
+ }
2396
+ return null;
2397
+ };
2398
+ if (stages[i] .verbatim) continue;
2399
+ for (const [, name] of String(stages[i] .text).matchAll(TOKEN)) {
2400
+ const reason = why(name );
2401
+ if (reason) return `${at} asks for {${name}}, ${reason}`;
2402
+ }
2403
+ }
2404
+ return null;
2405
+ }
2406
+
2407
+ // Runs the stages in order, keeping every prompt sent and every reply
2408
+ // received. Without the transcript a multi-stage run is undebuggable: you see
2409
+ // a different answer and cannot tell which stage changed it, or whether a
2410
+ // stage silently received an empty {items}.
2411
+ //
2412
+ // `call(prompt, dataUrl)` is the transport, injected because that is the only
2413
+ // part that differs between the callers: the tab posts to the relay so the
2414
+ // browser can reach a firewalled Ollama, the script goes to the endpoint
2415
+ // itself, and a check hands it a recorded reply and no network at all. What a
2416
+ // reply then means -- discarded, or these items -- is decided here, once. A
2417
+ // job whose stages run on different models passes an array, one transport
2418
+ // per stage; a reply carrying `conn` names the connection that gave it.
2419
+ //
2420
+ // A stage's `kind` is what its reply is, and names an OUTPUT_KINDS entry that
2421
+ // reads it: "text" is anything at all, forwarded untouched as {reply}, so a
2422
+ // stage can describe the image for the next one to list; a kind with a
2423
+ // shape -- "list", items -- is held to its own rules and forwarded under
2424
+ // the tokens it renders. The caller sets the last stage's kind, and that stage's
2425
+ // result is the one graded: a job that ends in text has no terms.
2426
+ //
2427
+ // A stage's `modifiers` change its value before it is handed on or tested.
2428
+ // A caller that gives no stage any gets the last kind's own defaults at
2429
+ // `mode`, which is how every caller before modifiers asked for a case.
2430
+ //
2431
+ // `text` is the text content, when that is what is being run: it joins stage
2432
+ // 1 the way `textPrompt` says, and any later stage that places {text}.
2433
+ async function runPipeline(stages , dataUrl , call ,
2434
+ opts = {}) {
2435
+ const { tokens = TOKEN_DEFAULTS, mode, text = null } = opts;
2436
+ const transcript = [];
2437
+ const problem = jobProblem(stages, call, tokens, text);
2438
+ if (problem) return { error: problem, terms: [], transcript, ms: 0 };
2439
+ const kinds = stages.map(kindOf);
2440
+ const explicit = stages.some(st => Array.isArray(st.modifiers));
2441
+ const mods = stages.map((st, i) => explicit ? (st.modifiers || [])
2442
+ : i === stages.length - 1 ? (OUTPUT_KINDS[kinds[i] ] .modifiers?.(mode) || []) : []);
2443
+ const done = [];
2444
+ // The most recent stage's value, for the result: the terms a job holds
2445
+ // are the ones its latest stage produced, even when a later one failed.
2446
+ let held = null, total = 0, echoed = 0, dropped = [];
2447
+ // Each stage's modifiers apply to its own value as it is read. A stage with
2448
+ // none hands the model's own spelling on, because "Eiffel Tower"
2449
+ // lowercased on the way through is a different question from the one the
2450
+ // first model answered; a caller that gives no stage any gets the last
2451
+ // kind's defaults on the last stage alone (above).
2452
+ const result = (more ) => {
2453
+ const reader = held && OUTPUT_KINDS[held.kind] ;
2454
+ return { terms: reader?.terms ? reader.terms(held .value) : [], transcript, ms: total, ...more };
2455
+ };
2456
+ for (let i = 0; i < stages.length; i++){
2457
+ const st = stages[i] , kind = kinds[i] , reader = OUTPUT_KINDS[kind] ;
2458
+ const named = (m ) => stages.length > 1 ? `stage ${i + 1}: ${m}` : m;
2459
+ const forwarded = new Map (text != null ? [["text", text]] : []);
2460
+ done.forEach((d, k) => {
2461
+ forwarded.set(`stage${k + 1}`, d.reply);
2462
+ for (const [name, v] of Object.entries(d.rendered)) forwarded.set(`stage${k + 1}.${name}`, v);
2463
+ });
2464
+ const prev = done[i - 1];
2465
+ if (prev) {
2466
+ forwarded.set("reply", prev.reply);
2467
+ for (const [name, v] of Object.entries(prev.rendered)) forwarded.set(name, v);
2468
+ }
2469
+ // A verbatim stage's words are the call's own template, sent as written.
2470
+ const instruction = st.verbatim ? st.text : resolvePrompt(st.text, tokenSet(tokens, i));
2471
+ const fill = (to ) => instruction.replace(TOKEN, (m, name ) =>
2472
+ forwarded.has(name) ? to(forwarded.get(name) ) : m);
2473
+ const sent = st.verbatim ? instruction : i === 0 && text != null ? textPrompt(instruction, text) : fill(v => v);
2474
+ // What the echo guard measures the reply against: the stage's own
2475
+ // instruction, with nothing that was forwarded into it. Forwarded material
2476
+ // is what the stage is about -- the text content, the items or the reply
2477
+ // an earlier stage produced -- so a reply that repeats it has answered, not
2478
+ // copied. Judged against the whole prompt, a text run lost every term it
2479
+ // correctly took out of the text (#310), and a stage asked to correct a
2480
+ // list lost every item it rightly kept (#408).
2481
+ const echo = st.verbatim ? instruction : fill(() => "");
2482
+ // A stage that does not say sends the item's image on the first stage
2483
+ // only; a job says, and is honoured, the first one included.
2484
+ const withImage = (st.withImage ?? i === 0) && dataUrl != null;
2485
+ const res = await (Array.isArray(call) ? call[i] : call )(sent, withImage ? dataUrl : null);
2486
+ total += res.ms;
2487
+ transcript.push({ n: i + 1, kind, conn: res.conn ?? null, withImage, sent: res.sent ?? sent,
2488
+ got: res.raw ?? "", ms: res.ms, error: res.error, ...(res.said != null ? { said: res.said } : {}) });
2489
+ if (res.error) return result({ error: named(res.error) });
2490
+ // A stage that failed is the end of the job: nothing after it is asked.
2491
+ // Handed nothing, the next stage still answers -- about an empty list --
2492
+ // and that answer is graded as though the job had worked.
2493
+ const handsOn = i < stages.length - 1;
2494
+ const against = res.readAgainst ?? echo;
2495
+ const read = reader.read(res.raw ?? "", { echo: against, parsing: opts.parsing, settings: st.settings });
2496
+ if (read.error) return result({ error: named(read.error) });
2497
+ const mod = applyModifiers(mods[i], kind, read.value, { instruction: against, body: read.body ?? res.raw ?? "" });
2498
+ if (mod.reject) return result({ error: named(`discarded: ${mod.reject}`) });
2499
+ const nothing = handsOn ? reader.handsOn?.(mod.value) : null;
2500
+ if (nothing) return result({ error: named(nothing) });
2501
+ const value = mod.value;
2502
+ held = { kind, value };
2503
+ const gone = [...(read.dropped || []), ...mod.dropped];
2504
+ echoed = (read.echoed || 0) + mod.dropped.filter(d => d.by === ECHO).length;
2505
+ done.push({ reply: res.raw ?? "", rendered: reader.render?.(value) || {} });
2506
+ if (!handsOn) dropped = gone;
2507
+ }
2508
+ return result({ echoed, dropped });
2509
+ }
2510
+
2511
+ /**
2512
+ * One reply read as a job's [out] reads it, and every modifier applied, as
2513
+ * a one-stage run would: the items it keeps, what the modifiers dropped and
2514
+ * why, how many repeated [instruction], and why it was thrown away, if it
2515
+ * was. The same code as runPipeline, without a transport -- what a check or
2516
+ * the page asks when it wants to know what a configuration makes of a reply.
2517
+ */
2518
+ function readList(reply , out , instruction = "")
2519
+ {
2520
+ const reader = OUTPUT_KINDS[out.kind];
2521
+ if (!reader) return { items: [], dropped: [], echoed: 0, error: `"${out.kind}" is not an output kind this lab has` };
2522
+ const { kind, modifiers, ...settings } = out;
2523
+ const read = reader.read(reply, { echo: instruction, settings });
2524
+ if (read.error) return { items: [], dropped: [], echoed: 0, error: read.error };
2525
+ const mod = applyModifiers(modifiers, kind, read.value, { instruction, body: read.body ?? reply });
2526
+ const items = reader.terms ? reader.terms(mod.value) : [];
2527
+ return {
2528
+ items: mod.reject ? [] : items, dropped: mod.dropped,
2529
+ echoed: mod.dropped.filter(d => d.by === ECHO).length,
2530
+ error: mod.reject ? `discarded: ${mod.reject}` : null,
2531
+ };
2532
+ }
2533
+
2534
+ /** A value through a list of modifiers, in order, each one that accepts its
2535
+ kind: the value they leave, what they removed, and the first rejection. A
2536
+ rejection ends the list -- nothing after it is asked about an answer that
2537
+ was thrown away. */
2538
+ function applyModifiers (list , kind , value ,
2539
+ ctx = { instruction: "", body: "" }) {
2540
+ let v = value;
2541
+ const dropped = [];
2542
+ for (const m of list || []) {
2543
+ const entry = MODIFIERS[m.type];
2544
+ if (!entry || (entry.accepts && !entry.accepts.includes(kind))) continue;
2545
+ const out = entry.apply(v, m, ctx) ;
2546
+ const outcome = out && typeof out === "object" && !Array.isArray(out) && "value" in (out )
2547
+ ? out : null;
2548
+ if (!outcome) { v = out ; continue; }
2549
+ v = outcome.value;
2550
+ dropped.push(...(outcome.dropped || []));
2551
+ if (outcome.reject) return { value: v, dropped, reject: outcome.reject };
2552
+ }
2553
+ return { value: v, dropped };
2554
+ }
2555
+
2556
+ // ---- The pipeline model ----------------------------------------------------
2557
+ // docs/pipeline-model.md. One document describes a run for the page, the
2558
+ // queue and run-evals.js alike, and every one of them reads it through the
2559
+ // functions below rather than through a translation of its own.
2560
+ //
2561
+ // The lab is generic, so what a reply is, what a test scores and how a value
2562
+ // is changed are registry entries. The ones here are the lab's own, and so
2563
+ // are kinds/list.ts's -- the List kind and its modifiers. A dataset
2564
+ // registers nothing: it is data a graded test names by id, and a run carries.
2565
+
2566
+ // 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
2567
+ // version 1 kept them as two maps (tokens: { values, blocks }).
2568
+ // 3: the version-2 list kind is gone; a job that answered in it answers
2569
+ // as a `list` with the modifiers that read a reply exactly as it did, and
2570
+ // its token is {items} (docs/design/generic-lab.md §3.1).
2571
+ // 4: every scenario has an id, and a cell may name the Prompt library prompt
2572
+ // it was picked from (docs/pipeline-model.md §14).
2573
+ // 5: the pipeline and every job have an id, minted for an older document by
2574
+ // position (docs/pipeline-model.md §7, #93); version 4's ids are kept.
2575
+ // 6: `tests` is an ordered list, each test with an id, a name and
2576
+ // continueOnFailure, where version 5 held one test or null (#98).
2577
+ // 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
2578
+ // 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
2579
+ // Call (Prompt: the image flag and token mappings), Read Reply (the output).
2580
+ // 9: a test is Metrics. A Single Test and a Graded set are read as the Metrics
2581
+ // they convert to (LEGACY_TESTS' toMetrics, proven equal by
2582
+ // metrics-parity-check.js).
2583
+ const PIPELINE_VERSION = 9 ;
2584
+
2585
+ // Plain objects, so an entry is added by assignment and a reader never needs
2586
+ // to know which registered it.
2587
+ const STEP_TYPES = Object.create(null);
2588
+ const CONTENT_TYPES = Object.create(null);
2589
+ const OUTPUT_KINDS = Object.create(null);
2590
+ const MODIFIERS = Object.create(null);
2591
+ const TEST_TYPES = Object.create(null);
2592
+ const METRICS = Object.create(null);
2593
+ const SOURCE_TYPES = Object.create(null);
2594
+
2595
+ // The type a Source that names none is: every Source was one before types.
2596
+ const DEFAULT_SOURCE_TYPE = "files";
2597
+ const sourceTypeOf = (src ) =>
2598
+ SOURCE_TYPES[isStr(src?.type) ? src .type : DEFAULT_SOURCE_TYPE];
2599
+
2600
+ // The kind a stage that names none answers in.
2601
+ let defaultOutputKind = "text";
2602
+ const defaultKind = () => defaultOutputKind;
2603
+ const kindOf = (st ) => st.kind ?? defaultOutputKind;
2604
+
2605
+ /** A module's entries into the registries, in one call. */
2606
+ function registerKinds(k ) {
2607
+ Object.assign(OUTPUT_KINDS, k.outputKinds || {});
2608
+ Object.assign(MODIFIERS, k.modifiers || {});
2609
+ Object.assign(TEST_TYPES, k.testTypes || {});
2610
+ Object.assign(SOURCE_TYPES, k.sourceTypes || {});
2611
+ Object.assign(METRICS, k.metrics || {});
2612
+ if (k.defaultKind) defaultOutputKind = k.defaultKind;
2613
+ }
2614
+
2615
+ /**
2616
+ * What a plugin's `register(lab)` is handed (plugin-api.d.ts): the same
2617
+ * registries the lab's own kinds go into, and nothing else. An id the lab
2618
+ * already has is refused, naming the plugin, rather than quietly replacing
2619
+ * what every run of it is read by.
2620
+ */
2621
+
2622
+
2623
+
2624
+
2625
+
2626
+
2627
+
2628
+
2629
+
2630
+ function pluginHost(pluginId ) {
2631
+ const taken = (table , what , ids ) => {
2632
+ const dup = ids.find(id => id in table);
2633
+ if (dup) throw new Error(`the plugin ${pluginId} registers the ${what} ${dup}, which the lab has already`);
2634
+ };
2635
+ return {
2636
+ labVersion: PIPELINE_VERSION,
2637
+ registerKinds(k) {
2638
+ taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
2639
+ taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
2640
+ taken(TEST_TYPES, "test type", Object.keys(k.testTypes || {}));
2641
+ taken(METRICS, "metric", Object.keys(k.metrics || {}));
2642
+ // The server keeps its own copy of the Source types, to refuse a row of
2643
+ // one it does not know, and a plugin never reaches the server's code.
2644
+ if (Object.keys((k ).sourceTypes || {}).length) {
2645
+ throw new Error(`the plugin ${pluginId} registers a Source type, which only the lab can`);
2646
+ }
2647
+ // A plugin adds what the lab lacks; the kind a new job answers in is the lab's.
2648
+ const { defaultKind: _lab, ...added } = k;
2649
+ registerKinds(added);
2650
+ },
2651
+ registerConnectionType(t) {
2652
+ taken(CONNECTION_TYPES, "connection type", [t.id]);
2653
+ CONNECTION_TYPES[t.id] = t;
2654
+ },
2655
+ };
2656
+ }
2657
+
2658
+ const isObj = (v ) => v != null && typeof v === "object" && !Array.isArray(v);
2659
+ const isStr = (v ) => typeof v === "string";
2660
+ const clone = (v ) => JSON.parse(JSON.stringify(v));
2661
+
2662
+ // A reply read as itself: anything at all, handed on untouched.
2663
+ OUTPUT_KINDS.text = {
2664
+ label: "Plain text", noun: "text", offers: [],
2665
+ description: "Keeps the reply as it came back, whole.",
2666
+ read: reply => ({ value: String(reply ?? "") }),
2667
+ handsOn: (value ) => (value.trim() ? null : "empty reply"),
2668
+ } ;
2669
+
2670
+ // ---- the connection a profile reference resolves to ------------------------
2671
+ // What a run carries of a Setup profile: everything a request needs and never
2672
+ // the key. A key follows the profile's id into $EVAL_API_KEY_<ID>, which the
2673
+ // server sets from its profiles store and the runner reads, so a name is only
2674
+ // a label and two profiles can share one. #46: the profile carries `type`,
2675
+ // and the run's resolved copy carries it too (docs/pipeline-model.md §7).
2676
+ const CONNECTION_FIELDS = ["name", "url", "model", "type", "temperature",
2677
+ "px", "format", "quality", "options"];
2678
+ const SETTING_KEYS = [...new Set(Object.values(CONNECTION_TYPES).flatMap(t => t.settings.map(s => s.key)))];
2679
+ const OPTION_FIELDS = ["seed", "nPredict", ...DECODING_KEYS];
2680
+ const IMAGE_FORMATS = ["image/jpeg", "image/webp", "image/png"];
2681
+ // Ids are what a key variable is spelled from, so they stay spellable: a
2682
+ // letter or digit run, with at most one hyphen.
2683
+ const PROFILE_ID = /^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?$/;
2684
+ const LOOKS_LIKE_A_KEY = /^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$/i;
2685
+
2686
+ /** The variable a profile's key is read from: pm1x8k2q is EVAL_API_KEY_PM1X8K2Q. */
2687
+ const keyVar = (id ) => "EVAL_API_KEY_"
2688
+ + String(id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
2689
+
2690
+ /** A stored profile as a run carries it: its request settings, no key. */
2691
+ function connectionOf(p ) {
2692
+ const type = typeOf(p);
2693
+ const settings = CONNECTION_TYPES[type] .settings.map(s => s.key);
2694
+ const options = {};
2695
+ // temperature is a common field, carried at the top like px/format/quality;
2696
+ // the options bag holds the type's own settings only.
2697
+ for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
2698
+ const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
2699
+ for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
2700
+ if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
2701
+ if (Object.keys(options).length) out.options = options;
2702
+ return out;
2703
+ }
2704
+
2705
+ /** A run's connection as one flat set of settings, the shape a profile has. */
2706
+ function connectionSettings(c ) {
2707
+ const { options, ...rest } = c || {};
2708
+ return { ...rest, ...(options || {}) };
2709
+ }
2710
+
2711
+ // The rest of what a connection's fields mean lives in the connection-type
2712
+ // registry above (#46): the runner asks a type how to build a request and how
2713
+ // to read a reply, and validation asks connectionProblems (below, with the run
2714
+ // document's other rules). A change to a connection's shape is a change to the
2715
+ // registry and to connectionProblems, not to anything that calls them.
2716
+
2717
+ /** How [conn] prepares an image: its budget, encoding and quality, null where unset. */
2718
+ function connectionImage(conn ) {
2719
+ const px = conn?.px === EDGE_448 ? EDGE_448 : (Number(conn?.px) > 0 ? Number(conn .px) : null);
2720
+ const format = IMAGE_FORMATS.includes(conn?.format ) ? conn .format : "image/jpeg";
2721
+ const quality = conn?.quality != null && Number(conn.quality) > 0 ? Number(conn.quality) : null;
2722
+ return { px, format, quality };
2723
+ }
2724
+
2725
+ /**
2726
+ * A profile's quality on the encoder's scale. A profile states it the way the
2727
+ * page's canvas took it, 0 to 1 -- 0.9, and 1 for the best -- while the
2728
+ * runner encodes with ImageMagick, whose scale is 1 to 100 and which ignores
2729
+ * anything under 1. Handed across unscaled, every quality a profile set was
2730
+ * silently dropped for the default, and 1, the best, became ImageMagick's
2731
+ * worst. A value already above 1 is taken as that scale already; none is 92,
2732
+ * the quality the container bakes its own copies at.
2733
+ */
2734
+ function encoderQuality(quality ) {
2735
+ if (quality == null || !(quality > 0)) return 92;
2736
+ return Math.round(quality <= 1 ? quality * 100 : Math.min(quality, 100));
2737
+ }
2738
+
2739
+ // ---- labels -----------------------------------------------------------------
2740
+ // A name is optional everywhere; a blank one is the number the canvas has
2741
+ // always shown, so a sentence names what the reader sees.
2742
+ const jobLabel = (doc , k ) =>
2743
+ (doc.jobs?.[k]?.name || "").trim() || `Job ${k + 1}`;
2744
+ const scenarioLabel = (doc , i ) =>
2745
+ (doc.scenarios?.[i]?.name || "").trim() || `Scenario ${i + 1}`;
2746
+ // The runner calls a job's link a stage; the model calls it a job.
2747
+ const jobWords = (s ) => s && String(s).replace(/\bstage[ ](\d+)/g, "job $1");
2748
+
2749
+ // ---- the steps --------------------------------------------------------------
2750
+
2751
+ /** Every field of [obj] not in [allowed], as a sentence -- a key by name. */
2752
+ function onlyFields(obj , at , allowed , bad , keyFrom ) {
2753
+ for (const k of Object.keys(obj)) {
2754
+ if (allowed.includes(k)) continue;
2755
+ bad.push(LOOKS_LIKE_A_KEY.test(k)
2756
+ ? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVAL_API_KEY_<ID>"} supplies it`
2757
+ : `${at} has "${k}", which is not a pipeline field`);
2758
+ }
2759
+ }
2760
+
2761
+ // A Source is referenced, with the two numbers that decide how many items a
2762
+ // run over it is: the first N files, and how many times over.
2763
+ // A folder of files: images sent as images, text files as {text}. The
2764
+ // extensions are server.py's FILE_TYPES, which the server holds uploads to.
2765
+ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
2766
+ SOURCE_TYPES.files = {
2767
+ label: "File Library",
2768
+ description: "A folder of images and text files you upload.",
2769
+ noun: "file",
2770
+ uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
2771
+ sendsImage: true,
2772
+ itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
2773
+ };
2774
+
2775
+ // A Power Automate cloud flow (docs/power-automate.md): its items are
2776
+ // records -- one call of a Power Automate step each: what its template read,
2777
+ // the request production sent and what came back (flows/record.ts) -- and it
2778
+ // keeps the flow's definition beside them. Records are imported from run
2779
+ // history, written by hand or uploaded; nothing it holds is an image.
2780
+ SOURCE_TYPES["power-automate"] = {
2781
+ label: "Power Automate workflow",
2782
+ description: "A cloud flow read from Microsoft 365, with records of its steps' calls from its run history.",
2783
+ noun: "record",
2784
+ uploads: [".json"],
2785
+ sendsImage: false,
2786
+ itemText: () => false,
2787
+ };
2788
+
2789
+ CONTENT_TYPES.source = {
2790
+ label: "Source",
2791
+ description: "Every file or record in a Source from the Library.",
2792
+ fields: ["type", "ref", "first", "loops"],
2793
+ runFields: ["files", "revs"],
2794
+ validate(c, ctx, bad){
2795
+ if (!isRef(c.ref)) return void bad.push("a Source content has to name its Source as { id, name }");
2796
+ if (c.first != null && !(Number.isInteger(c.first) && c.first > 0)) {
2797
+ bad.push("Run first wants a positive number, or empty for all");
2798
+ }
2799
+ if (c.loops != null && !(Number.isInteger(c.loops) && c.loops >= 1)) {
2800
+ bad.push("Loop wants a whole number of times, 1 or more");
2801
+ }
2802
+ if (ctx.run && (!Array.isArray(c.files) || !c.files.every(isStr))) {
2803
+ bad.push("a run's content.files has to be a list of file names");
2804
+ }
2805
+ if (ctx.run && c.revs != null && !(isObj(c.revs) && Object.values(c.revs).every(isStr))) {
2806
+ bad.push("a run's content.revs has to map file names to revisions");
2807
+ }
2808
+ if (ctx.sources) {
2809
+ const src = ctx.sources(c.ref.id);
2810
+ if (!src) bad.push(`Source ${c.ref.name || c.ref.id} not found`);
2811
+ else if (!(src.files || []).length) {
2812
+ bad.push(`${src.name || c.ref.name} has no ${sourceTypeOf(src)?.noun ?? "item"}s`);
2813
+ }
2814
+ }
2815
+ },
2816
+ // What a stage's {text} can be: a Source's text files join there, one item
2817
+ // each (run-evals.js's file-type registry), so {text} is asked for fairly
2818
+ // unless the files are known and none is text. Null is "none".
2819
+ text(c, ctx){
2820
+ const src = ctx.sources?.(c.ref?.id);
2821
+ const names = src ? (src.files || []).map(f => (f )?.name ?? f) : (c.files || []);
2822
+ if (!names.length) return "";
2823
+ const type = sourceTypeOf(src);
2824
+ return type && names.some(n => type.itemText(String(n))) ? "" : null;
2825
+ },
2826
+ // The Source's files, limited to the first N and looped, in order and with
2827
+ // their repeats -- the list the runner, the server and the page all count.
2828
+ expand(c, files){
2829
+ const limited = (files || []).slice(0, c.first != null && c.first > 0 ? c.first : undefined);
2830
+ const out = [];
2831
+ for (let l = 0; l < Math.max(1, c.loops || 1); l++) out.push(...limited);
2832
+ return out;
2833
+ },
2834
+ };
2835
+ CONTENT_TYPES.text = {
2836
+ label: "Text",
2837
+ description: "One item: the text written here.",
2838
+ inline: true,
2839
+ fields: ["type", "text"],
2840
+ runFields: [],
2841
+ text: c => c.text,
2842
+ validate(c, ctx, bad){
2843
+ if (!isStr(c.text)) bad.push("a text content has to hold its text");
2844
+ else if (!c.text.trim()) bad.push("the text content is empty");
2845
+ },
2846
+ };
2847
+
2848
+ // No item at all: one bare item, so a stage is sent its prompt alone -- a
2849
+ // model asked the prompt itself, or Echo answering with it, the prompt then
2850
+ // being the recorded reply.
2851
+ CONTENT_TYPES.prompt = {
2852
+ label: "Prompt only",
2853
+ description: "No item: each scenario's prompt is sent on its own.",
2854
+ inline: true,
2855
+ bare: true,
2856
+ fields: ["type"],
2857
+ runFields: [],
2858
+ text: () => null,
2859
+ validate(){},
2860
+ };
2861
+
2862
+ const isRef = (r ) => isObj(r) && isStr(r.id) && !!r.id && (r.name == null || isStr(r.name));
2863
+
2864
+ // The whole-run assertion: every item's reply pooled, one verdict per
2865
+ // scenario. Any kind: a kind with no terms is matched as its replies.
2866
+ // ---- tests retired in version 9 ------------------------------------------------
2867
+ // The Single Test and the Graded set, as version 8 and earlier held them. They
2868
+ // are no longer test types -- a pipeline has one kind of check, Metrics -- but
2869
+ // every document that holds one is read with it converted (`toMetrics`,
2870
+ // through upgradePipeline), and metrics-parity-check.js still reads a run
2871
+ // both ways to prove the conversion changes nothing.
2872
+ const LEGACY_TESTS = Object.create(null);
2873
+
2874
+ LEGACY_TESTS.single = {
2875
+ label: "Single Test",
2876
+ fields: ["type", "all", "any", "none", "parse", "exact", "length", "lengthOp", "count", "countOp"],
2877
+ accepts: null,
2878
+ defaults: () => ({ type: "single", all: [], any: [], none: [], parse: "csv", count: null,
2879
+ countOp: "exactly", exact: null, length: null, lengthOp: "exactly" }),
2880
+ validate(t, ctx, bad){
2881
+ for (const k of ["all", "any", "none"]) {
2882
+ if (t[k] != null && !(Array.isArray(t[k]) && t[k].every(isStr))) bad.push(`the Single Test's ${k} has to be a list of strings`);
2883
+ }
2884
+ for (const k of ["length", "count"]) {
2885
+ if (t[k] != null && !Number.isInteger(t[k])) bad.push(`the Single Test's ${k} has to be a whole number`);
2886
+ }
2887
+ if (t.parse != null && !["csv", "json", "xml"].includes(t.parse)) bad.push("the Single Test parses as csv, json or xml, or not at all");
2888
+ if (t.countOp != null && !COUNT_OP_LABEL.has(t.countOp)) bad.push(`the Single Test's count is compared ${[...COUNT_OP_LABEL.keys()].join(", ")}`);
2889
+ if (t.lengthOp != null && !LENGTH_OP_LABEL.has(t.lengthOp)) bad.push(`the Single Test's length is compared ${[...LENGTH_OP_LABEL.keys()].join(", ")}`);
2890
+ },
2891
+ // [ress] is one scenario's result per item, in order.
2892
+ verdict(t, ress, kind){
2893
+ const raws = [], terms = [];
2894
+ for (const r of ress) {
2895
+ if (!r) continue;
2896
+ raws.push(r.transcript?.[r.transcript.length - 1]?.got ?? r.raw ?? "");
2897
+ terms.push(...(r.terms || []));
2898
+ }
2899
+ const s = scoreSingle(t, terms, raws, !(kind != null && OUTPUT_KINDS[kind]?.terms));
2900
+ return { pass: s.pass, detail: s.pass ? "all matched" : s.reasons.join(" · "), ran: raws.length, checks: s.checks };
2901
+ },
2902
+ rules(t){
2903
+ return singleRules(t);
2904
+ },
2905
+ };
2906
+
2907
+ // A dataset's cases, scored item by item with scoreCase. The test
2908
+ // references the dataset the way a pipeline references a Source, and the run
2909
+ // carries the body it was submitted against. It scores the terms a value
2910
+ // yields, so it accepts every kind that yields any: plain text has none, and
2911
+ // would fail every case.
2912
+ LEGACY_TESTS.graded = {
2913
+ label: "Graded set",
2914
+ fields: ["type", "dataset"],
2915
+ accepts: () => Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms),
2916
+ validate(t, ctx, bad){
2917
+ const d = t.dataset;
2918
+ if (!isRef(d) || (d.version != null && !isStr(d.version))) {
2919
+ return void bad.push("a graded test has to name its dataset as { id, name }");
2920
+ }
2921
+ if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
2922
+ },
2923
+ score: (t, kase, res) => scoreCase(kase, res),
2924
+ };
2925
+
2926
+ // Metrics: checks of each reply, deterministic or model-graded, from the
2927
+ // METRICS registry (metrics/builtin.ts registers the lab's own), with the
2928
+ // test's own list for every item and a case's `metrics` for its item. Scored
2929
+ // all-must-pass -- every metric passes -- or weighted: points, each metric's
2930
+ // score times its weight, against a threshold (#150's points, a negative
2931
+ // weight taking them away).
2932
+ // `of: "said"` reads the API's own text rather than the reply as the job read
2933
+ // it -- an HTTP Request's raw reply beside the flow's reading of it.
2934
+ const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
2935
+ const SCORING_MODES = ["all", "weighted"];
2936
+
2937
+ /** What is wrong with a list of metrics, as sentences naming [at]. */
2938
+ function metricsProblems(list , at , bad ) {
2939
+ if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
2940
+ list.forEach((m , x ) => {
2941
+ const where = `${at}, metric ${x + 1}`;
2942
+ const entry = isObj(m) && METRICS[m.type];
2943
+ if (!entry) return void bad.push(`${where} is of type "${m?.type}", which is not a metric this lab has`);
2944
+ onlyFields(m, where, [...METRIC_FIELDS, ...entry.options.map(o => o.key)], bad);
2945
+ if (m.not != null && typeof m.not !== "boolean") bad.push(`${where}: not has to be true or false`);
2946
+ if (m.weight != null && typeof m.weight !== "number") bad.push(`${where}: weight has to be a number`);
2947
+ if (m.metric != null && !isStr(m.metric)) bad.push(`${where}: metric names a group, as text`);
2948
+ if (m.of != null && m.of !== "reply" && m.of !== "said") bad.push(`${where}: of is reply or said`);
2949
+ if (m.every != null && typeof m.every !== "boolean") bad.push(`${where}: every has to be true or false`);
2950
+ entry.validate?.(m , `${where} (${entry.label})`, bad);
2951
+ });
2952
+ }
2953
+
2954
+ /** Whether any of [list] asks a grader. */
2955
+ const gradedIn = (list ) => Array.isArray(list) && list.some((m ) => !!METRICS[m?.type]?.graded);
2956
+
2957
+ /** What a metric reads of one result. */
2958
+ function metricInput(res , kase , more ) {
2959
+ const last = res.transcript?.at(-1);
2960
+ const text = last?.got ?? res.raw ?? "";
2961
+ return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
2962
+ error: res.error ?? null, ms: res.ms ?? 0, production: more.production ?? null, kase };
2963
+ }
2964
+
2965
+ /** A whole run's replies as one input: every reply's text together, and
2966
+ every item's terms pooled, as the Single Test pooled them. */
2967
+ function runInput(ress , plain ) {
2968
+ const replies = [], terms = [];
2969
+ let ms = 0;
2970
+ for (const r of ress) {
2971
+ if (!r) continue;
2972
+ replies.push(r.transcript?.[r.transcript.length - 1]?.got ?? r.raw ?? "");
2973
+ terms.push(...(r.terms || []));
2974
+ ms += r.ms ?? 0;
2975
+ }
2976
+ return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, production: null, kase: null };
2977
+ }
2978
+
2979
+ /** One metric's reading of [input]: `of`, `not` and a failure all applied,
2980
+ or null where it has nothing to read. Synchronous where it can be -- a
2981
+ whole run's verdict is read on every render -- and a promise otherwise. */
2982
+ function readMetric(m , input , ctx ) {
2983
+ const entry = METRICS[m.type];
2984
+ const weight = typeof m.weight === "number" ? m.weight : 1;
2985
+ const label = (isStr(m.metric) && m.metric.trim()) || entry?.label || m.type;
2986
+ if (!entry) return { type: m.type, label, weight, pass: false, score: 0, reason: `"${m.type}" is not a metric this lab has` };
2987
+ const read = m.of === "said" ? { ...input, text: input.said ?? input.text } : input;
2988
+ const done = (o ) => {
2989
+ if (!o) return null;
2990
+ const turned = m.not ? { ...o, pass: !o.pass, score: entry.counts ? -o.score : 1 - o.score, reason: `not: ${o.reason}` } : o;
2991
+ return { type: m.type, label, weight, ...turned };
2992
+ };
2993
+ const failed = (e ) => ({ type: m.type, label, weight, pass: false, score: 0,
2994
+ reason: e instanceof Error ? e.message : String(e) });
2995
+ // A grader's call is timed: a run counts its calls and their time.
2996
+ const t0 = Date.now();
2997
+ const timed = (r ) =>
2998
+ r && entry.graded && ctx.ask ? { ...r, asked: Date.now() - t0 } : r;
2999
+ try {
3000
+ const o = entry.score(read, m, ctx);
3001
+ return o && typeof (o ).then === "function"
3002
+ ? (o ).then(done, failed).then(timed) : timed(done(o ));
3003
+ } catch (e) { return timed(failed(e)); }
3004
+ }
3005
+
3006
+ /** A test's score from its metrics' readings, the way [mode] says: all must
3007
+ pass, or weighted points against [threshold]. What a case metric found,
3008
+ missed and invented is carried up, for a run's totals. Null for none. */
3009
+ function scoreOf(metrics , mode = "all", threshold = null) {
3010
+ if (!metrics.length) return null;
3011
+ const detail = metrics.some(r => r.found || r.missed || r.invented) ? {
3012
+ found: metrics.flatMap(r => r.found ?? []), missed: metrics.flatMap(r => r.missed ?? []),
3013
+ invented: metrics.flatMap(r => r.invented ?? []) } : {};
3014
+ if (mode === "weighted") {
3015
+ const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
3016
+ return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
3017
+ }
3018
+ const counted = metrics.filter(r => r.weight !== 0);
3019
+ return { pass: counted.every(r => r.pass), score: counted.length ? counted.filter(r => r.pass).length / counted.length : null,
3020
+ metrics, ...detail };
3021
+ }
3022
+
3023
+ /** [list] read against one reply, scored the way [mode] says; null where no
3024
+ metric had anything to read. */
3025
+ async function readMetrics(list , input , ctx ,
3026
+ mode = "all", threshold = null) {
3027
+ const metrics = [];
3028
+ for (const m of list) {
3029
+ const r = await readMetric(m, input, ctx);
3030
+ if (r) metrics.push(r);
3031
+ }
3032
+ return scoreOf(metrics, mode, threshold);
3033
+ }
3034
+
3035
+ /** [list] read over a whole run, synchronously: each metric over the run's
3036
+ replies together, or -- `every` -- each reply alone, all of them passing. */
3037
+ function readRun(list , run ) {
3038
+ const sync = (r , m ) =>
3039
+ r && typeof (r ).then === "function"
3040
+ ? { type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: 1, pass: false, score: 0,
3041
+ reason: "a model-graded metric cannot read a whole run" }
3042
+ : r ;
3043
+ return list.flatMap(m => {
3044
+ if (!m.every) { const r = sync(readMetric(m, run, {}), m); return r ? [r] : []; }
3045
+ const each = run.replies.map(text => sync(readMetric(m, { ...run, text, replies: [text] }, {}), m))
3046
+ .filter((r) => !!r);
3047
+ const off = each.filter(r => !r.pass);
3048
+ const first = each[0];
3049
+ if (!first) return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
3050
+ pass: true, score: 1, reason: "no replies" }];
3051
+ return [{ ...first, pass: !off.length, score: each.length ? (each.length - off.length) / each.length : 1,
3052
+ reason: off.length ? `${off.length} of ${each.length} replies: ${off[0] .reason}` : `all ${each.length} replies` }];
3053
+ });
3054
+ }
3055
+
3056
+ TEST_TYPES.metrics = {
3057
+ label: "Metrics",
3058
+ description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3059
+ fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
3060
+ // A metric that reads terms (a case) reads only a kind that yields them.
3061
+ accepts: t => (Array.isArray(t.metrics) && t.metrics.some((m ) => METRICS[m?.type]?.needsTerms)
3062
+ ? Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms) : null),
3063
+ defaults: () => ({ type: "metrics", metrics: [], mode: "all", threshold: null, grader: null, dataset: null, over: "item" }),
3064
+ validate(t, ctx, bad){
3065
+ metricsProblems(t.metrics, "the Metrics", bad);
3066
+ if (t.over != null && t.over !== "item" && t.over !== "run") bad.push("the Metrics read each item, or the whole run");
3067
+ if (t.over === "run") {
3068
+ const one = (Array.isArray(t.metrics) ? t.metrics : []).find((m ) => METRICS[m?.type]?.graded || METRICS[m?.type]?.perItem);
3069
+ if (one) bad.push(`${METRICS[one.type] .label} reads one item at a time, so it cannot read a whole run`);
3070
+ }
3071
+ // The dataset whose cases a case metric reads, and whose own metrics join.
3072
+ if (t.dataset != null) {
3073
+ if (!isRef(t.dataset) || (t.dataset.version != null && !isStr(t.dataset.version))) bad.push("the Metrics name their dataset as { id, name }");
3074
+ else if (ctx.datasets && !ctx.datasets.some(x => x.id === t.dataset.id)) bad.push(`Dataset ${t.dataset.name || t.dataset.id} not found`);
3075
+ }
3076
+ if (!SCORING_MODES.includes(t.mode)) bad.push(`the Metrics are scored ${SCORING_MODES.join(" or ")}`);
3077
+ if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
3078
+ if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
3079
+ if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
3080
+ // The test's own grader, or the lab's.
3081
+ const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
3082
+ if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a Setup profile the lab's grader");
3083
+ // A grader is asked words, and needs a model to ask.
3084
+ const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3085
+ if (grader && ctx.profiles && !p) bad.push(`Setup profile ${grader.name || grader.id} not found`);
3086
+ else if (p) {
3087
+ const type = CONNECTION_TYPES[typeOf(p )];
3088
+ if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
3089
+ else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Setup profile");
3090
+ }
3091
+ },
3092
+ rules: t => (Array.isArray(t.metrics) ? t.metrics : []).map((m , x ) => ({
3093
+ key: `m${x}`, label: (isStr(m.metric) && m.metric) || METRICS[m.type]?.label || m.type,
3094
+ want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3095
+ .filter(Boolean).join(" ") })),
3096
+ profiles: t => (isRef(t.grader) ? [t.grader] : []),
3097
+ // A run carries the lab's grader on a test that names none and may ask one:
3098
+ // a model-graded metric of its own, or a case's, over a dataset.
3099
+ resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
3100
+ ? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
3101
+ wholeRun: t => t.over === "run",
3102
+ // Over the whole run: every metric over the replies together, or each alone.
3103
+ verdict(t, ress, kind){
3104
+ const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
3105
+ const metrics = readRun(t.metrics || [], run);
3106
+ const s = scoreOf(metrics, t.mode, t.threshold);
3107
+ const off = metrics.filter(r => !r.pass);
3108
+ return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
3109
+ ran: run.replies.length,
3110
+ checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3111
+ },
3112
+ // Rule m{x} is the test's own metric x, read in that place on every item.
3113
+ // A score stored before Metrics -- a Graded set's, read as its conversion --
3114
+ // has no readings of its own: its one metric's reading is the score's.
3115
+ ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3116
+ : Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
3117
+ // Every item, with its case's own metrics where it has a case.
3118
+ read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((kase?.metrics ) || [])],
3119
+ metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3120
+ };
3121
+
3122
+ // ---- the lab's own scorers, as metrics ----------------------------------------
3123
+ // What the Graded set and the Single Test check, one metric each, so a test of
3124
+ // either converts to Metrics that read a run exactly as it did
3125
+ // (metrics-parity-check.js). Here rather than in metrics/builtin.ts because
3126
+ // each is the core's own matcher.
3127
+
3128
+ // An item's case, as the Graded set scores it: its expectations, forbidden
3129
+ // terms and count bounds, with what it found, missed and invented kept.
3130
+ METRICS.case = {
3131
+ label: "Matches its case",
3132
+ description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
3133
+ perItem: true,
3134
+ needsTerms: true,
3135
+ options: [],
3136
+ defaults: () => ({}),
3137
+ score(input) {
3138
+ if (!input.kase) return null;
3139
+ const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
3140
+ return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
3141
+ found: s.found, missed: s.missed, invented: s.invented };
3142
+ },
3143
+ };
3144
+
3145
+ // Items the reply holds -- all of them, or any -- matched as the lab matches
3146
+ // a term; a kind that yields none is matched in the replies' text instead,
3147
+ // ignoring case, as the Single Test did.
3148
+ METRICS["has-items"] = {
3149
+ label: "Has items",
3150
+ description: "Passes when the reply holds all, or any, of the items listed.",
3151
+ options: [{ key: "values", label: "Items", type: "textarea" },
3152
+ { key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
3153
+ defaults: () => ({ values: "", need: "all" }),
3154
+ score(input, m) {
3155
+ const values = String(m.values ?? "").split("\n").map(v => v.trim()).filter(Boolean);
3156
+ const has = (v ) => input.plain ? input.replies.some(r => r.toLowerCase().includes(v.toLowerCase()))
3157
+ : input.terms.some(t => termIn([t], v));
3158
+ const hit = values.filter(has), missing = values.filter(v => !has(v));
3159
+ const pass = m.need === "any" ? hit.length > 0 : !missing.length;
3160
+ return { pass, score: values.length ? hit.length / values.length : 1,
3161
+ reason: m.need === "any" ? (hit.length ? `has ${hit.join(", ")}` : `has none of ${values.join(", ")}`)
3162
+ : missing.length ? `missed ${missing.join(", ")}` : `has ${values.join(", ")}` };
3163
+ },
3164
+ };
3165
+
3166
+ // A reply's length in characters, once trimmed.
3167
+ METRICS.length = {
3168
+ label: "Length",
3169
+ description: "Compares the reply's length in characters.",
3170
+ options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
3171
+ { key: "n", label: "Characters", type: "number" }],
3172
+ defaults: () => ({ op: "exactly", n: 0 }),
3173
+ score(input, m) {
3174
+ const got = input.text.trim().length;
3175
+ const pass = lengthHolds(String(m.op || "exactly"), got, Number(m.n));
3176
+ return { pass, score: pass ? 1 : 0, reason: `${got} characters` };
3177
+ },
3178
+ };
3179
+
3180
+ // A reply read as CSV, JSON or XML: it has to read, and hold as many results
3181
+ // as the count says.
3182
+ METRICS["parse-count"] = {
3183
+ label: "Parses",
3184
+ description: "Passes when the reply reads as the format chosen, holding the number of results set.",
3185
+ // Unformatted is one result a reply, as the Single Test counted it.
3186
+ options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
3187
+ { value: "json", label: "json" }, { value: "xml", label: "xml" }] },
3188
+ { key: "countOp", label: "Count is", type: "select", choices: COUNT_OPS.map(([value, label]) => ({ value, label })) },
3189
+ { key: "count", label: "Count", type: "number" }],
3190
+ defaults: () => ({ parse: "json", countOp: "exactly", count: null }),
3191
+ score(input, m) {
3192
+ const pc = parseCount(input.text, (m.parse ) || null);
3193
+ if (!pc.valid) return { pass: false, score: 0, reason: `invalid ${m.parse}` };
3194
+ const count = m.count == null || m.count === "" ? null : Number(m.count);
3195
+ if (count != null && !countHolds(String(m.countOp || "exactly"), pc.count, count)) {
3196
+ return { pass: false, score: 0, reason: `${pc.count} results` };
3197
+ }
3198
+ return { pass: true, score: 1, reason: pc.count == null ? `reads as ${m.parse}` : `${pc.count} results` };
3199
+ },
3200
+ };
3201
+
3202
+ /** What every test carries, kept across a conversion. */
3203
+ const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
3204
+
3205
+ // A Graded set is a Metrics test over the same dataset, its one metric the case.
3206
+ LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
3207
+ dataset: t.dataset, over: "item", metrics: [{ type: "case" }] });
3208
+
3209
+ // A Single Test is Metrics over the whole run: its lists over the run's items
3210
+ // together, and exact, length and parse over each reply alone.
3211
+ LEGACY_TESTS.single .toMetrics = t => {
3212
+ const list = (v ) => (Array.isArray(v) ? v : []).map(String).filter(x => x.trim()).join("\n");
3213
+ const metrics = [];
3214
+ if (list(t.all)) metrics.push({ type: "has-items", values: list(t.all), need: "all" });
3215
+ if (list(t.any)) metrics.push({ type: "has-items", values: list(t.any), need: "any" });
3216
+ if (list(t.none)) metrics.push({ type: "has-items", values: list(t.none), need: "any", not: true });
3217
+ if (t.exact != null && String(t.exact).trim() !== "") metrics.push({ type: "equals", value: String(t.exact), json: false, every: true });
3218
+ if (t.length != null) metrics.push({ type: "length", op: t.lengthOp || "exactly", n: t.length, every: true });
3219
+ if (t.parse === "json" || t.parse === "xml" || t.count != null) {
3220
+ metrics.push({ type: "parse-count", parse: t.parse ?? "", countOp: t.countOp || "exactly", count: t.count ?? null, every: true });
3221
+ }
3222
+ return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
3223
+ };
3224
+
3225
+ /** The grader a Metrics test names, reached through what the runner hands it. */
3226
+ function graderCtx(t , more ) {
3227
+ const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3228
+ return ask ? { ask } : {};
3229
+ }
3230
+
3231
+ /** A metric's options in a few words -- "watering", "^\\{" -- or nothing
3232
+ for one with none set: what History prints beside its name. */
3233
+ function metricSummary(m ) {
3234
+ const entry = METRICS[m.type];
3235
+ // Each option as it reads in the editor: a choice's label, a box that is
3236
+ // ticked by its own label, text by its first line.
3237
+ const said = (entry?.options || []).map(o => {
3238
+ const v = m[o.key];
3239
+ if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3240
+ if (o.type === "checkbox") return v ? o.label.toLowerCase() : "";
3241
+ // Text that runs to lines (a schema) reads as its first words, run together.
3242
+ return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3243
+ }).filter(Boolean);
3244
+ return said.join(", ");
3245
+ }
3246
+
3247
+ // ---- a job's steps (pipeline-model §3) -------------------------------------
3248
+
3249
+ // Attach Content: the items a run goes over, job 1's first step.
3250
+ STEP_TYPES.attachContent = {
3251
+ label: "Attach Content", slot: "content",
3252
+ description: "The items the run goes through: each is sent to every scenario.",
3253
+ in: null, out: "items",
3254
+ apply: "itemContent", // run-evals.js, through its file-type registry
3255
+ fields: ["type", "content"],
3256
+ validate(step, ctx, bad){
3257
+ const c = step?.content;
3258
+ if (!c) return void bad.push("set the content first");
3259
+ const type = isObj(c) && CONTENT_TYPES[c.type];
3260
+ if (!type) return void bad.push(`content of type "${c?.type}" is not one this lab has`);
3261
+ onlyFields(c, "the content", ctx.run ? [...type.fields, ...type.runFields] : type.fields, bad);
3262
+ type.validate(c, ctx, bad);
3263
+ },
3264
+ };
3265
+
3266
+ // Prompt: a Call that sends each scenario's words to its model.
3267
+ STEP_TYPES.prompt = {
3268
+ label: "Prompt", slot: "call", asks: "prompt", modelFrom: "profile",
3269
+ description: "Asks each scenario's model its prompt about the item.",
3270
+ in: "item", out: "text",
3271
+ apply: "runPipeline",
3272
+ fields: ["type", "withImage", "tokenMappings"],
3273
+ validate(step, _ctx, bad, at){
3274
+ if (typeof step.withImage !== "boolean") bad.push(`${at}: withImage has to be true or false`);
3275
+ tokenMappingsProblems(step.tokenMappings, at, bad);
3276
+ },
3277
+ };
3278
+
3279
+ // HTTP Request: a Call that sends a Power Automate step's own request,
3280
+ // rebuilt for each item from its record (docs/power-automate.md).
3281
+ const HTTP_METHODS = ["GET", "POST", "PUT", "PATCH", "DELETE"];
3282
+ /** The first field under [v] named like a key, as a path, or null. */
3283
+ function keyField(v , at = "body") {
3284
+ if (Array.isArray(v)) {
3285
+ for (let i = 0; i < v.length; i++) { const hit = keyField(v[i], `${at}[${i}]`); if (hit) return hit; }
3286
+ } else if (isObj(v)) {
3287
+ for (const [k, x] of Object.entries(v)) {
3288
+ if (LOOKS_LIKE_A_KEY.test(k) || /^(?:api[-_]?key|access[-_]?token|client[-_]?secret)$/i.test(k)) return `${at}.${k}`;
3289
+ const hit = keyField(x, `${at}.${k}`);
3290
+ if (hit) return hit;
3291
+ }
3292
+ }
3293
+ return null;
3294
+ }
3295
+ STEP_TYPES.httpRequest = {
3296
+ label: "HTTP Request", slot: "call", asks: "request", verbatim: true, modelFrom: "cell",
3297
+ description: "Sends a Power Automate step's own request, rebuilt from each record, with each scenario's words in it.",
3298
+ cellFields: ["system", "model", "settings", "prefill"],
3299
+ firstJobOnly: "an HTTP Request is sent from its item's record, so it is job 1's call",
3300
+ // Production's reply is the record's result, read as the flow reads it.
3301
+ production(call, record){
3302
+ const result = isObj(record) && isObj(record.result) ? record.result : null;
3303
+ if (!result || typeof result.status !== "number") return null;
3304
+ try {
3305
+ return httpReplyOf(call, { prompt: "" }, result.body, result.status,
3306
+ { scope: record.scope || {}, now: record.at ?? null }).raw;
3307
+ } catch { return null; }
3308
+ },
3309
+ // A new scenario starts from the flow's own words and fields.
3310
+ newCell: (call) => (HTTP_APIS[call.api] ?? HTTP_APIS.raw ).cellOf(call.body),
3311
+ // A cell's fields, where its HTTP API has a place for them.
3312
+ cellProblems(cell, call, at, bad){
3313
+ const api = HTTP_APIS[call.api];
3314
+ if (!api) return;
3315
+ for (const f of ["system", "model", "prefill"] ) {
3316
+ if (cell[f] == null) continue;
3317
+ if (!isStr(cell[f])) bad.push(`${at}: ${f} has to be text`);
3318
+ else if (!api.fields.includes(f)) bad.push(`${at}: ${api.label} has no place for a ${f}`);
3319
+ }
3320
+ if (cell.settings != null) {
3321
+ if (!isObj(cell.settings) || !Object.values(cell.settings).every(isStr)) bad.push(`${at}: settings are names to text`);
3322
+ else for (const k of Object.keys(cell.settings)) {
3323
+ if (!api.settings.includes(k)) bad.push(`${at}: ${api.label} has no setting ${k}`);
3324
+ }
3325
+ }
3326
+ if (isStr(cell.prompt)) {
3327
+ try { api.place(call.body, cell ); }
3328
+ catch (e) { bad.push(`${at}: ${(e ).message}`); }
3329
+ }
3330
+ },
3331
+ in: "item", out: "text",
3332
+ apply: "runPipeline",
3333
+ fields: ["type", "step", "api", "method", "path", "query", "body", "readAs", "loop"],
3334
+ validate(step, _ctx, bad, at){
3335
+ if (!isStr(step.step) || !step.step.trim()) bad.push(`${at}: an HTTP Request names the flow step it is`);
3336
+ if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
3337
+ if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
3338
+ if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
3339
+ if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
3340
+ if (step.readAs != null && !isStr(step.readAs)) bad.push(`${at}: readAs is an expression or null`);
3341
+ if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
3342
+ // A key is the Setup profile's, sent by its Key header; one in the
3343
+ // template would be kept in the pipeline and every run of it.
3344
+ const hit = keyField(step.body) ?? keyField(step.query, "query");
3345
+ if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Setup profile sends it`);
3346
+ },
3347
+ };
3348
+
3349
+ // Read Reply: how a job's answer is read -- its kind and modifiers.
3350
+ STEP_TYPES.readReply = {
3351
+ label: "Read Reply", slot: "reply",
3352
+ description: "How the reply is read before tests and later jobs see it.",
3353
+ in: "text", out: step => step.out?.kind,
3354
+ // Read inside the stage: runPipeline parses each reply as it comes back.
3355
+ apply: "runPipeline",
3356
+ fields: ["type", "out"],
3357
+ validate(step, _ctx, bad, at){
3358
+ const out = step.out;
3359
+ if (!isObj(out) || !OUTPUT_KINDS[out.kind]) {
3360
+ return void bad.push(`${at} answers as "${out?.kind}", which is not an output kind this lab has`);
3361
+ }
3362
+ const kindEntry = OUTPUT_KINDS[out.kind] ;
3363
+ onlyFields(out, `${at}'s output`, ["kind", "modifiers", ...(kindEntry.settings || []).map(o => o.key)], bad);
3364
+ kindEntry.validateSettings?.(out, `${at}'s output`, bad);
3365
+ if (!Array.isArray(out.modifiers)) return void bad.push(`${at}: modifiers has to be a list`);
3366
+ for (const m of out.modifiers) {
3367
+ const entry = isObj(m) && MODIFIERS[m.type];
3368
+ if (!entry) { bad.push(`${at}: "${m?.type}" is not a modifier this lab has`); continue; }
3369
+ if (entry.accepts && !entry.accepts.includes(out.kind)) {
3370
+ bad.push(`${at}: ${entry.label || m.type} does not apply to ${OUTPUT_KINDS[out.kind] .noun || out.kind}`);
3371
+ }
3372
+ entry.validate?.(m , `${at}: ${entry.label || m.type}`, bad);
3373
+ }
3374
+ },
3375
+ };
3376
+
3377
+ // ---- reading a job's steps ---------------------------------------------------
3378
+ // Every reader asks these, never a step's position: a job's slots are fixed,
3379
+ // but which of them a job has (Attach Content is job 1's alone) is its own.
3380
+
3381
+ const slotOf = (st ) =>
3382
+ (isObj(st) ? STEP_TYPES[st.type ]?.slot : undefined);
3383
+
3384
+ /** The step in [job]'s [slot], whatever type fills it. */
3385
+ const stepIn = (job , slot ) =>
3386
+ (job?.steps || []).find(st => slotOf(st) === slot) ;
3387
+
3388
+ /** The registry entry that fills [slot] by default: what a new step there is. */
3389
+ function slotEntry(slot ) {
3390
+ const hit = Object.entries(STEP_TYPES).find(([, e]) => e.slot === slot);
3391
+ return hit ? { type: hit[0], entry: hit[1] } : undefined;
3392
+ }
3393
+
3394
+ /** A pipeline's content: job 1's content step, or none. */
3395
+ function contentOf(doc ) {
3396
+ return stepIn (doc?.jobs?.[0], "content")?.content ?? null;
3397
+ }
3398
+
3399
+ /** [doc] with its content set -- job 1's content step added, replaced, or
3400
+ taken away for null. */
3401
+ function withContent (doc , content ) {
3402
+ const [first, ...rest] = doc.jobs;
3403
+ if (!first) return doc;
3404
+ const steps = first.steps.filter(st => slotOf(st) !== "content");
3405
+ const type = slotEntry("content") .type;
3406
+ return { ...doc, jobs: [{ ...first, steps: content ? [{ type, content } , ...steps] : steps }, ...rest] };
3407
+ }
3408
+
3409
+ /** A job's call: the step that asks something and is answered. */
3410
+ const callOf = (job ) =>
3411
+ stepIn (job, "call") ;
3412
+
3413
+ /** The registry entry of a job's call: what it asks, and of whom. */
3414
+ const callEntryOf = (job ) =>
3415
+ STEP_TYPES[(stepIn(job, "call") )?.type ?? ""];
3416
+
3417
+ /** The request an HTTP Request [step] sends for [cell] over an item whose
3418
+ record read [ctx]'s scope: the cell's words and fields placed in the
3419
+ flow's body by its HTTP API, then every expression evaluated. */
3420
+ function httpRequestOf(step , cell , ctx )
3421
+ {
3422
+ const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
3423
+ const at = { ...ctx, loop: step.loop };
3424
+ const path = String(evaluate(step.path, at));
3425
+ // A path an expression wrote whole -- an address -- keeps its path and query.
3426
+ const [p, q] = /^https?:\/\//i.test(path) ? (() => { const u = new URL(path); return [u.pathname, u.search]; })() : [path, ""];
3427
+ const query = Object.fromEntries(Object.entries(step.query).map(([k, v]) => [k, asText(evaluate(v, at))]));
3428
+ return { method: step.method, path: p + q, query, body: evaluate(api.place(step.body, cell), at) };
3429
+ }
3430
+
3431
+ /** What the flow reads of a reply [j] to [step]: its readAs evaluated with the
3432
+ reply as the step's body -- the prefill put back in front, as the flow has
3433
+ to -- or the HTTP API's own text. */
3434
+ function httpReplyOf(step , cell , j , status ,
3435
+ ctx ) {
3436
+ const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
3437
+ const whole = cell.prefill && api.withPrefill ? api.withPrefill(j, cell.prefill) : j;
3438
+ const { raw: said, finishReason } = api.reply(whole);
3439
+ if (!step.readAs) return { raw: said, said, finishReason };
3440
+ const scope = { ...ctx.scope, actions: { ...(ctx.scope.actions || {}), [step.step]: { body: whole, statusCode: status } } };
3441
+ return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop })), said, finishReason };
3442
+ }
3443
+
3444
+ /** A job's reply step. */
3445
+ const replyOf = (job ) =>
3446
+ stepIn (job, "reply") ;
3447
+
3448
+ /** How a job's reply is read: its kind, settings and modifiers. */
3449
+ const outOf = (job ) => replyOf(job)?.out;
3450
+
3451
+ /** [job] with the step in [slot] changed by [change]. */
3452
+ function withSlot(job , slot , change ) {
3453
+ return { ...job, steps: job.steps.map(st => (slotOf(st) === slot ? change(st) : st)) };
3454
+ }
3455
+
3456
+ /** [job] reading its reply as [out]. */
3457
+ const withOut = (job , out ) => withSlot(job, "reply", st => ({ ...st, out } ));
3458
+
3459
+ /** [job] with its call changed by [patch]. */
3460
+ const withCall = (job , patch ) => withSlot(job, "call", st => ({ ...st, ...patch } ));
3461
+ /** A new id: random, so one minted in one browser never collides with one
3462
+ minted in another. A pipeline, its jobs and its scenarios each carry one
3463
+ (#93), minted once and never shown. */
3464
+ function newId() {
3465
+ const bytes = new Uint8Array(6);
3466
+ globalThis.crypto.getRandomValues(bytes);
3467
+ return Array.from(bytes, (b) => b.toString(16).padStart(2, "0")).join("");
3468
+ }
3469
+
3470
+ /** A new job: the kind a stage answers in, that kind's own modifiers, and
3471
+ a copy of the token set. */
3472
+ function jobDefaults(kind = defaultOutputKind, tokens = TOKEN_DEFAULTS) {
3473
+ return {
3474
+ type: "job", id: newId(), name: "",
3475
+ steps: [
3476
+ { type: "prompt", withImage: false, tokenMappings: clone(tokens) },
3477
+ { type: "readReply",
3478
+ out: { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] } },
3479
+ ],
3480
+ };
3481
+ }
3482
+ STEP_TYPES.job = {
3483
+ in: "item", out: job => outOf(job )?.kind,
3484
+ apply: "runPipeline",
3485
+ fields: ["type", "id", "name", "steps"],
3486
+ defaults: jobDefaults,
3487
+ // A job's steps sit in fixed slots: Attach Content first, on job 1 only;
3488
+ // one Call; Read Reply last. Each step is its entry's to check.
3489
+ validate(c, ctx, bad, k, doc){
3490
+ const at = jobLabel(doc, k);
3491
+ if (!isObj(c) || c.type !== "job") return void bad.push(`${at} has to be a job step`);
3492
+ onlyFields(c, at, STEP_TYPES.job .fields , bad);
3493
+ // An id is a job's own, like a scenario's: stable across renames, so a
3494
+ // stored run keeps pointing at the job that actually ran (#93).
3495
+ if (!isStr(c.id) || !c.id.trim()) bad.push(`${at} has no id`);
3496
+ else if ((doc.jobs ).findIndex((o) => isObj(o) && o.id === c.id) !== k) bad.push(`${at} has the id of another job`);
3497
+ if (c.name != null && !isStr(c.name)) bad.push(`${at}: name has to be text`);
3498
+ if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
3499
+ const steps = c.steps;
3500
+ // A job's steps are the entries with a slot; one without (a job, the
3501
+ // tests) or of no type the lab has is not one.
3502
+ const unknown = steps.find(st => !slotOf(st));
3503
+ if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
3504
+ const slots = steps.map(st => slotOf(st) );
3505
+ const want = [...(k === 0 && slots[0] === "content" ? ["content" ] : []), "call", "reply"];
3506
+ if (slots.join() !== want.join()) {
3507
+ // One sentence for the one mistake, the likeliest first.
3508
+ const label = (slot ) => slotEntry(slot)?.entry.label ?? slot;
3509
+ bad.push(k > 0 && slots.includes("content")
3510
+ ? `${at}: only job 1 attaches content -- a later job is handed the reply before it`
3511
+ : slots.filter(x => x === "call").length !== 1 ? `${at} has to have one call step`
3512
+ : `${at}: steps are ${label("content")} (job 1), one call, then ${label("reply")}`);
3513
+ return;
3514
+ }
3515
+ for (const st of steps) {
3516
+ const entry = STEP_TYPES[st.type] ;
3517
+ if (k > 0 && entry.firstJobOnly) { bad.push(`${at}: ${entry.firstJobOnly}`); continue; }
3518
+ onlyFields(st, `${at}'s ${entry.label ?? st.type}`, entry.fields ?? [], bad);
3519
+ entry.validate(st, ctx, bad, at);
3520
+ }
3521
+ },
3522
+ };
3523
+ // What a test carries besides its type's own fields.
3524
+ const TEST_FIELDS = ["id", "name", "continueOnFailure"];
3525
+
3526
+ /** Test [j]'s name, or the number it has always shown. */
3527
+ const testLabel = (doc , j ) =>
3528
+ (doc.tests?.[j]?.name || "").trim() || `Test ${j + 1}`;
3529
+
3530
+ /** A test type that settles over the whole run rather than item by item. */
3531
+ const isWholeRun = (type , t ) =>
3532
+ type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
3533
+
3534
+ /**
3535
+ * A document's tests as a list, whatever version wrote it: a queue row keeps
3536
+ * the run document it was submitted with, so a run from before version 6
3537
+ * holds one test or null, read here as the upgrade reads it (testsList).
3538
+ */
3539
+ function testsOf(doc ) {
3540
+ const t = doc?.tests;
3541
+ if (Array.isArray(t)) return t ;
3542
+ return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3543
+ }
3544
+
3545
+ /** The dataset a document's tests grade against, where one does: a run
3546
+ grades against one (validatePipeline says so), so the first names it. */
3547
+ function testsDataset(doc ) {
3548
+ const t = testsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3549
+ return t && "dataset" in t ? t.dataset : null;
3550
+ }
3551
+
3552
+ STEP_TYPES.tests = {
3553
+ in: "results", out: "verdict",
3554
+ apply: "score",
3555
+ validate(tests, ctx, bad, lastKind, doc){
3556
+ if (!Array.isArray(tests)) return void bad.push("tests has to be a list, empty for an unscored run");
3557
+ const seen = new Set ();
3558
+ tests.forEach((t , j ) => {
3559
+ const at = testLabel(doc, j);
3560
+ const type = isObj(t) && TEST_TYPES[t.type];
3561
+ if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
3562
+ const before = bad.length;
3563
+ if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
3564
+ else if (seen.has(t.id)) bad.push(`${at} has the id of another test`);
3565
+ else seen.add(t.id);
3566
+ if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
3567
+ if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
3568
+ onlyFields(t, at, [...type.fields, ...TEST_FIELDS], bad);
3569
+ type.validate(t, ctx, bad);
3570
+ // Which kinds a test scores is only worth saying of a test that is whole.
3571
+ if (bad.length > before) return;
3572
+ const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
3573
+ if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
3574
+ bad.push(`${at}: the ${type.label} scores ${accepts.map(k => OUTPUT_KINDS[k]?.noun || k).join(" or ")}, `
3575
+ + `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
3576
+ }
3577
+ });
3578
+ // A run is handed one dataset's body to grade against (server-side-runs §4).
3579
+ const named = new Set(tests.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3580
+ if (named.size > 1) bad.push("the tests grade against one dataset at a time");
3581
+ },
3582
+ };
3583
+
3584
+ // ---- the document -------------------------------------------------------------
3585
+
3586
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "scenarios", "tests"];
3587
+ // What resolving adds, and nothing else: the profiles it resolved to and the
3588
+ // run's own comment, which belongs to the run and never to the pipeline.
3589
+ const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
3590
+
3591
+ /** The one rule for a document that is not this version: refuse it, by name. */
3592
+ function versionProblem(doc ) {
3593
+ if (!isObj(doc)) return "a pipeline is a JSON object";
3594
+ if (doc.version === PIPELINE_VERSION) return null;
3595
+ return doc.version == null
3596
+ ? `this pipeline carries no version, and the lab reads version ${PIPELINE_VERSION}`
3597
+ : `this pipeline is version ${JSON.stringify(doc.version)}, and the lab reads version ${PIPELINE_VERSION}`;
3598
+ }
3599
+
3600
+ /** What an upgrade from version 2 needs from outside the document: the rules
3601
+ of the dataset a pipeline was graded against, which the retired
3602
+ version-2 list kind read every reply under. Without them a job is upgraded with no
3603
+ rules, as a run with no graded test parsed. */
3604
+
3605
+
3606
+
3607
+
3608
+ /**
3609
+ * [doc] at this version, when it is an earlier one this lab can read. From
3610
+ * version 1, its chains' `tokens` maps become `tokenMappings` lists (values
3611
+ * first, then blocks, each in the order the map held them). From version 2,
3612
+ * a chain of the retired version-2 list kind becomes a `list` with the modifiers
3613
+ * that read a reply exactly as it did (legacyTagsOut, held to it by
3614
+ * parity-check.js), under the graded dataset's rules [ctx] supplies, keeping
3615
+ * its case; and that kind's token in every prompt is {items}. From versions 3
3616
+ * and 4, every id a current document must carry is minted for the ones it
3617
+ * lacks: a scenario's `s1`, `s2` ... by position and a chain's `j1`, `j2` ...
3618
+ * by position -- the same on every reading, so a pipeline read twice before
3619
+ * it is saved keeps them, and the Prompt library's recorded uses keep
3620
+ * matching -- and a new one for the pipeline itself. Anything else comes back
3621
+ * as it was, for versionProblem to name. A copy: the caller's document is not
3622
+ * touched. From version 5, its test -- or none -- becomes a list of one (or
3623
+ * none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
3624
+ * renames and every job's `type` becomes `"job"`.
3625
+ */
3626
+ /** Every id a current document must carry, minted for the ones [doc] lacks.
3627
+ An id it already has is kept. */
3628
+ function upgradeIds(doc ) {
3629
+ doc.version = PIPELINE_VERSION;
3630
+ if (!isStr(doc.id) || !doc.id) doc.id = newId();
3631
+ (Array.isArray(doc.chains) ? doc.chains : []).forEach((ch , k ) => {
3632
+ if (isObj(ch) && (!isStr(ch.id) || !ch.id)) ch.id = `j${k + 1}`;
3633
+ });
3634
+ (Array.isArray(doc.scenarios) ? doc.scenarios : []).forEach((sc , i ) => {
3635
+ if (isObj(sc) && (!isStr(sc.id) || !sc.id)) sc.id = `s${i + 1}`;
3636
+ });
3637
+ return doc;
3638
+ }
3639
+
3640
+ /** Every profile reference cut to { id, name }. The Runs tab once kept the
3641
+ whole Setup profile a scenario picked -- key and all -- so a document saved
3642
+ then is cleaned as it is read, and validation refuses one that is not. */
3643
+ function profileRefs(doc ) {
3644
+ const cut = (r ) => (isObj(r) && isStr(r.id) ? { id: r.id, name: isStr(r.name) ? r.name : "" } : r);
3645
+ for (const sc of Array.isArray(doc.scenarios) ? doc.scenarios : []) {
3646
+ if (!isObj(sc)) continue;
3647
+ if ("profile" in sc) sc.profile = cut(sc.profile);
3648
+ for (const cell of Array.isArray(sc.stages) ? sc.stages : []) {
3649
+ if (isObj(cell) && cell.profile != null) cell.profile = cut(cell.profile);
3650
+ }
3651
+ }
3652
+ }
3653
+
3654
+ /** Whether any profile reference in a pipeline holds more than { id, name }. */
3655
+ function fatProfileRef(doc ) {
3656
+ const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
3657
+ return isObj(doc) && Array.isArray(doc.scenarios) && doc.scenarios.some((sc ) =>
3658
+ isObj(sc) && (fat(sc.profile) || (Array.isArray(sc.stages) && sc.stages.some((c ) => isObj(c) && fat(c.profile)))));
3659
+ }
3660
+
3661
+ /** Version 5 to 6: the one test, or none, as a list. Its id is `t1`, so
3662
+ reading the same document twice gives the same id, and a stored result's
3663
+ score is read as that test's (upgradeResults). */
3664
+ function testsList(doc ) {
3665
+ doc.version = PIPELINE_VERSION;
3666
+ // Every upgrade ends here, so this is where a profile reference is cut.
3667
+ profileRefs(doc);
3668
+ if (Array.isArray(doc.tests)) return doc;
3669
+ doc.tests = isObj(doc.tests) ? [{ id: "t1", name: "", ...doc.tests, continueOnFailure: true }] : [];
3670
+ return doc;
3671
+ }
3672
+
3673
+ /** A new test of [type] at the lab's defaults, named [name], continuing on failure. */
3674
+ function newTest(type , name = "", fields = {}) {
3675
+ const own = TEST_TYPES[type]?.defaults?.() ?? { type };
3676
+ return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
3677
+ }
3678
+
3679
+ /** Version 6 to 7: chains are jobs. The field renames in place -- so an
3680
+ upgraded document reads the same, key for key -- and every job's `type`
3681
+ becomes `"job"`, so an older document reads as one of today's everywhere. */
3682
+ function jobsOf(next ) {
3683
+ next.version = PIPELINE_VERSION;
3684
+ if (Array.isArray(next.chains)) {
3685
+ const entries = Object.entries(next).map(([k, v]) => (k === "chains" ? ["jobs", v] : [k, v]));
3686
+ next = Object.fromEntries(entries);
3687
+ for (const job of next.jobs) if (isObj(job)) job.type = "job";
3688
+ }
3689
+ return next;
3690
+ }
3691
+
3692
+ /** Version 7 to 8: a job is its steps (pipeline-model §3). The pipeline's
3693
+ content becomes job 1's Attach Content; a job's image flag and token
3694
+ mappings its Prompt call; its output its Read Reply. What each held is
3695
+ moved as it was, so the run is the same run. */
3696
+ function stepsOf(next ) {
3697
+ next.version = PIPELINE_VERSION;
3698
+ const content = next.content ?? null;
3699
+ delete next.content;
3700
+ (Array.isArray(next.jobs) ? next.jobs : []).forEach((job , k ) => {
3701
+ if (!isObj(job) || Array.isArray(job.steps)) return;
3702
+ const { withImage, tokenMappings, out, ...rest } = job;
3703
+ for (const key of Object.keys(job)) delete job[key];
3704
+ Object.assign(job, rest, {
3705
+ steps: [
3706
+ ...(k === 0 && content ? [{ type: "attachContent", content }] : []),
3707
+ { type: "prompt", withImage: !!withImage, tokenMappings: tokenMappings ?? [] },
3708
+ { type: "readReply", out },
3709
+ ],
3710
+ });
3711
+ });
3712
+ return next;
3713
+ }
3714
+
3715
+ /** Version 8 to 9: every test is Metrics -- a Single Test or a Graded set
3716
+ converted, keeping its id, name and Continue on failure; one of a type
3717
+ this lab never had is left for validation to name. */
3718
+ function metricsOf(next ) {
3719
+ next.version = PIPELINE_VERSION;
3720
+ if (Array.isArray(next.tests)) {
3721
+ next.tests = next.tests.map((t ) => (isObj(t) && LEGACY_TESTS[t.type]?.toMetrics ? LEGACY_TESTS[t.type] .toMetrics (t) : t));
3722
+ }
3723
+ return next;
3724
+ }
3725
+
3726
+ function upgradePipeline (doc , ctx = {}) {
3727
+ // A current document is read as it is, but for a profile reference the Runs
3728
+ // tab saved whole (see profileRefs), which is cut back.
3729
+ if (isObj(doc) && doc.version === PIPELINE_VERSION) {
3730
+ if (!fatProfileRef(doc)) return doc;
3731
+ const cleaned = clone(doc) ;
3732
+ profileRefs(cleaned);
3733
+ return cleaned ;
3734
+ }
3735
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8].includes(doc.version )) return doc;
3736
+ let next = clone(doc) ;
3737
+ if (next.version === 8) return metricsOf(next) ;
3738
+ if (next.version === 7) return metricsOf(stepsOf(testsList(next))) ;
3739
+ if (next.version === 6) return metricsOf(stepsOf(jobsOf(next))) ;
3740
+ if (next.version === 5) return metricsOf(stepsOf(jobsOf(testsList(next)))) ;
3741
+ if (next.version === 3 || next.version === 4) return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next))))) ;
3742
+ if (next.version === 1) {
3743
+ next.version = 2;
3744
+ if (Array.isArray(next.chains)) {
3745
+ for (const ch of next.chains) {
3746
+ if (!isObj(ch) || !("tokens" in ch)) continue;
3747
+ ch.tokenMappings = tokenMappingsFromV1(ch.tokens);
3748
+ delete ch.tokens;
3749
+ }
3750
+ }
3751
+ }
3752
+ next.version = PIPELINE_VERSION;
3753
+ const graded = isObj(next.tests) && next.tests.type === "graded" && isObj(next.tests.dataset) ? next.tests.dataset.id : null;
3754
+ const rules = graded && ctx.rulesFor ? ctx.rulesFor(graded) ?? null : null;
3755
+ for (const ch of Array.isArray(next.chains) ? next.chains : []) {
3756
+ if (!isObj(ch) || !isObj(ch.out) || ch.out.kind !== "tags") continue; // vocab: the v2 kind's id
3757
+ const mode = (Array.isArray(ch.out.modifiers) ? ch.out.modifiers : []).find((m ) => m?.type === "case")?.mode ?? "NONE";
3758
+ ch.out = clone(legacyTagsOut(rules, mode));
3759
+ }
3760
+ const renamed = (text ) => typeof text === "string"
3761
+ ? text.replace(/\{((?:stage[1-9]\d*\.)?)tags\}/g, "{$1items}") : text; // vocab: the v2 kind's token
3762
+ for (const sc of Array.isArray(next.scenarios) ? next.scenarios : []) {
3763
+ for (const cell of isObj(sc) && Array.isArray(sc.stages) ? sc.stages : []) {
3764
+ if (isObj(cell)) cell.prompt = renamed(cell.prompt);
3765
+ }
3766
+ }
3767
+ return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next))))) ;
3768
+ }
3769
+
3770
+ /** Version 1's two maps as a list of mappings. */
3771
+ function tokenMappingsFromV1(set ) {
3772
+ if (!isObj(set)) return [];
3773
+ const v1 = set ;
3774
+ return [
3775
+ ...Object.entries(isObj(v1.values) ? v1.values : {}).map(([name, value]) => ({ name, type: "value", value: String(value) })),
3776
+ ...Object.entries(isObj(v1.blocks) ? v1.blocks : {}).map(([name, on]) => ({ name, type: "block", enabled: !!on })),
3777
+ ];
3778
+ }
3779
+
3780
+ /** A current-version document with one job and no scenario. Its job sends
3781
+ the item's image, as a job over a Source would; one over text content
3782
+ has none to send, and the page turns it off with the content. */
3783
+ function blankPipeline(opts = {}) {
3784
+ const job = withCall(jobDefaults(opts.kind, opts.tokens), { withImage: true });
3785
+ return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
3786
+ jobs: [job], scenarios: [], tests: [] };
3787
+ }
3788
+
3789
+ /**
3790
+ * Every reason [doc] cannot run, as sentences -- none when it can. The
3791
+ * reference lookups are [ctx]'s: `profiles(id)` and `sources(id)` answer with
3792
+ * what the id names, or nothing. A run document answers for its own profiles.
3793
+ */
3794
+ function validatePipeline(input , ctx = {}) {
3795
+ const v = versionProblem(input);
3796
+ if (v) return [v];
3797
+ const doc = input ;
3798
+ const bad = [];
3799
+ const run = doc.profiles !== undefined;
3800
+ const c = { ...ctx, run };
3801
+ if (run && !ctx.profiles) {
3802
+ c.profiles = (id ) => isObj(doc.profiles) && Object.hasOwn(doc.profiles, id) ? doc.profiles[id] : null;
3803
+ }
3804
+ onlyFields(doc, "the pipeline", run ? RUN_FIELDS : PIPELINE_FIELDS, bad);
3805
+ if (!isStr(doc.id) || !doc.id.trim()) bad.push("the pipeline has no id");
3806
+ if (doc.name != null && !isStr(doc.name)) bad.push("name has to be text");
3807
+ if (run && doc.comment != null && !isStr(doc.comment)) bad.push("comment has to be text");
3808
+ if (!Array.isArray(doc.jobs) || !doc.jobs.length) {
3809
+ return [...bad, "jobs has to be a list of at least one job"];
3810
+ }
3811
+ if (!Array.isArray(doc.scenarios) || !doc.scenarios.length) bad.push("add a scenario first");
3812
+ // Content is job 1's Attach Content step; a job 1 without one has none yet.
3813
+ if (!stepIn(doc.jobs[0], "content")) bad.push("set the content first");
3814
+ if (run) profilesProblems(doc.profiles, bad);
3815
+ doc.jobs.forEach((ch , k ) => STEP_TYPES.job .validate(ch, c, bad, k, doc));
3816
+ if (bad.length) return bad;
3817
+ const content = contentOf(doc) ;
3818
+
3819
+ const scenarios = doc.scenarios;
3820
+ const n = doc.jobs.length;
3821
+ scenarios.forEach((sc, i) => {
3822
+ const at = scenarioLabel(doc, i);
3823
+ if (!isObj(sc)) return void bad.push(`${at} has to be an object`);
3824
+ onlyFields(sc, at, ["id", "name", "profile", "stages"], bad);
3825
+ if (!isStr(sc.id) || !sc.id.trim()) bad.push(`${at} has no id`);
3826
+ else if (scenarios.findIndex((o) => isObj(o) && o.id === sc.id) !== i) bad.push(`${at} has the id of another scenario`);
3827
+ if (sc.name != null && !isStr(sc.name)) bad.push(`${at}: name has to be text`);
3828
+ if (!isRef(sc.profile)) bad.push(`${at} has to name its Setup profile as { id, name }`);
3829
+ else onlyFields(sc.profile, `${at}'s profile`, ["id", "name"], bad);
3830
+ if (!Array.isArray(sc.stages) || sc.stages.length !== n) {
3831
+ bad.push(`${at} has ${Array.isArray(sc.stages) ? sc.stages.length : "no"} prompts for ${n} job${n === 1 ? "" : "s"}`);
3832
+ return;
3833
+ }
3834
+ sc.stages.forEach((cell , k ) => {
3835
+ if (!isObj(cell)) return void bad.push(`${at}, ${jobLabel(doc, k)} has to be an object`);
3836
+ const call = callEntryOf(doc.jobs[k]);
3837
+ onlyFields(cell, `${at}, ${jobLabel(doc, k)}`, ["prompt", "profile", "from", ...(call?.cellFields || [])], bad);
3838
+ call?.cellProblems?.(cell, callOf(doc.jobs[k]), `${at}, ${jobLabel(doc, k)}`, bad);
3839
+ if (cell.profile != null && !isRef(cell.profile)) bad.push(`${at}, ${jobLabel(doc, k)} has to name its Setup profile as { id, name }`);
3840
+ else if (cell.profile != null) onlyFields(cell.profile, `${at}, ${jobLabel(doc, k)}'s profile`, ["id", "name"], bad);
3841
+ if (cell.from != null && !isRef(cell.from)) bad.push(`${at}, ${jobLabel(doc, k)} has to name the prompt it was picked from as { id, name }`);
3842
+ });
3843
+ });
3844
+ if (bad.length) return bad;
3845
+
3846
+ // A reference that resolves to nothing, once per reference.
3847
+ const missing = new Set ();
3848
+ const lookup = (ref ) => {
3849
+ const p = c.profiles ? c.profiles(ref.id) : {};
3850
+ if (!p && !missing.has(ref.id)) {
3851
+ missing.add(ref.id);
3852
+ bad.push(`Setup profile ${ref.name || ref.id} not found`);
3853
+ }
3854
+ return p;
3855
+ };
3856
+ // A type that answers from the item (Echo) has no model, answers only
3857
+ // job 1, and answers only a text item.
3858
+ const local = (p ) => !!p && typeof p === "object" && !!CONNECTION_TYPES[typeOf(p )]?.local;
3859
+ // Every scenario needs a prompt in every job. A blank one is an error
3860
+ // rather than a stage dropped, which would hand the job before it
3861
+ // straight to the job after it, under numbers nobody sees. The one it
3862
+ // may leave blank is job 1 answered from the item's own text (Echo over
3863
+ // a Source or Text): the reply is the item, and a prompt would only be
3864
+ // what it is read against. Over Prompt only the prompt is the reply.
3865
+ const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
3866
+ for (let k = 0; k < n; k++) {
3867
+ const i = scenarios.findIndex(sc => (!isStr(sc.stages[k].prompt) || !sc.stages[k].prompt.trim())
3868
+ && !(k === 0 && itemsHaveText && local(c.profiles ? c.profiles((sc.stages[0].profile || sc.profile).id) : null)));
3869
+ if (i >= 0) bad.push(n > 1 ? `${jobLabel(doc, k)}, ${scenarioLabel(doc, i)} has no prompt`
3870
+ : `${scenarioLabel(doc, i)} has no prompt`);
3871
+ }
3872
+ const contentFiles = content?.type === "source"
3873
+ ? (c.sources?.(content.ref?.id)?.files ?? content.files ?? null) : null;
3874
+ const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
3875
+ .find(n => !/\.(?:txt|md|csv)$/i.test(n));
3876
+ scenarios.forEach((sc, i) => {
3877
+ const conns = sc.stages.map((cell ) => lookup(cell.profile || sc.profile));
3878
+ conns.forEach((p , k ) => {
3879
+ if (!local(p)) return;
3880
+ const label = CONNECTION_TYPES[typeOf(p )] .label;
3881
+ if (k > 0) bad.push(`${jobLabel(doc, k)}, ${scenarioLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
3882
+ else if (nonText) bad.push(`${scenarioLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
3883
+ });
3884
+ // A connection answers what its job's call asks: words, or a whole request.
3885
+ conns.forEach((p , k ) => {
3886
+ const call = callEntryOf(doc.jobs[k]);
3887
+ // Only a profile that was looked up says what it answers.
3888
+ if (!c.profiles || !p || !isObj(p) || !call) return;
3889
+ const type = CONNECTION_TYPES[typeOf(p )];
3890
+ if (type && !(type.answers ?? ["prompt"]).includes(call.asks ?? "prompt")) {
3891
+ bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${scenarioLabel(doc, i)}: ${call.label ?? "this call"} needs a profile that answers it, and ${type.label} does not`);
3892
+ }
3893
+ });
3894
+ // The model is the profile's unless the call's cell carries it.
3895
+ const bare = conns.findIndex((p , k ) => p && c.profiles && !local(p)
3896
+ && callEntryOf(doc.jobs[k])?.modelFrom !== "cell" && !String(p.model || "").trim());
3897
+ if (bare >= 0) {
3898
+ bad.push(n > 1
3899
+ ? `no model on the profile ${jobLabel(doc, bare)} of ${scenarioLabel(doc, i)} uses — manage profiles on the Setup tab`
3900
+ : `no model on ${scenarioLabel(doc, i)}'s Setup profile — manage profiles on the Setup tab`);
3901
+ }
3902
+ });
3903
+ STEP_TYPES.tests .validate(doc.tests, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
3904
+ if (bad.length) return bad;
3905
+
3906
+ // Asked before anything is sent, so a misspelt token costs nothing and
3907
+ // says which job, instead of failing every item the same way.
3908
+ const text = CONTENT_TYPES[content?.type]?.text?.(content, c) ?? null;
3909
+ scenarios.forEach((sc, i) => {
3910
+ const stages = doc.jobs.map((ch , k ) => ({ text: sc.stages[k].prompt, kind: outOf(ch).kind,
3911
+ verbatim: !!callEntryOf(ch)?.verbatim }));
3912
+ const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) => callOf(ch).tokenMappings), text);
3913
+ if (problem) bad.push(`${scenarioLabel(doc, i)}: ${jobWords(problem)}`);
3914
+ });
3915
+ return bad;
3916
+ }
3917
+
3918
+ /** A run document's profiles table: id → connection, keyless, spellable. */
3919
+ function profilesProblems(profiles , bad ) {
3920
+ if (!isObj(profiles)) return void bad.push("profiles has to be an object of id → connection");
3921
+ const spelt = new Map ();
3922
+ for (const [id, conn] of Object.entries(profiles)) {
3923
+ if (!PROFILE_ID.test(id)) {
3924
+ bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVAL_API_KEY_<ID>`);
3925
+ continue;
3926
+ }
3927
+ const from = keyVar(id);
3928
+ if (spelt.has(from)) bad.push(`profiles ${spelt.get(from)} and ${id} both take their key from $${from}`);
3929
+ spelt.set(from, id);
3930
+ const at = `profile ${id}`;
3931
+ if (!isObj(conn)) { bad.push(`${at} has to be an object`); continue; }
3932
+ connectionProblems(conn, at, from, bad);
3933
+ }
3934
+ }
3935
+
3936
+ /** What is wrong with one connection of a run, as sentences naming it [at]. */
3937
+ function connectionProblems(conn , at , from , bad ) {
3938
+ onlyFields(conn, at, CONNECTION_FIELDS, bad, from);
3939
+ const type = conn.type ?? profileType(conn);
3940
+ if (!isStr(conn.type) || !CONNECTION_TYPES[type]) {
3941
+ bad.push(`${at}: type has to be one of ${Object.keys(CONNECTION_TYPES).join(", ")}`);
3942
+ }
3943
+ if (!isStr(conn.name)) bad.push(`${at}: name has to be text`);
3944
+ const url = conn.url ?? "";
3945
+ if (!isStr(url)) bad.push(`${at}: url has to be text, blank for the lab's Ollama`);
3946
+ else if (url.trim()) {
3947
+ let u = null;
3948
+ try { u = new URL(/:\/\//.test(url) ? url : "http://" + url); }
3949
+ catch { bad.push(`${at}: ${url} cannot be read as an address`); }
3950
+ // A key in an address is written into every log line and error message
3951
+ // that prints it.
3952
+ if (u && (u.username || u.password)) {
3953
+ bad.push(`${at} has a key in its address, and a key never goes in a pipeline — $${from} supplies it`);
3954
+ }
3955
+ }
3956
+ if (!isStr(conn.model)) bad.push(`${at}: model has to be text`);
3957
+ if (conn.temperature != null && !isStr(conn.temperature) && typeof conn.temperature !== "number") {
3958
+ bad.push(`${at}: temperature has to be a number`);
3959
+ }
3960
+ if (conn.px != null && !(conn.px === EDGE_448
3961
+ || (typeof conn.px === "number" ? conn.px > 0 : isStr(conn.px) && conn.px.trim() && Number(conn.px) > 0))) {
3962
+ bad.push(`${at}: px has to be a pixel-area number, or "edge448"`);
3963
+ }
3964
+ if (conn.format != null && !IMAGE_FORMATS.includes(conn.format)) {
3965
+ bad.push(`${at}: format has to be one of ${IMAGE_FORMATS.join(", ")}`);
3966
+ }
3967
+ if (conn.quality != null && !(typeof conn.quality === "number"
3968
+ ? conn.quality > 0 && conn.quality <= 1 : isStr(conn.quality) && Number(conn.quality) > 0)) {
3969
+ bad.push(`${at}: quality has to be between 0 and 1`);
3970
+ }
3971
+ // A setting the type does not carry is not a setting it can send.
3972
+ if (conn.options != null) {
3973
+ if (!isObj(conn.options)) bad.push(`${at}: options has to be an object`);
3974
+ else {
3975
+ const keys = CONNECTION_TYPES[type] ? CONNECTION_TYPES[type].settings.map(s => s.key) : OPTION_FIELDS;
3976
+ onlyFields(conn.options, `${at}'s options`, keys, bad, from);
3977
+ for (const [k, val] of Object.entries(conn.options)) {
3978
+ if (!isStr(val) && typeof val !== "number") bad.push(`${at}: options.${k} has to be text or a number`);
3979
+ }
3980
+ }
3981
+ }
3982
+ // What a request of this type needs before it can be built.
3983
+ if (CONNECTION_TYPES[type]?.url && !String(url || "").trim()) {
3984
+ bad.push(`${at}: ${CONNECTION_TYPES[type].label} needs its address, ${CONNECTION_TYPES[type].url}`);
3985
+ }
3986
+ if (CONNECTION_TYPES[type]?.refusesHosted && isHostedUrl(url)) {
3987
+ bad.push(`${at}: llama.cpp requires its llama-server, not a hosted model`);
3988
+ }
3989
+ for (const why of CONNECTION_TYPES[type]?.settingsProblems?.(connectionSettings(conn)) ?? []) bad.push(`${at} ${why}`);
3990
+ }
3991
+
3992
+ /** The profile ids a pipeline references, scenarios first, in order. */
3993
+ function profileIds(doc ) {
3994
+ const ids = [];
3995
+ for (const sc of doc.scenarios || []) {
3996
+ for (const ref of [sc.profile, ...(sc.stages || []).map((c ) => c.profile)]) {
3997
+ if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
3998
+ }
3999
+ }
4000
+ // Then the ones a test asks (a grader), so a run carries them too.
4001
+ for (const t of testsOf(doc)) {
4002
+ for (const ref of TEST_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
4003
+ }
4004
+ return ids;
4005
+ }
4006
+
4007
+ /**
4008
+ * The run document: [doc] plus what its references resolved to, once, at
4009
+ * submit. `ctx.profiles(id)` answers with a stored profile, `ctx.files` is
4010
+ * the Source's file list as it reads now, `ctx.comment` is the run's own.
4011
+ */
4012
+ function resolvePipeline(doc ,
4013
+ ctx = {}) {
4014
+ const run = clone(doc) ;
4015
+ // What the lab supplies a test -- its grader -- before the profiles it
4016
+ // asks are carried.
4017
+ run.tests = (Array.isArray(run.tests) ? run.tests : []).map((t ) => TEST_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
4018
+ run.profiles = {};
4019
+ for (const id of profileIds(run)) {
4020
+ const p = ctx.profiles?.(id);
4021
+ if (p) run.profiles[id] = connectionOf(p );
4022
+ }
4023
+ const content = contentOf(run) ;
4024
+ if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4025
+ if (ctx.datasetVersion) {
4026
+ for (const t of run.tests || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4027
+ }
4028
+ if (isStr(ctx.comment)) run.comment = ctx.comment;
4029
+ return run;
4030
+ }
4031
+
4032
+ /** A run put back on the canvas: the run document with its additions dropped. */
4033
+ function pipelineOfRun(run , ctx = {}) {
4034
+ const doc = clone(upgradePipeline(run, ctx)) ;
4035
+ delete doc.profiles;
4036
+ delete doc.comment;
4037
+ delete doc.plugins;
4038
+ const content = contentOf(doc) ;
4039
+ if (content) { delete content.files; delete content.revs; }
4040
+ for (const t of doc.tests || []) if (isObj(t.dataset)) delete t.dataset.version;
4041
+ return doc;
4042
+ }
4043
+
4044
+ // ---- the pipeline's file form (#29) ------------------------------------------
4045
+ //
4046
+ // YAML is what a pipeline is written as and read from: a browser has no YAML
4047
+ // parser, and the server is stdlib-only Python, so the vendored js-yaml next
4048
+ // to this file does it -- its ES module build, which the page's bundle and
4049
+ // node import alike. Only these functions touch it.
4050
+
4051
+
4052
+ /** [doc] as YAML text, the pipeline's file form -- references only, so an
4053
+ export carries no key, no URL, no file list and no file bytes. */
4054
+ function pipelineToYaml(doc ) {
4055
+ return yaml.dump(doc, { lineWidth: -1, noRefs: true });
4056
+ }
4057
+
4058
+ /**
4059
+ * [doc] as this lab would hold it: every reference matched by id, then by
4060
+ * name (pipeline-model §5), so a pipeline written in another lab -- ids that
4061
+ * mean nothing here -- finds its Source and Setup profiles by the names they
4062
+ * were exported under. [ctx]'s lists are what the lab holds:
4063
+ * `{ profiles, sources, datasets }`, each `{ id, name }[]`. Returns
4064
+ * `{ doc, missing }`, where `missing` is one sentence per reference neither
4065
+ * id nor name matched, or `{ error }` when the document is refused: not a
4066
+ * version the lab reads, or not something the lab can edit and run as it
4067
+ * stands.
4068
+ */
4069
+ function importPipeline(input , ctx = {}) {
4070
+ // A file exported before version 2 still imports: it is upgraded, not refused.
4071
+ const doc = upgradePipeline(input, ctx);
4072
+ const v = versionProblem(doc);
4073
+ if (v) return { error: v };
4074
+ const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], datasets = ctx.datasets ?? [];
4075
+ const next = clone(doc) ;
4076
+ const missing = [], seen = new Set ();
4077
+ const lists = { profile: profiles, source: sources, dataset: datasets };
4078
+ const words = {
4079
+ profile: (ref ) => `Setup profile ${ref.name || ref.id} not found`,
4080
+ source: (ref ) => `Source ${ref.name || ref.id} not found`,
4081
+ dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
4082
+ };
4083
+ const remap = (ref , kind ) => {
4084
+ const hit = lists[kind].find(x => x.id === ref.id) || lists[kind].find(x => x.name === ref.name);
4085
+ if (hit) return { id: hit.id, name: hit.name };
4086
+ const sentence = words[kind](ref);
4087
+ if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
4088
+ return ref;
4089
+ };
4090
+ const content = contentOf(next) ;
4091
+ if (content?.type === "source") content.ref = remap(content.ref, "source");
4092
+ for (const sc of next.scenarios || []) {
4093
+ sc.profile = remap(sc.profile, "profile");
4094
+ for (const cell of sc.stages || []) {
4095
+ if (cell.profile) cell.profile = remap(cell.profile, "profile");
4096
+ }
4097
+ }
4098
+ for (const t of Array.isArray(next.tests) ? next.tests : []) {
4099
+ if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4100
+ }
4101
+ // A hand-written file needs no id of its own: mint the ones it lacks, and
4102
+ // keep the ones it carries, so export → import → export is the same
4103
+ // document and a re-import can recognise the same pipeline (#93).
4104
+ mintIds(next);
4105
+ // What a document may still lack on import is its references -- matched
4106
+ // above and listed as missing. Anything else the model says is wrong is a
4107
+ // document the lab cannot edit and run as it stands, so it is refused by
4108
+ // the first such sentence. Validated with no lookups, so a reference a
4109
+ // later step will satisfy is never the refusal.
4110
+ const bad = validatePipeline(next).filter(s => !/ not found$/.test(s));
4111
+ if (bad.length) return { error: bad[0] };
4112
+ return { doc: next, missing };
4113
+ }
4114
+
4115
+ /** Mint, in place, the ids a current-version document must have that it
4116
+ lacks: a hand-written YAML carries none, and one it carries is kept. */
4117
+ function mintIds(doc ) {
4118
+ if (!isStr(doc.id) || !doc.id) doc.id = newId();
4119
+ for (const ch of Array.isArray(doc.jobs) ? doc.jobs : []) {
4120
+ if (isObj(ch) && (!isStr(ch.id) || !ch.id)) ch.id = newId();
4121
+ }
4122
+ for (const sc of Array.isArray(doc.scenarios) ? doc.scenarios : []) {
4123
+ if (isObj(sc) && (!isStr(sc.id) || !sc.id)) sc.id = newId();
4124
+ }
4125
+ for (const t of Array.isArray(doc.tests) ? doc.tests : []) {
4126
+ if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4127
+ }
4128
+ return doc;
4129
+ }
4130
+
4131
+ /** [text] as a pipeline of this lab: YAML parsed, references remapped by id
4132
+ then by name. importPipeline's `{ doc, missing }`, or `{ error }` when the
4133
+ text is not YAML or the document is refused. */
4134
+ function yamlToPipeline(text , ctx ) {
4135
+ let doc ;
4136
+ try { doc = yaml.load(text); }
4137
+ catch { return { error: "this file is not YAML, so nothing was imported" }; }
4138
+ return importPipeline(doc, ctx);
4139
+ }
4140
+
4141
+ /**
4142
+ * Scenario [i] of a run document as runPipeline and a transport take it:
4143
+ * each stage's wording, kind, image flag and modifiers, the token set of
4144
+ * the job it sits in, and the connection it asks -- its own profile, or
4145
+ * its scenario's.
4146
+ */
4147
+ function stagesFor(run , i )
4148
+
4149
+ {
4150
+ const sc = run.scenarios[i] ;
4151
+ const stages = run.jobs.map((ch, k) => {
4152
+ const { kind, modifiers, ...settings } = outOf(ch);
4153
+ return { text: sc.stages[k] .prompt, kind, withImage: !!callOf(ch).withImage,
4154
+ verbatim: !!callEntryOf(ch)?.verbatim, modifiers: modifiers || [], settings };
4155
+ });
4156
+ const connections = run.jobs.map((ch, k) => {
4157
+ const id = sc.stages[k] .profile?.id ?? sc.profile.id;
4158
+ return { id, ...(run.profiles?.[id] || {}) };
4159
+ });
4160
+ // Each job's call step and this scenario's cell in it, for a transport
4161
+ // that builds its own request from them (an HTTP Request).
4162
+ return { stages, tokens: run.jobs.map(ch => callOf(ch).tokenMappings), connections,
4163
+ calls: run.jobs.map(ch => stepIn(ch, "call") ), cells: sc.stages };
4164
+ }
4165
+
4166
+ /** The Setup profile scenario [i] of a run ran under, as History shows it: keyless. */
4167
+ function scenarioProfile(run , i ) {
4168
+ const id = run.scenarios?.[i]?.profile?.id;
4169
+ const conn = id != null ? run.profiles?.[id] : null;
4170
+ return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
4171
+ }
4172
+
4173
+ /** The kind a run's result is read as: its last job's. */
4174
+ const lastKind = (doc ) => outOf(doc.jobs?.at(-1))?.kind ?? null;
4175
+
4176
+ /** A value of a modifier on a run's last job -- the case mode History prints. */
4177
+ const lastModifier = (doc , type ) =>
4178
+ (outOf(doc.jobs?.at(-1))?.modifiers || []).find(m => m.type === type) || null;
4179
+
4180
+ /** What a modifier is set to, in a few words, from its entry's own options:
4181
+ a choice's label, a number, "3 rules" -- or "on" for one with none. What
4182
+ History's column for its type and a failing row's task text print, so
4183
+ neither names a modifier of its own. */
4184
+ function modifierSummary(m ) {
4185
+ const entry = MODIFIERS[m.type];
4186
+ const said = (entry?.options || []).map(o => {
4187
+ const v = (m )[o.key];
4188
+ if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? String(v ?? "");
4189
+ if (o.type === "rules") return Array.isArray(v) ? `${v.length} ${v.length === 1 ? "rule" : "rules"}` : "";
4190
+ return v == null ? "" : String(v);
4191
+ }).filter(Boolean);
4192
+ return said.length ? said.join(", ") : "on";
4193
+ }
4194
+
4195
+ // ---- the tests, in order (docs/pipeline-model.md §3) ---------------------------
4196
+ // Tests only read a run: none changes what a later one sees, and none stops
4197
+ // the model being sent the next item. What order changes is Continue on
4198
+ // failure. A per-item test that fails and does not continue stops the tests
4199
+ // after it for that item alone -- they read Skipped there, and a whole-run
4200
+ // test after it pools the items it did not stop. A whole-run test settles
4201
+ // once every item is in, and one that fails then and does not continue
4202
+ // leaves every test after it Skipped.
4203
+
4204
+ const SKIPPED = Object.freeze({ skipped: true });
4205
+ const isSkipped = (s ) => isObj(s) && s.skipped === true;
4206
+ /** A reading that failed: a score that did not pass. Skipped is not one. */
4207
+ const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
4208
+
4209
+ /**
4210
+ * One reply's scores under a run's per-item tests, by test id, in order: a
4211
+ * test with no case to score leaves no entry, and one after a failure that
4212
+ * does not continue reads Skipped. Null where nothing was scored.
4213
+ */
4214
+ function itemScores(run , kase , res ) {
4215
+ if (!kase) return null;
4216
+ const out = {};
4217
+ let stopped = false;
4218
+ for (const t of testsOf(run)) {
4219
+ const type = TEST_TYPES[t.type];
4220
+ if (!type?.score) continue;
4221
+ if (stopped) { out[t.id] = SKIPPED; continue; }
4222
+ const s = type.score(t, kase, res);
4223
+ out[t.id] = s;
4224
+ if (!s.pass && !t.continueOnFailure) stopped = true;
4225
+ }
4226
+ return Object.keys(out).length ? out : null;
4227
+ }
4228
+
4229
+ /** What production replied to an item, for a run's metrics: its last job's
4230
+ call says, from the item's record, or nobody does. */
4231
+ function productionOf(run , record ) {
4232
+ const last = run.jobs.at(-1);
4233
+ const call = stepIn(last, "call");
4234
+ return record ? STEP_TYPES[(call )?.type ?? ""]?.production?.(call, record) ?? null : null;
4235
+ }
4236
+
4237
+ /**
4238
+ * itemScores for the runner, which can wait: a test that reads every item
4239
+ * (`read`: the Metrics, which may ask a grader) scores one with no case too.
4240
+ */
4241
+ async function itemScoresAsync(run , kase , res ,
4242
+ more = {}) {
4243
+ const out = {};
4244
+ let stopped = false;
4245
+ for (const t of testsOf(run)) {
4246
+ const type = TEST_TYPES[t.type];
4247
+ if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
4248
+ if (stopped) { out[t.id] = SKIPPED; continue; }
4249
+ // A test with nothing to read on this item leaves no entry, as a graded
4250
+ // test does on an item with no case.
4251
+ const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
4252
+ : type.score (t, kase , res);
4253
+ if (!s) continue;
4254
+ out[t.id] = s;
4255
+ if (!s.pass && !t.continueOnFailure) stopped = true;
4256
+ }
4257
+ return Object.keys(out).length ? out : null;
4258
+ }
4259
+
4260
+ /**
4261
+ * Every test's reading of scenario [i] of a run, in the run's order, from
4262
+ * the items so far. [settled] says every item is in: only then has a
4263
+ * whole-run test settled, so only then does its failure skip the tests
4264
+ * after it.
4265
+ */
4266
+ function scenarioTests(run , i , items ,
4267
+ settled = true) {
4268
+ const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
4269
+ // Which items a per-item test has stopped so far, for the tests after it.
4270
+ const stopped = cells.map(() => false);
4271
+ let skipRest = false;
4272
+ const tests = testsOf(run);
4273
+ return tests.map((t, j) => {
4274
+ const type = TEST_TYPES[t.type];
4275
+ const whole = isWholeRun(type, t);
4276
+ const base = { id: t.id, label: testLabel({ tests }, j), whole, skipped: skipRest,
4277
+ verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
4278
+ ran: 0, passed: 0, skippedItems: 0 };
4279
+ if (skipRest) {
4280
+ if (!whole) base.items = cells.map(c => (c?.res ? SKIPPED : null));
4281
+ return base;
4282
+ }
4283
+ if (whole) {
4284
+ const ress = cells.map((c, x) => (stopped[x] ? undefined : c?.res));
4285
+ base.verdict = type .verdict (t, ress, lastKind(run));
4286
+ if (settled && !base.verdict.pass && !t.continueOnFailure) skipRest = true;
4287
+ return base;
4288
+ }
4289
+ cells.forEach((c, x) => {
4290
+ const s = c?.scores?.[t.id] ?? null;
4291
+ const read = stopped[x] && c?.res ? SKIPPED : s;
4292
+ base.items[x] = read;
4293
+ if (!read) return;
4294
+ if (isSkipped(read)) { base.skippedItems++; return; }
4295
+ base.ran++;
4296
+ if (read.pass) base.passed++;
4297
+ else if (!t.continueOnFailure) stopped[x] = true;
4298
+ });
4299
+ return base;
4300
+ });
4301
+ }
4302
+
4303
+ /** A scenario's pass or fail over every test that read it: null where none
4304
+ has anything to say yet. */
4305
+ function scenarioPasses(outcomes ) {
4306
+ let said = false;
4307
+ for (const o of outcomes) {
4308
+ if (o.skipped) continue;
4309
+ if (o.whole) {
4310
+ if (!o.verdict?.ran) continue;
4311
+ said = true;
4312
+ if (!o.verdict.pass) return false;
4313
+ } else if (o.ran) {
4314
+ said = true;
4315
+ if (o.passed < o.ran) return false;
4316
+ }
4317
+ }
4318
+ return said ? true : null;
4319
+ }
4320
+
4321
+ /**
4322
+ * Every rule the committed graded set has to satisfy, as sentences.
4323
+ *
4324
+ * Shared rather than asserted in CI alone, because the lab now lets somebody
4325
+ * edit the set in the browser: rules that live only in `evals-check.js` are
4326
+ * rules you discover after committing, and the whole point of editing in the
4327
+ * page is not having to commit to find out. So the check calls this, the page
4328
+ * calls this, and an edit the page accepts is an edit CI accepts.
4329
+ *
4330
+ * [files] is what the lab can show and what it cannot -- `{ shown, hidden }`,
4331
+ * either as file names. Omitted, the coverage rules are skipped rather than
4332
+ * failed: `run-evals.js` grades whatever set it is given and has no business
4333
+ * asserting that a repository directory matches it.
4334
+ *
4335
+ * Returns sentences, not a boolean. A refusal that does not say which case is
4336
+ * wrong sends somebody back to a 2,000-line file to guess.
4337
+ */
4338
+ function validateEvals(ev , files ) {
4339
+ const bad = [];
4340
+ if (!ev || typeof ev !== "object" || Array.isArray(ev)) {
4341
+ return ["the graded set has to be a JSON object with a `cases` list"];
4342
+ }
4343
+ if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
4344
+ if (bad.length) return bad;
4345
+
4346
+ const graded = ev.cases || [];
4347
+ // Identity first: every rule below reports which case is at fault, so a
4348
+ // case with no usable id makes the rest of the report unreadable.
4349
+ const seen = new Map ();
4350
+ for (const c of graded) {
4351
+ {
4352
+ if (!c || typeof c !== "object" || Array.isArray(c)) {
4353
+ bad.push("cases holds something that is not a case");
4354
+ continue;
4355
+ }
4356
+ if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
4357
+ else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
4358
+ else seen.set(c.id, "cases");
4359
+ if (!caseFile(c).trim()) {
4360
+ bad.push(`${c.id || "a case"} names no file`);
4361
+ }
4362
+ for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
4363
+ if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
4364
+ }
4365
+ for (const key of ["minCount", "maxCount"]) {
4366
+ if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
4367
+ }
4368
+ if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
4369
+ // A case's own metrics, which a Metrics test adds to its own for this item.
4370
+ if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
4371
+ }
4372
+ }
4373
+ if (bad.length) return bad;
4374
+
4375
+ // A term the scorer cannot match is a term that can never be produced, so
4376
+ // the case is unpassable however well the model answers.
4377
+ const matchable = (term ) => termIn([String(term).trim()], term);
4378
+
4379
+ for (const c of graded) {
4380
+ if (c.todo) continue;
4381
+ const say = (m ) => bad.push(`${c.id}: ${m}`);
4382
+ const expect = c.expect || [], forbid = c.forbid || [], anyOf = c.anyOf || [];
4383
+
4384
+ if (c.discarded === true) {
4385
+ // What is discarded holds nothing, so there is nothing else to expect.
4386
+ if (expect.length || anyOf.length || forbid.length || c.minCount != null || c.maxCount != null) {
4387
+ say("expects its answer discarded, and states something the answer should hold too");
4388
+ }
4389
+ continue;
4390
+ }
4391
+ const metrics = Array.isArray(c.metrics) ? c.metrics : [];
4392
+ if (!expect.length && !anyOf.length && !metrics.length) say("graded but states nothing to expect");
4393
+ for (const t of expect) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
4394
+ for (const g of anyOf) {
4395
+ if (!Array.isArray(g)) { say("an anyOf group has to be a list of terms"); continue; }
4396
+ if (!g.length) say("an anyOf group is empty, so nothing can satisfy it");
4397
+ for (const t of g) if (!matchable(t)) say(`offers ${t}, which its own scorer cannot match`);
4398
+ }
4399
+ // An exception excuses only the forbidden terms inside it, so one that
4400
+ // holds none of them changes nothing and reads as if it did.
4401
+ for (const a of c.allow || []) {
4402
+ if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
4403
+ }
4404
+ // A term on both lists cannot be produced and cannot be withheld.
4405
+ for (const t of [...expect, ...anyOf.flat()]) {
4406
+ if (forbid.includes(t)) say(`${t} is both expected and forbidden`);
4407
+ }
4408
+ const { minCount: lo, maxCount: hi } = c;
4409
+ if (lo != null && hi != null && lo > hi) say(`minCount ${lo} is above maxCount ${hi}`);
4410
+ if (lo != null && lo < 1) say(`minCount ${lo} is not a bound`);
4411
+ // The mirror of the floor: a ceiling below one says no answer is
4412
+ // acceptable, which is an entry that can never pass rather than a
4413
+ // strict one.
4414
+ if (hi != null && hi < 1) say(`maxCount ${hi} leaves no answer that could pass`);
4415
+ // A case that names more distinct things than the reply may carry, in
4416
+ // the scorer's own counting of a thing (#486: an expect term is one, an
4417
+ // anyOf group is one), can never pass however well the model answers.
4418
+ const things = expect.length + anyOf.length;
4419
+ if (hi != null && things > hi) {
4420
+ say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
4421
+ }
4422
+ }
4423
+
4424
+ // One file, one set of expectations. A file here twice is two sets of
4425
+ // them, graded separately, and both would be listed.
4426
+ const where = new Map ();
4427
+ for (const c of graded) {
4428
+ const name = caseFile(c);
4429
+ const counted = where.get(name);
4430
+ if (counted) {
4431
+ bad.push(`${name} is graded twice — one file, `
4432
+ + `two sets of expectations. Grade it once.`);
4433
+ } else where.set(name, true);
4434
+ }
4435
+ const gradedAt = (name ) => where.has(name);
4436
+
4437
+ if (!files) return bad;
4438
+ const shown = files.shown || [], hidden = files.hidden || [];
4439
+ for (const file of where.keys()) {
4440
+ if (!shown.includes(file) && !hidden.includes(file)) {
4441
+ bad.push(`${file} is graded and is not in the Source`);
4442
+ }
4443
+ if (hidden.includes(file)) {
4444
+ bad.push(`${file} is graded, and the lab cannot display it `
4445
+ + `— the row would have no image and nothing to send`);
4446
+ }
4447
+ }
4448
+ // The other direction, which is issue #270: a file dropped into the
4449
+ // set and graded by nobody is one nobody discovers is ungraded.
4450
+ for (const f of shown) {
4451
+ if (!gradedAt(f)) {
4452
+ bad.push(`${f} is in the Source and this dataset does not grade it, `
4453
+ + `so the Evals tab does not list it`);
4454
+ }
4455
+ }
4456
+ return bad;
4457
+ }
4458
+
4459
+ /**
4460
+ * Warnings, not refusals: things a graded set may still be right about.
4461
+ *
4462
+ * The one it carries is the count [prompt] -- the dataset's -- asks for. A number
4463
+ * parsed out of the wording (`up to 8 …`) is a hint rather than a
4464
+ * rule -- a prompt without a number carries no hint, and silence follows --
4465
+ * so a case that asks for more things than the prompt requests is told,
4466
+ * never stopped.
4467
+ */
4468
+ function evalsWarnings(ev , prompt ) {
4469
+ if (!ev || typeof ev !== "object" || Array.isArray(ev)) return [];
4470
+ const n = (String(prompt ?? "").match(/up to (\d+) \w/i) || [])[1];
4471
+ if (!n) return [];
4472
+ const want = Number(n), warn = [];
4473
+ for (const c of ev.cases || []) {
4474
+ if (!c || c.todo) continue;
4475
+ const things = (c.expect || []).length + (c.anyOf || []).length;
4476
+ if (things > want) {
4477
+ warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
4478
+ }
4479
+ }
4480
+ return warn;
4481
+ }
4482
+
4483
+ /**
4484
+ * The graded set as the repository holds it, byte for byte.
4485
+ *
4486
+ * Two-space JSON with a trailing newline and its punctuation written as
4487
+ * characters rather than escapes, which is what the committed file is -- one
4488
+ * line of it disagreed and was normalised when this was written. Byte-identity
4489
+ * is the whole point: an export of a set you edited one case of has to produce
4490
+ * a `git diff` of that one case, or nobody will trust it enough to commit it.
4491
+ * `evals-check.js` asserts the round trip against the real file.
4492
+ */
4493
+ function evalsJson(ev ) {
4494
+ return JSON.stringify(canonicalCases(ev), null, 2) + "\n";
4495
+ }
4496
+
4497
+
4498
+ // ---- Loading -------------------------------------------------------------
4499
+ // The list kind registers its entries from its own file, handed what it
4500
+ // registers with rather than importing it, so the two modules are no cycle
4501
+ // and every caller that imports this has them without a line of its own.
4502
+ list({ registerKinds });
4503
+ registerMetrics({ registerKinds });
4504
+
4505
+ export {
4506
+ TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
4507
+ loopReplyError, preparedSize,
4508
+ termIn, forbiddenIn, scoreCase, gradedSetFrom, caseFile, canonicalCases, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4509
+ SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
4510
+ CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, callEntryOf, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4511
+ EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
4512
+ tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
4513
+ scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
4514
+ COUNT_OP_LABEL, LENGTH_OP_LABEL,
4515
+ PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, TEST_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4516
+ SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4517
+ registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
4518
+ connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
4519
+ CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
4520
+ contentOf, withContent, callOf, replyOf, outOf, withOut, withCall, slotEntry, SLOTS,
4521
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newTest, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
4522
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioTests, scenarioPasses, isSkipped, failedScore,
4523
+ testLabel, testsOf, testsDataset, isWholeRun, TEST_FIELDS,
4524
+ pipelineToYaml, importPipeline, yamlToPipeline,
4525
+ pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
4526
+ };