evals-lab 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/LICENSE.md +57 -0
  2. package/README.md +98 -0
  3. package/bin/evals-lab.js +196 -0
  4. package/lab/demo/CREDITS.md +92 -0
  5. package/lab/demo/datasets/demo-1.json +1839 -0
  6. package/lab/demo/datasets/demo-2.json +1683 -0
  7. package/lab/demo/manifest.json +12 -0
  8. package/lab/demo/pipelines/demo-1.json +104 -0
  9. package/lab/demo/pipelines/demo-2.json +104 -0
  10. package/lab/demo/sources/Demo 1/basketball-hoop.jpg +0 -0
  11. package/lab/demo/sources/Demo 1/blue-boardwalk.jpg +0 -0
  12. package/lab/demo/sources/Demo 1/busy-beach.jpg +0 -0
  13. package/lab/demo/sources/Demo 1/butcher-sign.jpg +0 -0
  14. package/lab/demo/sources/Demo 1/cactus-flower.jpg +0 -0
  15. package/lab/demo/sources/Demo 1/corner-shop.jpg +0 -0
  16. package/lab/demo/sources/Demo 1/crosswalk-cyclist-panorama.jpg +0 -0
  17. package/lab/demo/sources/Demo 1/cyanotype-room.jpg +0 -0
  18. package/lab/demo/sources/Demo 1/dead-end-sign.jpg +0 -0
  19. package/lab/demo/sources/Demo 1/dog-in-snow.jpg +0 -0
  20. package/lab/demo/sources/Demo 1/empty-bedroom.jpg +0 -0
  21. package/lab/demo/sources/Demo 1/farmers-market-stall.jpg +0 -0
  22. package/lab/demo/sources/Demo 1/four-jets.jpg +0 -0
  23. package/lab/demo/sources/Demo 1/german-shepherd.jpg +0 -0
  24. package/lab/demo/sources/Demo 1/harbour-bridge-dusk.jpg +0 -0
  25. package/lab/demo/sources/Demo 1/harrow-on-the-hill-sign.jpg +0 -0
  26. package/lab/demo/sources/Demo 1/hong-kong-street.jpg +0 -0
  27. package/lab/demo/sources/Demo 1/house-salisbury-street.jpg +0 -0
  28. package/lab/demo/sources/Demo 1/infrared-orchard.jpg +0 -0
  29. package/lab/demo/sources/Demo 1/keyboard-desk.jpg +0 -0
  30. package/lab/demo/sources/Demo 1/lighthouse-dunes.jpg +0 -0
  31. package/lab/demo/sources/Demo 1/mountain-lake.jpg +0 -0
  32. package/lab/demo/sources/Demo 1/museum-skeleton.jpg +0 -0
  33. package/lab/demo/sources/Demo 1/music-room.jpg +0 -0
  34. package/lab/demo/sources/Demo 1/old-red-car.jpg +0 -0
  35. package/lab/demo/sources/Demo 1/parked-car-plate.jpg +0 -0
  36. package/lab/demo/sources/Demo 1/phone-and-wallet.png +0 -0
  37. package/lab/demo/sources/Demo 1/phone-box.jpg +0 -0
  38. package/lab/demo/sources/Demo 1/pink-flamingo.jpg +0 -0
  39. package/lab/demo/sources/Demo 1/postcards.jpg +0 -0
  40. package/lab/demo/sources/Demo 1/red-roof-church.jpg +0 -0
  41. package/lab/demo/sources/Demo 1/roadside-mailboxes.jpg +0 -0
  42. package/lab/demo/sources/Demo 1/rodeo.jpg +0 -0
  43. package/lab/demo/sources/Demo 1/running-tap.tif +0 -0
  44. package/lab/demo/sources/Demo 1/snail-on-stem.jpg +0 -0
  45. package/lab/demo/sources/Demo 1/two-horses-field.jpg +0 -0
  46. package/lab/demo/sources/Demo 1/university-sign.jpg +0 -0
  47. package/lab/demo/sources/Demo 1/vegetable-crates.jpg +0 -0
  48. package/lab/demo/sources/Demo 1/watermarked-car.jpg +0 -0
  49. package/lab/demo/sources/Demo 1/watermarked-pills.jpg +0 -0
  50. package/lab/demo/sources/Demo 1/whiteboard-delegate.jpg +0 -0
  51. package/lab/demo/sources/Demo 1/whiteboard-germ-layers.jpg +0 -0
  52. package/lab/demo/sources/Demo 2/aerial-city.jpg +0 -0
  53. package/lab/demo/sources/Demo 2/alligator-pen.jpg +0 -0
  54. package/lab/demo/sources/Demo 2/arm-tattoo.jpg +0 -0
  55. package/lab/demo/sources/Demo 2/bed-and-plant.jpg +0 -0
  56. package/lab/demo/sources/Demo 2/bright-bedroom.jpg +0 -0
  57. package/lab/demo/sources/Demo 2/car-headlight.jpg +0 -0
  58. package/lab/demo/sources/Demo 2/child-in-surf.jpg +0 -0
  59. package/lab/demo/sources/Demo 2/city-highway.jpg +0 -0
  60. package/lab/demo/sources/Demo 2/corner-bakery.jpg +0 -0
  61. package/lab/demo/sources/Demo 2/cyclist-yellow-jacket.jpg +0 -0
  62. package/lab/demo/sources/Demo 2/excavator-street-signs.jpg +0 -0
  63. package/lab/demo/sources/Demo 2/globe-closeup.jpg +0 -0
  64. package/lab/demo/sources/Demo 2/harbour-village.jpg +0 -0
  65. package/lab/demo/sources/Demo 2/high-street-walkers.jpg +0 -0
  66. package/lab/demo/sources/Demo 2/hillside-rooftops.jpg +0 -0
  67. package/lab/demo/sources/Demo 2/house-at-night.jpg +0 -0
  68. package/lab/demo/sources/Demo 2/ivy-leaves.jpg +0 -0
  69. package/lab/demo/sources/Demo 2/man-with-alligator.jpg +0 -0
  70. package/lab/demo/sources/Demo 2/orange-mailbox-hedge.jpg +0 -0
  71. package/lab/demo/sources/Demo 2/pedal-boats.jpg +0 -0
  72. package/lab/demo/sources/Demo 2/pink-trees-vignette.jpg +0 -0
  73. package/lab/demo/sources/Demo 2/please-leave-quietly-sign.jpg +0 -0
  74. package/lab/demo/sources/Demo 2/railroad-crossing-sign.jpg +0 -0
  75. package/lab/demo/sources/Demo 2/roadworks-sign.jpg +0 -0
  76. package/lab/demo/sources/Demo 2/roundabout-sign.jpg +0 -0
  77. package/lab/demo/sources/Demo 2/sea-cave.jpg +0 -0
  78. package/lab/demo/sources/Demo 2/shadow-on-sand.jpg +0 -0
  79. package/lab/demo/sources/Demo 2/signpost-a404.jpg +0 -0
  80. package/lab/demo/sources/Demo 2/street-fruit-cart.jpg +0 -0
  81. package/lab/demo/sources/Demo 2/street-sign-parnassusweg.jpg +0 -0
  82. package/lab/demo/sources/Demo 2/two-horses-close.jpg +0 -0
  83. package/lab/demo/sources/Demo 2/victorian-house.jpg +0 -0
  84. package/lab/demo/sources/Demo 2/vintage-dashboard.jpg +0 -0
  85. package/lab/demo/sources/Demo 2/volcano-at-dusk.jpg +0 -0
  86. package/lab/demo/sources/Demo 2/wall-camera.jpg +0 -0
  87. package/lab/demo/sources/Demo 2/watch-for-rocks-sign.jpg +0 -0
  88. package/lab/demo/sources/Demo 2/waterfall.jpg +0 -0
  89. package/lab/demo/sources/Demo 2/white-domes.jpg +0 -0
  90. package/lab/demo/sources/Demo 2/whiteboard-meeting-notes.jpg +0 -0
  91. package/lab/demo/sources/Demo 2/whiteboard-messages.jpg +0 -0
  92. package/lab/demo/sources/Demo 2/wooden-house-fence.jpg +0 -0
  93. package/lab/evals-core.mjs +4526 -0
  94. package/lab/flows/wdl.mjs +373 -0
  95. package/lab/js-yaml.mjs +3851 -0
  96. package/lab/kinds/list.mjs +411 -0
  97. package/lab/metrics/builtin.mjs +353 -0
  98. package/lab/presets.json +62 -0
  99. package/lab/run-evals.js +1262 -0
  100. package/lab/server.py +5044 -0
  101. package/lab/web/dist/assets/dist-DI3ewZYj.js +1 -0
  102. package/lab/web/dist/assets/gallery-DipkRvqJ.js +3 -0
  103. package/lab/web/dist/assets/gallery-o7c4lfpn.css +1 -0
  104. package/lab/web/dist/assets/main-CpVssvrb.js +18 -0
  105. package/lab/web/dist/assets/main-DjQQums6.css +1 -0
  106. package/lab/web/dist/assets/tokens-B9intIuT.js +51 -0
  107. package/lab/web/dist/assets/tokens-s6I-RMVq.css +1 -0
  108. package/lab/web/dist/gallery.html +18 -0
  109. package/lab/web/dist/index.html +23 -0
  110. package/package.json +20 -0
@@ -0,0 +1,1262 @@
1
+ #!/usr/bin/env node
2
+ // Runs the graded set headlessly. The second of the three callers of
3
+ // evals-core.ts -- the Prompt Lab tab is the first, CI is the third.
4
+ //
5
+ // An agent is a first-class caller here, not an afterthought, so this says
6
+ // what it did twice: a short human report, and a JSON one with every verdict
7
+ // in it. The exit code is the third statement and the coarsest:
8
+ //
9
+ // 0 every graded case passed
10
+ // 1 the set ran and at least one case failed
11
+ // 2 the set did not run in full -- nothing is graded, or an image or a
12
+ // recorded reply was missing. NOT a pass. An eval set that scores well
13
+ // because it never ran is the failure this whole ticket is about.
14
+ // 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
15
+ //
16
+ // The gating policy -- which of those a pull request may merge on -- is #250's
17
+ // to decide, and it has the JSON to decide it from. This only reports.
18
+ //
19
+ // Ollama is firewalled to a handful of hosts, so a live run has to happen on
20
+ // one of them. That is a fact about the network and not something a flag here
21
+ // can arrange; `--replies` is the mode that needs no model at all.
22
+ //
23
+ // The server-side-runs modes (#525, design §4 and §11.4) are here rather
24
+ // than in a second executor, because the executor *is* the CLI: a run the
25
+ // server queues and a run a terminal starts have to be the same thing or the
26
+ // parity check measures nothing.
27
+ //
28
+ // --source/--files the files come out of a Source: the snapshot's file
29
+ // list, read off the directory the orchestrator handed
30
+ // over. A case whose image missed the snapshot is
31
+ // unrun, not guessed around -- a report that cannot be
32
+ // reproduced is not a report.
33
+ // --progress JSON lines, one item per line, on a stream of their
34
+ // own: stdout, unless the final report is on stdout,
35
+ // in which case stderr -- so one cannot swallow the
36
+ // other. With --progress the human report moves to
37
+ // stderr, keeping stdout pure lines for the tail.
38
+ // --timeout a per-request cap, like the relay's, so a mute model
39
+ // becomes a failed item rather than a frozen queue.
40
+ // --resume continue a run from its first unfinished item: the
41
+ // cases the earlier report scored are carried whole and
42
+ // the rest are run. Verdicts are never recomputed.
43
+ // --item one case against the run's snapshot.
44
+ // --rescore a run's stored results, re-scored against the dataset
45
+ // it names with --dataset: the replies are kept -- they
46
+ // happened, and asking again is a different question --
47
+ // and only the verdicts are recomputed. No model, no
48
+ // files. A run whose target results are for text is the
49
+ // one item a --run names; this mode needs the snapshot
50
+ // (`--run`), the stored results (`--results-file`) and,
51
+ // for a graded run, the dataset (`--dataset`), and takes
52
+ // `--only <n>` for one item.
53
+ //
54
+ // A dataset is never read from the lab's own directories. The queue hands the
55
+ // worker the body a run was submitted with (--dataset), so a run grades
56
+ // against exactly that, whatever the dataset holds by the time it runs.
57
+ //
58
+ // What a file IS to a run is decided by the type→handler registry, not the
59
+ // queue: text joins at {text}, images go through the pixel budget, and a
60
+ // file with no handler runs anyway, flagged. See HANDLERS below.
61
+ const fs = require("fs");
62
+ const path = require("path");
63
+ const { execFileSync } = require("child_process");
64
+ const core = require("./evals-core.mjs");
65
+
66
+ const HERE = __dirname;
67
+ const ROOT = path.join(HERE, "..", "..");
68
+
69
+ // Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
70
+ // into 448px tiles and the tile COUNT is what costs. The same budget the tab
71
+ // offers by default and the container bakes its copies to.
72
+ const PIXELS = 921600;
73
+
74
+ // The connection a caller that names no address means, exactly as in
75
+ // server.py -- Ollama on this machine.
76
+ const OLLAMA = process.env.OLLAMA_URL || "http://127.0.0.1:11434";
77
+
78
+ const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
79
+
80
+ // The relay's own per-request cap (server.py `_proxy`). A request that
81
+ // outlives it is a failed item rather than a model that mutes and holds a
82
+ // single-flight queue for good; `--timeout` can shorten it, never lengthen.
83
+ const REQUEST_CAP = 600;
84
+
85
+ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
86
+
87
+ --model <id> the model to grade. Required for a live run.
88
+ --url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
89
+ Ollama on this machine. A key comes from $EVAL_API_KEY, never argv.
90
+ --prompt <text> the prompt to grade, as a template: its tokens resolve
91
+ under --tokens. Required without --pipeline: a dataset
92
+ holds no prompt (the lab's Prompt library does).
93
+ --prompt-file <p> the same, read from a file.
94
+ --plugins <dir> where the lab keeps its plugins: <dir>/<id>/<version>/,
95
+ each with its manifest.json and entry. A --run loads
96
+ the plugins it recorded; anything else loads every one
97
+ there. Each registers into the core before the run is
98
+ read, as the page's plugins do (docs/packs.md).
99
+ --tokens <file> the token mappings that prompt is resolved with, as
100
+ JSON: a list of { name, type, value | enabled }, as a
101
+ job's tokenMappings. Default: none, so a prompt
102
+ naming a token is refused, naming it.
103
+ --pipeline <file> a run document (docs/pipeline-model.md) with one
104
+ scenario and a graded test, graded case by case against
105
+ the set. Instead of --prompt, --model and --url. Each
106
+ profile's key comes from $EVAL_API_KEY_<ID>, never the
107
+ file; its content.files, when not empty, is the file
108
+ list a case has to be in, as --files says.
109
+ --samples <dir> where the images are. Default: the lab's samples/.
110
+ --dataset <file> the dataset a graded run is scored against: its body
111
+ ({"cases"}), or the file the lab's Export writes.
112
+ Versions 1 to 3 are read too (a prompt they hold is
113
+ not used: --prompt names it); a version-2
114
+ body's rules are what an older run's jobs, and a run
115
+ without --pipeline, read replies under. Its cases grade.
116
+ Required for anything graded.
117
+ --source <dir> the same, named the way a run names it: the directory a
118
+ Source is stored at. --samples and --source are one
119
+ flag by two names; both together are refused.
120
+ --files <file> the run's file-list snapshot, as JSON: a list of names,
121
+ or {"files": [...]}. Only files it names are read from
122
+ the Source; a case whose file missed the list is unrun.
123
+ --replies <file> score recorded replies instead of calling a model:
124
+ {"<case id>": "<reply>"}, or for a pipeline one reply per
125
+ stage: {"<case id>": ["<stage 1>", "<stage 2>"]}. No
126
+ network, no images.
127
+ --item <id> run one case from the set, nothing else.
128
+ --resume <file> a report from an earlier attempt: the cases it scored
129
+ are carried whole, the rest run from the first
130
+ unfinished one.
131
+ --timeout <s> cap each request at this many seconds, at most 600 --
132
+ the relay's cap, and a longer value cannot extend it.
133
+ Default: 600.
134
+ --progress write one JSON line per finished item, so a caller
135
+ can tail a run in progress. On stdout unless the
136
+ report is on stdout too, in which case stderr; and
137
+ then the human report moves to stderr.
138
+ --cancel-file <p> a path whose existence is checked between items: the run
139
+ stops after the item in flight, never mid-reply.
140
+ --run <file> a server run: the run document the queue hands the
141
+ worker, every scenario over every item of its content;
142
+ the files themselves live in --source. Replaces
143
+ --pipeline/--model/--replies. Keys as --pipeline.
144
+ --progress-file <p> JSON-lines progress for a --run, one line per item as it
145
+ finishes, on a stream of its own -- never the stream the
146
+ final report takes, so one cannot swallow the other.
147
+ --results-file <f> a --run's results so far (resume and re-run item).
148
+ --from <n> a --run starts at item n (resume), keeping earlier results.
149
+ --only <n> a --run runs item n alone (re-run item), replacing it.
150
+ --rescore re-score a run's stored results against --dataset.
151
+ Needs --run and --results-file, and --dataset for a
152
+ graded run; an optional --only <n> re-scores item n
153
+ alone. No model, no files, no network.
154
+ --json <path|-> write the machine-readable report. "-" means stdout, and
155
+ sends the human report to stderr.
156
+ `;
157
+
158
+ // NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
159
+ // pipe, and exiting discards whatever has not drained: measured, a report of
160
+ // 78,872 bytes came out of `--json - | ...` cut to 65,431 -- the pipe buffer
161
+ // -- and unparseable, while the same run redirected to a file was whole. An
162
+ // agent piping the machine-readable report is the caller this script was
163
+ // written for, so the code is set and the process is left to finish writing
164
+ // and leave on its own.
165
+ //
166
+ // Whether the loss shows up depends on how fast the reader drains -- the same
167
+ // report piped to `wc -c` arrived whole -- so it is not a fault a run can be
168
+ // relied on to reveal. `runner-check.js` asserts the absence of the call
169
+ // instead, and that is why.
170
+ //
171
+ // The price is that a live run sits for a few seconds after printing its
172
+ // report: undici keeps a socket to the endpoint alive and that holds the event
173
+ // loop open. Measured at 6.3s against an endpoint still listening and 0.3s
174
+ // once it closes; a --replies run makes no request and leaves at once. It is a
175
+ // wait and not a hang, the exit code is already set by then, and a run that
176
+ // spent minutes on the images will not notice. Do not "fix" it with a
177
+ // process.exit() -- that is the truncation above, straight back.
178
+ //
179
+ // Which means stopping early has to be a throw rather than an exit, since a
180
+ // `broken()` that returned would let its caller carry on with the argument it
181
+ // had just refused.
182
+ class Stop extends Error {
183
+ constructor(message, code) { super(message); this.code = code; }
184
+ }
185
+
186
+ function broken(msg) {
187
+ throw new Stop(msg, EXIT.broken);
188
+ }
189
+
190
+ function parseArgs(argv) {
191
+ const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
192
+ replies: "", json: "", source: "", files: "", item: "", resume: "",
193
+ timeout: null, progress: false, cancelFile: "",
194
+ run: "", progressFile: "", resultsFile: "", from: null, only: null,
195
+ rescore: false, tokens: "", plugins: "" };
196
+ for (let i = 0; i < argv.length; i++) {
197
+ const a = argv[i];
198
+ const value = () => {
199
+ const v = argv[++i];
200
+ if (v === undefined) broken(`${a} needs a value\n\n${USAGE}`);
201
+ return v;
202
+ };
203
+ switch (a) {
204
+ case "--model": o.model = value(); break;
205
+ case "--url": o.url = value(); break;
206
+ case "--prompt": o.prompt = value(); break;
207
+ case "--prompt-file": o.prompt = fs.readFileSync(value(), "utf8").trim(); break;
208
+ case "--tokens": o.tokens = value(); break;
209
+ case "--plugins": o.plugins = value(); break;
210
+ case "--pipeline": o.pipeline = value(); break;
211
+ case "--samples": o.samples = value(); break;
212
+ case "--dataset": o.dataset = value(); break;
213
+ case "--replies": o.replies = value(); break;
214
+ case "--source": o.source = value(); break;
215
+ case "--files": o.files = value(); break;
216
+ case "--item": o.item = value(); break;
217
+ case "--resume": o.resume = value(); break;
218
+ case "--timeout": o.timeout = value(); break;
219
+ case "--progress": o.progress = true; break;
220
+ case "--cancel-file": o.cancelFile = value(); break;
221
+ case "--run": o.run = value(); break;
222
+ case "--progress-file": o.progressFile = value(); break;
223
+ case "--results-file": o.resultsFile = value(); break;
224
+ case "--from": o.from = value(); break;
225
+ case "--only": o.only = value(); break;
226
+ case "--rescore": o.rescore = true; break;
227
+ case "--json": o.json = value(); break;
228
+ case "-h": case "--help":
229
+ process.stdout.write(USAGE);
230
+ throw new Stop("", EXIT.passed);
231
+ default: broken(`unknown option ${a}\n\n${USAGE}`);
232
+ }
233
+ }
234
+ return o;
235
+ }
236
+
237
+ // api_base in server.py, in JavaScript. A bare host and port, a base URL, a
238
+ // /v1 and the full chat path are all the same endpoint, and all four are
239
+ // reasonable things to type. The tab reaches a model through the relay, which
240
+ // does this for it; this script does not, so it does it here -- one copy in
241
+ // the core, shared with the page's Test connection.
242
+ const apiBase = core.apiBase;
243
+
244
+ const firstFile = (...paths) => paths.find(p => fs.existsSync(p)) || "";
245
+
246
+ // Repo-relative where it is in the repo, and as given where it is not: a
247
+ // report that names the set as ../../../../tmp/x.json is a report nobody can
248
+ // tell was run against the committed one.
249
+ const inRepo = p => {
250
+ const rel = path.relative(ROOT, p);
251
+ return rel && !rel.startsWith("..") ? rel : p;
252
+ };
253
+
254
+ // ---- A run document -------------------------------------------------------
255
+ // What the queue hands the worker and what CI commits: a pipeline resolved
256
+ // (docs/pipeline-model.md §5). evals-core.ts validates it -- every field on a
257
+ // list, a key refused by name -- so this script, the page and #29's import
258
+ // refuse the same file with the same sentence.
259
+ const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
260
+ const isText = v => typeof v === "string";
261
+
262
+ // [dataset] is the body --dataset handed over, or null: the one dataset a
263
+ // graded run can name, whatever id it names it by, since the queue hands the
264
+ // worker the body of exactly the dataset the run was submitted against.
265
+ function readRun(file, dataset) {
266
+ let doc;
267
+ try {
268
+ // A run queued before the current version is read as one of today's:
269
+ // resuming or re-running an old row works on the document it was
270
+ // submitted with. A v2 run's list jobs read their replies under the
271
+ // rules of the dataset it was graded against -- the body handed over.
272
+ doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
273
+ { rulesFor: () => rulesOf(dataset) });
274
+ } catch (e) {
275
+ broken(`${file}: ${e.message}`);
276
+ }
277
+ const named = isObject(doc) ? core.testsDataset(doc) : null;
278
+ const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
279
+ if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
280
+ return doc;
281
+ }
282
+
283
+ // The rules a version-2 body held, by the body readDataset made of it: what
284
+ // an older run's jobs read replies under. A body of today's holds none.
285
+ const heldRules = new WeakMap();
286
+ const rulesOf = (dataset) => (dataset && heldRules.get(dataset)) ?? null;
287
+
288
+ // The dataset --dataset names: a dataset's body, or the file the lab's
289
+ // Export writes for one, as one of today's.
290
+ function readDataset(file) {
291
+ if (!file) return null;
292
+ let doc;
293
+ try {
294
+ doc = JSON.parse(fs.readFileSync(file, "utf8"));
295
+ } catch (e) {
296
+ broken(`${file}: ${e.message}`);
297
+ }
298
+ if (isObject(doc) && doc.format === "evals-lab/dataset") {
299
+ if (![1, 2, 3, 4].includes(doc.version)) {
300
+ broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 4`);
301
+ }
302
+ doc = isObject(doc.dataset) ? doc.dataset.body : null;
303
+ }
304
+ if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
305
+ broken(`${file}: rules has to be null or { "rules": [...] }`);
306
+ }
307
+ // A body from an earlier version -- a run queued then, resumed now -- reads
308
+ // as one of today's, its rules taken first.
309
+ const rules = core.datasetRules(doc);
310
+ doc = core.upgradeDatasetBody(doc);
311
+ if (!isObject(doc) || !Array.isArray(doc.cases)) {
312
+ broken(`${file} is not a dataset: its body has a cases list, or it is the lab's Export of one`);
313
+ }
314
+ if (rules) heldRules.set(doc, rules);
315
+ return doc;
316
+ }
317
+
318
+ // The cases a graded run is scored against, by the file each names.
319
+ function gradedBy(dataset) {
320
+ const out = new Map();
321
+ for (const c of core.gradedSetFrom(dataset || {})) {
322
+ if (!c.todo) out.set(c.filename, c);
323
+ }
324
+ return out;
325
+ }
326
+
327
+ // ---- Preparing an image ----------------------------------------------
328
+ // The tab does this on a canvas, which is the one thing in the run that cannot
329
+ // be shared: there is no canvas in Node and adding one would mean a native
330
+ // dependency in a lab whose whole premise is that it has none.
331
+ //
332
+ // So the arithmetic is shared instead, by being the same arithmetic:
333
+ // core.preparedSize is the tab's own capping -- including the 448-longest-edge
334
+ // mode an area cannot express -- and the conversion that follows matches the
335
+ // canvas's toDataURL.
336
+ //
337
+ // The stage's profile says how far an image is capped and by which
338
+ // encoding; absent, the 720p JPEG quality 92 the container bakes its copies
339
+ // to, which is what this stage always ran at.
340
+ function imageMagick() {
341
+ // Not `convert` on Windows: that name is the system's own disk tool there.
342
+ for (const bin of process.platform === "win32" ? ["magick"] : ["magick", "convert"]) {
343
+ try {
344
+ execFileSync(bin, ["-version"], { stdio: "ignore" });
345
+ return bin;
346
+ } catch { /* not installed under that name */ }
347
+ }
348
+ return "";
349
+ }
350
+
351
+ function prepare(bin, file, c) {
352
+ const out = execFileSync(bin, [file, "-auto-orient", "-format", "%w %h", "info:"],
353
+ { maxBuffer: 1 << 20 }).toString().trim();
354
+ let [w, h] = out.split(/\s+/).map(Number);
355
+ if (!(w > 0 && h > 0)) throw new Error(`${path.basename(file)}: no dimensions`);
356
+ // What the connection says, from the one place that reads its fields.
357
+ const image = core.connectionImage(c);
358
+ [w, h] = core.preparedSize(w, h, image.px ?? PIXELS);
359
+ const fmt = { "image/jpeg": ["jpg", "image/jpeg"], "image/png": ["png", "image/png"],
360
+ "image/webp": ["webp", "image/webp"] }[image.format];
361
+ const quality = String(core.encoderQuality(image.quality));
362
+ const jpeg = execFileSync(
363
+ bin, [file, "-auto-orient", "-strip", "-resize", `${w}x${h}!`, "-quality", quality, `${fmt[0]}:-`],
364
+ { maxBuffer: 64 << 20 });
365
+ return { dataUrl: `data:${fmt[1]};base64,` + jpeg.toString("base64"), dims: `${w}x${h}` };
366
+ }
367
+
368
+ // ---- The type → handler registry ------------------------------------------
369
+ // Design §4. What a file *is* to a run is decided here and only here, so the
370
+ // queue and the page never learn about file types, and pdf and xlsx are
371
+ // later entries that touch nothing but this map.
372
+ //
373
+ // A file with no mapper is passed through anyway -- decoded as UTF-8 with
374
+ // replacement and joined at {text} like the text handler, and flagged in the
375
+ // transcript. Stating the failure mode beats hiding it: a PDF's text layer
376
+ // is not extracted, so its streams arrive as arbitrary bytes, and a run
377
+ // that read them "well" would be a finding about nothing.
378
+ const HANDLERS = new Map();
379
+ [
380
+ // Images: prepared to the profile's pixel budget by ImageMagick and sent
381
+ // as the image content part. The formats are server.py's FILE_TYPES
382
+ // image half, so nothing accepted for upload has no executor to run it.
383
+ ["image", "jpg", "jpeg", "png", "heic", "heif", "dng", "tif", "tiff"],
384
+ // Text-ish: decoded UTF-8, joining the stage at {text} -- the text-content
385
+ // semantics the canvas already has (core.textPrompt).
386
+ ["text", "txt", "md", "csv"],
387
+ ].forEach((row) => {
388
+ const [kind, ...exts] = row;
389
+ for (const ext of exts) HANDLERS.set(ext, kind);
390
+ });
391
+ // The extension of a file in the Source, lower-cased. Names arrive already
392
+ // clean from the upload path; this only classifies.
393
+ const extOf = name => (path.extname(String(name)) || "").slice(1).toLowerCase();
394
+
395
+ const UNMAPPED = "the file has no handler, so it was passed through as raw "
396
+ + "bytes decoded as UTF-8 with replacement -- whatever a "
397
+ + "real reader of this type would have extracted was not "
398
+ + "extracted, and the content is mojibake";
399
+
400
+ /**
401
+ * One file of the Source made into what the job consumes: an image as the
402
+ * dataUrl at [conn]'s budget, or text at {text}. A file with no mapper
403
+ * still runs, and says so.
404
+ */
405
+ function itemContent(bin, file, conn) {
406
+ const kind = HANDLERS.get(extOf(file));
407
+ if (kind === "image") {
408
+ const { dataUrl, dims } = prepare(bin, file, conn);
409
+ return { kind, dataUrl, text: null, flags: [] };
410
+ }
411
+ // TextDecoder with fatal unset replaces malformed bytes with U+FFFD --
412
+ // what "decoded as UTF-8 with replacement" means, for a mapped-text file
413
+ // and an unmapped one alike.
414
+ const text = new TextDecoder().decode(fs.readFileSync(file));
415
+ return { kind: kind ?? "unmapped", dataUrl: null, text, flags: kind ? [] : [UNMAPPED] };
416
+ }
417
+
418
+ // ---- Reaching a model ----------------------------------------------------
419
+
420
+ /**
421
+ * Where a connection goes and the key it takes, refused before anything is
422
+ * sent if the key cannot go there. [from] is the variable the key is read
423
+ * from, named in the report in place of the key.
424
+ */
425
+ function reach(c, from) {
426
+ const base = core.connectionBase({ ...c, url: c.url || OLLAMA });
427
+ const key = (process.env[from] || "").trim();
428
+ if (key) {
429
+ // server.py's header_safe: fetch rejects such a header with the value in
430
+ // its message, which would put the key in the report.
431
+ if (![...key].every(ch => ch >= " " && ch <= "~")) {
432
+ broken(`$${from} has characters a header cannot carry, so the key was not sent.`);
433
+ }
434
+ // server.py's rule, for server.py's reason: a key that bills a real card
435
+ // should not cross a network in the clear. Loopback never leaves the
436
+ // machine.
437
+ const host = new URL(base).hostname;
438
+ if (!base.startsWith("https://") && !["127.0.0.1", "[::1]", "localhost"].includes(host)) {
439
+ broken(`$${from} is set and ${base} is not https, so the key was not sent.`);
440
+ }
441
+ }
442
+ return { name: c.name, base, key, from };
443
+ }
444
+
445
+ // The page's callModel, less the relay: the same body, and on llama.cpp the
446
+ // same check for a reply that looped. A request that outlives the cap is
447
+ // aborted and comes back as the stage's error, so the queue it belongs to
448
+ // keeps moving instead of waiting on a model that mutes.
449
+ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
450
+
451
+ // Each stage's transport for one item. A type that answers from the item
452
+ // itself (Echo, #97) is asked nothing: its reply is the item's own text --
453
+ // or, over Prompt only, the prompt -- and the stage reads it as it would a
454
+ // model's.
455
+ //
456
+ // A call that asks for a whole request (an HTTP Request, #flows) builds it
457
+ // from its step, the scenario's cell and the item's record, and is handed
458
+ // only its words: the transport is where the three meet.
459
+ function callsFor(connections, links, text, plan, record) {
460
+ return connections.map((c, k) => {
461
+ const step = plan?.calls?.[k];
462
+ if (core.STEP_TYPES[step?.type]?.asks === "request") {
463
+ return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
464
+ }
465
+ return core.CONNECTION_TYPES[core.typeOf(c)]?.local
466
+ ? async sent => core.localAnswer(c, text, sent)
467
+ : (sent, url) => ask(links[k], c, sent, url);
468
+ });
469
+ }
470
+
471
+ /** A Setup profile the run carries, asked as a grader: its reply's text, or
472
+ the failure as an error the metric reports. One transport per profile. */
473
+ const graders = new WeakMap();
474
+ function graderFor(run) {
475
+ if (!graders.has(run)) {
476
+ const made = new Map();
477
+ graders.set(run, ref => {
478
+ const c = run.profiles?.[ref.id];
479
+ if (!c) return undefined;
480
+ if (!made.has(ref.id)) {
481
+ const link = reach(c, core.keyVar(ref.id));
482
+ made.set(ref.id, async prompt => {
483
+ const r = await ask(link, { id: ref.id, ...c }, prompt, null);
484
+ if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
485
+ return r.raw ?? "";
486
+ });
487
+ }
488
+ return made.get(ref.id);
489
+ });
490
+ }
491
+ return graders.get(run);
492
+ }
493
+
494
+ /** A text item as a record's scope: its text is the trigger's body. */
495
+ const textRecord = (text) => ({ scope: { trigger: { body: text } }, at: null });
496
+
497
+ /** An HTTP Request: the flow's request rebuilt from [record], sent to [to],
498
+ and the reply read as the flow reads it. */
499
+ async function send(to, c, step, cell, record) {
500
+ const t0 = Date.now();
501
+ const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
502
+ const ctx = { scope: record.scope || {}, now: record.at || null };
503
+ let req;
504
+ try {
505
+ req = core.connectionSend(c, core.httpRequestOf(step, cell, ctx), to.base, to.key);
506
+ } catch (e) {
507
+ return said({ error: `the request could not be built -- ${e.message}` });
508
+ }
509
+ // A body that is text -- an Any endpoint flow's -- goes as it is written;
510
+ // anything else as JSON, the way the flow sent it.
511
+ const isText = typeof req.body === "string";
512
+ const payload = req.method === "GET" ? null : isText ? req.body : JSON.stringify(req.body);
513
+ const sent = `${req.method || "POST"} ${req.url}${payload == null ? "" : `\n\n${payload}`}`;
514
+ const done = more => said({ sent, ...more });
515
+ try {
516
+ const r = await fetch(req.url, {
517
+ method: req.method || "POST",
518
+ headers: { "Content-Type": isText ? "text/plain; charset=utf-8" : "application/json", ...req.headers },
519
+ redirect: "error",
520
+ signal: AbortSignal.timeout(seconds * 1000),
521
+ ...(payload == null ? {} : { body: payload }),
522
+ });
523
+ const text = await r.text();
524
+ let j = text;
525
+ try { j = JSON.parse(text); } catch { /* the body as text */ }
526
+ if (!r.ok) return done({ error: j?.error?.message || (typeof j?.error === "string" ? j.error : null) || `HTTP ${r.status}` });
527
+ const read = core.httpReplyOf(step, cell, j, r.status, ctx);
528
+ return done({ raw: read.raw, said: read.said });
529
+ } catch (e) {
530
+ if (e?.name === "TimeoutError" || e?.name === "AbortError") {
531
+ return done({ error: `the endpoint gave no reply within ${seconds}s — timed out` });
532
+ }
533
+ return done({ error: String(e?.message || e) });
534
+ }
535
+ }
536
+
537
+ async function ask(to, c, prompt, dataUrl) {
538
+ const t0 = Date.now();
539
+ const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
540
+ try {
541
+ // The connection's type builds the whole request: the URL, the headers
542
+ // (Anthropic's key travels in x-api-key), and the body.
543
+ const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
544
+ const r = await fetch(req.url, {
545
+ method: "POST",
546
+ headers: { "Content-Type": "application/json", ...req.headers },
547
+ signal: AbortSignal.timeout(seconds * 1000),
548
+ body: JSON.stringify(req.body),
549
+ });
550
+ const j = await r.json().catch(() => ({}));
551
+ if (!r.ok) return said({ error: j.error?.message || j.error || `HTTP ${r.status}` });
552
+ const { raw, finishReason } = core.connectionReply(c, j);
553
+ const error = core.connectionReplyProblem(c, raw, finishReason);
554
+ return said(error ? { raw, error } : { raw });
555
+ } catch (e) {
556
+ if (e?.name === "TimeoutError" || e?.name === "AbortError") {
557
+ return said({ error: `the model gave no reply within ${seconds}s — timed out` });
558
+ }
559
+ return said({ error: String(e?.message || e) });
560
+ }
561
+ }
562
+
563
+ // ---- The run -------------------------------------------------------------
564
+
565
+ // ---- A server run: what the queue hands the worker (#531) ------------------
566
+ // The run document carries everything a run was submitted with -- the
567
+ // scenarios, the profiles they resolved to, and the content: a file list into
568
+ // --source, in order and with its repeats, or the inline text. Each item runs
569
+ // through every scenario's jobs as an image does on the canvas; a
570
+ // Source deleted between submit and here shows as a failed run, and a file
571
+ // missing from it is an incomplete item, never a pass.
572
+ //
573
+ // Progress is JSON lines on a stream of its own -- --progress-file, never the
574
+ // report -- so the queue can tail it into the run row as items finish. The
575
+ // report is written whole at the end, from the results the run held plus what
576
+ // this invocation did, so resume and re-run item agree with a fresh run.
577
+
578
+ // What a file *is* to a run, the same registry §4 gives the graded set: an
579
+ // image is prepared to the profile's pixel budget; a text file is decoded
580
+ // UTF-8 and joined at {text}; anything else is passed through as UTF-8 with
581
+ // replacement and flagged, so the failure mode is stated rather than hidden.
582
+ const RUN_IMAGE_TYPES = new Set([".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff"]);
583
+ const RUN_TEXT_TYPES = new Set([".txt", ".md", ".csv"]);
584
+
585
+ function readUtf8(file) {
586
+ return fs.readFileSync(file).toString("utf8");
587
+ }
588
+
589
+ // Every test's reading of each scenario, keyed by the test's id, exactly as
590
+ // the page reads it: a whole-run test's verdict, settled once every item is
591
+ // in, and a per-item test's counts. [settled] is false for a run that
592
+ // stopped short, whose whole-run tests have not settled.
593
+ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
594
+ Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
595
+ name: o.label, skipped: o.skipped,
596
+ ...(o.whole
597
+ ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
598
+ : { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
599
+ }])));
600
+
601
+ // The run as the report restates it: each scenario's stages with the
602
+ // connection each one asked.
603
+ const scenariosOf = run => run.scenarios.map((sc, i) => {
604
+ const { stages, connections } = core.stagesFor(run, i);
605
+ return {
606
+ n: i + 1, name: sc.name || null,
607
+ stages: stages.map((s, k) => {
608
+ const { id, ...connection } = connections[k];
609
+ return { n: k + 1, prompt: s.text, kind: s.kind, withImage: s.withImage, profile: id, connection };
610
+ }),
611
+ };
612
+ });
613
+
614
+ async function runSnapshot(o) {
615
+ const dataset = readDataset(o.dataset);
616
+ const run = readRun(o.run, dataset);
617
+ // One item a pipeline holds itself (Text, or Prompt only's bare one), or
618
+ // the Source's files.
619
+ const content = core.contentOf(run);
620
+ const inline = core.CONTENT_TYPES[content.type];
621
+ const bare = !!inline?.bare;
622
+ const text = inline?.inline && !bare ? inline.text?.(content, {}) ?? "" : "";
623
+ const files = content.type === "source" ? content.files || [] : [];
624
+ if (!bare && !files.length && !text.trim()) {
625
+ broken(`${o.run} names no files and no text, so there is nothing to run`);
626
+ }
627
+
628
+ // The dataset the run was submitted against, for scoring its items.
629
+ const graded = gradedBy(dataset);
630
+
631
+ const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
632
+ : text.trim() ? [{ name: null, kind: "text", text }]
633
+ : files.map(name => {
634
+ const ext = path.extname(name).toLowerCase();
635
+ if (RUN_IMAGE_TYPES.has(ext)) return { name, kind: "image" };
636
+ // A record (flows/record.ts): one call of a Power Automate step, what
637
+ // an HTTP Request is rebuilt from.
638
+ if (ext === ".json") return { name, kind: "record" };
639
+ return { name, kind: "text", unmapped: !RUN_TEXT_TYPES.has(ext) };
640
+ });
641
+ const total = items.length;
642
+ // Only the run's own image items need ImageMagick: a text-only Source runs
643
+ // with nothing on PATH but the node running this.
644
+ const bin = items.some(it => it.kind === "image") ? imageMagick() : null;
645
+ if (items.some(it => it.kind === "image") && !bin) {
646
+ broken("a run over images needs ImageMagick, to cap each image to the pixel budget the tab caps it to on a canvas.");
647
+ }
648
+
649
+ // Each scenario once: its stages, the jobs' token sets, and a transport
650
+ // per stage to the profile that stage resolves to, keyed by that
651
+ // profile's id.
652
+ const plans = run.scenarios.map((_, i) => {
653
+ const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
654
+ const links = connections.map(c => reach(c, core.keyVar(c.id)));
655
+ return { stages, tokens, connections, links, calls, cells };
656
+ });
657
+
658
+ let results = [];
659
+ if (o.resultsFile) {
660
+ try { results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8")); } catch { /* fresh run */ }
661
+ // Kept as they were written; read as today's.
662
+ results = Array.isArray(results) ? core.upgradeResults(results) : [];
663
+ }
664
+ const only = o.only != null ? parseInt(o.only, 10) : null;
665
+ const from = only != null ? only : (o.from != null ? parseInt(o.from, 10) : 0);
666
+
667
+ const progressFd = o.progressFile ? fs.openSync(o.progressFile, "a") : null;
668
+ const progress = obj => { if (progressFd != null) fs.writeSync(progressFd, JSON.stringify(obj) + "\n"); };
669
+
670
+ const unrun = [];
671
+ let cancelled = false;
672
+
673
+ for (let i = 0; i < total; i++) {
674
+ if (only == null && i < from) continue;
675
+ if (o.cancelFile && fs.existsSync(o.cancelFile)) { cancelled = true; break; }
676
+ const item = items[i];
677
+ // Said before the work, so a reader knows what is in flight rather than
678
+ // what last finished; the row below it is the same item, done.
679
+ progress({ event: "start", item: i, name: item.name });
680
+ const t0 = Date.now();
681
+ const sides = [];
682
+ let production = null;
683
+ let failed = null;
684
+ for (const plan of plans) {
685
+ let dataUrl = null;
686
+ if (item.kind === "image") {
687
+ const file = path.join(o.source, item.name);
688
+ if (!fs.existsSync(file)) { failed = `the file ${item.name} is not in the source`; break; }
689
+ try {
690
+ ({ dataUrl } = prepare(bin, file, plan.connections[0]));
691
+ } catch (e) {
692
+ failed = `${item.name} could not be prepared — ${e.message}`;
693
+ break;
694
+ }
695
+ }
696
+ let record = null;
697
+ if (item.kind === "record") {
698
+ const file = path.join(o.source, item.name);
699
+ try { record = JSON.parse(readUtf8(file)); } catch (e) {
700
+ failed = fs.existsSync(file) ? `${item.name} is not a record -- ${e.message}` : `the file ${item.name} is not in the source`;
701
+ break;
702
+ }
703
+ }
704
+ const textOf = item.kind === "text" && !item.bare
705
+ ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
706
+ const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
707
+ { tokens: plan.tokens, text: textOf });
708
+ // A case is found by its file's name, whatever kind of file it is: a
709
+ // text item a dataset grades is graded like an image.
710
+ const kase = item.name ? graded.get(item.name) ?? null : null;
711
+ production = core.productionOf(run, record);
712
+ sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run) }) });
713
+ }
714
+ // Production's reply to the item, where its record holds one: what a
715
+ // metric compares with, kept so a re-score reads it again.
716
+ const out = { item: i, name: item.name, kind: item.kind, ...(production != null ? { production } : {}),
717
+ unmapped: item.unmapped || null, ms: Date.now() - t0,
718
+ scenarios: sides, ...(failed ? { unrun: failed } : {}) };
719
+ results[i] = out;
720
+ progress(out);
721
+ if (failed) unrun.push(`${item.name || "the text"}: ${failed}`);
722
+ }
723
+ if (progressFd != null) fs.closeSync(progressFd);
724
+
725
+ const report = {
726
+ runner: "tools/prompt-lab/run-evals.js",
727
+ ranAt: new Date().toISOString(),
728
+ verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
729
+ run: {
730
+ name: run.name ?? null,
731
+ tests: run.tests,
732
+ verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
733
+ scenarios: scenariosOf(run),
734
+ items: results.slice(0, total),
735
+ },
736
+ unrun,
737
+ };
738
+
739
+ const say = o.json === "-" ? console.error : console.log;
740
+ say(`${report.runner} — run ${report.run.name ?? ""}`);
741
+ for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
742
+ if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
743
+ say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled" : report.verdict}`);
744
+
745
+ if (o.json) {
746
+ const text2 = JSON.stringify(report, null, 2);
747
+ if (o.json === "-") process.stdout.write(text2 + "\n");
748
+ else fs.writeFileSync(o.json, text2 + "\n");
749
+ }
750
+ process.exitCode = cancelled ? EXIT.passed : (unrun.length ? EXIT.incomplete : EXIT.passed);
751
+ }
752
+
753
+ /** Re-score a run's stored results against the dataset --dataset hands over,
754
+ * keeping the replies that were stored: what changes a verdict is the scorer
755
+ * or the cases, never a new question to the model. The report is shaped like
756
+ * a --run's, with the stored results as its items, and a whole-run test's
757
+ * verdict settled the same way. */
758
+ async function runRescore(o){
759
+ const dataset = readDataset(o.dataset);
760
+ const run = readRun(o.run, dataset);
761
+ let results;
762
+ try {
763
+ results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
764
+ } catch (e) {
765
+ broken(`${e.path ?? "a read"}: ${e.message}`);
766
+ }
767
+ if (!Array.isArray(results)) {
768
+ broken(`${o.resultsFile}: a run's results are a list of items`);
769
+ }
770
+ results = core.upgradeResults(results);
771
+
772
+ // The cases the run's items are scored against. A file the set does
773
+ // not grade keeps its reply and has no score -- honestly: nothing else would
774
+ // say what the set covers.
775
+ const graded = gradedBy(dataset);
776
+
777
+ const only = o.only != null ? parseInt(o.only, 10) : null;
778
+ // A test that reads every item (the Metrics) re-reads one with no case
779
+ // too, against the production reply the item kept.
780
+ const items = await Promise.all(results.map(async (it, i) => {
781
+ if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
782
+ const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
783
+ const more = { production: it.production ?? null, grader: graderFor(run) };
784
+ return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
785
+ (isObject(side) && side.res)
786
+ ? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
787
+ }));
788
+
789
+ const report = {
790
+ runner: "tools/prompt-lab/run-evals.js",
791
+ ranAt: new Date().toISOString(),
792
+ verdict: "done",
793
+ rescored: true,
794
+ run: {
795
+ name: run.name ?? null,
796
+ tests: run.tests,
797
+ verdicts: verdictsOf(run, items, true),
798
+ scenarios: scenariosOf(run),
799
+ items,
800
+ },
801
+ };
802
+
803
+ if (o.json) {
804
+ const text = JSON.stringify(report, null, 2);
805
+ if (o.json === "-") process.stdout.write(text + "\n");
806
+ else fs.writeFileSync(o.json, text + "\n");
807
+ }
808
+ process.exitCode = EXIT.passed;
809
+ }
810
+
811
+ /**
812
+ * The run --prompt, --model and --url describe, as a run document: one job
813
+ * that sees the image and answers in the default kind, one scenario,
814
+ * graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
815
+ * connection, as it never did, and its key is $EVAL_API_KEY.
816
+ */
817
+ function cliRun(o, dataset) {
818
+ let doc = core.blankPipeline();
819
+ // Graded, so the job answers with a list: the configuration that reads a
820
+ // reply as this command always has (legacyTagsOut), under the dataset's
821
+ // rules. A --pipeline states its own.
822
+ doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
823
+ if (o.tokens) {
824
+ try {
825
+ doc.jobs[0] = core.withCall(doc.jobs[0], { tokenMappings: JSON.parse(fs.readFileSync(o.tokens, "utf8")) });
826
+ } catch (e) {
827
+ broken(`${o.tokens}: ${e.message}`);
828
+ }
829
+ }
830
+ doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
831
+ doc.scenarios = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
832
+ stages: [{ prompt: o.prompt ?? "" }] }];
833
+ doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
834
+ mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
835
+ doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
836
+ return doc;
837
+ }
838
+
839
+ // The plugins a run is read under, registered before anything reads it: a
840
+ // --run's recorded list, at the versions it recorded, or every plugin in the
841
+ // folder. A plugin that will not load stops the run, naming it -- a run read
842
+ // without the code it was submitted under grades something else.
843
+ async function loadPlugins(o) {
844
+ if (!o.plugins) return;
845
+ const { pathToFileURL } = require("url");
846
+ let wanted = null;
847
+ if (o.run) {
848
+ try { wanted = JSON.parse(fs.readFileSync(o.run, "utf8")).plugins ?? []; } catch { wanted = []; }
849
+ }
850
+ const dirs = wanted
851
+ ? wanted.map(p => path.join(o.plugins, String(p.id), String(p.version)))
852
+ : fs.readdirSync(o.plugins, { withFileTypes: true }).filter(d => d.isDirectory() && !d.name.startsWith("tmp"))
853
+ .flatMap(d => fs.readdirSync(path.join(o.plugins, d.name)).map(v => path.join(o.plugins, d.name, v)));
854
+ for (const dir of dirs) {
855
+ let manifest;
856
+ try {
857
+ manifest = JSON.parse(fs.readFileSync(path.join(dir, "manifest.json"), "utf8"));
858
+ const mod = await import(pathToFileURL(path.join(dir, manifest.entry)).href);
859
+ await (mod.default ?? mod.register)(core.pluginHost(manifest.id));
860
+ } catch (e) {
861
+ broken(`the plugin in ${dir} did not load: ${e.message}`);
862
+ }
863
+ }
864
+ }
865
+
866
+ async function main() {
867
+ const o = parseArgs(process.argv.slice(2));
868
+ await loadPlugins(o);
869
+ if (o.rescore) {
870
+ if (!o.run || !o.resultsFile) {
871
+ broken("--rescore re-scores a stored run, so it needs --run and --results-file, "
872
+ + "and for a graded run the dataset it scores against, --dataset.");
873
+ }
874
+ if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies
875
+ || o.samples || o.source || o.files || o.resume || o.item || o.progress) {
876
+ broken("--rescore re-scores what the run already holds, so it cannot be given "
877
+ + "with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples, --source, "
878
+ + "--files, --resume, --item or --progress.");
879
+ }
880
+ return runRescore(o);
881
+ }
882
+ if (o.run) {
883
+ if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
884
+ broken("--run names everything a run takes -- scenarios, files and text -- so it "
885
+ + "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
886
+ }
887
+ const n = Number(o.timeout ?? REQUEST_CAP);
888
+ if (!(n > 0)) broken(`--timeout has to be a number of seconds above zero`);
889
+ seconds = Math.min(n, REQUEST_CAP);
890
+ return runSnapshot(o);
891
+ }
892
+ if (o.pipeline && (o.prompt != null || o.tokens || o.model || o.url)) {
893
+ broken("--pipeline names every stage's prompt, tokens and connection, so it cannot be "
894
+ + "given with --prompt, --prompt-file, --tokens, --model or --url.");
895
+ }
896
+ if (o.pipeline && o.files) {
897
+ broken("--pipeline's run document carries its file list in content.files, so --files "
898
+ + "would say it twice.");
899
+ }
900
+ if (o.samples && o.source) {
901
+ broken("--samples and --source name the same directory, so both together "
902
+ + "say two things.");
903
+ }
904
+ if (o.resume && o.item) {
905
+ broken("--resume and --item say different things about what to run.");
906
+ }
907
+
908
+ // The dataset the run is graded against. Everything below grades, so there
909
+ // is no run without one.
910
+ if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
911
+ const dataset = readDataset(o.dataset);
912
+
913
+ // The run to grade. A --pipeline document is one scenario graded against
914
+ // its dataset; without one it is the stage this script always ran -- the
915
+ // prompt, seeing the image, on --url with $EVAL_API_KEY -- stated as
916
+ // the same kind of document, so both go through one reading of it.
917
+ if (!o.pipeline && !(o.prompt ?? "").trim()) {
918
+ broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
919
+ + "the lab's Prompt library does.");
920
+ }
921
+ const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
922
+ if (o.pipeline) {
923
+ if (run.scenarios.length !== 1) {
924
+ broken(`${o.pipeline} has ${run.scenarios.length} scenarios, and --pipeline grades one `
925
+ + "against the set -- --run runs them all.");
926
+ }
927
+ if (!core.testsDataset(run)) {
928
+ broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
929
+ }
930
+ if (core.contentOf(run).type !== "source") {
931
+ broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
932
+ }
933
+ }
934
+ // Case by case: each item against its case, as the case metric of the
935
+ // test that names the dataset reads it (scoreCase).
936
+ const { stages, tokens, connections } = core.stagesFor(run, 0);
937
+ const prompt = core.resolvePrompt(stages[0].text, core.tokenSet(tokens, 0));
938
+
939
+ // The graded half, through the function the tab builds its own list with.
940
+ const set = core.gradedSetFrom(dataset);
941
+ let graded = set.filter(c => !c.todo);
942
+
943
+ const limit = (() => {
944
+ // The relay's cap applies no matter what the caller asked for: a request
945
+ // that outlives it is a failed item rather than a model that mutes and
946
+ // holds a single-flight queue for good.
947
+ const n = Number(o.timeout ?? REQUEST_CAP);
948
+ if (!(n > 0)) broken(`--timeout has to be a number of seconds above zero`);
949
+ return Math.min(n, REQUEST_CAP);
950
+ })();
951
+ seconds = limit;
952
+
953
+ const itemsOnly = (() => {
954
+ if (!o.item) return null;
955
+ const hit = graded.filter(c => c.id === o.item);
956
+ if (!hit.length) broken(`no graded case is named ${o.item}`);
957
+ return hit;
958
+ })();
959
+
960
+ // The earlier report a resume carries its finished items from. Both shapes
961
+ // a caller can leave behind: a full --json report, or the bare list of
962
+ // scores of rows. The earlier attempt's unrun is deliberately not honoured:
963
+ // those are the unfinished items a resume exists to run.
964
+ const carried = (() => {
965
+ if (!o.resume) return null;
966
+ let doc;
967
+ try {
968
+ doc = JSON.parse(fs.readFileSync(o.resume, "utf8"));
969
+ } catch (e) {
970
+ broken(`${o.resume}: ${e.message}`);
971
+ }
972
+ const rows = Array.isArray(doc) ? doc : doc?.cases;
973
+ if (!Array.isArray(rows) || !rows.every(r => isObject(r) && isText(r.id))) {
974
+ broken(`${o.resume}: a report carries its results under "cases" as a list of rows`);
975
+ }
976
+ const out = new Map();
977
+ for (const r of rows) if (!out.has(r.id)) out.set(r.id, r);
978
+ return out;
979
+ })();
980
+
981
+ // The snapshot a run was submitted with: only files it names are read, so
982
+ // files added to the Source afterwards cannot join midway and a report
983
+ // names exactly what it graded. A run document's own list is that
984
+ // snapshot; an empty one names none, and every graded case runs.
985
+ const snapshotSet = (() => {
986
+ if (o.pipeline) return core.contentOf(run).files?.length ? new Set(core.contentOf(run).files) : null;
987
+ if (!o.files) return null;
988
+ let doc;
989
+ try {
990
+ doc = JSON.parse(fs.readFileSync(o.files, "utf8"));
991
+ } catch (e) {
992
+ broken(`${o.files}: ${e.message}`);
993
+ }
994
+ const list = Array.isArray(doc) ? doc : isObject(doc) && Array.isArray(doc.files) ? doc.files : null;
995
+ if (!list || !list.every(isText)) {
996
+ broken(`${o.files}: a file-list snapshot is a list of names, or {"files": [...]}`);
997
+ }
998
+ return new Set(list);
999
+ })();
1000
+
1001
+ let replies = null, bin = "", samples = "", links = [];
1002
+ if (o.replies) {
1003
+ try {
1004
+ replies = JSON.parse(fs.readFileSync(o.replies, "utf8"));
1005
+ } catch (e) {
1006
+ broken(`${o.replies}: ${e.message}`);
1007
+ }
1008
+ } else {
1009
+ if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
1010
+ links = connections.map((c, i) => {
1011
+ if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
1012
+ const to = reach(c, o.pipeline ? core.keyVar(c.id) : "EVAL_API_KEY");
1013
+ // Once here rather than once per image: a llama.cpp stage on a
1014
+ // hosted model is a request that cannot be built at all.
1015
+ try {
1016
+ core.connectionRequest(c, "", null, to.base, to.key);
1017
+ } catch (e) {
1018
+ broken(`stage ${i + 1}: ${e.message}`);
1019
+ }
1020
+ return to;
1021
+ });
1022
+ samples = o.samples || o.source || firstFile(path.join(HERE, "samples"));
1023
+ if (!samples) broken("no images: there is no samples/ beside this script. "
1024
+ + "--samples or --source names a directory.");
1025
+ const allowed = itemsOnly ?? graded;
1026
+ // $snapshotSet above: only a Source run sees it; --replies reads no file.
1027
+ const inSnapshot = c => !snapshotSet || snapshotSet.has(c.filename);
1028
+ // Only the run's own image items need ImageMagick: a text-only Source
1029
+ // -- the .txt, .csv case -- runs with nothing on PATH but the node
1030
+ // running this, which is the point of having text in the registry.
1031
+ const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.filename)) === "image");
1032
+ if (pictures.length) {
1033
+ bin = imageMagick();
1034
+ if (!bin) broken("a run over images needs ImageMagick, to cap each "
1035
+ + "one to the pixel budget the tab caps it to on a canvas. "
1036
+ + "Install it, or use --replies to score recorded replies "
1037
+ + "without a model.");
1038
+ }
1039
+ }
1040
+
1041
+ // ---- Progress: items on a stream of their own ---------------------------
1042
+ // §5's JSON lines, one per finished item, for a caller to tail. Which
1043
+ // stream is decided by where the final report goes: if the report is on
1044
+ // stdout, the lines take the other one, so neither can swallow the other.
1045
+ // With --progress the human report moves to stderr regardless, so a line
1046
+ // on stdout is an item and nothing else.
1047
+ const progress = o.progress ? (o.json === "-" ? process.stderr : process.stdout) : null;
1048
+
1049
+ const rows = [];
1050
+ const unrun = [];
1051
+
1052
+ // The work: everything graded except a single-item run's one case, and,
1053
+ // on a resume, minus the earlier attempt's finished items. Those are
1054
+ // carried whole -- their verdicts already happened and a resume that
1055
+ // recomputed them would be deciding the numbers twice.
1056
+ const toRun = (itemsOnly ?? graded).filter(c => !carried?.has(c.id));
1057
+ if (carried) {
1058
+ for (const [id, r] of carried) {
1059
+ if (graded.some(c => c.id === id)) rows.push(r);
1060
+ }
1061
+ }
1062
+
1063
+ // Only files the snapshot names are read from the Source.
1064
+ const wanted = c => !snapshotSet || snapshotSet.has(c.filename);
1065
+ // The cancel marker is checked between items, so a cancel never loses the
1066
+ // reply in flight: the item that was being asked finishes, and the rest
1067
+ // are unrun rather than half-asked.
1068
+ let cancelled = false, done = 0;
1069
+ for (const kase of toRun) {
1070
+ if (o.cancelFile && fs.existsSync(o.cancelFile)) { cancelled = true; break; }
1071
+ done++;
1072
+ let item, calls;
1073
+ if (replies) {
1074
+ const got = replies[kase.id];
1075
+ const said = isText(got) ? [got] : got;
1076
+ if (!Array.isArray(said) || !said.every(isText)) {
1077
+ unrun.push(`${kase.id}: no recorded reply`);
1078
+ continue;
1079
+ }
1080
+ if (said.length !== stages.length) {
1081
+ unrun.push(`${kase.id}: ${said.length} recorded repl${said.length === 1 ? "y" : "ies"} `
1082
+ + `for ${stages.length} stages`);
1083
+ continue;
1084
+ }
1085
+ calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
1086
+ } else {
1087
+ if (!wanted(kase)) {
1088
+ unrun.push(`${kase.id}: ${kase.filename} is not in the run's file list`);
1089
+ continue;
1090
+ }
1091
+ const file = path.join(samples, kase.filename);
1092
+ if (!fs.existsSync(file)) {
1093
+ unrun.push(`${kase.id}: ${kase.filename} is not in ${samples}`);
1094
+ continue;
1095
+ }
1096
+ try {
1097
+ item = itemContent(bin, file, connections[0]);
1098
+ } catch (e) {
1099
+ unrun.push(`${kase.id}: ${kase.filename} could not be prepared — ${e.message}`);
1100
+ continue;
1101
+ }
1102
+ calls = callsFor(connections, links, item?.text ?? null);
1103
+ }
1104
+
1105
+ const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
1106
+ { tokens, text: item?.text ?? null });
1107
+ const s = core.scoreCase(kase, res);
1108
+ // A file with no mapper ran anyway; the flag keeps the transcript honest
1109
+ // about what the model was actually shown. §4: stated, not hidden.
1110
+ if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
1111
+ rows.push({
1112
+ id: kase.id, filename: kase.filename, half: kase.half,
1113
+ pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
1114
+ found: s.found, missed: s.missed, invented: s.invented, unmet: s.unmet,
1115
+ under: s.under, over: s.over, count: s.count,
1116
+ watchFound: s.watchFound, watchTotal: s.watchTotal,
1117
+ terms: res.terms || [], ms: res.ms,
1118
+ // What a file type was to this run -- "image", "text", "unmapped".
1119
+ type: item?.kind ?? null,
1120
+ // What the model actually said. #248 wants a failure written out as a
1121
+ // task an agent can act on, and "missed water" without the reply that
1122
+ // missed it is not one.
1123
+ reply: res.transcript?.[res.transcript.length - 1]?.got ?? "",
1124
+ // Every stage's prompt, reply and time, as the page's transcript shows
1125
+ // them: the final reply alone cannot say which stage went wrong.
1126
+ transcript: res.transcript,
1127
+ });
1128
+ if (progress) {
1129
+ progress.write(JSON.stringify({
1130
+ event: "item", id: kase.id, filename: kase.filename, n: rows.length,
1131
+ pass: s.pass, score: s.score, discarded: s.discarded,
1132
+ error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
1133
+ ms: res.ms,
1134
+ }) + "\n");
1135
+ }
1136
+ }
1137
+ // A cancelled run says which items never ran, so the report names what it
1138
+ // did not cover rather than letting the tail guess.
1139
+ if (cancelled) {
1140
+ for (const kase of toRun.slice(done)) {
1141
+ unrun.push(`${kase.id}: not run — cancelled after the item in flight`);
1142
+ }
1143
+ }
1144
+
1145
+ // The sum is made here, from the rows, in one place: a fresh run's rows
1146
+ // and a resume's carried ones are scored the same way, and no number is
1147
+ // worked out twice.
1148
+ const tally = core.emptyTally();
1149
+ for (const r of rows) {
1150
+ tally.ran++;
1151
+ if (r.pass) tally.passed++;
1152
+ tally.found += r.found.length;
1153
+ tally.of += r.found.length + r.missed.length + r.invented.length;
1154
+ }
1155
+
1156
+ // Anything short of the whole graded set is not a result. This is the
1157
+ // constraint in the issue: a set that did not run must not read as one that
1158
+ // passed, and 49 items absent with the fiftieth green is exactly how
1159
+ // that happens.
1160
+ const complete = graded.length > 0 && unrun.length === 0;
1161
+ const verdict = !complete ? "incomplete" : tally.passed === tally.ran ? "pass" : "fail";
1162
+
1163
+ const report = {
1164
+ runner: "tools/prompt-lab/run-evals.js",
1165
+ ranAt: new Date().toISOString(),
1166
+ dataset: inRepo(o.dataset),
1167
+ prompt,
1168
+ ...(o.pipeline ? { pipeline: {
1169
+ file: inRepo(o.pipeline), name: run.name || null,
1170
+ stages: stages.map((s, i) => {
1171
+ const { id, ...connection } = connections[i];
1172
+ return { n: i + 1, kind: s.kind, withImage: s.withImage,
1173
+ prompt: core.resolvePrompt(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1174
+ }),
1175
+ } } : {}),
1176
+ // Which variable a key came from, never the key.
1177
+ transport: replies ? { kind: "replies", file: o.replies }
1178
+ : o.pipeline ? { kind: "model", stages: links.map((to, i) => ({
1179
+ n: i + 1, connection: to.name, model: connections[i].model,
1180
+ endpoint: to.base, keyFrom: to.key ? to.from : null })) }
1181
+ : { kind: "model", model: o.model, endpoint: links[0].base },
1182
+ // How long a request could run, as given; a higher ask is capped at
1183
+ // the relay's 600.
1184
+ timeout: seconds,
1185
+ // The modes a server-side run reaches the worker through.
1186
+ ...(o.item ? { item: o.item } : {}),
1187
+ ...(o.resume ? { resume: o.resume, carried: carried.size } : {}),
1188
+ ...(snapshotSet ? { files: o.files || inRepo(o.pipeline),
1189
+ filesListed: snapshotSet.size } : {}),
1190
+ ...(o.source ? { source: inRepo(o.source) } : {}),
1191
+ verdict,
1192
+ graded: (itemsOnly ?? graded).length,
1193
+ ungraded: set.length - graded.length,
1194
+ ran: tally.ran,
1195
+ passed: tally.passed,
1196
+ failed: tally.ran - tally.passed,
1197
+ unrun,
1198
+ terms: { found: tally.found, of: tally.of, percent: core.tallyPercent(tally) },
1199
+ cases: rows,
1200
+ };
1201
+
1202
+ const toStdout = o.json === "-";
1203
+ // With --progress the chosen stream stays pure JSON, so what was prose
1204
+ // moves to stderr whether the report took stdout or not.
1205
+ const say = o.progress ? console.error : toStdout ? console.error : console.log;
1206
+ const pct = core.tallyPercent(tally);
1207
+
1208
+ say(`${report.runner} — ${report.dataset}`);
1209
+ if (o.pipeline) {
1210
+ say(`pipeline ${report.pipeline.file}${run.name ? ` — ${run.name}` : ""}`);
1211
+ if (replies) say(`recorded replies from ${o.replies}`);
1212
+ for (const [i, s] of report.pipeline.stages.entries()) {
1213
+ const to = links[i];
1214
+ say(`stage ${s.n} · ${s.kind} · ${s.withImage ? "with the image" : "text only"}`
1215
+ + (to ? ` · ${to.name}: ${s.connection.model} at ${to.base}`
1216
+ + (to.key ? ` · key from $${to.from}` : "") : ""));
1217
+ say(` ${s.prompt.replace(/\n/g, "\n ")}`);
1218
+ }
1219
+ } else {
1220
+ say(replies ? `recorded replies from ${o.replies}`
1221
+ : `${o.model} at ${links[0].base}`);
1222
+ say(`prompt: ${prompt}`);
1223
+ }
1224
+ say("");
1225
+ for (const r of rows.filter(r => !r.pass)) {
1226
+ const n = r.score == null ? " —" : `${String(Math.round(r.score * 100)).padStart(3)}%`;
1227
+ say(`fail ${r.id.padEnd(16)} ${n} ${r.reasons.join(" · ")}`);
1228
+ }
1229
+ // Capped: a run where nothing was found says so in one screen, and the
1230
+ // JSON report carries all of them for anything that needs the list.
1231
+ for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
1232
+ if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
1233
+ say("");
1234
+ say(`${tally.passed} of ${report.graded} graded cases passed`
1235
+ + ` · ${tally.found} of ${tally.of} terms${pct == null ? "" : ` (${pct}%)`}`
1236
+ + ` · ${report.ungraded} not graded yet`);
1237
+ if (!complete) {
1238
+ say(rows.length === 0 && !unrun.length
1239
+ ? "NOTHING RAN: the set grades no items. That is not a pass."
1240
+ : `NOTHING SCORED for ${unrun.length} of ${report.graded} graded cases, `
1241
+ + `so this is a partial run and not a result.`);
1242
+ }
1243
+
1244
+ if (o.json) {
1245
+ const text = JSON.stringify(report, null, 2);
1246
+ if (toStdout) process.stdout.write(text + "\n");
1247
+ else fs.writeFileSync(o.json, text + "\n");
1248
+ }
1249
+
1250
+ process.exitCode = !complete ? EXIT.incomplete
1251
+ : report.failed ? EXIT.failed : EXIT.passed;
1252
+ }
1253
+
1254
+ main().catch(e => {
1255
+ if (e instanceof Stop) {
1256
+ if (e.message) console.error(e.message);
1257
+ process.exitCode = e.code;
1258
+ return;
1259
+ }
1260
+ console.error(`the run failed: ${e?.stack || e}`);
1261
+ process.exitCode = EXIT.broken;
1262
+ });