evals-lab 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +57 -0
- package/README.md +98 -0
- package/bin/evals-lab.js +196 -0
- package/lab/demo/CREDITS.md +92 -0
- package/lab/demo/datasets/demo-1.json +1839 -0
- package/lab/demo/datasets/demo-2.json +1683 -0
- package/lab/demo/manifest.json +12 -0
- package/lab/demo/pipelines/demo-1.json +104 -0
- package/lab/demo/pipelines/demo-2.json +104 -0
- package/lab/demo/sources/Demo 1/basketball-hoop.jpg +0 -0
- package/lab/demo/sources/Demo 1/blue-boardwalk.jpg +0 -0
- package/lab/demo/sources/Demo 1/busy-beach.jpg +0 -0
- package/lab/demo/sources/Demo 1/butcher-sign.jpg +0 -0
- package/lab/demo/sources/Demo 1/cactus-flower.jpg +0 -0
- package/lab/demo/sources/Demo 1/corner-shop.jpg +0 -0
- package/lab/demo/sources/Demo 1/crosswalk-cyclist-panorama.jpg +0 -0
- package/lab/demo/sources/Demo 1/cyanotype-room.jpg +0 -0
- package/lab/demo/sources/Demo 1/dead-end-sign.jpg +0 -0
- package/lab/demo/sources/Demo 1/dog-in-snow.jpg +0 -0
- package/lab/demo/sources/Demo 1/empty-bedroom.jpg +0 -0
- package/lab/demo/sources/Demo 1/farmers-market-stall.jpg +0 -0
- package/lab/demo/sources/Demo 1/four-jets.jpg +0 -0
- package/lab/demo/sources/Demo 1/german-shepherd.jpg +0 -0
- package/lab/demo/sources/Demo 1/harbour-bridge-dusk.jpg +0 -0
- package/lab/demo/sources/Demo 1/harrow-on-the-hill-sign.jpg +0 -0
- package/lab/demo/sources/Demo 1/hong-kong-street.jpg +0 -0
- package/lab/demo/sources/Demo 1/house-salisbury-street.jpg +0 -0
- package/lab/demo/sources/Demo 1/infrared-orchard.jpg +0 -0
- package/lab/demo/sources/Demo 1/keyboard-desk.jpg +0 -0
- package/lab/demo/sources/Demo 1/lighthouse-dunes.jpg +0 -0
- package/lab/demo/sources/Demo 1/mountain-lake.jpg +0 -0
- package/lab/demo/sources/Demo 1/museum-skeleton.jpg +0 -0
- package/lab/demo/sources/Demo 1/music-room.jpg +0 -0
- package/lab/demo/sources/Demo 1/old-red-car.jpg +0 -0
- package/lab/demo/sources/Demo 1/parked-car-plate.jpg +0 -0
- package/lab/demo/sources/Demo 1/phone-and-wallet.png +0 -0
- package/lab/demo/sources/Demo 1/phone-box.jpg +0 -0
- package/lab/demo/sources/Demo 1/pink-flamingo.jpg +0 -0
- package/lab/demo/sources/Demo 1/postcards.jpg +0 -0
- package/lab/demo/sources/Demo 1/red-roof-church.jpg +0 -0
- package/lab/demo/sources/Demo 1/roadside-mailboxes.jpg +0 -0
- package/lab/demo/sources/Demo 1/rodeo.jpg +0 -0
- package/lab/demo/sources/Demo 1/running-tap.tif +0 -0
- package/lab/demo/sources/Demo 1/snail-on-stem.jpg +0 -0
- package/lab/demo/sources/Demo 1/two-horses-field.jpg +0 -0
- package/lab/demo/sources/Demo 1/university-sign.jpg +0 -0
- package/lab/demo/sources/Demo 1/vegetable-crates.jpg +0 -0
- package/lab/demo/sources/Demo 1/watermarked-car.jpg +0 -0
- package/lab/demo/sources/Demo 1/watermarked-pills.jpg +0 -0
- package/lab/demo/sources/Demo 1/whiteboard-delegate.jpg +0 -0
- package/lab/demo/sources/Demo 1/whiteboard-germ-layers.jpg +0 -0
- package/lab/demo/sources/Demo 2/aerial-city.jpg +0 -0
- package/lab/demo/sources/Demo 2/alligator-pen.jpg +0 -0
- package/lab/demo/sources/Demo 2/arm-tattoo.jpg +0 -0
- package/lab/demo/sources/Demo 2/bed-and-plant.jpg +0 -0
- package/lab/demo/sources/Demo 2/bright-bedroom.jpg +0 -0
- package/lab/demo/sources/Demo 2/car-headlight.jpg +0 -0
- package/lab/demo/sources/Demo 2/child-in-surf.jpg +0 -0
- package/lab/demo/sources/Demo 2/city-highway.jpg +0 -0
- package/lab/demo/sources/Demo 2/corner-bakery.jpg +0 -0
- package/lab/demo/sources/Demo 2/cyclist-yellow-jacket.jpg +0 -0
- package/lab/demo/sources/Demo 2/excavator-street-signs.jpg +0 -0
- package/lab/demo/sources/Demo 2/globe-closeup.jpg +0 -0
- package/lab/demo/sources/Demo 2/harbour-village.jpg +0 -0
- package/lab/demo/sources/Demo 2/high-street-walkers.jpg +0 -0
- package/lab/demo/sources/Demo 2/hillside-rooftops.jpg +0 -0
- package/lab/demo/sources/Demo 2/house-at-night.jpg +0 -0
- package/lab/demo/sources/Demo 2/ivy-leaves.jpg +0 -0
- package/lab/demo/sources/Demo 2/man-with-alligator.jpg +0 -0
- package/lab/demo/sources/Demo 2/orange-mailbox-hedge.jpg +0 -0
- package/lab/demo/sources/Demo 2/pedal-boats.jpg +0 -0
- package/lab/demo/sources/Demo 2/pink-trees-vignette.jpg +0 -0
- package/lab/demo/sources/Demo 2/please-leave-quietly-sign.jpg +0 -0
- package/lab/demo/sources/Demo 2/railroad-crossing-sign.jpg +0 -0
- package/lab/demo/sources/Demo 2/roadworks-sign.jpg +0 -0
- package/lab/demo/sources/Demo 2/roundabout-sign.jpg +0 -0
- package/lab/demo/sources/Demo 2/sea-cave.jpg +0 -0
- package/lab/demo/sources/Demo 2/shadow-on-sand.jpg +0 -0
- package/lab/demo/sources/Demo 2/signpost-a404.jpg +0 -0
- package/lab/demo/sources/Demo 2/street-fruit-cart.jpg +0 -0
- package/lab/demo/sources/Demo 2/street-sign-parnassusweg.jpg +0 -0
- package/lab/demo/sources/Demo 2/two-horses-close.jpg +0 -0
- package/lab/demo/sources/Demo 2/victorian-house.jpg +0 -0
- package/lab/demo/sources/Demo 2/vintage-dashboard.jpg +0 -0
- package/lab/demo/sources/Demo 2/volcano-at-dusk.jpg +0 -0
- package/lab/demo/sources/Demo 2/wall-camera.jpg +0 -0
- package/lab/demo/sources/Demo 2/watch-for-rocks-sign.jpg +0 -0
- package/lab/demo/sources/Demo 2/waterfall.jpg +0 -0
- package/lab/demo/sources/Demo 2/white-domes.jpg +0 -0
- package/lab/demo/sources/Demo 2/whiteboard-meeting-notes.jpg +0 -0
- package/lab/demo/sources/Demo 2/whiteboard-messages.jpg +0 -0
- package/lab/demo/sources/Demo 2/wooden-house-fence.jpg +0 -0
- package/lab/evals-core.mjs +4526 -0
- package/lab/flows/wdl.mjs +373 -0
- package/lab/js-yaml.mjs +3851 -0
- package/lab/kinds/list.mjs +411 -0
- package/lab/metrics/builtin.mjs +353 -0
- package/lab/presets.json +62 -0
- package/lab/run-evals.js +1262 -0
- package/lab/server.py +5044 -0
- package/lab/web/dist/assets/dist-DI3ewZYj.js +1 -0
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +3 -0
- package/lab/web/dist/assets/gallery-o7c4lfpn.css +1 -0
- package/lab/web/dist/assets/main-CpVssvrb.js +18 -0
- package/lab/web/dist/assets/main-DjQQums6.css +1 -0
- package/lab/web/dist/assets/tokens-B9intIuT.js +51 -0
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +1 -0
- package/lab/web/dist/gallery.html +18 -0
- package/lab/web/dist/index.html +23 -0
- package/package.json +20 -0
package/lab/run-evals.js
ADDED
|
@@ -0,0 +1,1262 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Runs the graded set headlessly. The second of the three callers of
|
|
3
|
+
// evals-core.ts -- the Prompt Lab tab is the first, CI is the third.
|
|
4
|
+
//
|
|
5
|
+
// An agent is a first-class caller here, not an afterthought, so this says
|
|
6
|
+
// what it did twice: a short human report, and a JSON one with every verdict
|
|
7
|
+
// in it. The exit code is the third statement and the coarsest:
|
|
8
|
+
//
|
|
9
|
+
// 0 every graded case passed
|
|
10
|
+
// 1 the set ran and at least one case failed
|
|
11
|
+
// 2 the set did not run in full -- nothing is graded, or an image or a
|
|
12
|
+
// recorded reply was missing. NOT a pass. An eval set that scores well
|
|
13
|
+
// because it never ran is the failure this whole ticket is about.
|
|
14
|
+
// 3 the run could not be attempted: bad arguments, no ImageMagick, no set.
|
|
15
|
+
//
|
|
16
|
+
// The gating policy -- which of those a pull request may merge on -- is #250's
|
|
17
|
+
// to decide, and it has the JSON to decide it from. This only reports.
|
|
18
|
+
//
|
|
19
|
+
// Ollama is firewalled to a handful of hosts, so a live run has to happen on
|
|
20
|
+
// one of them. That is a fact about the network and not something a flag here
|
|
21
|
+
// can arrange; `--replies` is the mode that needs no model at all.
|
|
22
|
+
//
|
|
23
|
+
// The server-side-runs modes (#525, design §4 and §11.4) are here rather
|
|
24
|
+
// than in a second executor, because the executor *is* the CLI: a run the
|
|
25
|
+
// server queues and a run a terminal starts have to be the same thing or the
|
|
26
|
+
// parity check measures nothing.
|
|
27
|
+
//
|
|
28
|
+
// --source/--files the files come out of a Source: the snapshot's file
|
|
29
|
+
// list, read off the directory the orchestrator handed
|
|
30
|
+
// over. A case whose image missed the snapshot is
|
|
31
|
+
// unrun, not guessed around -- a report that cannot be
|
|
32
|
+
// reproduced is not a report.
|
|
33
|
+
// --progress JSON lines, one item per line, on a stream of their
|
|
34
|
+
// own: stdout, unless the final report is on stdout,
|
|
35
|
+
// in which case stderr -- so one cannot swallow the
|
|
36
|
+
// other. With --progress the human report moves to
|
|
37
|
+
// stderr, keeping stdout pure lines for the tail.
|
|
38
|
+
// --timeout a per-request cap, like the relay's, so a mute model
|
|
39
|
+
// becomes a failed item rather than a frozen queue.
|
|
40
|
+
// --resume continue a run from its first unfinished item: the
|
|
41
|
+
// cases the earlier report scored are carried whole and
|
|
42
|
+
// the rest are run. Verdicts are never recomputed.
|
|
43
|
+
// --item one case against the run's snapshot.
|
|
44
|
+
// --rescore a run's stored results, re-scored against the dataset
|
|
45
|
+
// it names with --dataset: the replies are kept -- they
|
|
46
|
+
// happened, and asking again is a different question --
|
|
47
|
+
// and only the verdicts are recomputed. No model, no
|
|
48
|
+
// files. A run whose target results are for text is the
|
|
49
|
+
// one item a --run names; this mode needs the snapshot
|
|
50
|
+
// (`--run`), the stored results (`--results-file`) and,
|
|
51
|
+
// for a graded run, the dataset (`--dataset`), and takes
|
|
52
|
+
// `--only <n>` for one item.
|
|
53
|
+
//
|
|
54
|
+
// A dataset is never read from the lab's own directories. The queue hands the
|
|
55
|
+
// worker the body a run was submitted with (--dataset), so a run grades
|
|
56
|
+
// against exactly that, whatever the dataset holds by the time it runs.
|
|
57
|
+
//
|
|
58
|
+
// What a file IS to a run is decided by the type→handler registry, not the
|
|
59
|
+
// queue: text joins at {text}, images go through the pixel budget, and a
|
|
60
|
+
// file with no handler runs anyway, flagged. See HANDLERS below.
|
|
61
|
+
const fs = require("fs");
|
|
62
|
+
const path = require("path");
|
|
63
|
+
const { execFileSync } = require("child_process");
|
|
64
|
+
const core = require("./evals-core.mjs");
|
|
65
|
+
|
|
66
|
+
const HERE = __dirname;
|
|
67
|
+
const ROOT = path.join(HERE, "..", "..");
|
|
68
|
+
|
|
69
|
+
// Tagger.PIXELS_720P, as an area rather than a longest edge: llama.cpp slices
|
|
70
|
+
// into 448px tiles and the tile COUNT is what costs. The same budget the tab
|
|
71
|
+
// offers by default and the container bakes its copies to.
|
|
72
|
+
const PIXELS = 921600;
|
|
73
|
+
|
|
74
|
+
// The connection a caller that names no address means, exactly as in
|
|
75
|
+
// server.py -- Ollama on this machine.
|
|
76
|
+
const OLLAMA = process.env.OLLAMA_URL || "http://127.0.0.1:11434";
|
|
77
|
+
|
|
78
|
+
const EXIT = { passed: 0, failed: 1, incomplete: 2, broken: 3 };
|
|
79
|
+
|
|
80
|
+
// The relay's own per-request cap (server.py `_proxy`). A request that
|
|
81
|
+
// outlives it is a failed item rather than a model that mutes and holds a
|
|
82
|
+
// single-flight queue for good; `--timeout` can shorten it, never lengthen.
|
|
83
|
+
const REQUEST_CAP = 600;
|
|
84
|
+
|
|
85
|
+
const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
86
|
+
|
|
87
|
+
--model <id> the model to grade. Required for a live run.
|
|
88
|
+
--url <base> an OpenAI-shaped endpoint. Default: $OLLAMA_URL, else
|
|
89
|
+
Ollama on this machine. A key comes from $EVAL_API_KEY, never argv.
|
|
90
|
+
--prompt <text> the prompt to grade, as a template: its tokens resolve
|
|
91
|
+
under --tokens. Required without --pipeline: a dataset
|
|
92
|
+
holds no prompt (the lab's Prompt library does).
|
|
93
|
+
--prompt-file <p> the same, read from a file.
|
|
94
|
+
--plugins <dir> where the lab keeps its plugins: <dir>/<id>/<version>/,
|
|
95
|
+
each with its manifest.json and entry. A --run loads
|
|
96
|
+
the plugins it recorded; anything else loads every one
|
|
97
|
+
there. Each registers into the core before the run is
|
|
98
|
+
read, as the page's plugins do (docs/packs.md).
|
|
99
|
+
--tokens <file> the token mappings that prompt is resolved with, as
|
|
100
|
+
JSON: a list of { name, type, value | enabled }, as a
|
|
101
|
+
job's tokenMappings. Default: none, so a prompt
|
|
102
|
+
naming a token is refused, naming it.
|
|
103
|
+
--pipeline <file> a run document (docs/pipeline-model.md) with one
|
|
104
|
+
scenario and a graded test, graded case by case against
|
|
105
|
+
the set. Instead of --prompt, --model and --url. Each
|
|
106
|
+
profile's key comes from $EVAL_API_KEY_<ID>, never the
|
|
107
|
+
file; its content.files, when not empty, is the file
|
|
108
|
+
list a case has to be in, as --files says.
|
|
109
|
+
--samples <dir> where the images are. Default: the lab's samples/.
|
|
110
|
+
--dataset <file> the dataset a graded run is scored against: its body
|
|
111
|
+
({"cases"}), or the file the lab's Export writes.
|
|
112
|
+
Versions 1 to 3 are read too (a prompt they hold is
|
|
113
|
+
not used: --prompt names it); a version-2
|
|
114
|
+
body's rules are what an older run's jobs, and a run
|
|
115
|
+
without --pipeline, read replies under. Its cases grade.
|
|
116
|
+
Required for anything graded.
|
|
117
|
+
--source <dir> the same, named the way a run names it: the directory a
|
|
118
|
+
Source is stored at. --samples and --source are one
|
|
119
|
+
flag by two names; both together are refused.
|
|
120
|
+
--files <file> the run's file-list snapshot, as JSON: a list of names,
|
|
121
|
+
or {"files": [...]}. Only files it names are read from
|
|
122
|
+
the Source; a case whose file missed the list is unrun.
|
|
123
|
+
--replies <file> score recorded replies instead of calling a model:
|
|
124
|
+
{"<case id>": "<reply>"}, or for a pipeline one reply per
|
|
125
|
+
stage: {"<case id>": ["<stage 1>", "<stage 2>"]}. No
|
|
126
|
+
network, no images.
|
|
127
|
+
--item <id> run one case from the set, nothing else.
|
|
128
|
+
--resume <file> a report from an earlier attempt: the cases it scored
|
|
129
|
+
are carried whole, the rest run from the first
|
|
130
|
+
unfinished one.
|
|
131
|
+
--timeout <s> cap each request at this many seconds, at most 600 --
|
|
132
|
+
the relay's cap, and a longer value cannot extend it.
|
|
133
|
+
Default: 600.
|
|
134
|
+
--progress write one JSON line per finished item, so a caller
|
|
135
|
+
can tail a run in progress. On stdout unless the
|
|
136
|
+
report is on stdout too, in which case stderr; and
|
|
137
|
+
then the human report moves to stderr.
|
|
138
|
+
--cancel-file <p> a path whose existence is checked between items: the run
|
|
139
|
+
stops after the item in flight, never mid-reply.
|
|
140
|
+
--run <file> a server run: the run document the queue hands the
|
|
141
|
+
worker, every scenario over every item of its content;
|
|
142
|
+
the files themselves live in --source. Replaces
|
|
143
|
+
--pipeline/--model/--replies. Keys as --pipeline.
|
|
144
|
+
--progress-file <p> JSON-lines progress for a --run, one line per item as it
|
|
145
|
+
finishes, on a stream of its own -- never the stream the
|
|
146
|
+
final report takes, so one cannot swallow the other.
|
|
147
|
+
--results-file <f> a --run's results so far (resume and re-run item).
|
|
148
|
+
--from <n> a --run starts at item n (resume), keeping earlier results.
|
|
149
|
+
--only <n> a --run runs item n alone (re-run item), replacing it.
|
|
150
|
+
--rescore re-score a run's stored results against --dataset.
|
|
151
|
+
Needs --run and --results-file, and --dataset for a
|
|
152
|
+
graded run; an optional --only <n> re-scores item n
|
|
153
|
+
alone. No model, no files, no network.
|
|
154
|
+
--json <path|-> write the machine-readable report. "-" means stdout, and
|
|
155
|
+
sends the human report to stderr.
|
|
156
|
+
`;
|
|
157
|
+
|
|
158
|
+
// NOTHING HERE CALLS process.exit(). Node's stdout is asynchronous down a
|
|
159
|
+
// pipe, and exiting discards whatever has not drained: measured, a report of
|
|
160
|
+
// 78,872 bytes came out of `--json - | ...` cut to 65,431 -- the pipe buffer
|
|
161
|
+
// -- and unparseable, while the same run redirected to a file was whole. An
|
|
162
|
+
// agent piping the machine-readable report is the caller this script was
|
|
163
|
+
// written for, so the code is set and the process is left to finish writing
|
|
164
|
+
// and leave on its own.
|
|
165
|
+
//
|
|
166
|
+
// Whether the loss shows up depends on how fast the reader drains -- the same
|
|
167
|
+
// report piped to `wc -c` arrived whole -- so it is not a fault a run can be
|
|
168
|
+
// relied on to reveal. `runner-check.js` asserts the absence of the call
|
|
169
|
+
// instead, and that is why.
|
|
170
|
+
//
|
|
171
|
+
// The price is that a live run sits for a few seconds after printing its
|
|
172
|
+
// report: undici keeps a socket to the endpoint alive and that holds the event
|
|
173
|
+
// loop open. Measured at 6.3s against an endpoint still listening and 0.3s
|
|
174
|
+
// once it closes; a --replies run makes no request and leaves at once. It is a
|
|
175
|
+
// wait and not a hang, the exit code is already set by then, and a run that
|
|
176
|
+
// spent minutes on the images will not notice. Do not "fix" it with a
|
|
177
|
+
// process.exit() -- that is the truncation above, straight back.
|
|
178
|
+
//
|
|
179
|
+
// Which means stopping early has to be a throw rather than an exit, since a
|
|
180
|
+
// `broken()` that returned would let its caller carry on with the argument it
|
|
181
|
+
// had just refused.
|
|
182
|
+
class Stop extends Error {
|
|
183
|
+
constructor(message, code) { super(message); this.code = code; }
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
function broken(msg) {
|
|
187
|
+
throw new Stop(msg, EXIT.broken);
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
function parseArgs(argv) {
|
|
191
|
+
const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
|
|
192
|
+
replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
193
|
+
timeout: null, progress: false, cancelFile: "",
|
|
194
|
+
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
195
|
+
rescore: false, tokens: "", plugins: "" };
|
|
196
|
+
for (let i = 0; i < argv.length; i++) {
|
|
197
|
+
const a = argv[i];
|
|
198
|
+
const value = () => {
|
|
199
|
+
const v = argv[++i];
|
|
200
|
+
if (v === undefined) broken(`${a} needs a value\n\n${USAGE}`);
|
|
201
|
+
return v;
|
|
202
|
+
};
|
|
203
|
+
switch (a) {
|
|
204
|
+
case "--model": o.model = value(); break;
|
|
205
|
+
case "--url": o.url = value(); break;
|
|
206
|
+
case "--prompt": o.prompt = value(); break;
|
|
207
|
+
case "--prompt-file": o.prompt = fs.readFileSync(value(), "utf8").trim(); break;
|
|
208
|
+
case "--tokens": o.tokens = value(); break;
|
|
209
|
+
case "--plugins": o.plugins = value(); break;
|
|
210
|
+
case "--pipeline": o.pipeline = value(); break;
|
|
211
|
+
case "--samples": o.samples = value(); break;
|
|
212
|
+
case "--dataset": o.dataset = value(); break;
|
|
213
|
+
case "--replies": o.replies = value(); break;
|
|
214
|
+
case "--source": o.source = value(); break;
|
|
215
|
+
case "--files": o.files = value(); break;
|
|
216
|
+
case "--item": o.item = value(); break;
|
|
217
|
+
case "--resume": o.resume = value(); break;
|
|
218
|
+
case "--timeout": o.timeout = value(); break;
|
|
219
|
+
case "--progress": o.progress = true; break;
|
|
220
|
+
case "--cancel-file": o.cancelFile = value(); break;
|
|
221
|
+
case "--run": o.run = value(); break;
|
|
222
|
+
case "--progress-file": o.progressFile = value(); break;
|
|
223
|
+
case "--results-file": o.resultsFile = value(); break;
|
|
224
|
+
case "--from": o.from = value(); break;
|
|
225
|
+
case "--only": o.only = value(); break;
|
|
226
|
+
case "--rescore": o.rescore = true; break;
|
|
227
|
+
case "--json": o.json = value(); break;
|
|
228
|
+
case "-h": case "--help":
|
|
229
|
+
process.stdout.write(USAGE);
|
|
230
|
+
throw new Stop("", EXIT.passed);
|
|
231
|
+
default: broken(`unknown option ${a}\n\n${USAGE}`);
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
return o;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// api_base in server.py, in JavaScript. A bare host and port, a base URL, a
|
|
238
|
+
// /v1 and the full chat path are all the same endpoint, and all four are
|
|
239
|
+
// reasonable things to type. The tab reaches a model through the relay, which
|
|
240
|
+
// does this for it; this script does not, so it does it here -- one copy in
|
|
241
|
+
// the core, shared with the page's Test connection.
|
|
242
|
+
const apiBase = core.apiBase;
|
|
243
|
+
|
|
244
|
+
const firstFile = (...paths) => paths.find(p => fs.existsSync(p)) || "";
|
|
245
|
+
|
|
246
|
+
// Repo-relative where it is in the repo, and as given where it is not: a
|
|
247
|
+
// report that names the set as ../../../../tmp/x.json is a report nobody can
|
|
248
|
+
// tell was run against the committed one.
|
|
249
|
+
const inRepo = p => {
|
|
250
|
+
const rel = path.relative(ROOT, p);
|
|
251
|
+
return rel && !rel.startsWith("..") ? rel : p;
|
|
252
|
+
};
|
|
253
|
+
|
|
254
|
+
// ---- A run document -------------------------------------------------------
|
|
255
|
+
// What the queue hands the worker and what CI commits: a pipeline resolved
|
|
256
|
+
// (docs/pipeline-model.md §5). evals-core.ts validates it -- every field on a
|
|
257
|
+
// list, a key refused by name -- so this script, the page and #29's import
|
|
258
|
+
// refuse the same file with the same sentence.
|
|
259
|
+
const isObject = v => v != null && typeof v === "object" && !Array.isArray(v);
|
|
260
|
+
const isText = v => typeof v === "string";
|
|
261
|
+
|
|
262
|
+
// [dataset] is the body --dataset handed over, or null: the one dataset a
|
|
263
|
+
// graded run can name, whatever id it names it by, since the queue hands the
|
|
264
|
+
// worker the body of exactly the dataset the run was submitted against.
|
|
265
|
+
function readRun(file, dataset) {
|
|
266
|
+
let doc;
|
|
267
|
+
try {
|
|
268
|
+
// A run queued before the current version is read as one of today's:
|
|
269
|
+
// resuming or re-running an old row works on the document it was
|
|
270
|
+
// submitted with. A v2 run's list jobs read their replies under the
|
|
271
|
+
// rules of the dataset it was graded against -- the body handed over.
|
|
272
|
+
doc = core.upgradePipeline(JSON.parse(fs.readFileSync(file, "utf8")),
|
|
273
|
+
{ rulesFor: () => rulesOf(dataset) });
|
|
274
|
+
} catch (e) {
|
|
275
|
+
broken(`${file}: ${e.message}`);
|
|
276
|
+
}
|
|
277
|
+
const named = isObject(doc) ? core.testsDataset(doc) : null;
|
|
278
|
+
const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
|
|
279
|
+
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
280
|
+
return doc;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
// The rules a version-2 body held, by the body readDataset made of it: what
|
|
284
|
+
// an older run's jobs read replies under. A body of today's holds none.
|
|
285
|
+
const heldRules = new WeakMap();
|
|
286
|
+
const rulesOf = (dataset) => (dataset && heldRules.get(dataset)) ?? null;
|
|
287
|
+
|
|
288
|
+
// The dataset --dataset names: a dataset's body, or the file the lab's
|
|
289
|
+
// Export writes for one, as one of today's.
|
|
290
|
+
function readDataset(file) {
|
|
291
|
+
if (!file) return null;
|
|
292
|
+
let doc;
|
|
293
|
+
try {
|
|
294
|
+
doc = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
295
|
+
} catch (e) {
|
|
296
|
+
broken(`${file}: ${e.message}`);
|
|
297
|
+
}
|
|
298
|
+
if (isObject(doc) && doc.format === "evals-lab/dataset") {
|
|
299
|
+
if (![1, 2, 3, 4].includes(doc.version)) {
|
|
300
|
+
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 4`);
|
|
301
|
+
}
|
|
302
|
+
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
303
|
+
}
|
|
304
|
+
if (isObject(doc) && doc.rules != null && !(isObject(doc.rules) && Array.isArray(doc.rules.rules))) {
|
|
305
|
+
broken(`${file}: rules has to be null or { "rules": [...] }`);
|
|
306
|
+
}
|
|
307
|
+
// A body from an earlier version -- a run queued then, resumed now -- reads
|
|
308
|
+
// as one of today's, its rules taken first.
|
|
309
|
+
const rules = core.datasetRules(doc);
|
|
310
|
+
doc = core.upgradeDatasetBody(doc);
|
|
311
|
+
if (!isObject(doc) || !Array.isArray(doc.cases)) {
|
|
312
|
+
broken(`${file} is not a dataset: its body has a cases list, or it is the lab's Export of one`);
|
|
313
|
+
}
|
|
314
|
+
if (rules) heldRules.set(doc, rules);
|
|
315
|
+
return doc;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
// The cases a graded run is scored against, by the file each names.
|
|
319
|
+
function gradedBy(dataset) {
|
|
320
|
+
const out = new Map();
|
|
321
|
+
for (const c of core.gradedSetFrom(dataset || {})) {
|
|
322
|
+
if (!c.todo) out.set(c.filename, c);
|
|
323
|
+
}
|
|
324
|
+
return out;
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
// ---- Preparing an image ----------------------------------------------
|
|
328
|
+
// The tab does this on a canvas, which is the one thing in the run that cannot
|
|
329
|
+
// be shared: there is no canvas in Node and adding one would mean a native
|
|
330
|
+
// dependency in a lab whose whole premise is that it has none.
|
|
331
|
+
//
|
|
332
|
+
// So the arithmetic is shared instead, by being the same arithmetic:
|
|
333
|
+
// core.preparedSize is the tab's own capping -- including the 448-longest-edge
|
|
334
|
+
// mode an area cannot express -- and the conversion that follows matches the
|
|
335
|
+
// canvas's toDataURL.
|
|
336
|
+
//
|
|
337
|
+
// The stage's profile says how far an image is capped and by which
|
|
338
|
+
// encoding; absent, the 720p JPEG quality 92 the container bakes its copies
|
|
339
|
+
// to, which is what this stage always ran at.
|
|
340
|
+
function imageMagick() {
|
|
341
|
+
// Not `convert` on Windows: that name is the system's own disk tool there.
|
|
342
|
+
for (const bin of process.platform === "win32" ? ["magick"] : ["magick", "convert"]) {
|
|
343
|
+
try {
|
|
344
|
+
execFileSync(bin, ["-version"], { stdio: "ignore" });
|
|
345
|
+
return bin;
|
|
346
|
+
} catch { /* not installed under that name */ }
|
|
347
|
+
}
|
|
348
|
+
return "";
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
function prepare(bin, file, c) {
|
|
352
|
+
const out = execFileSync(bin, [file, "-auto-orient", "-format", "%w %h", "info:"],
|
|
353
|
+
{ maxBuffer: 1 << 20 }).toString().trim();
|
|
354
|
+
let [w, h] = out.split(/\s+/).map(Number);
|
|
355
|
+
if (!(w > 0 && h > 0)) throw new Error(`${path.basename(file)}: no dimensions`);
|
|
356
|
+
// What the connection says, from the one place that reads its fields.
|
|
357
|
+
const image = core.connectionImage(c);
|
|
358
|
+
[w, h] = core.preparedSize(w, h, image.px ?? PIXELS);
|
|
359
|
+
const fmt = { "image/jpeg": ["jpg", "image/jpeg"], "image/png": ["png", "image/png"],
|
|
360
|
+
"image/webp": ["webp", "image/webp"] }[image.format];
|
|
361
|
+
const quality = String(core.encoderQuality(image.quality));
|
|
362
|
+
const jpeg = execFileSync(
|
|
363
|
+
bin, [file, "-auto-orient", "-strip", "-resize", `${w}x${h}!`, "-quality", quality, `${fmt[0]}:-`],
|
|
364
|
+
{ maxBuffer: 64 << 20 });
|
|
365
|
+
return { dataUrl: `data:${fmt[1]};base64,` + jpeg.toString("base64"), dims: `${w}x${h}` };
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
// ---- The type → handler registry ------------------------------------------
|
|
369
|
+
// Design §4. What a file *is* to a run is decided here and only here, so the
|
|
370
|
+
// queue and the page never learn about file types, and pdf and xlsx are
|
|
371
|
+
// later entries that touch nothing but this map.
|
|
372
|
+
//
|
|
373
|
+
// A file with no mapper is passed through anyway -- decoded as UTF-8 with
|
|
374
|
+
// replacement and joined at {text} like the text handler, and flagged in the
|
|
375
|
+
// transcript. Stating the failure mode beats hiding it: a PDF's text layer
|
|
376
|
+
// is not extracted, so its streams arrive as arbitrary bytes, and a run
|
|
377
|
+
// that read them "well" would be a finding about nothing.
|
|
378
|
+
const HANDLERS = new Map();
|
|
379
|
+
[
|
|
380
|
+
// Images: prepared to the profile's pixel budget by ImageMagick and sent
|
|
381
|
+
// as the image content part. The formats are server.py's FILE_TYPES
|
|
382
|
+
// image half, so nothing accepted for upload has no executor to run it.
|
|
383
|
+
["image", "jpg", "jpeg", "png", "heic", "heif", "dng", "tif", "tiff"],
|
|
384
|
+
// Text-ish: decoded UTF-8, joining the stage at {text} -- the text-content
|
|
385
|
+
// semantics the canvas already has (core.textPrompt).
|
|
386
|
+
["text", "txt", "md", "csv"],
|
|
387
|
+
].forEach((row) => {
|
|
388
|
+
const [kind, ...exts] = row;
|
|
389
|
+
for (const ext of exts) HANDLERS.set(ext, kind);
|
|
390
|
+
});
|
|
391
|
+
// The extension of a file in the Source, lower-cased. Names arrive already
|
|
392
|
+
// clean from the upload path; this only classifies.
|
|
393
|
+
const extOf = name => (path.extname(String(name)) || "").slice(1).toLowerCase();
|
|
394
|
+
|
|
395
|
+
const UNMAPPED = "the file has no handler, so it was passed through as raw "
|
|
396
|
+
+ "bytes decoded as UTF-8 with replacement -- whatever a "
|
|
397
|
+
+ "real reader of this type would have extracted was not "
|
|
398
|
+
+ "extracted, and the content is mojibake";
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* One file of the Source made into what the job consumes: an image as the
|
|
402
|
+
* dataUrl at [conn]'s budget, or text at {text}. A file with no mapper
|
|
403
|
+
* still runs, and says so.
|
|
404
|
+
*/
|
|
405
|
+
function itemContent(bin, file, conn) {
|
|
406
|
+
const kind = HANDLERS.get(extOf(file));
|
|
407
|
+
if (kind === "image") {
|
|
408
|
+
const { dataUrl, dims } = prepare(bin, file, conn);
|
|
409
|
+
return { kind, dataUrl, text: null, flags: [] };
|
|
410
|
+
}
|
|
411
|
+
// TextDecoder with fatal unset replaces malformed bytes with U+FFFD --
|
|
412
|
+
// what "decoded as UTF-8 with replacement" means, for a mapped-text file
|
|
413
|
+
// and an unmapped one alike.
|
|
414
|
+
const text = new TextDecoder().decode(fs.readFileSync(file));
|
|
415
|
+
return { kind: kind ?? "unmapped", dataUrl: null, text, flags: kind ? [] : [UNMAPPED] };
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
// ---- Reaching a model ----------------------------------------------------
|
|
419
|
+
|
|
420
|
+
/**
|
|
421
|
+
* Where a connection goes and the key it takes, refused before anything is
|
|
422
|
+
* sent if the key cannot go there. [from] is the variable the key is read
|
|
423
|
+
* from, named in the report in place of the key.
|
|
424
|
+
*/
|
|
425
|
+
function reach(c, from) {
|
|
426
|
+
const base = core.connectionBase({ ...c, url: c.url || OLLAMA });
|
|
427
|
+
const key = (process.env[from] || "").trim();
|
|
428
|
+
if (key) {
|
|
429
|
+
// server.py's header_safe: fetch rejects such a header with the value in
|
|
430
|
+
// its message, which would put the key in the report.
|
|
431
|
+
if (![...key].every(ch => ch >= " " && ch <= "~")) {
|
|
432
|
+
broken(`$${from} has characters a header cannot carry, so the key was not sent.`);
|
|
433
|
+
}
|
|
434
|
+
// server.py's rule, for server.py's reason: a key that bills a real card
|
|
435
|
+
// should not cross a network in the clear. Loopback never leaves the
|
|
436
|
+
// machine.
|
|
437
|
+
const host = new URL(base).hostname;
|
|
438
|
+
if (!base.startsWith("https://") && !["127.0.0.1", "[::1]", "localhost"].includes(host)) {
|
|
439
|
+
broken(`$${from} is set and ${base} is not https, so the key was not sent.`);
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
return { name: c.name, base, key, from };
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
// The page's callModel, less the relay: the same body, and on llama.cpp the
|
|
446
|
+
// same check for a reply that looped. A request that outlives the cap is
|
|
447
|
+
// aborted and comes back as the stage's error, so the queue it belongs to
|
|
448
|
+
// keeps moving instead of waiting on a model that mutes.
|
|
449
|
+
let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
|
|
450
|
+
|
|
451
|
+
// Each stage's transport for one item. A type that answers from the item
|
|
452
|
+
// itself (Echo, #97) is asked nothing: its reply is the item's own text --
|
|
453
|
+
// or, over Prompt only, the prompt -- and the stage reads it as it would a
|
|
454
|
+
// model's.
|
|
455
|
+
//
|
|
456
|
+
// A call that asks for a whole request (an HTTP Request, #flows) builds it
|
|
457
|
+
// from its step, the scenario's cell and the item's record, and is handed
|
|
458
|
+
// only its words: the transport is where the three meet.
|
|
459
|
+
function callsFor(connections, links, text, plan, record) {
|
|
460
|
+
return connections.map((c, k) => {
|
|
461
|
+
const step = plan?.calls?.[k];
|
|
462
|
+
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
463
|
+
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
464
|
+
}
|
|
465
|
+
return core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
466
|
+
? async sent => core.localAnswer(c, text, sent)
|
|
467
|
+
: (sent, url) => ask(links[k], c, sent, url);
|
|
468
|
+
});
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
/** A Setup profile the run carries, asked as a grader: its reply's text, or
|
|
472
|
+
the failure as an error the metric reports. One transport per profile. */
|
|
473
|
+
const graders = new WeakMap();
|
|
474
|
+
function graderFor(run) {
|
|
475
|
+
if (!graders.has(run)) {
|
|
476
|
+
const made = new Map();
|
|
477
|
+
graders.set(run, ref => {
|
|
478
|
+
const c = run.profiles?.[ref.id];
|
|
479
|
+
if (!c) return undefined;
|
|
480
|
+
if (!made.has(ref.id)) {
|
|
481
|
+
const link = reach(c, core.keyVar(ref.id));
|
|
482
|
+
made.set(ref.id, async prompt => {
|
|
483
|
+
const r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
484
|
+
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
485
|
+
return r.raw ?? "";
|
|
486
|
+
});
|
|
487
|
+
}
|
|
488
|
+
return made.get(ref.id);
|
|
489
|
+
});
|
|
490
|
+
}
|
|
491
|
+
return graders.get(run);
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
/** A text item as a record's scope: its text is the trigger's body. */
|
|
495
|
+
const textRecord = (text) => ({ scope: { trigger: { body: text } }, at: null });
|
|
496
|
+
|
|
497
|
+
/** An HTTP Request: the flow's request rebuilt from [record], sent to [to],
|
|
498
|
+
and the reply read as the flow reads it. */
|
|
499
|
+
async function send(to, c, step, cell, record) {
|
|
500
|
+
const t0 = Date.now();
|
|
501
|
+
const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
|
|
502
|
+
const ctx = { scope: record.scope || {}, now: record.at || null };
|
|
503
|
+
let req;
|
|
504
|
+
try {
|
|
505
|
+
req = core.connectionSend(c, core.httpRequestOf(step, cell, ctx), to.base, to.key);
|
|
506
|
+
} catch (e) {
|
|
507
|
+
return said({ error: `the request could not be built -- ${e.message}` });
|
|
508
|
+
}
|
|
509
|
+
// A body that is text -- an Any endpoint flow's -- goes as it is written;
|
|
510
|
+
// anything else as JSON, the way the flow sent it.
|
|
511
|
+
const isText = typeof req.body === "string";
|
|
512
|
+
const payload = req.method === "GET" ? null : isText ? req.body : JSON.stringify(req.body);
|
|
513
|
+
const sent = `${req.method || "POST"} ${req.url}${payload == null ? "" : `\n\n${payload}`}`;
|
|
514
|
+
const done = more => said({ sent, ...more });
|
|
515
|
+
try {
|
|
516
|
+
const r = await fetch(req.url, {
|
|
517
|
+
method: req.method || "POST",
|
|
518
|
+
headers: { "Content-Type": isText ? "text/plain; charset=utf-8" : "application/json", ...req.headers },
|
|
519
|
+
redirect: "error",
|
|
520
|
+
signal: AbortSignal.timeout(seconds * 1000),
|
|
521
|
+
...(payload == null ? {} : { body: payload }),
|
|
522
|
+
});
|
|
523
|
+
const text = await r.text();
|
|
524
|
+
let j = text;
|
|
525
|
+
try { j = JSON.parse(text); } catch { /* the body as text */ }
|
|
526
|
+
if (!r.ok) return done({ error: j?.error?.message || (typeof j?.error === "string" ? j.error : null) || `HTTP ${r.status}` });
|
|
527
|
+
const read = core.httpReplyOf(step, cell, j, r.status, ctx);
|
|
528
|
+
return done({ raw: read.raw, said: read.said });
|
|
529
|
+
} catch (e) {
|
|
530
|
+
if (e?.name === "TimeoutError" || e?.name === "AbortError") {
|
|
531
|
+
return done({ error: `the endpoint gave no reply within ${seconds}s — timed out` });
|
|
532
|
+
}
|
|
533
|
+
return done({ error: String(e?.message || e) });
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
async function ask(to, c, prompt, dataUrl) {
|
|
538
|
+
const t0 = Date.now();
|
|
539
|
+
const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
|
|
540
|
+
try {
|
|
541
|
+
// The connection's type builds the whole request: the URL, the headers
|
|
542
|
+
// (Anthropic's key travels in x-api-key), and the body.
|
|
543
|
+
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
|
|
544
|
+
const r = await fetch(req.url, {
|
|
545
|
+
method: "POST",
|
|
546
|
+
headers: { "Content-Type": "application/json", ...req.headers },
|
|
547
|
+
signal: AbortSignal.timeout(seconds * 1000),
|
|
548
|
+
body: JSON.stringify(req.body),
|
|
549
|
+
});
|
|
550
|
+
const j = await r.json().catch(() => ({}));
|
|
551
|
+
if (!r.ok) return said({ error: j.error?.message || j.error || `HTTP ${r.status}` });
|
|
552
|
+
const { raw, finishReason } = core.connectionReply(c, j);
|
|
553
|
+
const error = core.connectionReplyProblem(c, raw, finishReason);
|
|
554
|
+
return said(error ? { raw, error } : { raw });
|
|
555
|
+
} catch (e) {
|
|
556
|
+
if (e?.name === "TimeoutError" || e?.name === "AbortError") {
|
|
557
|
+
return said({ error: `the model gave no reply within ${seconds}s — timed out` });
|
|
558
|
+
}
|
|
559
|
+
return said({ error: String(e?.message || e) });
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
// ---- The run -------------------------------------------------------------
|
|
564
|
+
|
|
565
|
+
// ---- A server run: what the queue hands the worker (#531) ------------------
|
|
566
|
+
// The run document carries everything a run was submitted with -- the
|
|
567
|
+
// scenarios, the profiles they resolved to, and the content: a file list into
|
|
568
|
+
// --source, in order and with its repeats, or the inline text. Each item runs
|
|
569
|
+
// through every scenario's jobs as an image does on the canvas; a
|
|
570
|
+
// Source deleted between submit and here shows as a failed run, and a file
|
|
571
|
+
// missing from it is an incomplete item, never a pass.
|
|
572
|
+
//
|
|
573
|
+
// Progress is JSON lines on a stream of its own -- --progress-file, never the
|
|
574
|
+
// report -- so the queue can tail it into the run row as items finish. The
|
|
575
|
+
// report is written whole at the end, from the results the run held plus what
|
|
576
|
+
// this invocation did, so resume and re-run item agree with a fresh run.
|
|
577
|
+
|
|
578
|
+
// What a file *is* to a run, the same registry §4 gives the graded set: an
|
|
579
|
+
// image is prepared to the profile's pixel budget; a text file is decoded
|
|
580
|
+
// UTF-8 and joined at {text}; anything else is passed through as UTF-8 with
|
|
581
|
+
// replacement and flagged, so the failure mode is stated rather than hidden.
|
|
582
|
+
const RUN_IMAGE_TYPES = new Set([".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff"]);
|
|
583
|
+
const RUN_TEXT_TYPES = new Set([".txt", ".md", ".csv"]);
|
|
584
|
+
|
|
585
|
+
function readUtf8(file) {
|
|
586
|
+
return fs.readFileSync(file).toString("utf8");
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
// Every test's reading of each scenario, keyed by the test's id, exactly as
|
|
590
|
+
// the page reads it: a whole-run test's verdict, settled once every item is
|
|
591
|
+
// in, and a per-item test's counts. [settled] is false for a run that
|
|
592
|
+
// stopped short, whose whole-run tests have not settled.
|
|
593
|
+
const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
|
|
594
|
+
Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
|
|
595
|
+
name: o.label, skipped: o.skipped,
|
|
596
|
+
...(o.whole
|
|
597
|
+
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
598
|
+
: { ran: o.ran, passed: o.passed, skippedItems: o.skippedItems }),
|
|
599
|
+
}])));
|
|
600
|
+
|
|
601
|
+
// The run as the report restates it: each scenario's stages with the
|
|
602
|
+
// connection each one asked.
|
|
603
|
+
const scenariosOf = run => run.scenarios.map((sc, i) => {
|
|
604
|
+
const { stages, connections } = core.stagesFor(run, i);
|
|
605
|
+
return {
|
|
606
|
+
n: i + 1, name: sc.name || null,
|
|
607
|
+
stages: stages.map((s, k) => {
|
|
608
|
+
const { id, ...connection } = connections[k];
|
|
609
|
+
return { n: k + 1, prompt: s.text, kind: s.kind, withImage: s.withImage, profile: id, connection };
|
|
610
|
+
}),
|
|
611
|
+
};
|
|
612
|
+
});
|
|
613
|
+
|
|
614
|
+
async function runSnapshot(o) {
|
|
615
|
+
const dataset = readDataset(o.dataset);
|
|
616
|
+
const run = readRun(o.run, dataset);
|
|
617
|
+
// One item a pipeline holds itself (Text, or Prompt only's bare one), or
|
|
618
|
+
// the Source's files.
|
|
619
|
+
const content = core.contentOf(run);
|
|
620
|
+
const inline = core.CONTENT_TYPES[content.type];
|
|
621
|
+
const bare = !!inline?.bare;
|
|
622
|
+
const text = inline?.inline && !bare ? inline.text?.(content, {}) ?? "" : "";
|
|
623
|
+
const files = content.type === "source" ? content.files || [] : [];
|
|
624
|
+
if (!bare && !files.length && !text.trim()) {
|
|
625
|
+
broken(`${o.run} names no files and no text, so there is nothing to run`);
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
// The dataset the run was submitted against, for scoring its items.
|
|
629
|
+
const graded = gradedBy(dataset);
|
|
630
|
+
|
|
631
|
+
const items = bare ? [{ name: null, kind: "text", text: null, bare: true }]
|
|
632
|
+
: text.trim() ? [{ name: null, kind: "text", text }]
|
|
633
|
+
: files.map(name => {
|
|
634
|
+
const ext = path.extname(name).toLowerCase();
|
|
635
|
+
if (RUN_IMAGE_TYPES.has(ext)) return { name, kind: "image" };
|
|
636
|
+
// A record (flows/record.ts): one call of a Power Automate step, what
|
|
637
|
+
// an HTTP Request is rebuilt from.
|
|
638
|
+
if (ext === ".json") return { name, kind: "record" };
|
|
639
|
+
return { name, kind: "text", unmapped: !RUN_TEXT_TYPES.has(ext) };
|
|
640
|
+
});
|
|
641
|
+
const total = items.length;
|
|
642
|
+
// Only the run's own image items need ImageMagick: a text-only Source runs
|
|
643
|
+
// with nothing on PATH but the node running this.
|
|
644
|
+
const bin = items.some(it => it.kind === "image") ? imageMagick() : null;
|
|
645
|
+
if (items.some(it => it.kind === "image") && !bin) {
|
|
646
|
+
broken("a run over images needs ImageMagick, to cap each image to the pixel budget the tab caps it to on a canvas.");
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
// Each scenario once: its stages, the jobs' token sets, and a transport
|
|
650
|
+
// per stage to the profile that stage resolves to, keyed by that
|
|
651
|
+
// profile's id.
|
|
652
|
+
const plans = run.scenarios.map((_, i) => {
|
|
653
|
+
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
654
|
+
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
655
|
+
return { stages, tokens, connections, links, calls, cells };
|
|
656
|
+
});
|
|
657
|
+
|
|
658
|
+
let results = [];
|
|
659
|
+
if (o.resultsFile) {
|
|
660
|
+
try { results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8")); } catch { /* fresh run */ }
|
|
661
|
+
// Kept as they were written; read as today's.
|
|
662
|
+
results = Array.isArray(results) ? core.upgradeResults(results) : [];
|
|
663
|
+
}
|
|
664
|
+
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
665
|
+
const from = only != null ? only : (o.from != null ? parseInt(o.from, 10) : 0);
|
|
666
|
+
|
|
667
|
+
const progressFd = o.progressFile ? fs.openSync(o.progressFile, "a") : null;
|
|
668
|
+
const progress = obj => { if (progressFd != null) fs.writeSync(progressFd, JSON.stringify(obj) + "\n"); };
|
|
669
|
+
|
|
670
|
+
const unrun = [];
|
|
671
|
+
let cancelled = false;
|
|
672
|
+
|
|
673
|
+
for (let i = 0; i < total; i++) {
|
|
674
|
+
if (only == null && i < from) continue;
|
|
675
|
+
if (o.cancelFile && fs.existsSync(o.cancelFile)) { cancelled = true; break; }
|
|
676
|
+
const item = items[i];
|
|
677
|
+
// Said before the work, so a reader knows what is in flight rather than
|
|
678
|
+
// what last finished; the row below it is the same item, done.
|
|
679
|
+
progress({ event: "start", item: i, name: item.name });
|
|
680
|
+
const t0 = Date.now();
|
|
681
|
+
const sides = [];
|
|
682
|
+
let production = null;
|
|
683
|
+
let failed = null;
|
|
684
|
+
for (const plan of plans) {
|
|
685
|
+
let dataUrl = null;
|
|
686
|
+
if (item.kind === "image") {
|
|
687
|
+
const file = path.join(o.source, item.name);
|
|
688
|
+
if (!fs.existsSync(file)) { failed = `the file ${item.name} is not in the source`; break; }
|
|
689
|
+
try {
|
|
690
|
+
({ dataUrl } = prepare(bin, file, plan.connections[0]));
|
|
691
|
+
} catch (e) {
|
|
692
|
+
failed = `${item.name} could not be prepared — ${e.message}`;
|
|
693
|
+
break;
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
let record = null;
|
|
697
|
+
if (item.kind === "record") {
|
|
698
|
+
const file = path.join(o.source, item.name);
|
|
699
|
+
try { record = JSON.parse(readUtf8(file)); } catch (e) {
|
|
700
|
+
failed = fs.existsSync(file) ? `${item.name} is not a record -- ${e.message}` : `the file ${item.name} is not in the source`;
|
|
701
|
+
break;
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
const textOf = item.kind === "text" && !item.bare
|
|
705
|
+
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
706
|
+
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
707
|
+
{ tokens: plan.tokens, text: textOf });
|
|
708
|
+
// A case is found by its file's name, whatever kind of file it is: a
|
|
709
|
+
// text item a dataset grades is graded like an image.
|
|
710
|
+
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
711
|
+
production = core.productionOf(run, record);
|
|
712
|
+
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res, { production, grader: graderFor(run) }) });
|
|
713
|
+
}
|
|
714
|
+
// Production's reply to the item, where its record holds one: what a
|
|
715
|
+
// metric compares with, kept so a re-score reads it again.
|
|
716
|
+
const out = { item: i, name: item.name, kind: item.kind, ...(production != null ? { production } : {}),
|
|
717
|
+
unmapped: item.unmapped || null, ms: Date.now() - t0,
|
|
718
|
+
scenarios: sides, ...(failed ? { unrun: failed } : {}) };
|
|
719
|
+
results[i] = out;
|
|
720
|
+
progress(out);
|
|
721
|
+
if (failed) unrun.push(`${item.name || "the text"}: ${failed}`);
|
|
722
|
+
}
|
|
723
|
+
if (progressFd != null) fs.closeSync(progressFd);
|
|
724
|
+
|
|
725
|
+
const report = {
|
|
726
|
+
runner: "tools/prompt-lab/run-evals.js",
|
|
727
|
+
ranAt: new Date().toISOString(),
|
|
728
|
+
verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
|
|
729
|
+
run: {
|
|
730
|
+
name: run.name ?? null,
|
|
731
|
+
tests: run.tests,
|
|
732
|
+
verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
|
|
733
|
+
scenarios: scenariosOf(run),
|
|
734
|
+
items: results.slice(0, total),
|
|
735
|
+
},
|
|
736
|
+
unrun,
|
|
737
|
+
};
|
|
738
|
+
|
|
739
|
+
const say = o.json === "-" ? console.error : console.log;
|
|
740
|
+
say(`${report.runner} — run ${report.run.name ?? ""}`);
|
|
741
|
+
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
742
|
+
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
743
|
+
say(`${results.filter(Boolean).length} of ${total} items ran · ${cancelled ? "cancelled" : report.verdict}`);
|
|
744
|
+
|
|
745
|
+
if (o.json) {
|
|
746
|
+
const text2 = JSON.stringify(report, null, 2);
|
|
747
|
+
if (o.json === "-") process.stdout.write(text2 + "\n");
|
|
748
|
+
else fs.writeFileSync(o.json, text2 + "\n");
|
|
749
|
+
}
|
|
750
|
+
process.exitCode = cancelled ? EXIT.passed : (unrun.length ? EXIT.incomplete : EXIT.passed);
|
|
751
|
+
}
|
|
752
|
+
|
|
753
|
+
/** Re-score a run's stored results against the dataset --dataset hands over,
|
|
754
|
+
* keeping the replies that were stored: what changes a verdict is the scorer
|
|
755
|
+
* or the cases, never a new question to the model. The report is shaped like
|
|
756
|
+
* a --run's, with the stored results as its items, and a whole-run test's
|
|
757
|
+
* verdict settled the same way. */
|
|
758
|
+
async function runRescore(o){
|
|
759
|
+
const dataset = readDataset(o.dataset);
|
|
760
|
+
const run = readRun(o.run, dataset);
|
|
761
|
+
let results;
|
|
762
|
+
try {
|
|
763
|
+
results = JSON.parse(fs.readFileSync(o.resultsFile, "utf8"));
|
|
764
|
+
} catch (e) {
|
|
765
|
+
broken(`${e.path ?? "a read"}: ${e.message}`);
|
|
766
|
+
}
|
|
767
|
+
if (!Array.isArray(results)) {
|
|
768
|
+
broken(`${o.resultsFile}: a run's results are a list of items`);
|
|
769
|
+
}
|
|
770
|
+
results = core.upgradeResults(results);
|
|
771
|
+
|
|
772
|
+
// The cases the run's items are scored against. A file the set does
|
|
773
|
+
// not grade keeps its reply and has no score -- honestly: nothing else would
|
|
774
|
+
// say what the set covers.
|
|
775
|
+
const graded = gradedBy(dataset);
|
|
776
|
+
|
|
777
|
+
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
778
|
+
// A test that reads every item (the Metrics) re-reads one with no case
|
|
779
|
+
// too, against the production reply the item kept.
|
|
780
|
+
const items = await Promise.all(results.map(async (it, i) => {
|
|
781
|
+
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
782
|
+
const kase = isText(it.name) ? graded.get(it.name) ?? null : null;
|
|
783
|
+
const more = { production: it.production ?? null, grader: graderFor(run) };
|
|
784
|
+
return { ...it, scenarios: await Promise.all(it.scenarios.map(async side =>
|
|
785
|
+
(isObject(side) && side.res)
|
|
786
|
+
? { ...side, scores: await core.itemScoresAsync(run, kase, side.res, more) } : side)) };
|
|
787
|
+
}));
|
|
788
|
+
|
|
789
|
+
const report = {
|
|
790
|
+
runner: "tools/prompt-lab/run-evals.js",
|
|
791
|
+
ranAt: new Date().toISOString(),
|
|
792
|
+
verdict: "done",
|
|
793
|
+
rescored: true,
|
|
794
|
+
run: {
|
|
795
|
+
name: run.name ?? null,
|
|
796
|
+
tests: run.tests,
|
|
797
|
+
verdicts: verdictsOf(run, items, true),
|
|
798
|
+
scenarios: scenariosOf(run),
|
|
799
|
+
items,
|
|
800
|
+
},
|
|
801
|
+
};
|
|
802
|
+
|
|
803
|
+
if (o.json) {
|
|
804
|
+
const text = JSON.stringify(report, null, 2);
|
|
805
|
+
if (o.json === "-") process.stdout.write(text + "\n");
|
|
806
|
+
else fs.writeFileSync(o.json, text + "\n");
|
|
807
|
+
}
|
|
808
|
+
process.exitCode = EXIT.passed;
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
/**
|
|
812
|
+
* The run --prompt, --model and --url describe, as a run document: one job
|
|
813
|
+
* that sees the image and answers in the default kind, one scenario,
|
|
814
|
+
* graded against --dataset, asking --prompt. Its profile has no name, so the transcript names no
|
|
815
|
+
* connection, as it never did, and its key is $EVAL_API_KEY.
|
|
816
|
+
*/
|
|
817
|
+
function cliRun(o, dataset) {
|
|
818
|
+
let doc = core.blankPipeline();
|
|
819
|
+
// Graded, so the job answers with a list: the configuration that reads a
|
|
820
|
+
// reply as this command always has (legacyTagsOut), under the dataset's
|
|
821
|
+
// rules. A --pipeline states its own.
|
|
822
|
+
doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
|
|
823
|
+
if (o.tokens) {
|
|
824
|
+
try {
|
|
825
|
+
doc.jobs[0] = core.withCall(doc.jobs[0], { tokenMappings: JSON.parse(fs.readFileSync(o.tokens, "utf8")) });
|
|
826
|
+
} catch (e) {
|
|
827
|
+
broken(`${o.tokens}: ${e.message}`);
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
831
|
+
doc.scenarios = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
832
|
+
stages: [{ prompt: o.prompt ?? "" }] }];
|
|
833
|
+
doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
|
|
834
|
+
mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
|
|
835
|
+
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
836
|
+
return doc;
|
|
837
|
+
}
|
|
838
|
+
|
|
839
|
+
// The plugins a run is read under, registered before anything reads it: a
|
|
840
|
+
// --run's recorded list, at the versions it recorded, or every plugin in the
|
|
841
|
+
// folder. A plugin that will not load stops the run, naming it -- a run read
|
|
842
|
+
// without the code it was submitted under grades something else.
|
|
843
|
+
async function loadPlugins(o) {
|
|
844
|
+
if (!o.plugins) return;
|
|
845
|
+
const { pathToFileURL } = require("url");
|
|
846
|
+
let wanted = null;
|
|
847
|
+
if (o.run) {
|
|
848
|
+
try { wanted = JSON.parse(fs.readFileSync(o.run, "utf8")).plugins ?? []; } catch { wanted = []; }
|
|
849
|
+
}
|
|
850
|
+
const dirs = wanted
|
|
851
|
+
? wanted.map(p => path.join(o.plugins, String(p.id), String(p.version)))
|
|
852
|
+
: fs.readdirSync(o.plugins, { withFileTypes: true }).filter(d => d.isDirectory() && !d.name.startsWith("tmp"))
|
|
853
|
+
.flatMap(d => fs.readdirSync(path.join(o.plugins, d.name)).map(v => path.join(o.plugins, d.name, v)));
|
|
854
|
+
for (const dir of dirs) {
|
|
855
|
+
let manifest;
|
|
856
|
+
try {
|
|
857
|
+
manifest = JSON.parse(fs.readFileSync(path.join(dir, "manifest.json"), "utf8"));
|
|
858
|
+
const mod = await import(pathToFileURL(path.join(dir, manifest.entry)).href);
|
|
859
|
+
await (mod.default ?? mod.register)(core.pluginHost(manifest.id));
|
|
860
|
+
} catch (e) {
|
|
861
|
+
broken(`the plugin in ${dir} did not load: ${e.message}`);
|
|
862
|
+
}
|
|
863
|
+
}
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
async function main() {
|
|
867
|
+
const o = parseArgs(process.argv.slice(2));
|
|
868
|
+
await loadPlugins(o);
|
|
869
|
+
if (o.rescore) {
|
|
870
|
+
if (!o.run || !o.resultsFile) {
|
|
871
|
+
broken("--rescore re-scores a stored run, so it needs --run and --results-file, "
|
|
872
|
+
+ "and for a graded run the dataset it scores against, --dataset.");
|
|
873
|
+
}
|
|
874
|
+
if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies
|
|
875
|
+
|| o.samples || o.source || o.files || o.resume || o.item || o.progress) {
|
|
876
|
+
broken("--rescore re-scores what the run already holds, so it cannot be given "
|
|
877
|
+
+ "with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples, --source, "
|
|
878
|
+
+ "--files, --resume, --item or --progress.");
|
|
879
|
+
}
|
|
880
|
+
return runRescore(o);
|
|
881
|
+
}
|
|
882
|
+
if (o.run) {
|
|
883
|
+
if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
|
|
884
|
+
broken("--run names everything a run takes -- scenarios, files and text -- so it "
|
|
885
|
+
+ "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
|
|
886
|
+
}
|
|
887
|
+
const n = Number(o.timeout ?? REQUEST_CAP);
|
|
888
|
+
if (!(n > 0)) broken(`--timeout has to be a number of seconds above zero`);
|
|
889
|
+
seconds = Math.min(n, REQUEST_CAP);
|
|
890
|
+
return runSnapshot(o);
|
|
891
|
+
}
|
|
892
|
+
if (o.pipeline && (o.prompt != null || o.tokens || o.model || o.url)) {
|
|
893
|
+
broken("--pipeline names every stage's prompt, tokens and connection, so it cannot be "
|
|
894
|
+
+ "given with --prompt, --prompt-file, --tokens, --model or --url.");
|
|
895
|
+
}
|
|
896
|
+
if (o.pipeline && o.files) {
|
|
897
|
+
broken("--pipeline's run document carries its file list in content.files, so --files "
|
|
898
|
+
+ "would say it twice.");
|
|
899
|
+
}
|
|
900
|
+
if (o.samples && o.source) {
|
|
901
|
+
broken("--samples and --source name the same directory, so both together "
|
|
902
|
+
+ "say two things.");
|
|
903
|
+
}
|
|
904
|
+
if (o.resume && o.item) {
|
|
905
|
+
broken("--resume and --item say different things about what to run.");
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
// The dataset the run is graded against. Everything below grades, so there
|
|
909
|
+
// is no run without one.
|
|
910
|
+
if (!o.dataset) broken(`--dataset names the dataset to grade against.\n\n${USAGE}`);
|
|
911
|
+
const dataset = readDataset(o.dataset);
|
|
912
|
+
|
|
913
|
+
// The run to grade. A --pipeline document is one scenario graded against
|
|
914
|
+
// its dataset; without one it is the stage this script always ran -- the
|
|
915
|
+
// prompt, seeing the image, on --url with $EVAL_API_KEY -- stated as
|
|
916
|
+
// the same kind of document, so both go through one reading of it.
|
|
917
|
+
if (!o.pipeline && !(o.prompt ?? "").trim()) {
|
|
918
|
+
broken("--prompt or --prompt-file names the prompt to grade: a dataset holds none, "
|
|
919
|
+
+ "the lab's Prompt library does.");
|
|
920
|
+
}
|
|
921
|
+
const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
|
|
922
|
+
if (o.pipeline) {
|
|
923
|
+
if (run.scenarios.length !== 1) {
|
|
924
|
+
broken(`${o.pipeline} has ${run.scenarios.length} scenarios, and --pipeline grades one `
|
|
925
|
+
+ "against the set -- --run runs them all.");
|
|
926
|
+
}
|
|
927
|
+
if (!core.testsDataset(run)) {
|
|
928
|
+
broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
|
|
929
|
+
}
|
|
930
|
+
if (core.contentOf(run).type !== "source") {
|
|
931
|
+
broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
|
|
932
|
+
}
|
|
933
|
+
}
|
|
934
|
+
// Case by case: each item against its case, as the case metric of the
|
|
935
|
+
// test that names the dataset reads it (scoreCase).
|
|
936
|
+
const { stages, tokens, connections } = core.stagesFor(run, 0);
|
|
937
|
+
const prompt = core.resolvePrompt(stages[0].text, core.tokenSet(tokens, 0));
|
|
938
|
+
|
|
939
|
+
// The graded half, through the function the tab builds its own list with.
|
|
940
|
+
const set = core.gradedSetFrom(dataset);
|
|
941
|
+
let graded = set.filter(c => !c.todo);
|
|
942
|
+
|
|
943
|
+
const limit = (() => {
|
|
944
|
+
// The relay's cap applies no matter what the caller asked for: a request
|
|
945
|
+
// that outlives it is a failed item rather than a model that mutes and
|
|
946
|
+
// holds a single-flight queue for good.
|
|
947
|
+
const n = Number(o.timeout ?? REQUEST_CAP);
|
|
948
|
+
if (!(n > 0)) broken(`--timeout has to be a number of seconds above zero`);
|
|
949
|
+
return Math.min(n, REQUEST_CAP);
|
|
950
|
+
})();
|
|
951
|
+
seconds = limit;
|
|
952
|
+
|
|
953
|
+
const itemsOnly = (() => {
|
|
954
|
+
if (!o.item) return null;
|
|
955
|
+
const hit = graded.filter(c => c.id === o.item);
|
|
956
|
+
if (!hit.length) broken(`no graded case is named ${o.item}`);
|
|
957
|
+
return hit;
|
|
958
|
+
})();
|
|
959
|
+
|
|
960
|
+
// The earlier report a resume carries its finished items from. Both shapes
|
|
961
|
+
// a caller can leave behind: a full --json report, or the bare list of
|
|
962
|
+
// scores of rows. The earlier attempt's unrun is deliberately not honoured:
|
|
963
|
+
// those are the unfinished items a resume exists to run.
|
|
964
|
+
const carried = (() => {
|
|
965
|
+
if (!o.resume) return null;
|
|
966
|
+
let doc;
|
|
967
|
+
try {
|
|
968
|
+
doc = JSON.parse(fs.readFileSync(o.resume, "utf8"));
|
|
969
|
+
} catch (e) {
|
|
970
|
+
broken(`${o.resume}: ${e.message}`);
|
|
971
|
+
}
|
|
972
|
+
const rows = Array.isArray(doc) ? doc : doc?.cases;
|
|
973
|
+
if (!Array.isArray(rows) || !rows.every(r => isObject(r) && isText(r.id))) {
|
|
974
|
+
broken(`${o.resume}: a report carries its results under "cases" as a list of rows`);
|
|
975
|
+
}
|
|
976
|
+
const out = new Map();
|
|
977
|
+
for (const r of rows) if (!out.has(r.id)) out.set(r.id, r);
|
|
978
|
+
return out;
|
|
979
|
+
})();
|
|
980
|
+
|
|
981
|
+
// The snapshot a run was submitted with: only files it names are read, so
|
|
982
|
+
// files added to the Source afterwards cannot join midway and a report
|
|
983
|
+
// names exactly what it graded. A run document's own list is that
|
|
984
|
+
// snapshot; an empty one names none, and every graded case runs.
|
|
985
|
+
const snapshotSet = (() => {
|
|
986
|
+
if (o.pipeline) return core.contentOf(run).files?.length ? new Set(core.contentOf(run).files) : null;
|
|
987
|
+
if (!o.files) return null;
|
|
988
|
+
let doc;
|
|
989
|
+
try {
|
|
990
|
+
doc = JSON.parse(fs.readFileSync(o.files, "utf8"));
|
|
991
|
+
} catch (e) {
|
|
992
|
+
broken(`${o.files}: ${e.message}`);
|
|
993
|
+
}
|
|
994
|
+
const list = Array.isArray(doc) ? doc : isObject(doc) && Array.isArray(doc.files) ? doc.files : null;
|
|
995
|
+
if (!list || !list.every(isText)) {
|
|
996
|
+
broken(`${o.files}: a file-list snapshot is a list of names, or {"files": [...]}`);
|
|
997
|
+
}
|
|
998
|
+
return new Set(list);
|
|
999
|
+
})();
|
|
1000
|
+
|
|
1001
|
+
let replies = null, bin = "", samples = "", links = [];
|
|
1002
|
+
if (o.replies) {
|
|
1003
|
+
try {
|
|
1004
|
+
replies = JSON.parse(fs.readFileSync(o.replies, "utf8"));
|
|
1005
|
+
} catch (e) {
|
|
1006
|
+
broken(`${o.replies}: ${e.message}`);
|
|
1007
|
+
}
|
|
1008
|
+
} else {
|
|
1009
|
+
if (!o.pipeline && !o.model) broken(`--model names the model to grade.\n\n${USAGE}`);
|
|
1010
|
+
links = connections.map((c, i) => {
|
|
1011
|
+
if (!String(c.model || "").trim()) broken(`stage ${i + 1}'s connection names no model.`);
|
|
1012
|
+
const to = reach(c, o.pipeline ? core.keyVar(c.id) : "EVAL_API_KEY");
|
|
1013
|
+
// Once here rather than once per image: a llama.cpp stage on a
|
|
1014
|
+
// hosted model is a request that cannot be built at all.
|
|
1015
|
+
try {
|
|
1016
|
+
core.connectionRequest(c, "", null, to.base, to.key);
|
|
1017
|
+
} catch (e) {
|
|
1018
|
+
broken(`stage ${i + 1}: ${e.message}`);
|
|
1019
|
+
}
|
|
1020
|
+
return to;
|
|
1021
|
+
});
|
|
1022
|
+
samples = o.samples || o.source || firstFile(path.join(HERE, "samples"));
|
|
1023
|
+
if (!samples) broken("no images: there is no samples/ beside this script. "
|
|
1024
|
+
+ "--samples or --source names a directory.");
|
|
1025
|
+
const allowed = itemsOnly ?? graded;
|
|
1026
|
+
// $snapshotSet above: only a Source run sees it; --replies reads no file.
|
|
1027
|
+
const inSnapshot = c => !snapshotSet || snapshotSet.has(c.filename);
|
|
1028
|
+
// Only the run's own image items need ImageMagick: a text-only Source
|
|
1029
|
+
// -- the .txt, .csv case -- runs with nothing on PATH but the node
|
|
1030
|
+
// running this, which is the point of having text in the registry.
|
|
1031
|
+
const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.filename)) === "image");
|
|
1032
|
+
if (pictures.length) {
|
|
1033
|
+
bin = imageMagick();
|
|
1034
|
+
if (!bin) broken("a run over images needs ImageMagick, to cap each "
|
|
1035
|
+
+ "one to the pixel budget the tab caps it to on a canvas. "
|
|
1036
|
+
+ "Install it, or use --replies to score recorded replies "
|
|
1037
|
+
+ "without a model.");
|
|
1038
|
+
}
|
|
1039
|
+
}
|
|
1040
|
+
|
|
1041
|
+
// ---- Progress: items on a stream of their own ---------------------------
|
|
1042
|
+
// §5's JSON lines, one per finished item, for a caller to tail. Which
|
|
1043
|
+
// stream is decided by where the final report goes: if the report is on
|
|
1044
|
+
// stdout, the lines take the other one, so neither can swallow the other.
|
|
1045
|
+
// With --progress the human report moves to stderr regardless, so a line
|
|
1046
|
+
// on stdout is an item and nothing else.
|
|
1047
|
+
const progress = o.progress ? (o.json === "-" ? process.stderr : process.stdout) : null;
|
|
1048
|
+
|
|
1049
|
+
const rows = [];
|
|
1050
|
+
const unrun = [];
|
|
1051
|
+
|
|
1052
|
+
// The work: everything graded except a single-item run's one case, and,
|
|
1053
|
+
// on a resume, minus the earlier attempt's finished items. Those are
|
|
1054
|
+
// carried whole -- their verdicts already happened and a resume that
|
|
1055
|
+
// recomputed them would be deciding the numbers twice.
|
|
1056
|
+
const toRun = (itemsOnly ?? graded).filter(c => !carried?.has(c.id));
|
|
1057
|
+
if (carried) {
|
|
1058
|
+
for (const [id, r] of carried) {
|
|
1059
|
+
if (graded.some(c => c.id === id)) rows.push(r);
|
|
1060
|
+
}
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1063
|
+
// Only files the snapshot names are read from the Source.
|
|
1064
|
+
const wanted = c => !snapshotSet || snapshotSet.has(c.filename);
|
|
1065
|
+
// The cancel marker is checked between items, so a cancel never loses the
|
|
1066
|
+
// reply in flight: the item that was being asked finishes, and the rest
|
|
1067
|
+
// are unrun rather than half-asked.
|
|
1068
|
+
let cancelled = false, done = 0;
|
|
1069
|
+
for (const kase of toRun) {
|
|
1070
|
+
if (o.cancelFile && fs.existsSync(o.cancelFile)) { cancelled = true; break; }
|
|
1071
|
+
done++;
|
|
1072
|
+
let item, calls;
|
|
1073
|
+
if (replies) {
|
|
1074
|
+
const got = replies[kase.id];
|
|
1075
|
+
const said = isText(got) ? [got] : got;
|
|
1076
|
+
if (!Array.isArray(said) || !said.every(isText)) {
|
|
1077
|
+
unrun.push(`${kase.id}: no recorded reply`);
|
|
1078
|
+
continue;
|
|
1079
|
+
}
|
|
1080
|
+
if (said.length !== stages.length) {
|
|
1081
|
+
unrun.push(`${kase.id}: ${said.length} recorded repl${said.length === 1 ? "y" : "ies"} `
|
|
1082
|
+
+ `for ${stages.length} stages`);
|
|
1083
|
+
continue;
|
|
1084
|
+
}
|
|
1085
|
+
calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
|
|
1086
|
+
} else {
|
|
1087
|
+
if (!wanted(kase)) {
|
|
1088
|
+
unrun.push(`${kase.id}: ${kase.filename} is not in the run's file list`);
|
|
1089
|
+
continue;
|
|
1090
|
+
}
|
|
1091
|
+
const file = path.join(samples, kase.filename);
|
|
1092
|
+
if (!fs.existsSync(file)) {
|
|
1093
|
+
unrun.push(`${kase.id}: ${kase.filename} is not in ${samples}`);
|
|
1094
|
+
continue;
|
|
1095
|
+
}
|
|
1096
|
+
try {
|
|
1097
|
+
item = itemContent(bin, file, connections[0]);
|
|
1098
|
+
} catch (e) {
|
|
1099
|
+
unrun.push(`${kase.id}: ${kase.filename} could not be prepared — ${e.message}`);
|
|
1100
|
+
continue;
|
|
1101
|
+
}
|
|
1102
|
+
calls = callsFor(connections, links, item?.text ?? null);
|
|
1103
|
+
}
|
|
1104
|
+
|
|
1105
|
+
const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
|
|
1106
|
+
{ tokens, text: item?.text ?? null });
|
|
1107
|
+
const s = core.scoreCase(kase, res);
|
|
1108
|
+
// A file with no mapper ran anyway; the flag keeps the transcript honest
|
|
1109
|
+
// about what the model was actually shown. §4: stated, not hidden.
|
|
1110
|
+
if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
|
|
1111
|
+
rows.push({
|
|
1112
|
+
id: kase.id, filename: kase.filename, half: kase.half,
|
|
1113
|
+
pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
|
|
1114
|
+
found: s.found, missed: s.missed, invented: s.invented, unmet: s.unmet,
|
|
1115
|
+
under: s.under, over: s.over, count: s.count,
|
|
1116
|
+
watchFound: s.watchFound, watchTotal: s.watchTotal,
|
|
1117
|
+
terms: res.terms || [], ms: res.ms,
|
|
1118
|
+
// What a file type was to this run -- "image", "text", "unmapped".
|
|
1119
|
+
type: item?.kind ?? null,
|
|
1120
|
+
// What the model actually said. #248 wants a failure written out as a
|
|
1121
|
+
// task an agent can act on, and "missed water" without the reply that
|
|
1122
|
+
// missed it is not one.
|
|
1123
|
+
reply: res.transcript?.[res.transcript.length - 1]?.got ?? "",
|
|
1124
|
+
// Every stage's prompt, reply and time, as the page's transcript shows
|
|
1125
|
+
// them: the final reply alone cannot say which stage went wrong.
|
|
1126
|
+
transcript: res.transcript,
|
|
1127
|
+
});
|
|
1128
|
+
if (progress) {
|
|
1129
|
+
progress.write(JSON.stringify({
|
|
1130
|
+
event: "item", id: kase.id, filename: kase.filename, n: rows.length,
|
|
1131
|
+
pass: s.pass, score: s.score, discarded: s.discarded,
|
|
1132
|
+
error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
|
|
1133
|
+
ms: res.ms,
|
|
1134
|
+
}) + "\n");
|
|
1135
|
+
}
|
|
1136
|
+
}
|
|
1137
|
+
// A cancelled run says which items never ran, so the report names what it
|
|
1138
|
+
// did not cover rather than letting the tail guess.
|
|
1139
|
+
if (cancelled) {
|
|
1140
|
+
for (const kase of toRun.slice(done)) {
|
|
1141
|
+
unrun.push(`${kase.id}: not run — cancelled after the item in flight`);
|
|
1142
|
+
}
|
|
1143
|
+
}
|
|
1144
|
+
|
|
1145
|
+
// The sum is made here, from the rows, in one place: a fresh run's rows
|
|
1146
|
+
// and a resume's carried ones are scored the same way, and no number is
|
|
1147
|
+
// worked out twice.
|
|
1148
|
+
const tally = core.emptyTally();
|
|
1149
|
+
for (const r of rows) {
|
|
1150
|
+
tally.ran++;
|
|
1151
|
+
if (r.pass) tally.passed++;
|
|
1152
|
+
tally.found += r.found.length;
|
|
1153
|
+
tally.of += r.found.length + r.missed.length + r.invented.length;
|
|
1154
|
+
}
|
|
1155
|
+
|
|
1156
|
+
// Anything short of the whole graded set is not a result. This is the
|
|
1157
|
+
// constraint in the issue: a set that did not run must not read as one that
|
|
1158
|
+
// passed, and 49 items absent with the fiftieth green is exactly how
|
|
1159
|
+
// that happens.
|
|
1160
|
+
const complete = graded.length > 0 && unrun.length === 0;
|
|
1161
|
+
const verdict = !complete ? "incomplete" : tally.passed === tally.ran ? "pass" : "fail";
|
|
1162
|
+
|
|
1163
|
+
const report = {
|
|
1164
|
+
runner: "tools/prompt-lab/run-evals.js",
|
|
1165
|
+
ranAt: new Date().toISOString(),
|
|
1166
|
+
dataset: inRepo(o.dataset),
|
|
1167
|
+
prompt,
|
|
1168
|
+
...(o.pipeline ? { pipeline: {
|
|
1169
|
+
file: inRepo(o.pipeline), name: run.name || null,
|
|
1170
|
+
stages: stages.map((s, i) => {
|
|
1171
|
+
const { id, ...connection } = connections[i];
|
|
1172
|
+
return { n: i + 1, kind: s.kind, withImage: s.withImage,
|
|
1173
|
+
prompt: core.resolvePrompt(s.text, core.tokenSet(tokens, i)), profile: id, connection };
|
|
1174
|
+
}),
|
|
1175
|
+
} } : {}),
|
|
1176
|
+
// Which variable a key came from, never the key.
|
|
1177
|
+
transport: replies ? { kind: "replies", file: o.replies }
|
|
1178
|
+
: o.pipeline ? { kind: "model", stages: links.map((to, i) => ({
|
|
1179
|
+
n: i + 1, connection: to.name, model: connections[i].model,
|
|
1180
|
+
endpoint: to.base, keyFrom: to.key ? to.from : null })) }
|
|
1181
|
+
: { kind: "model", model: o.model, endpoint: links[0].base },
|
|
1182
|
+
// How long a request could run, as given; a higher ask is capped at
|
|
1183
|
+
// the relay's 600.
|
|
1184
|
+
timeout: seconds,
|
|
1185
|
+
// The modes a server-side run reaches the worker through.
|
|
1186
|
+
...(o.item ? { item: o.item } : {}),
|
|
1187
|
+
...(o.resume ? { resume: o.resume, carried: carried.size } : {}),
|
|
1188
|
+
...(snapshotSet ? { files: o.files || inRepo(o.pipeline),
|
|
1189
|
+
filesListed: snapshotSet.size } : {}),
|
|
1190
|
+
...(o.source ? { source: inRepo(o.source) } : {}),
|
|
1191
|
+
verdict,
|
|
1192
|
+
graded: (itemsOnly ?? graded).length,
|
|
1193
|
+
ungraded: set.length - graded.length,
|
|
1194
|
+
ran: tally.ran,
|
|
1195
|
+
passed: tally.passed,
|
|
1196
|
+
failed: tally.ran - tally.passed,
|
|
1197
|
+
unrun,
|
|
1198
|
+
terms: { found: tally.found, of: tally.of, percent: core.tallyPercent(tally) },
|
|
1199
|
+
cases: rows,
|
|
1200
|
+
};
|
|
1201
|
+
|
|
1202
|
+
const toStdout = o.json === "-";
|
|
1203
|
+
// With --progress the chosen stream stays pure JSON, so what was prose
|
|
1204
|
+
// moves to stderr whether the report took stdout or not.
|
|
1205
|
+
const say = o.progress ? console.error : toStdout ? console.error : console.log;
|
|
1206
|
+
const pct = core.tallyPercent(tally);
|
|
1207
|
+
|
|
1208
|
+
say(`${report.runner} — ${report.dataset}`);
|
|
1209
|
+
if (o.pipeline) {
|
|
1210
|
+
say(`pipeline ${report.pipeline.file}${run.name ? ` — ${run.name}` : ""}`);
|
|
1211
|
+
if (replies) say(`recorded replies from ${o.replies}`);
|
|
1212
|
+
for (const [i, s] of report.pipeline.stages.entries()) {
|
|
1213
|
+
const to = links[i];
|
|
1214
|
+
say(`stage ${s.n} · ${s.kind} · ${s.withImage ? "with the image" : "text only"}`
|
|
1215
|
+
+ (to ? ` · ${to.name}: ${s.connection.model} at ${to.base}`
|
|
1216
|
+
+ (to.key ? ` · key from $${to.from}` : "") : ""));
|
|
1217
|
+
say(` ${s.prompt.replace(/\n/g, "\n ")}`);
|
|
1218
|
+
}
|
|
1219
|
+
} else {
|
|
1220
|
+
say(replies ? `recorded replies from ${o.replies}`
|
|
1221
|
+
: `${o.model} at ${links[0].base}`);
|
|
1222
|
+
say(`prompt: ${prompt}`);
|
|
1223
|
+
}
|
|
1224
|
+
say("");
|
|
1225
|
+
for (const r of rows.filter(r => !r.pass)) {
|
|
1226
|
+
const n = r.score == null ? " —" : `${String(Math.round(r.score * 100)).padStart(3)}%`;
|
|
1227
|
+
say(`fail ${r.id.padEnd(16)} ${n} ${r.reasons.join(" · ")}`);
|
|
1228
|
+
}
|
|
1229
|
+
// Capped: a run where nothing was found says so in one screen, and the
|
|
1230
|
+
// JSON report carries all of them for anything that needs the list.
|
|
1231
|
+
for (const u of unrun.slice(0, 10)) say(`---- ${u}`);
|
|
1232
|
+
if (unrun.length > 10) say(`---- and ${unrun.length - 10} more, all of them in the JSON report`);
|
|
1233
|
+
say("");
|
|
1234
|
+
say(`${tally.passed} of ${report.graded} graded cases passed`
|
|
1235
|
+
+ ` · ${tally.found} of ${tally.of} terms${pct == null ? "" : ` (${pct}%)`}`
|
|
1236
|
+
+ ` · ${report.ungraded} not graded yet`);
|
|
1237
|
+
if (!complete) {
|
|
1238
|
+
say(rows.length === 0 && !unrun.length
|
|
1239
|
+
? "NOTHING RAN: the set grades no items. That is not a pass."
|
|
1240
|
+
: `NOTHING SCORED for ${unrun.length} of ${report.graded} graded cases, `
|
|
1241
|
+
+ `so this is a partial run and not a result.`);
|
|
1242
|
+
}
|
|
1243
|
+
|
|
1244
|
+
if (o.json) {
|
|
1245
|
+
const text = JSON.stringify(report, null, 2);
|
|
1246
|
+
if (toStdout) process.stdout.write(text + "\n");
|
|
1247
|
+
else fs.writeFileSync(o.json, text + "\n");
|
|
1248
|
+
}
|
|
1249
|
+
|
|
1250
|
+
process.exitCode = !complete ? EXIT.incomplete
|
|
1251
|
+
: report.failed ? EXIT.failed : EXIT.passed;
|
|
1252
|
+
}
|
|
1253
|
+
|
|
1254
|
+
main().catch(e => {
|
|
1255
|
+
if (e instanceof Stop) {
|
|
1256
|
+
if (e.message) console.error(e.message);
|
|
1257
|
+
process.exitCode = e.code;
|
|
1258
|
+
return;
|
|
1259
|
+
}
|
|
1260
|
+
console.error(`the run failed: ${e?.stack || e}`);
|
|
1261
|
+
process.exitCode = EXIT.broken;
|
|
1262
|
+
});
|