evals-lab 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +43 -0
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +36 -30
- package/lab/demo/pipelines/demo-2.json +36 -30
- package/lab/evals-core.mjs +1122 -710
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +77 -49
- package/lab/server.py +544 -154
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-61NS6C4m.js +0 -18
- package/lab/web/dist/assets/main-DjQQums6.css +0 -1
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/kinds/list.mjs
CHANGED
|
@@ -332,7 +332,7 @@ const LIST_MODIFIERS = {
|
|
|
332
332
|
|
|
333
333
|
const LIST_KIND = {
|
|
334
334
|
label: "List", noun: "items", offers: ["items"],
|
|
335
|
-
description: "Splits the reply into items (by commas, lines or a JSON array) that
|
|
335
|
+
description: "Splits the reply into items (by commas, lines or a JSON array) that evals and later jobs read one by one.",
|
|
336
336
|
settings: [
|
|
337
337
|
{ key: "parse", label: "Parse as", type: "select", choices: [
|
|
338
338
|
{ value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
|
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -3,13 +3,13 @@
|
|
|
3
3
|
// read the text, ones that read it against what production replied, and
|
|
4
4
|
// model-graded ones that ask a grader. Each is one METRICS entry the core
|
|
5
5
|
// registers, as it registers kinds/list.ts; a plugin adds more the same way.
|
|
6
|
-
|
|
6
|
+
|
|
7
7
|
|
|
8
8
|
const ok = (pass , reason , score = pass ? 1 : 0) => ({ pass, score, reason });
|
|
9
9
|
const str = (v ) => (v == null ? "" : String(v));
|
|
10
10
|
const num = (v ) => (typeof v === "number" ? v : str(v).trim() && !Number.isNaN(Number(v)) ? Number(v) : null);
|
|
11
|
-
/** A list option: one entry a line. */
|
|
12
|
-
const lines = (v ) => str(v).split("\n").map((l) => l.trim()).filter(Boolean);
|
|
11
|
+
/** A list option: one entry a line, or a list of them. */
|
|
12
|
+
const lines = (v ) => (Array.isArray(v) ? v.map(str) : str(v).split("\n")).map((l) => l.trim()).filter(Boolean);
|
|
13
13
|
const short = (s ) => (s.length > 60 ? `${s.slice(0, 57)}…` : s);
|
|
14
14
|
const json = (s ) => {
|
|
15
15
|
try { return { value: JSON.parse(s) }; } catch (e) { return { error: e instanceof Error ? e.message : String(e) }; }
|
|
@@ -81,6 +81,45 @@ function distance(a , b ) {
|
|
|
81
81
|
return d[b.length] ;
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
// ---- matching what a reply holds ---------------------------------------------
|
|
85
|
+
// A kind that yields items (a list) is matched item by item, the way the lab
|
|
86
|
+
// matches a term (docs/datasets.md § Matching): a value's words in some item,
|
|
87
|
+
// in order and adjacent -- "cat" is in "tabby cat" and not in "cathedral" --
|
|
88
|
+
// and ignoring case where the metric's Ignore case is on, as it is by default.
|
|
89
|
+
// A reply the job threw away yields none. A reply read as text, or the API's
|
|
90
|
+
// own text (`of: "said"`), is matched as text.
|
|
91
|
+
|
|
92
|
+
/** The words of [s], lowercased unless [keepCase]: the lab's own reading of a term. */
|
|
93
|
+
const words = (s , keepCase = false) => (keepCase ? s : s.toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
|
|
94
|
+
const termIn = (items , term , ignoreCase = true) => {
|
|
95
|
+
const t = words(term, !ignoreCase);
|
|
96
|
+
if (!t.length) return false;
|
|
97
|
+
return items.some((item) => {
|
|
98
|
+
const w = words(item, !ignoreCase);
|
|
99
|
+
for (let i = 0; i + t.length <= w.length; i++) if (t.every((x, j) => w[i + j] === x)) return true;
|
|
100
|
+
return false;
|
|
101
|
+
});
|
|
102
|
+
};
|
|
103
|
+
/** Whether [input] is matched item by item. */
|
|
104
|
+
const byItem = (input , m ) => !input.plain && m.of !== "said";
|
|
105
|
+
/** Whether [input] holds [value], as [m] says to match it: in its items, or
|
|
106
|
+
in its text -- there, outside any of [except]. */
|
|
107
|
+
function holds(input , m , value , except = []) {
|
|
108
|
+
if (byItem(input, m)) {
|
|
109
|
+
const items = input.error ? [] : input.terms;
|
|
110
|
+
const ic = m.ignoreCase === true;
|
|
111
|
+
// An item holding an exception that holds the value does not count; the
|
|
112
|
+
// value anywhere else still does.
|
|
113
|
+
return items.some((item) => termIn([item], value, ic) && !except.some((e) => termIn([e], value, ic) && termIn([item], e, ic)));
|
|
114
|
+
}
|
|
115
|
+
const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
|
|
116
|
+
let text = fold(input.text);
|
|
117
|
+
for (const e of except) if (fold(e).includes(fold(value))) text = text.split(fold(e)).join(" ");
|
|
118
|
+
return text.includes(fold(value));
|
|
119
|
+
}
|
|
120
|
+
/** A requirement as Results names it: a term, or the group it was. */
|
|
121
|
+
const spoken = (g ) => (g.length === 1 ? g[0] : g);
|
|
122
|
+
|
|
84
123
|
/** Production's reply, or the reason a metric that compares with it cannot. */
|
|
85
124
|
const production = (input ) => {
|
|
86
125
|
if (input.production == null) throw new Error("this item has no production reply to compare with");
|
|
@@ -107,8 +146,12 @@ const gradedPass = (v , threshold )
|
|
|
107
146
|
};
|
|
108
147
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
109
148
|
|
|
149
|
+
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
|
+
const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
|
|
151
|
+
|
|
110
152
|
const metrics = {
|
|
111
153
|
equals: {
|
|
154
|
+
family: TEXT,
|
|
112
155
|
label: "Equals",
|
|
113
156
|
description: "Passes when the reply is exactly the value; as JSON, the same value with keys in any order.",
|
|
114
157
|
// As JSON, a reply equals the value as a value -- keys in any order;
|
|
@@ -123,44 +166,53 @@ const metrics = {
|
|
|
123
166
|
},
|
|
124
167
|
},
|
|
125
168
|
contains: {
|
|
169
|
+
family: TEXT,
|
|
126
170
|
label: "Contains",
|
|
127
|
-
description: "Passes when the reply contains the value.",
|
|
128
|
-
options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }
|
|
129
|
-
|
|
171
|
+
description: "Passes when the reply contains the value; turned round, an exception excuses the value inside it.",
|
|
172
|
+
options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" },
|
|
173
|
+
{ key: "except", label: "Except", type: "textarea" }],
|
|
174
|
+
defaults: () => ({ value: "", ignoreCase: true }),
|
|
130
175
|
validate: (m, at, bad) => needs(m, ["value"], at, bad),
|
|
176
|
+
// Turned round (`not`), what it found is what the reply invented; a
|
|
177
|
+
// reading of items says so, for a run's totals.
|
|
131
178
|
score: (input, m) => {
|
|
132
|
-
const
|
|
133
|
-
const hit =
|
|
134
|
-
|
|
179
|
+
const v = str(m.value);
|
|
180
|
+
const hit = holds(input, m, v, lines(m.except));
|
|
181
|
+
const detail = !byItem(input, m) ? {} : m.not ? { invented: hit ? [v] : [] } : hit ? { found: [v], missed: [] } : { found: [], missed: [v] };
|
|
182
|
+
return { ...ok(hit, hit ? `has ${short(v)}` : `has no ${short(v)}`), ...detail };
|
|
135
183
|
},
|
|
136
184
|
},
|
|
137
185
|
"contains-any": {
|
|
186
|
+
family: TEXT,
|
|
138
187
|
label: "Contains any",
|
|
139
188
|
description: "Passes when the reply contains at least one of the values, one a line.",
|
|
140
189
|
options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
|
|
141
|
-
defaults: () => ({ values: "", ignoreCase:
|
|
190
|
+
defaults: () => ({ values: "", ignoreCase: true }),
|
|
142
191
|
validate: (m, at, bad) => needs(m, ["values"], at, bad),
|
|
143
192
|
score: (input, m) => {
|
|
144
|
-
const
|
|
145
|
-
const hit =
|
|
146
|
-
|
|
193
|
+
const all = lines(m.values);
|
|
194
|
+
const hit = all.find((v) => holds(input, m, v));
|
|
195
|
+
const detail = !byItem(input, m) ? {} : hit ? { found: [spoken(all)], missed: [] } : { found: [], missed: [spoken(all)] };
|
|
196
|
+
return { ...ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${all.join(", ")}`), ...detail };
|
|
147
197
|
},
|
|
148
198
|
},
|
|
149
199
|
"contains-all": {
|
|
200
|
+
family: TEXT,
|
|
150
201
|
label: "Contains all",
|
|
151
202
|
description: "Passes when the reply contains every value, one a line.",
|
|
152
203
|
options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
|
|
153
|
-
defaults: () => ({ values: "", ignoreCase:
|
|
204
|
+
defaults: () => ({ values: "", ignoreCase: true }),
|
|
154
205
|
validate: (m, at, bad) => needs(m, ["values"], at, bad),
|
|
155
206
|
score: (input, m) => {
|
|
156
|
-
const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
|
|
157
207
|
const all = lines(m.values);
|
|
158
|
-
const missing = all.filter((v) => !
|
|
159
|
-
|
|
160
|
-
|
|
208
|
+
const missing = all.filter((v) => !holds(input, m, v));
|
|
209
|
+
const detail = !byItem(input, m) ? {} : { found: all.filter((v) => !missing.includes(v)), missed: missing };
|
|
210
|
+
return { ...ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
|
|
211
|
+
all.length ? (all.length - missing.length) / all.length : 1), ...detail };
|
|
161
212
|
},
|
|
162
213
|
},
|
|
163
214
|
regex: {
|
|
215
|
+
family: TEXT,
|
|
164
216
|
label: "Matches",
|
|
165
217
|
description: "Passes when the reply matches the regular expression.",
|
|
166
218
|
options: [{ key: "pattern", label: "Pattern", type: "text" }, { key: "flags", label: "Flags", type: "text" }],
|
|
@@ -172,6 +224,7 @@ const metrics = {
|
|
|
172
224
|
},
|
|
173
225
|
},
|
|
174
226
|
"starts-with": {
|
|
227
|
+
family: TEXT,
|
|
175
228
|
label: "Starts with",
|
|
176
229
|
description: "Passes when the reply starts with the value.",
|
|
177
230
|
options: [{ key: "value", label: "Value", type: "text" }],
|
|
@@ -183,6 +236,7 @@ const metrics = {
|
|
|
183
236
|
},
|
|
184
237
|
},
|
|
185
238
|
"is-json": {
|
|
239
|
+
family: TEXT,
|
|
186
240
|
label: "Is JSON",
|
|
187
241
|
description: "Passes when the reply is JSON and, with a schema, holds to it.",
|
|
188
242
|
options: [{ key: "schema", label: "Schema", type: "textarea" }],
|
|
@@ -197,6 +251,7 @@ const metrics = {
|
|
|
197
251
|
},
|
|
198
252
|
},
|
|
199
253
|
"json-field": {
|
|
254
|
+
family: TEXT,
|
|
200
255
|
label: "JSON field",
|
|
201
256
|
description: "Reads one field of a JSON reply and compares it.",
|
|
202
257
|
options: [
|
|
@@ -219,7 +274,21 @@ const metrics = {
|
|
|
219
274
|
return ok(pass, `${str(m.path)} is ${short(text)}`);
|
|
220
275
|
},
|
|
221
276
|
},
|
|
277
|
+
// A case that expects its answer thrown away: the reply a job's Reject
|
|
278
|
+
// rules discarded is the finding.
|
|
279
|
+
discarded: {
|
|
280
|
+
family: RESULT,
|
|
281
|
+
label: "Discarded",
|
|
282
|
+
description: "Passes only when the job's rules discarded the reply.",
|
|
283
|
+
options: [],
|
|
284
|
+
defaults: () => ({}),
|
|
285
|
+
score: (input) => {
|
|
286
|
+
const thrown = !!input.error && input.error.startsWith("discarded: ");
|
|
287
|
+
return ok(thrown, thrown ? "discarded" : input.error ? input.error : `kept ${input.terms.length} items`);
|
|
288
|
+
},
|
|
289
|
+
},
|
|
222
290
|
"item-count": {
|
|
291
|
+
family: RESULT,
|
|
223
292
|
label: "Item count",
|
|
224
293
|
description: "Passes when the number of items is within the bounds.",
|
|
225
294
|
options: [{ key: "min", label: "At least", type: "number" }, { key: "max", label: "At most", type: "number" }],
|
|
@@ -231,6 +300,7 @@ const metrics = {
|
|
|
231
300
|
},
|
|
232
301
|
},
|
|
233
302
|
"count-matches": {
|
|
303
|
+
family: RESULT,
|
|
234
304
|
label: "Count matches",
|
|
235
305
|
description: "Counts matches of the pattern: points, for the weighted mode.",
|
|
236
306
|
counts: true,
|
|
@@ -246,6 +316,7 @@ const metrics = {
|
|
|
246
316
|
},
|
|
247
317
|
},
|
|
248
318
|
latency: {
|
|
319
|
+
family: RESULT,
|
|
249
320
|
label: "Latency",
|
|
250
321
|
description: "Passes when the reply came back within the time set.",
|
|
251
322
|
options: [{ key: "max", label: "At most (ms)", type: "number" }],
|
|
@@ -254,6 +325,7 @@ const metrics = {
|
|
|
254
325
|
score: (input, m) => ok(input.ms <= (num(m.max) ?? 0), `${input.ms} ms`),
|
|
255
326
|
},
|
|
256
327
|
levenshtein: {
|
|
328
|
+
family: TEXT,
|
|
257
329
|
label: "Near",
|
|
258
330
|
description: "Passes when the reply is within a few edits of the value.",
|
|
259
331
|
options: [{ key: "value", label: "Value", type: "textarea" }, { key: "max", label: "Edits at most", type: "number" }],
|
|
@@ -267,6 +339,7 @@ const metrics = {
|
|
|
267
339
|
},
|
|
268
340
|
// ---- against production ----------------------------------------------------
|
|
269
341
|
"equals-production": {
|
|
342
|
+
family: PROD,
|
|
270
343
|
label: "Same as production",
|
|
271
344
|
description: "Passes when the reply is what production replied to the same item.",
|
|
272
345
|
options: [],
|
|
@@ -279,6 +352,7 @@ const metrics = {
|
|
|
279
352
|
},
|
|
280
353
|
},
|
|
281
354
|
"fields-equal-production": {
|
|
355
|
+
family: PROD,
|
|
282
356
|
label: "Fields as production",
|
|
283
357
|
description: "Passes when the fields listed are what production's reply had.",
|
|
284
358
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
@@ -295,6 +369,7 @@ const metrics = {
|
|
|
295
369
|
},
|
|
296
370
|
},
|
|
297
371
|
"same-parse-outcome": {
|
|
372
|
+
family: PROD,
|
|
298
373
|
label: "Parses as production",
|
|
299
374
|
description: "Passes when the reply parses as JSON exactly when production's did.",
|
|
300
375
|
options: [],
|
|
@@ -306,6 +381,7 @@ const metrics = {
|
|
|
306
381
|
},
|
|
307
382
|
// ---- model-graded -------------------------------------------------------------
|
|
308
383
|
"llm-rubric": {
|
|
384
|
+
family: GRADED,
|
|
309
385
|
label: "Rubric",
|
|
310
386
|
description: "Asks the grader whether the reply meets the rubric.",
|
|
311
387
|
graded: true,
|
|
@@ -317,6 +393,7 @@ const metrics = {
|
|
|
317
393
|
num(m.threshold) ?? 0.5),
|
|
318
394
|
},
|
|
319
395
|
factuality: {
|
|
396
|
+
family: GRADED,
|
|
320
397
|
label: "Factual",
|
|
321
398
|
description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
|
|
322
399
|
graded: true,
|
|
@@ -329,6 +406,7 @@ const metrics = {
|
|
|
329
406
|
+ `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
|
|
330
407
|
},
|
|
331
408
|
"judge-vs-production": {
|
|
409
|
+
family: GRADED,
|
|
332
410
|
label: "Judged against production",
|
|
333
411
|
description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
|
|
334
412
|
graded: true,
|
package/lab/run-evals.js
CHANGED
|
@@ -101,7 +101,7 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
101
101
|
job's tokenMappings. Default: none, so a prompt
|
|
102
102
|
naming a token is refused, naming it.
|
|
103
103
|
--pipeline <file> a run document (docs/pipeline-model.md) with one
|
|
104
|
-
scenario and a graded
|
|
104
|
+
scenario and a graded eval, graded case by case against
|
|
105
105
|
the set. Instead of --prompt, --model and --url. Each
|
|
106
106
|
profile's key comes from $EVAL_API_KEY_<ID>, never the
|
|
107
107
|
file; its content.files, when not empty, is the file
|
|
@@ -274,7 +274,7 @@ function readRun(file, dataset) {
|
|
|
274
274
|
} catch (e) {
|
|
275
275
|
broken(`${file}: ${e.message}`);
|
|
276
276
|
}
|
|
277
|
-
const named = isObject(doc) ? core.
|
|
277
|
+
const named = isObject(doc) ? core.evalsDataset(doc) : null;
|
|
278
278
|
const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
|
|
279
279
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
280
280
|
return doc;
|
|
@@ -296,8 +296,8 @@ function readDataset(file) {
|
|
|
296
296
|
broken(`${file}: ${e.message}`);
|
|
297
297
|
}
|
|
298
298
|
if (isObject(doc) && doc.format === "evals-lab/dataset") {
|
|
299
|
-
if (![1, 2, 3, 4].includes(doc.version)) {
|
|
300
|
-
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to
|
|
299
|
+
if (![1, 2, 3, 4, 5].includes(doc.version)) {
|
|
300
|
+
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 5`);
|
|
301
301
|
}
|
|
302
302
|
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
303
303
|
}
|
|
@@ -315,11 +315,12 @@ function readDataset(file) {
|
|
|
315
315
|
return doc;
|
|
316
316
|
}
|
|
317
317
|
|
|
318
|
-
// The cases a graded run is scored against, by the
|
|
318
|
+
// The cases a graded run is scored against, by the item each names: exactly,
|
|
319
|
+
// as the Source names it.
|
|
319
320
|
function gradedBy(dataset) {
|
|
320
321
|
const out = new Map();
|
|
321
322
|
for (const c of core.gradedSetFrom(dataset || {})) {
|
|
322
|
-
if (!c.todo) out.set(c.
|
|
323
|
+
if (!c.todo) out.set(c.item, c);
|
|
323
324
|
}
|
|
324
325
|
return out;
|
|
325
326
|
}
|
|
@@ -453,21 +454,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
|
|
|
453
454
|
// or, over Prompt only, the prompt -- and the stage reads it as it would a
|
|
454
455
|
// model's.
|
|
455
456
|
//
|
|
456
|
-
// A
|
|
457
|
-
// from
|
|
458
|
-
// only its words: the transport is where the three meet.
|
|
457
|
+
// A target step that asks for a whole request (an HTTP Request, #flows)
|
|
458
|
+
// builds it from the step, the job's flow step and the item's record, and is
|
|
459
|
+
// handed only its words: the transport is where the three meet.
|
|
460
|
+
//
|
|
461
|
+
// Any other step's reply, in a job that reads the flow's step through a Read
|
|
462
|
+
// as, is read the same way: put in the body the flow's API would have sent
|
|
463
|
+
// it in, then read as the flow reads it -- so a model asked in words and
|
|
464
|
+
// production are compared alike (docs/pipeline-model.md §16 › Targets).
|
|
459
465
|
function callsFor(connections, links, text, plan, record) {
|
|
460
466
|
return connections.map((c, k) => {
|
|
461
467
|
const step = plan?.calls?.[k];
|
|
462
468
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
463
469
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
464
470
|
}
|
|
465
|
-
|
|
471
|
+
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
466
472
|
? async sent => core.localAnswer(c, text, sent)
|
|
467
473
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
474
|
+
if (!step?.readAs || !step?.step) return call;
|
|
475
|
+
return async (sent, url) => {
|
|
476
|
+
const r = await call(sent, url);
|
|
477
|
+
if (r.error || r.raw == null) return r;
|
|
478
|
+
try {
|
|
479
|
+
return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
|
|
480
|
+
} catch (e) {
|
|
481
|
+
return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
|
|
482
|
+
}
|
|
483
|
+
};
|
|
468
484
|
});
|
|
469
485
|
}
|
|
470
486
|
|
|
487
|
+
// What an expression token reads: the item's record, or a text item read as
|
|
488
|
+
// one, and the loop job 1's flow step sits in.
|
|
489
|
+
const tokenRecord = (run, record, text) => {
|
|
490
|
+
const r = record ?? (text != null ? textRecord(text) : null);
|
|
491
|
+
return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
|
|
492
|
+
};
|
|
493
|
+
|
|
494
|
+
// A prompt as a report restates it: an expression token reads an item's
|
|
495
|
+
// record, and a report has none to name, so its words stay as written.
|
|
496
|
+
const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
|
|
497
|
+
|
|
471
498
|
/** A Setup profile the run carries, asked as a grader: its reply's text, or
|
|
472
499
|
the failure as an error the metric reports. One transport per profile. */
|
|
473
500
|
const graders = new WeakMap();
|
|
@@ -586,12 +613,12 @@ function readUtf8(file) {
|
|
|
586
613
|
return fs.readFileSync(file).toString("utf8");
|
|
587
614
|
}
|
|
588
615
|
|
|
589
|
-
// Every
|
|
590
|
-
// the page reads it: a whole-run
|
|
591
|
-
// in, and a per-item
|
|
592
|
-
// stopped short, whose whole-run
|
|
593
|
-
const verdictsOf = (run, items, settled) => run.
|
|
594
|
-
Object.fromEntries(core.
|
|
616
|
+
// Every eval's reading of each scenario, keyed by the eval's id, exactly as
|
|
617
|
+
// the page reads it: a whole-run eval's verdict, settled once every item is
|
|
618
|
+
// in, and a per-item eval's counts. [settled] is false for a run that
|
|
619
|
+
// stopped short, whose whole-run evals have not settled.
|
|
620
|
+
const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
|
|
621
|
+
Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
|
|
595
622
|
name: o.label, skipped: o.skipped,
|
|
596
623
|
...(o.whole
|
|
597
624
|
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
@@ -600,7 +627,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
|
|
|
600
627
|
|
|
601
628
|
// The run as the report restates it: each scenario's stages with the
|
|
602
629
|
// connection each one asked.
|
|
603
|
-
const scenariosOf = run => run.
|
|
630
|
+
const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
604
631
|
const { stages, connections } = core.stagesFor(run, i);
|
|
605
632
|
return {
|
|
606
633
|
n: i + 1, name: sc.name || null,
|
|
@@ -649,7 +676,7 @@ async function runSnapshot(o) {
|
|
|
649
676
|
// Each scenario once: its stages, the jobs' token sets, and a transport
|
|
650
677
|
// per stage to the profile that stage resolves to, keyed by that
|
|
651
678
|
// profile's id.
|
|
652
|
-
const plans = run.
|
|
679
|
+
const plans = core.targetsOf(run).map((_, i) => {
|
|
653
680
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
654
681
|
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
655
682
|
return { stages, tokens, connections, links, calls, cells };
|
|
@@ -704,7 +731,7 @@ async function runSnapshot(o) {
|
|
|
704
731
|
const textOf = item.kind === "text" && !item.bare
|
|
705
732
|
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
706
733
|
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
707
|
-
{ tokens: plan.tokens, text: textOf });
|
|
734
|
+
{ tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
|
|
708
735
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
709
736
|
// text item a dataset grades is graded like an image.
|
|
710
737
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
@@ -728,7 +755,7 @@ async function runSnapshot(o) {
|
|
|
728
755
|
verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
|
|
729
756
|
run: {
|
|
730
757
|
name: run.name ?? null,
|
|
731
|
-
|
|
758
|
+
evals: run.evals,
|
|
732
759
|
verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
|
|
733
760
|
scenarios: scenariosOf(run),
|
|
734
761
|
items: results.slice(0, total),
|
|
@@ -753,7 +780,7 @@ async function runSnapshot(o) {
|
|
|
753
780
|
/** Re-score a run's stored results against the dataset --dataset hands over,
|
|
754
781
|
* keeping the replies that were stored: what changes a verdict is the scorer
|
|
755
782
|
* or the cases, never a new question to the model. The report is shaped like
|
|
756
|
-
* a --run's, with the stored results as its items, and a whole-run
|
|
783
|
+
* a --run's, with the stored results as its items, and a whole-run eval's
|
|
757
784
|
* verdict settled the same way. */
|
|
758
785
|
async function runRescore(o){
|
|
759
786
|
const dataset = readDataset(o.dataset);
|
|
@@ -775,7 +802,7 @@ async function runRescore(o){
|
|
|
775
802
|
const graded = gradedBy(dataset);
|
|
776
803
|
|
|
777
804
|
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
778
|
-
//
|
|
805
|
+
// An eval that reads every item (the Metrics) re-reads one with no case
|
|
779
806
|
// too, against the production reply the item kept.
|
|
780
807
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
781
808
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
@@ -793,7 +820,7 @@ async function runRescore(o){
|
|
|
793
820
|
rescored: true,
|
|
794
821
|
run: {
|
|
795
822
|
name: run.name ?? null,
|
|
796
|
-
|
|
823
|
+
evals: run.evals,
|
|
797
824
|
verdicts: verdictsOf(run, items, true),
|
|
798
825
|
scenarios: scenariosOf(run),
|
|
799
826
|
items,
|
|
@@ -822,16 +849,16 @@ function cliRun(o, dataset) {
|
|
|
822
849
|
doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
|
|
823
850
|
if (o.tokens) {
|
|
824
851
|
try {
|
|
825
|
-
doc.jobs[0] = core.
|
|
852
|
+
doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
|
|
826
853
|
} catch (e) {
|
|
827
854
|
broken(`${o.tokens}: ${e.message}`);
|
|
828
855
|
}
|
|
829
856
|
}
|
|
830
857
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
831
|
-
doc.
|
|
832
|
-
|
|
833
|
-
doc.
|
|
834
|
-
mode: "all", threshold: null, grader: null, over: "item", metrics: [
|
|
858
|
+
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
859
|
+
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
860
|
+
doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
|
|
861
|
+
mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
|
|
835
862
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
836
863
|
return doc;
|
|
837
864
|
}
|
|
@@ -881,7 +908,7 @@ async function main() {
|
|
|
881
908
|
}
|
|
882
909
|
if (o.run) {
|
|
883
910
|
if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
|
|
884
|
-
broken("--run names everything a run takes --
|
|
911
|
+
broken("--run names everything a run takes -- targets, files and text -- so it "
|
|
885
912
|
+ "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
|
|
886
913
|
}
|
|
887
914
|
const n = Number(o.timeout ?? REQUEST_CAP);
|
|
@@ -920,21 +947,23 @@ async function main() {
|
|
|
920
947
|
}
|
|
921
948
|
const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
|
|
922
949
|
if (o.pipeline) {
|
|
923
|
-
if (run.
|
|
924
|
-
broken(`${o.pipeline} has ${run.
|
|
950
|
+
if (core.targetsOf(run).length !== 1) {
|
|
951
|
+
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
|
925
952
|
+ "against the set -- --run runs them all.");
|
|
926
953
|
}
|
|
927
|
-
if (!core.
|
|
954
|
+
if (!core.evalsDataset(run)) {
|
|
928
955
|
broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
|
|
929
956
|
}
|
|
930
957
|
if (core.contentOf(run).type !== "source") {
|
|
931
958
|
broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
|
|
932
959
|
}
|
|
933
960
|
}
|
|
934
|
-
// Case by case: each item against its case, as the
|
|
935
|
-
//
|
|
961
|
+
// Case by case: each item against its case's metrics, as the eval that
|
|
962
|
+
// names the dataset reads them (core.readCase).
|
|
936
963
|
const { stages, tokens, connections } = core.stagesFor(run, 0);
|
|
937
|
-
const prompt =
|
|
964
|
+
const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
|
|
965
|
+
// A last stage that yields no items is matched as its text.
|
|
966
|
+
const plain = !core.OUTPUT_KINDS[stages.at(-1)?.kind ?? ""]?.terms;
|
|
938
967
|
|
|
939
968
|
// The graded half, through the function the tab builds its own list with.
|
|
940
969
|
const set = core.gradedSetFrom(dataset);
|
|
@@ -1024,11 +1053,11 @@ async function main() {
|
|
|
1024
1053
|
+ "--samples or --source names a directory.");
|
|
1025
1054
|
const allowed = itemsOnly ?? graded;
|
|
1026
1055
|
// $snapshotSet above: only a Source run sees it; --replies reads no file.
|
|
1027
|
-
const inSnapshot = c => !snapshotSet || snapshotSet.has(c.
|
|
1056
|
+
const inSnapshot = c => !snapshotSet || snapshotSet.has(c.item);
|
|
1028
1057
|
// Only the run's own image items need ImageMagick: a text-only Source
|
|
1029
1058
|
// -- the .txt, .csv case -- runs with nothing on PATH but the node
|
|
1030
1059
|
// running this, which is the point of having text in the registry.
|
|
1031
|
-
const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.
|
|
1060
|
+
const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.item)) === "image");
|
|
1032
1061
|
if (pictures.length) {
|
|
1033
1062
|
bin = imageMagick();
|
|
1034
1063
|
if (!bin) broken("a run over images needs ImageMagick, to cap each "
|
|
@@ -1061,7 +1090,7 @@ async function main() {
|
|
|
1061
1090
|
}
|
|
1062
1091
|
|
|
1063
1092
|
// Only files the snapshot names are read from the Source.
|
|
1064
|
-
const wanted = c => !snapshotSet || snapshotSet.has(c.
|
|
1093
|
+
const wanted = c => !snapshotSet || snapshotSet.has(c.item);
|
|
1065
1094
|
// The cancel marker is checked between items, so a cancel never loses the
|
|
1066
1095
|
// reply in flight: the item that was being asked finishes, and the rest
|
|
1067
1096
|
// are unrun rather than half-asked.
|
|
@@ -1085,18 +1114,18 @@ async function main() {
|
|
|
1085
1114
|
calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
|
|
1086
1115
|
} else {
|
|
1087
1116
|
if (!wanted(kase)) {
|
|
1088
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1117
|
+
unrun.push(`${kase.id}: ${kase.item} is not in the run's file list`);
|
|
1089
1118
|
continue;
|
|
1090
1119
|
}
|
|
1091
|
-
const file = path.join(samples, kase.
|
|
1120
|
+
const file = path.join(samples, kase.item);
|
|
1092
1121
|
if (!fs.existsSync(file)) {
|
|
1093
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1122
|
+
unrun.push(`${kase.id}: ${kase.item} is not in ${samples}`);
|
|
1094
1123
|
continue;
|
|
1095
1124
|
}
|
|
1096
1125
|
try {
|
|
1097
1126
|
item = itemContent(bin, file, connections[0]);
|
|
1098
1127
|
} catch (e) {
|
|
1099
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1128
|
+
unrun.push(`${kase.id}: ${kase.item} could not be prepared — ${e.message}`);
|
|
1100
1129
|
continue;
|
|
1101
1130
|
}
|
|
1102
1131
|
calls = callsFor(connections, links, item?.text ?? null);
|
|
@@ -1104,16 +1133,15 @@ async function main() {
|
|
|
1104
1133
|
|
|
1105
1134
|
const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
|
|
1106
1135
|
{ tokens, text: item?.text ?? null });
|
|
1107
|
-
const s = core.
|
|
1136
|
+
const s = core.readCase(kase, res, plain);
|
|
1108
1137
|
// A file with no mapper ran anyway; the flag keeps the transcript honest
|
|
1109
1138
|
// about what the model was actually shown. §4: stated, not hidden.
|
|
1110
1139
|
if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
|
|
1111
1140
|
rows.push({
|
|
1112
|
-
id: kase.id,
|
|
1141
|
+
id: kase.id, item: kase.item, half: kase.half,
|
|
1113
1142
|
pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
|
|
1114
|
-
found: s.found, missed: s.missed, invented: s.invented,
|
|
1115
|
-
|
|
1116
|
-
watchFound: s.watchFound, watchTotal: s.watchTotal,
|
|
1143
|
+
found: s.found, missed: s.missed, invented: s.invented, count: s.count,
|
|
1144
|
+
metrics: s.metrics, watch: s.watch,
|
|
1117
1145
|
terms: res.terms || [], ms: res.ms,
|
|
1118
1146
|
// What a file type was to this run -- "image", "text", "unmapped".
|
|
1119
1147
|
type: item?.kind ?? null,
|
|
@@ -1127,7 +1155,7 @@ async function main() {
|
|
|
1127
1155
|
});
|
|
1128
1156
|
if (progress) {
|
|
1129
1157
|
progress.write(JSON.stringify({
|
|
1130
|
-
event: "item", id: kase.id,
|
|
1158
|
+
event: "item", id: kase.id, item: kase.item, n: rows.length,
|
|
1131
1159
|
pass: s.pass, score: s.score, discarded: s.discarded,
|
|
1132
1160
|
error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
|
|
1133
1161
|
ms: res.ms,
|
|
@@ -1170,7 +1198,7 @@ async function main() {
|
|
|
1170
1198
|
stages: stages.map((s, i) => {
|
|
1171
1199
|
const { id, ...connection } = connections[i];
|
|
1172
1200
|
return { n: i + 1, kind: s.kind, withImage: s.withImage,
|
|
1173
|
-
prompt:
|
|
1201
|
+
prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
|
|
1174
1202
|
}),
|
|
1175
1203
|
} } : {}),
|
|
1176
1204
|
// Which variable a key came from, never the key.
|