evals-lab 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +17 -9
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +3 -7
- package/lab/demo/pipelines/demo-2.json +3 -7
- package/lab/evals-core.mjs +429 -434
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +35 -33
- package/lab/server.py +296 -82
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DsetJSXv.js +0 -3
- package/lab/web/dist/assets/main-B-VtDGxC.css +0 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +0 -19
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +0 -51
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +0 -1
package/lab/kinds/list.mjs
CHANGED
|
@@ -332,7 +332,7 @@ const LIST_MODIFIERS = {
|
|
|
332
332
|
|
|
333
333
|
const LIST_KIND = {
|
|
334
334
|
label: "List", noun: "items", offers: ["items"],
|
|
335
|
-
description: "Splits the reply into items (by commas, lines or a JSON array) that
|
|
335
|
+
description: "Splits the reply into items (by commas, lines or a JSON array) that evals and later jobs read one by one.",
|
|
336
336
|
settings: [
|
|
337
337
|
{ key: "parse", label: "Parse as", type: "select", choices: [
|
|
338
338
|
{ value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
|
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -3,13 +3,13 @@
|
|
|
3
3
|
// read the text, ones that read it against what production replied, and
|
|
4
4
|
// model-graded ones that ask a grader. Each is one METRICS entry the core
|
|
5
5
|
// registers, as it registers kinds/list.ts; a plugin adds more the same way.
|
|
6
|
-
|
|
6
|
+
|
|
7
7
|
|
|
8
8
|
const ok = (pass , reason , score = pass ? 1 : 0) => ({ pass, score, reason });
|
|
9
9
|
const str = (v ) => (v == null ? "" : String(v));
|
|
10
10
|
const num = (v ) => (typeof v === "number" ? v : str(v).trim() && !Number.isNaN(Number(v)) ? Number(v) : null);
|
|
11
|
-
/** A list option: one entry a line. */
|
|
12
|
-
const lines = (v ) => str(v).split("\n").map((l) => l.trim()).filter(Boolean);
|
|
11
|
+
/** A list option: one entry a line, or a list of them. */
|
|
12
|
+
const lines = (v ) => (Array.isArray(v) ? v.map(str) : str(v).split("\n")).map((l) => l.trim()).filter(Boolean);
|
|
13
13
|
const short = (s ) => (s.length > 60 ? `${s.slice(0, 57)}…` : s);
|
|
14
14
|
const json = (s ) => {
|
|
15
15
|
try { return { value: JSON.parse(s) }; } catch (e) { return { error: e instanceof Error ? e.message : String(e) }; }
|
|
@@ -81,6 +81,45 @@ function distance(a , b ) {
|
|
|
81
81
|
return d[b.length] ;
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
// ---- matching what a reply holds ---------------------------------------------
|
|
85
|
+
// A kind that yields items (a list) is matched item by item, the way the lab
|
|
86
|
+
// matches a term (docs/datasets.md § Matching): a value's words in some item,
|
|
87
|
+
// in order and adjacent -- "cat" is in "tabby cat" and not in "cathedral" --
|
|
88
|
+
// and ignoring case where the metric's Ignore case is on, as it is by default.
|
|
89
|
+
// A reply the job threw away yields none. A reply read as text, or the API's
|
|
90
|
+
// own text (`of: "said"`), is matched as text.
|
|
91
|
+
|
|
92
|
+
/** The words of [s], lowercased unless [keepCase]: the lab's own reading of a term. */
|
|
93
|
+
const words = (s , keepCase = false) => (keepCase ? s : s.toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
|
|
94
|
+
const termIn = (items , term , ignoreCase = true) => {
|
|
95
|
+
const t = words(term, !ignoreCase);
|
|
96
|
+
if (!t.length) return false;
|
|
97
|
+
return items.some((item) => {
|
|
98
|
+
const w = words(item, !ignoreCase);
|
|
99
|
+
for (let i = 0; i + t.length <= w.length; i++) if (t.every((x, j) => w[i + j] === x)) return true;
|
|
100
|
+
return false;
|
|
101
|
+
});
|
|
102
|
+
};
|
|
103
|
+
/** Whether [input] is matched item by item. */
|
|
104
|
+
const byItem = (input , m ) => !input.plain && m.of !== "said";
|
|
105
|
+
/** Whether [input] holds [value], as [m] says to match it: in its items, or
|
|
106
|
+
in its text -- there, outside any of [except]. */
|
|
107
|
+
function holds(input , m , value , except = []) {
|
|
108
|
+
if (byItem(input, m)) {
|
|
109
|
+
const items = input.error ? [] : input.terms;
|
|
110
|
+
const ic = m.ignoreCase === true;
|
|
111
|
+
// An item holding an exception that holds the value does not count; the
|
|
112
|
+
// value anywhere else still does.
|
|
113
|
+
return items.some((item) => termIn([item], value, ic) && !except.some((e) => termIn([e], value, ic) && termIn([item], e, ic)));
|
|
114
|
+
}
|
|
115
|
+
const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
|
|
116
|
+
let text = fold(input.text);
|
|
117
|
+
for (const e of except) if (fold(e).includes(fold(value))) text = text.split(fold(e)).join(" ");
|
|
118
|
+
return text.includes(fold(value));
|
|
119
|
+
}
|
|
120
|
+
/** A requirement as Results names it: a term, or the group it was. */
|
|
121
|
+
const spoken = (g ) => (g.length === 1 ? g[0] : g);
|
|
122
|
+
|
|
84
123
|
/** Production's reply, or the reason a metric that compares with it cannot. */
|
|
85
124
|
const production = (input ) => {
|
|
86
125
|
if (input.production == null) throw new Error("this item has no production reply to compare with");
|
|
@@ -107,8 +146,12 @@ const gradedPass = (v , threshold )
|
|
|
107
146
|
};
|
|
108
147
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
109
148
|
|
|
149
|
+
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
|
+
const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
|
|
151
|
+
|
|
110
152
|
const metrics = {
|
|
111
153
|
equals: {
|
|
154
|
+
family: TEXT,
|
|
112
155
|
label: "Equals",
|
|
113
156
|
description: "Passes when the reply is exactly the value; as JSON, the same value with keys in any order.",
|
|
114
157
|
// As JSON, a reply equals the value as a value -- keys in any order;
|
|
@@ -123,44 +166,53 @@ const metrics = {
|
|
|
123
166
|
},
|
|
124
167
|
},
|
|
125
168
|
contains: {
|
|
169
|
+
family: TEXT,
|
|
126
170
|
label: "Contains",
|
|
127
|
-
description: "Passes when the reply contains the value.",
|
|
128
|
-
options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }
|
|
129
|
-
|
|
171
|
+
description: "Passes when the reply contains the value; turned round, an exception excuses the value inside it.",
|
|
172
|
+
options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" },
|
|
173
|
+
{ key: "except", label: "Except", type: "textarea" }],
|
|
174
|
+
defaults: () => ({ value: "", ignoreCase: true }),
|
|
130
175
|
validate: (m, at, bad) => needs(m, ["value"], at, bad),
|
|
176
|
+
// Turned round (`not`), what it found is what the reply invented; a
|
|
177
|
+
// reading of items says so, for a run's totals.
|
|
131
178
|
score: (input, m) => {
|
|
132
|
-
const
|
|
133
|
-
const hit =
|
|
134
|
-
|
|
179
|
+
const v = str(m.value);
|
|
180
|
+
const hit = holds(input, m, v, lines(m.except));
|
|
181
|
+
const detail = !byItem(input, m) ? {} : m.not ? { invented: hit ? [v] : [] } : hit ? { found: [v], missed: [] } : { found: [], missed: [v] };
|
|
182
|
+
return { ...ok(hit, hit ? `has ${short(v)}` : `has no ${short(v)}`), ...detail };
|
|
135
183
|
},
|
|
136
184
|
},
|
|
137
185
|
"contains-any": {
|
|
186
|
+
family: TEXT,
|
|
138
187
|
label: "Contains any",
|
|
139
188
|
description: "Passes when the reply contains at least one of the values, one a line.",
|
|
140
189
|
options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
|
|
141
|
-
defaults: () => ({ values: "", ignoreCase:
|
|
190
|
+
defaults: () => ({ values: "", ignoreCase: true }),
|
|
142
191
|
validate: (m, at, bad) => needs(m, ["values"], at, bad),
|
|
143
192
|
score: (input, m) => {
|
|
144
|
-
const
|
|
145
|
-
const hit =
|
|
146
|
-
|
|
193
|
+
const all = lines(m.values);
|
|
194
|
+
const hit = all.find((v) => holds(input, m, v));
|
|
195
|
+
const detail = !byItem(input, m) ? {} : hit ? { found: [spoken(all)], missed: [] } : { found: [], missed: [spoken(all)] };
|
|
196
|
+
return { ...ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${all.join(", ")}`), ...detail };
|
|
147
197
|
},
|
|
148
198
|
},
|
|
149
199
|
"contains-all": {
|
|
200
|
+
family: TEXT,
|
|
150
201
|
label: "Contains all",
|
|
151
202
|
description: "Passes when the reply contains every value, one a line.",
|
|
152
203
|
options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
|
|
153
|
-
defaults: () => ({ values: "", ignoreCase:
|
|
204
|
+
defaults: () => ({ values: "", ignoreCase: true }),
|
|
154
205
|
validate: (m, at, bad) => needs(m, ["values"], at, bad),
|
|
155
206
|
score: (input, m) => {
|
|
156
|
-
const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
|
|
157
207
|
const all = lines(m.values);
|
|
158
|
-
const missing = all.filter((v) => !
|
|
159
|
-
|
|
160
|
-
|
|
208
|
+
const missing = all.filter((v) => !holds(input, m, v));
|
|
209
|
+
const detail = !byItem(input, m) ? {} : { found: all.filter((v) => !missing.includes(v)), missed: missing };
|
|
210
|
+
return { ...ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
|
|
211
|
+
all.length ? (all.length - missing.length) / all.length : 1), ...detail };
|
|
161
212
|
},
|
|
162
213
|
},
|
|
163
214
|
regex: {
|
|
215
|
+
family: TEXT,
|
|
164
216
|
label: "Matches",
|
|
165
217
|
description: "Passes when the reply matches the regular expression.",
|
|
166
218
|
options: [{ key: "pattern", label: "Pattern", type: "text" }, { key: "flags", label: "Flags", type: "text" }],
|
|
@@ -172,6 +224,7 @@ const metrics = {
|
|
|
172
224
|
},
|
|
173
225
|
},
|
|
174
226
|
"starts-with": {
|
|
227
|
+
family: TEXT,
|
|
175
228
|
label: "Starts with",
|
|
176
229
|
description: "Passes when the reply starts with the value.",
|
|
177
230
|
options: [{ key: "value", label: "Value", type: "text" }],
|
|
@@ -183,6 +236,7 @@ const metrics = {
|
|
|
183
236
|
},
|
|
184
237
|
},
|
|
185
238
|
"is-json": {
|
|
239
|
+
family: TEXT,
|
|
186
240
|
label: "Is JSON",
|
|
187
241
|
description: "Passes when the reply is JSON and, with a schema, holds to it.",
|
|
188
242
|
options: [{ key: "schema", label: "Schema", type: "textarea" }],
|
|
@@ -197,6 +251,7 @@ const metrics = {
|
|
|
197
251
|
},
|
|
198
252
|
},
|
|
199
253
|
"json-field": {
|
|
254
|
+
family: TEXT,
|
|
200
255
|
label: "JSON field",
|
|
201
256
|
description: "Reads one field of a JSON reply and compares it.",
|
|
202
257
|
options: [
|
|
@@ -219,7 +274,21 @@ const metrics = {
|
|
|
219
274
|
return ok(pass, `${str(m.path)} is ${short(text)}`);
|
|
220
275
|
},
|
|
221
276
|
},
|
|
277
|
+
// A case that expects its answer thrown away: the reply a job's Reject
|
|
278
|
+
// rules discarded is the finding.
|
|
279
|
+
discarded: {
|
|
280
|
+
family: RESULT,
|
|
281
|
+
label: "Discarded",
|
|
282
|
+
description: "Passes only when the job's rules discarded the reply.",
|
|
283
|
+
options: [],
|
|
284
|
+
defaults: () => ({}),
|
|
285
|
+
score: (input) => {
|
|
286
|
+
const thrown = !!input.error && input.error.startsWith("discarded: ");
|
|
287
|
+
return ok(thrown, thrown ? "discarded" : input.error ? input.error : `kept ${input.terms.length} items`);
|
|
288
|
+
},
|
|
289
|
+
},
|
|
222
290
|
"item-count": {
|
|
291
|
+
family: RESULT,
|
|
223
292
|
label: "Item count",
|
|
224
293
|
description: "Passes when the number of items is within the bounds.",
|
|
225
294
|
options: [{ key: "min", label: "At least", type: "number" }, { key: "max", label: "At most", type: "number" }],
|
|
@@ -231,6 +300,7 @@ const metrics = {
|
|
|
231
300
|
},
|
|
232
301
|
},
|
|
233
302
|
"count-matches": {
|
|
303
|
+
family: RESULT,
|
|
234
304
|
label: "Count matches",
|
|
235
305
|
description: "Counts matches of the pattern: points, for the weighted mode.",
|
|
236
306
|
counts: true,
|
|
@@ -246,6 +316,7 @@ const metrics = {
|
|
|
246
316
|
},
|
|
247
317
|
},
|
|
248
318
|
latency: {
|
|
319
|
+
family: RESULT,
|
|
249
320
|
label: "Latency",
|
|
250
321
|
description: "Passes when the reply came back within the time set.",
|
|
251
322
|
options: [{ key: "max", label: "At most (ms)", type: "number" }],
|
|
@@ -254,6 +325,7 @@ const metrics = {
|
|
|
254
325
|
score: (input, m) => ok(input.ms <= (num(m.max) ?? 0), `${input.ms} ms`),
|
|
255
326
|
},
|
|
256
327
|
levenshtein: {
|
|
328
|
+
family: TEXT,
|
|
257
329
|
label: "Near",
|
|
258
330
|
description: "Passes when the reply is within a few edits of the value.",
|
|
259
331
|
options: [{ key: "value", label: "Value", type: "textarea" }, { key: "max", label: "Edits at most", type: "number" }],
|
|
@@ -267,6 +339,7 @@ const metrics = {
|
|
|
267
339
|
},
|
|
268
340
|
// ---- against production ----------------------------------------------------
|
|
269
341
|
"equals-production": {
|
|
342
|
+
family: PROD,
|
|
270
343
|
label: "Same as production",
|
|
271
344
|
description: "Passes when the reply is what production replied to the same item.",
|
|
272
345
|
options: [],
|
|
@@ -279,6 +352,7 @@ const metrics = {
|
|
|
279
352
|
},
|
|
280
353
|
},
|
|
281
354
|
"fields-equal-production": {
|
|
355
|
+
family: PROD,
|
|
282
356
|
label: "Fields as production",
|
|
283
357
|
description: "Passes when the fields listed are what production's reply had.",
|
|
284
358
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
@@ -295,6 +369,7 @@ const metrics = {
|
|
|
295
369
|
},
|
|
296
370
|
},
|
|
297
371
|
"same-parse-outcome": {
|
|
372
|
+
family: PROD,
|
|
298
373
|
label: "Parses as production",
|
|
299
374
|
description: "Passes when the reply parses as JSON exactly when production's did.",
|
|
300
375
|
options: [],
|
|
@@ -306,6 +381,7 @@ const metrics = {
|
|
|
306
381
|
},
|
|
307
382
|
// ---- model-graded -------------------------------------------------------------
|
|
308
383
|
"llm-rubric": {
|
|
384
|
+
family: GRADED,
|
|
309
385
|
label: "Rubric",
|
|
310
386
|
description: "Asks the grader whether the reply meets the rubric.",
|
|
311
387
|
graded: true,
|
|
@@ -317,6 +393,7 @@ const metrics = {
|
|
|
317
393
|
num(m.threshold) ?? 0.5),
|
|
318
394
|
},
|
|
319
395
|
factuality: {
|
|
396
|
+
family: GRADED,
|
|
320
397
|
label: "Factual",
|
|
321
398
|
description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
|
|
322
399
|
graded: true,
|
|
@@ -329,6 +406,7 @@ const metrics = {
|
|
|
329
406
|
+ `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
|
|
330
407
|
},
|
|
331
408
|
"judge-vs-production": {
|
|
409
|
+
family: GRADED,
|
|
332
410
|
label: "Judged against production",
|
|
333
411
|
description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
|
|
334
412
|
graded: true,
|
package/lab/run-evals.js
CHANGED
|
@@ -101,7 +101,7 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
|
|
|
101
101
|
job's tokenMappings. Default: none, so a prompt
|
|
102
102
|
naming a token is refused, naming it.
|
|
103
103
|
--pipeline <file> a run document (docs/pipeline-model.md) with one
|
|
104
|
-
scenario and a graded
|
|
104
|
+
scenario and a graded eval, graded case by case against
|
|
105
105
|
the set. Instead of --prompt, --model and --url. Each
|
|
106
106
|
profile's key comes from $EVAL_API_KEY_<ID>, never the
|
|
107
107
|
file; its content.files, when not empty, is the file
|
|
@@ -274,7 +274,7 @@ function readRun(file, dataset) {
|
|
|
274
274
|
} catch (e) {
|
|
275
275
|
broken(`${file}: ${e.message}`);
|
|
276
276
|
}
|
|
277
|
-
const named = isObject(doc) ? core.
|
|
277
|
+
const named = isObject(doc) ? core.evalsDataset(doc) : null;
|
|
278
278
|
const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
|
|
279
279
|
if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
|
|
280
280
|
return doc;
|
|
@@ -296,8 +296,8 @@ function readDataset(file) {
|
|
|
296
296
|
broken(`${file}: ${e.message}`);
|
|
297
297
|
}
|
|
298
298
|
if (isObject(doc) && doc.format === "evals-lab/dataset") {
|
|
299
|
-
if (![1, 2, 3, 4].includes(doc.version)) {
|
|
300
|
-
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to
|
|
299
|
+
if (![1, 2, 3, 4, 5].includes(doc.version)) {
|
|
300
|
+
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 5`);
|
|
301
301
|
}
|
|
302
302
|
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
303
303
|
}
|
|
@@ -315,11 +315,12 @@ function readDataset(file) {
|
|
|
315
315
|
return doc;
|
|
316
316
|
}
|
|
317
317
|
|
|
318
|
-
// The cases a graded run is scored against, by the
|
|
318
|
+
// The cases a graded run is scored against, by the item each names: exactly,
|
|
319
|
+
// as the Source names it.
|
|
319
320
|
function gradedBy(dataset) {
|
|
320
321
|
const out = new Map();
|
|
321
322
|
for (const c of core.gradedSetFrom(dataset || {})) {
|
|
322
|
-
if (!c.todo) out.set(c.
|
|
323
|
+
if (!c.todo) out.set(c.item, c);
|
|
323
324
|
}
|
|
324
325
|
return out;
|
|
325
326
|
}
|
|
@@ -612,12 +613,12 @@ function readUtf8(file) {
|
|
|
612
613
|
return fs.readFileSync(file).toString("utf8");
|
|
613
614
|
}
|
|
614
615
|
|
|
615
|
-
// Every
|
|
616
|
-
// the page reads it: a whole-run
|
|
617
|
-
// in, and a per-item
|
|
618
|
-
// stopped short, whose whole-run
|
|
616
|
+
// Every eval's reading of each scenario, keyed by the eval's id, exactly as
|
|
617
|
+
// the page reads it: a whole-run eval's verdict, settled once every item is
|
|
618
|
+
// in, and a per-item eval's counts. [settled] is false for a run that
|
|
619
|
+
// stopped short, whose whole-run evals have not settled.
|
|
619
620
|
const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
|
|
620
|
-
Object.fromEntries(core.
|
|
621
|
+
Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
|
|
621
622
|
name: o.label, skipped: o.skipped,
|
|
622
623
|
...(o.whole
|
|
623
624
|
? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
|
|
@@ -754,7 +755,7 @@ async function runSnapshot(o) {
|
|
|
754
755
|
verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
|
|
755
756
|
run: {
|
|
756
757
|
name: run.name ?? null,
|
|
757
|
-
|
|
758
|
+
evals: run.evals,
|
|
758
759
|
verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
|
|
759
760
|
scenarios: scenariosOf(run),
|
|
760
761
|
items: results.slice(0, total),
|
|
@@ -779,7 +780,7 @@ async function runSnapshot(o) {
|
|
|
779
780
|
/** Re-score a run's stored results against the dataset --dataset hands over,
|
|
780
781
|
* keeping the replies that were stored: what changes a verdict is the scorer
|
|
781
782
|
* or the cases, never a new question to the model. The report is shaped like
|
|
782
|
-
* a --run's, with the stored results as its items, and a whole-run
|
|
783
|
+
* a --run's, with the stored results as its items, and a whole-run eval's
|
|
783
784
|
* verdict settled the same way. */
|
|
784
785
|
async function runRescore(o){
|
|
785
786
|
const dataset = readDataset(o.dataset);
|
|
@@ -801,7 +802,7 @@ async function runRescore(o){
|
|
|
801
802
|
const graded = gradedBy(dataset);
|
|
802
803
|
|
|
803
804
|
const only = o.only != null ? parseInt(o.only, 10) : null;
|
|
804
|
-
//
|
|
805
|
+
// An eval that reads every item (the Metrics) re-reads one with no case
|
|
805
806
|
// too, against the production reply the item kept.
|
|
806
807
|
const items = await Promise.all(results.map(async (it, i) => {
|
|
807
808
|
if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
|
|
@@ -819,7 +820,7 @@ async function runRescore(o){
|
|
|
819
820
|
rescored: true,
|
|
820
821
|
run: {
|
|
821
822
|
name: run.name ?? null,
|
|
822
|
-
|
|
823
|
+
evals: run.evals,
|
|
823
824
|
verdicts: verdictsOf(run, items, true),
|
|
824
825
|
scenarios: scenariosOf(run),
|
|
825
826
|
items,
|
|
@@ -856,8 +857,8 @@ function cliRun(o, dataset) {
|
|
|
856
857
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
857
858
|
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
858
859
|
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
859
|
-
doc.
|
|
860
|
-
mode: "all", threshold: null, grader: null, over: "item", metrics: [
|
|
860
|
+
doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
|
|
861
|
+
mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
|
|
861
862
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
862
863
|
return doc;
|
|
863
864
|
}
|
|
@@ -950,17 +951,19 @@ async function main() {
|
|
|
950
951
|
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
|
951
952
|
+ "against the set -- --run runs them all.");
|
|
952
953
|
}
|
|
953
|
-
if (!core.
|
|
954
|
+
if (!core.evalsDataset(run)) {
|
|
954
955
|
broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
|
|
955
956
|
}
|
|
956
957
|
if (core.contentOf(run).type !== "source") {
|
|
957
958
|
broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
|
|
958
959
|
}
|
|
959
960
|
}
|
|
960
|
-
// Case by case: each item against its case, as the
|
|
961
|
-
//
|
|
961
|
+
// Case by case: each item against its case's metrics, as the eval that
|
|
962
|
+
// names the dataset reads them (core.readCase).
|
|
962
963
|
const { stages, tokens, connections } = core.stagesFor(run, 0);
|
|
963
964
|
const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
|
|
965
|
+
// A last stage that yields no items is matched as its text.
|
|
966
|
+
const plain = !core.OUTPUT_KINDS[stages.at(-1)?.kind ?? ""]?.terms;
|
|
964
967
|
|
|
965
968
|
// The graded half, through the function the tab builds its own list with.
|
|
966
969
|
const set = core.gradedSetFrom(dataset);
|
|
@@ -1050,11 +1053,11 @@ async function main() {
|
|
|
1050
1053
|
+ "--samples or --source names a directory.");
|
|
1051
1054
|
const allowed = itemsOnly ?? graded;
|
|
1052
1055
|
// $snapshotSet above: only a Source run sees it; --replies reads no file.
|
|
1053
|
-
const inSnapshot = c => !snapshotSet || snapshotSet.has(c.
|
|
1056
|
+
const inSnapshot = c => !snapshotSet || snapshotSet.has(c.item);
|
|
1054
1057
|
// Only the run's own image items need ImageMagick: a text-only Source
|
|
1055
1058
|
// -- the .txt, .csv case -- runs with nothing on PATH but the node
|
|
1056
1059
|
// running this, which is the point of having text in the registry.
|
|
1057
|
-
const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.
|
|
1060
|
+
const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.item)) === "image");
|
|
1058
1061
|
if (pictures.length) {
|
|
1059
1062
|
bin = imageMagick();
|
|
1060
1063
|
if (!bin) broken("a run over images needs ImageMagick, to cap each "
|
|
@@ -1087,7 +1090,7 @@ async function main() {
|
|
|
1087
1090
|
}
|
|
1088
1091
|
|
|
1089
1092
|
// Only files the snapshot names are read from the Source.
|
|
1090
|
-
const wanted = c => !snapshotSet || snapshotSet.has(c.
|
|
1093
|
+
const wanted = c => !snapshotSet || snapshotSet.has(c.item);
|
|
1091
1094
|
// The cancel marker is checked between items, so a cancel never loses the
|
|
1092
1095
|
// reply in flight: the item that was being asked finishes, and the rest
|
|
1093
1096
|
// are unrun rather than half-asked.
|
|
@@ -1111,18 +1114,18 @@ async function main() {
|
|
|
1111
1114
|
calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
|
|
1112
1115
|
} else {
|
|
1113
1116
|
if (!wanted(kase)) {
|
|
1114
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1117
|
+
unrun.push(`${kase.id}: ${kase.item} is not in the run's file list`);
|
|
1115
1118
|
continue;
|
|
1116
1119
|
}
|
|
1117
|
-
const file = path.join(samples, kase.
|
|
1120
|
+
const file = path.join(samples, kase.item);
|
|
1118
1121
|
if (!fs.existsSync(file)) {
|
|
1119
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1122
|
+
unrun.push(`${kase.id}: ${kase.item} is not in ${samples}`);
|
|
1120
1123
|
continue;
|
|
1121
1124
|
}
|
|
1122
1125
|
try {
|
|
1123
1126
|
item = itemContent(bin, file, connections[0]);
|
|
1124
1127
|
} catch (e) {
|
|
1125
|
-
unrun.push(`${kase.id}: ${kase.
|
|
1128
|
+
unrun.push(`${kase.id}: ${kase.item} could not be prepared — ${e.message}`);
|
|
1126
1129
|
continue;
|
|
1127
1130
|
}
|
|
1128
1131
|
calls = callsFor(connections, links, item?.text ?? null);
|
|
@@ -1130,16 +1133,15 @@ async function main() {
|
|
|
1130
1133
|
|
|
1131
1134
|
const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
|
|
1132
1135
|
{ tokens, text: item?.text ?? null });
|
|
1133
|
-
const s = core.
|
|
1136
|
+
const s = core.readCase(kase, res, plain);
|
|
1134
1137
|
// A file with no mapper ran anyway; the flag keeps the transcript honest
|
|
1135
1138
|
// about what the model was actually shown. §4: stated, not hidden.
|
|
1136
1139
|
if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
|
|
1137
1140
|
rows.push({
|
|
1138
|
-
id: kase.id,
|
|
1141
|
+
id: kase.id, item: kase.item, half: kase.half,
|
|
1139
1142
|
pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
|
|
1140
|
-
found: s.found, missed: s.missed, invented: s.invented,
|
|
1141
|
-
|
|
1142
|
-
watchFound: s.watchFound, watchTotal: s.watchTotal,
|
|
1143
|
+
found: s.found, missed: s.missed, invented: s.invented, count: s.count,
|
|
1144
|
+
metrics: s.metrics, watch: s.watch,
|
|
1143
1145
|
terms: res.terms || [], ms: res.ms,
|
|
1144
1146
|
// What a file type was to this run -- "image", "text", "unmapped".
|
|
1145
1147
|
type: item?.kind ?? null,
|
|
@@ -1153,7 +1155,7 @@ async function main() {
|
|
|
1153
1155
|
});
|
|
1154
1156
|
if (progress) {
|
|
1155
1157
|
progress.write(JSON.stringify({
|
|
1156
|
-
event: "item", id: kase.id,
|
|
1158
|
+
event: "item", id: kase.id, item: kase.item, n: rows.length,
|
|
1157
1159
|
pass: s.pass, score: s.score, discarded: s.discarded,
|
|
1158
1160
|
error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
|
|
1159
1161
|
ms: res.ms,
|