evals-lab 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -332,7 +332,7 @@ const LIST_MODIFIERS = {
332
332
 
333
333
  const LIST_KIND = {
334
334
  label: "List", noun: "items", offers: ["items"],
335
- description: "Splits the reply into items (by commas, lines or a JSON array) that tests and later jobs read one by one.",
335
+ description: "Splits the reply into items (by commas, lines or a JSON array) that evals and later jobs read one by one.",
336
336
  settings: [
337
337
  { key: "parse", label: "Parse as", type: "select", choices: [
338
338
  { value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
@@ -3,13 +3,13 @@
3
3
  // read the text, ones that read it against what production replied, and
4
4
  // model-graded ones that ask a grader. Each is one METRICS entry the core
5
5
  // registers, as it registers kinds/list.ts; a plugin adds more the same way.
6
-
6
+
7
7
 
8
8
  const ok = (pass , reason , score = pass ? 1 : 0) => ({ pass, score, reason });
9
9
  const str = (v ) => (v == null ? "" : String(v));
10
10
  const num = (v ) => (typeof v === "number" ? v : str(v).trim() && !Number.isNaN(Number(v)) ? Number(v) : null);
11
- /** A list option: one entry a line. */
12
- const lines = (v ) => str(v).split("\n").map((l) => l.trim()).filter(Boolean);
11
+ /** A list option: one entry a line, or a list of them. */
12
+ const lines = (v ) => (Array.isArray(v) ? v.map(str) : str(v).split("\n")).map((l) => l.trim()).filter(Boolean);
13
13
  const short = (s ) => (s.length > 60 ? `${s.slice(0, 57)}…` : s);
14
14
  const json = (s ) => {
15
15
  try { return { value: JSON.parse(s) }; } catch (e) { return { error: e instanceof Error ? e.message : String(e) }; }
@@ -81,6 +81,45 @@ function distance(a , b ) {
81
81
  return d[b.length] ;
82
82
  }
83
83
 
84
+ // ---- matching what a reply holds ---------------------------------------------
85
+ // A kind that yields items (a list) is matched item by item, the way the lab
86
+ // matches a term (docs/datasets.md § Matching): a value's words in some item,
87
+ // in order and adjacent -- "cat" is in "tabby cat" and not in "cathedral" --
88
+ // and ignoring case where the metric's Ignore case is on, as it is by default.
89
+ // A reply the job threw away yields none. A reply read as text, or the API's
90
+ // own text (`of: "said"`), is matched as text.
91
+
92
+ /** The words of [s], lowercased unless [keepCase]: the lab's own reading of a term. */
93
+ const words = (s , keepCase = false) => (keepCase ? s : s.toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
94
+ const termIn = (items , term , ignoreCase = true) => {
95
+ const t = words(term, !ignoreCase);
96
+ if (!t.length) return false;
97
+ return items.some((item) => {
98
+ const w = words(item, !ignoreCase);
99
+ for (let i = 0; i + t.length <= w.length; i++) if (t.every((x, j) => w[i + j] === x)) return true;
100
+ return false;
101
+ });
102
+ };
103
+ /** Whether [input] is matched item by item. */
104
+ const byItem = (input , m ) => !input.plain && m.of !== "said";
105
+ /** Whether [input] holds [value], as [m] says to match it: in its items, or
106
+ in its text -- there, outside any of [except]. */
107
+ function holds(input , m , value , except = []) {
108
+ if (byItem(input, m)) {
109
+ const items = input.error ? [] : input.terms;
110
+ const ic = m.ignoreCase === true;
111
+ // An item holding an exception that holds the value does not count; the
112
+ // value anywhere else still does.
113
+ return items.some((item) => termIn([item], value, ic) && !except.some((e) => termIn([e], value, ic) && termIn([item], e, ic)));
114
+ }
115
+ const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
116
+ let text = fold(input.text);
117
+ for (const e of except) if (fold(e).includes(fold(value))) text = text.split(fold(e)).join(" ");
118
+ return text.includes(fold(value));
119
+ }
120
+ /** A requirement as Results names it: a term, or the group it was. */
121
+ const spoken = (g ) => (g.length === 1 ? g[0] : g);
122
+
84
123
  /** Production's reply, or the reason a metric that compares with it cannot. */
85
124
  const production = (input ) => {
86
125
  if (input.production == null) throw new Error("this item has no production reply to compare with");
@@ -107,8 +146,12 @@ const gradedPass = (v , threshold )
107
146
  };
108
147
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
109
148
 
149
+ // The families the Add metric picker lists the metrics under: what each reads.
150
+ const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
151
+
110
152
  const metrics = {
111
153
  equals: {
154
+ family: TEXT,
112
155
  label: "Equals",
113
156
  description: "Passes when the reply is exactly the value; as JSON, the same value with keys in any order.",
114
157
  // As JSON, a reply equals the value as a value -- keys in any order;
@@ -123,44 +166,53 @@ const metrics = {
123
166
  },
124
167
  },
125
168
  contains: {
169
+ family: TEXT,
126
170
  label: "Contains",
127
- description: "Passes when the reply contains the value.",
128
- options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
129
- defaults: () => ({ value: "", ignoreCase: false }),
171
+ description: "Passes when the reply contains the value; turned round, an exception excuses the value inside it.",
172
+ options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" },
173
+ { key: "except", label: "Except", type: "textarea" }],
174
+ defaults: () => ({ value: "", ignoreCase: true }),
130
175
  validate: (m, at, bad) => needs(m, ["value"], at, bad),
176
+ // Turned round (`not`), what it found is what the reply invented; a
177
+ // reading of items says so, for a run's totals.
131
178
  score: (input, m) => {
132
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
133
- const hit = fold(input.text).includes(fold(str(m.value)));
134
- return ok(hit, hit ? `has ${short(str(m.value))}` : `has no ${short(str(m.value))}`);
179
+ const v = str(m.value);
180
+ const hit = holds(input, m, v, lines(m.except));
181
+ const detail = !byItem(input, m) ? {} : m.not ? { invented: hit ? [v] : [] } : hit ? { found: [v], missed: [] } : { found: [], missed: [v] };
182
+ return { ...ok(hit, hit ? `has ${short(v)}` : `has no ${short(v)}`), ...detail };
135
183
  },
136
184
  },
137
185
  "contains-any": {
186
+ family: TEXT,
138
187
  label: "Contains any",
139
188
  description: "Passes when the reply contains at least one of the values, one a line.",
140
189
  options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
141
- defaults: () => ({ values: "", ignoreCase: false }),
190
+ defaults: () => ({ values: "", ignoreCase: true }),
142
191
  validate: (m, at, bad) => needs(m, ["values"], at, bad),
143
192
  score: (input, m) => {
144
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
145
- const hit = lines(m.values).find((v) => fold(input.text).includes(fold(v)));
146
- return ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${lines(m.values).join(", ")}`);
193
+ const all = lines(m.values);
194
+ const hit = all.find((v) => holds(input, m, v));
195
+ const detail = !byItem(input, m) ? {} : hit ? { found: [spoken(all)], missed: [] } : { found: [], missed: [spoken(all)] };
196
+ return { ...ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${all.join(", ")}`), ...detail };
147
197
  },
148
198
  },
149
199
  "contains-all": {
200
+ family: TEXT,
150
201
  label: "Contains all",
151
202
  description: "Passes when the reply contains every value, one a line.",
152
203
  options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
153
- defaults: () => ({ values: "", ignoreCase: false }),
204
+ defaults: () => ({ values: "", ignoreCase: true }),
154
205
  validate: (m, at, bad) => needs(m, ["values"], at, bad),
155
206
  score: (input, m) => {
156
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
157
207
  const all = lines(m.values);
158
- const missing = all.filter((v) => !fold(input.text).includes(fold(v)));
159
- return ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
160
- all.length ? (all.length - missing.length) / all.length : 1);
208
+ const missing = all.filter((v) => !holds(input, m, v));
209
+ const detail = !byItem(input, m) ? {} : { found: all.filter((v) => !missing.includes(v)), missed: missing };
210
+ return { ...ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
211
+ all.length ? (all.length - missing.length) / all.length : 1), ...detail };
161
212
  },
162
213
  },
163
214
  regex: {
215
+ family: TEXT,
164
216
  label: "Matches",
165
217
  description: "Passes when the reply matches the regular expression.",
166
218
  options: [{ key: "pattern", label: "Pattern", type: "text" }, { key: "flags", label: "Flags", type: "text" }],
@@ -172,6 +224,7 @@ const metrics = {
172
224
  },
173
225
  },
174
226
  "starts-with": {
227
+ family: TEXT,
175
228
  label: "Starts with",
176
229
  description: "Passes when the reply starts with the value.",
177
230
  options: [{ key: "value", label: "Value", type: "text" }],
@@ -183,6 +236,7 @@ const metrics = {
183
236
  },
184
237
  },
185
238
  "is-json": {
239
+ family: TEXT,
186
240
  label: "Is JSON",
187
241
  description: "Passes when the reply is JSON and, with a schema, holds to it.",
188
242
  options: [{ key: "schema", label: "Schema", type: "textarea" }],
@@ -197,6 +251,7 @@ const metrics = {
197
251
  },
198
252
  },
199
253
  "json-field": {
254
+ family: TEXT,
200
255
  label: "JSON field",
201
256
  description: "Reads one field of a JSON reply and compares it.",
202
257
  options: [
@@ -219,7 +274,21 @@ const metrics = {
219
274
  return ok(pass, `${str(m.path)} is ${short(text)}`);
220
275
  },
221
276
  },
277
+ // A case that expects its answer thrown away: the reply a job's Reject
278
+ // rules discarded is the finding.
279
+ discarded: {
280
+ family: RESULT,
281
+ label: "Discarded",
282
+ description: "Passes only when the job's rules discarded the reply.",
283
+ options: [],
284
+ defaults: () => ({}),
285
+ score: (input) => {
286
+ const thrown = !!input.error && input.error.startsWith("discarded: ");
287
+ return ok(thrown, thrown ? "discarded" : input.error ? input.error : `kept ${input.terms.length} items`);
288
+ },
289
+ },
222
290
  "item-count": {
291
+ family: RESULT,
223
292
  label: "Item count",
224
293
  description: "Passes when the number of items is within the bounds.",
225
294
  options: [{ key: "min", label: "At least", type: "number" }, { key: "max", label: "At most", type: "number" }],
@@ -231,6 +300,7 @@ const metrics = {
231
300
  },
232
301
  },
233
302
  "count-matches": {
303
+ family: RESULT,
234
304
  label: "Count matches",
235
305
  description: "Counts matches of the pattern: points, for the weighted mode.",
236
306
  counts: true,
@@ -246,6 +316,7 @@ const metrics = {
246
316
  },
247
317
  },
248
318
  latency: {
319
+ family: RESULT,
249
320
  label: "Latency",
250
321
  description: "Passes when the reply came back within the time set.",
251
322
  options: [{ key: "max", label: "At most (ms)", type: "number" }],
@@ -254,6 +325,7 @@ const metrics = {
254
325
  score: (input, m) => ok(input.ms <= (num(m.max) ?? 0), `${input.ms} ms`),
255
326
  },
256
327
  levenshtein: {
328
+ family: TEXT,
257
329
  label: "Near",
258
330
  description: "Passes when the reply is within a few edits of the value.",
259
331
  options: [{ key: "value", label: "Value", type: "textarea" }, { key: "max", label: "Edits at most", type: "number" }],
@@ -267,6 +339,7 @@ const metrics = {
267
339
  },
268
340
  // ---- against production ----------------------------------------------------
269
341
  "equals-production": {
342
+ family: PROD,
270
343
  label: "Same as production",
271
344
  description: "Passes when the reply is what production replied to the same item.",
272
345
  options: [],
@@ -279,6 +352,7 @@ const metrics = {
279
352
  },
280
353
  },
281
354
  "fields-equal-production": {
355
+ family: PROD,
282
356
  label: "Fields as production",
283
357
  description: "Passes when the fields listed are what production's reply had.",
284
358
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
@@ -295,6 +369,7 @@ const metrics = {
295
369
  },
296
370
  },
297
371
  "same-parse-outcome": {
372
+ family: PROD,
298
373
  label: "Parses as production",
299
374
  description: "Passes when the reply parses as JSON exactly when production's did.",
300
375
  options: [],
@@ -306,6 +381,7 @@ const metrics = {
306
381
  },
307
382
  // ---- model-graded -------------------------------------------------------------
308
383
  "llm-rubric": {
384
+ family: GRADED,
309
385
  label: "Rubric",
310
386
  description: "Asks the grader whether the reply meets the rubric.",
311
387
  graded: true,
@@ -317,6 +393,7 @@ const metrics = {
317
393
  num(m.threshold) ?? 0.5),
318
394
  },
319
395
  factuality: {
396
+ family: GRADED,
320
397
  label: "Factual",
321
398
  description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
322
399
  graded: true,
@@ -329,6 +406,7 @@ const metrics = {
329
406
  + `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
330
407
  },
331
408
  "judge-vs-production": {
409
+ family: GRADED,
332
410
  label: "Judged against production",
333
411
  description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
334
412
  graded: true,
package/lab/run-evals.js CHANGED
@@ -101,7 +101,7 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
101
101
  job's tokenMappings. Default: none, so a prompt
102
102
  naming a token is refused, naming it.
103
103
  --pipeline <file> a run document (docs/pipeline-model.md) with one
104
- scenario and a graded test, graded case by case against
104
+ scenario and a graded eval, graded case by case against
105
105
  the set. Instead of --prompt, --model and --url. Each
106
106
  profile's key comes from $EVAL_API_KEY_<ID>, never the
107
107
  file; its content.files, when not empty, is the file
@@ -274,7 +274,7 @@ function readRun(file, dataset) {
274
274
  } catch (e) {
275
275
  broken(`${file}: ${e.message}`);
276
276
  }
277
- const named = isObject(doc) ? core.testsDataset(doc) : null;
277
+ const named = isObject(doc) ? core.evalsDataset(doc) : null;
278
278
  const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
279
279
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
280
280
  return doc;
@@ -296,8 +296,8 @@ function readDataset(file) {
296
296
  broken(`${file}: ${e.message}`);
297
297
  }
298
298
  if (isObject(doc) && doc.format === "evals-lab/dataset") {
299
- if (![1, 2, 3, 4].includes(doc.version)) {
300
- broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 4`);
299
+ if (![1, 2, 3, 4, 5].includes(doc.version)) {
300
+ broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 5`);
301
301
  }
302
302
  doc = isObject(doc.dataset) ? doc.dataset.body : null;
303
303
  }
@@ -315,11 +315,12 @@ function readDataset(file) {
315
315
  return doc;
316
316
  }
317
317
 
318
- // The cases a graded run is scored against, by the file each names.
318
+ // The cases a graded run is scored against, by the item each names: exactly,
319
+ // as the Source names it.
319
320
  function gradedBy(dataset) {
320
321
  const out = new Map();
321
322
  for (const c of core.gradedSetFrom(dataset || {})) {
322
- if (!c.todo) out.set(c.filename, c);
323
+ if (!c.todo) out.set(c.item, c);
323
324
  }
324
325
  return out;
325
326
  }
@@ -612,12 +613,12 @@ function readUtf8(file) {
612
613
  return fs.readFileSync(file).toString("utf8");
613
614
  }
614
615
 
615
- // Every test's reading of each scenario, keyed by the test's id, exactly as
616
- // the page reads it: a whole-run test's verdict, settled once every item is
617
- // in, and a per-item test's counts. [settled] is false for a run that
618
- // stopped short, whose whole-run tests have not settled.
616
+ // Every eval's reading of each scenario, keyed by the eval's id, exactly as
617
+ // the page reads it: a whole-run eval's verdict, settled once every item is
618
+ // in, and a per-item eval's counts. [settled] is false for a run that
619
+ // stopped short, whose whole-run evals have not settled.
619
620
  const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
620
- Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
621
+ Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
621
622
  name: o.label, skipped: o.skipped,
622
623
  ...(o.whole
623
624
  ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
@@ -754,7 +755,7 @@ async function runSnapshot(o) {
754
755
  verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
755
756
  run: {
756
757
  name: run.name ?? null,
757
- tests: run.tests,
758
+ evals: run.evals,
758
759
  verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
759
760
  scenarios: scenariosOf(run),
760
761
  items: results.slice(0, total),
@@ -779,7 +780,7 @@ async function runSnapshot(o) {
779
780
  /** Re-score a run's stored results against the dataset --dataset hands over,
780
781
  * keeping the replies that were stored: what changes a verdict is the scorer
781
782
  * or the cases, never a new question to the model. The report is shaped like
782
- * a --run's, with the stored results as its items, and a whole-run test's
783
+ * a --run's, with the stored results as its items, and a whole-run eval's
783
784
  * verdict settled the same way. */
784
785
  async function runRescore(o){
785
786
  const dataset = readDataset(o.dataset);
@@ -801,7 +802,7 @@ async function runRescore(o){
801
802
  const graded = gradedBy(dataset);
802
803
 
803
804
  const only = o.only != null ? parseInt(o.only, 10) : null;
804
- // A test that reads every item (the Metrics) re-reads one with no case
805
+ // An eval that reads every item (the Metrics) re-reads one with no case
805
806
  // too, against the production reply the item kept.
806
807
  const items = await Promise.all(results.map(async (it, i) => {
807
808
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
@@ -819,7 +820,7 @@ async function runRescore(o){
819
820
  rescored: true,
820
821
  run: {
821
822
  name: run.name ?? null,
822
- tests: run.tests,
823
+ evals: run.evals,
823
824
  verdicts: verdictsOf(run, items, true),
824
825
  scenarios: scenariosOf(run),
825
826
  items,
@@ -856,8 +857,8 @@ function cliRun(o, dataset) {
856
857
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
857
858
  doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
858
859
  steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
859
- doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
860
- mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
860
+ doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
861
+ mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
861
862
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
862
863
  return doc;
863
864
  }
@@ -950,17 +951,19 @@ async function main() {
950
951
  broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
951
952
  + "against the set -- --run runs them all.");
952
953
  }
953
- if (!core.testsDataset(run)) {
954
+ if (!core.evalsDataset(run)) {
954
955
  broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
955
956
  }
956
957
  if (core.contentOf(run).type !== "source") {
957
958
  broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
958
959
  }
959
960
  }
960
- // Case by case: each item against its case, as the case metric of the
961
- // test that names the dataset reads it (scoreCase).
961
+ // Case by case: each item against its case's metrics, as the eval that
962
+ // names the dataset reads them (core.readCase).
962
963
  const { stages, tokens, connections } = core.stagesFor(run, 0);
963
964
  const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
965
+ // A last stage that yields no items is matched as its text.
966
+ const plain = !core.OUTPUT_KINDS[stages.at(-1)?.kind ?? ""]?.terms;
964
967
 
965
968
  // The graded half, through the function the tab builds its own list with.
966
969
  const set = core.gradedSetFrom(dataset);
@@ -1050,11 +1053,11 @@ async function main() {
1050
1053
  + "--samples or --source names a directory.");
1051
1054
  const allowed = itemsOnly ?? graded;
1052
1055
  // $snapshotSet above: only a Source run sees it; --replies reads no file.
1053
- const inSnapshot = c => !snapshotSet || snapshotSet.has(c.filename);
1056
+ const inSnapshot = c => !snapshotSet || snapshotSet.has(c.item);
1054
1057
  // Only the run's own image items need ImageMagick: a text-only Source
1055
1058
  // -- the .txt, .csv case -- runs with nothing on PATH but the node
1056
1059
  // running this, which is the point of having text in the registry.
1057
- const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.filename)) === "image");
1060
+ const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.item)) === "image");
1058
1061
  if (pictures.length) {
1059
1062
  bin = imageMagick();
1060
1063
  if (!bin) broken("a run over images needs ImageMagick, to cap each "
@@ -1087,7 +1090,7 @@ async function main() {
1087
1090
  }
1088
1091
 
1089
1092
  // Only files the snapshot names are read from the Source.
1090
- const wanted = c => !snapshotSet || snapshotSet.has(c.filename);
1093
+ const wanted = c => !snapshotSet || snapshotSet.has(c.item);
1091
1094
  // The cancel marker is checked between items, so a cancel never loses the
1092
1095
  // reply in flight: the item that was being asked finishes, and the rest
1093
1096
  // are unrun rather than half-asked.
@@ -1111,18 +1114,18 @@ async function main() {
1111
1114
  calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
1112
1115
  } else {
1113
1116
  if (!wanted(kase)) {
1114
- unrun.push(`${kase.id}: ${kase.filename} is not in the run's file list`);
1117
+ unrun.push(`${kase.id}: ${kase.item} is not in the run's file list`);
1115
1118
  continue;
1116
1119
  }
1117
- const file = path.join(samples, kase.filename);
1120
+ const file = path.join(samples, kase.item);
1118
1121
  if (!fs.existsSync(file)) {
1119
- unrun.push(`${kase.id}: ${kase.filename} is not in ${samples}`);
1122
+ unrun.push(`${kase.id}: ${kase.item} is not in ${samples}`);
1120
1123
  continue;
1121
1124
  }
1122
1125
  try {
1123
1126
  item = itemContent(bin, file, connections[0]);
1124
1127
  } catch (e) {
1125
- unrun.push(`${kase.id}: ${kase.filename} could not be prepared — ${e.message}`);
1128
+ unrun.push(`${kase.id}: ${kase.item} could not be prepared — ${e.message}`);
1126
1129
  continue;
1127
1130
  }
1128
1131
  calls = callsFor(connections, links, item?.text ?? null);
@@ -1130,16 +1133,15 @@ async function main() {
1130
1133
 
1131
1134
  const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
1132
1135
  { tokens, text: item?.text ?? null });
1133
- const s = core.scoreCase(kase, res);
1136
+ const s = core.readCase(kase, res, plain);
1134
1137
  // A file with no mapper ran anyway; the flag keeps the transcript honest
1135
1138
  // about what the model was actually shown. §4: stated, not hidden.
1136
1139
  if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
1137
1140
  rows.push({
1138
- id: kase.id, filename: kase.filename, half: kase.half,
1141
+ id: kase.id, item: kase.item, half: kase.half,
1139
1142
  pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
1140
- found: s.found, missed: s.missed, invented: s.invented, unmet: s.unmet,
1141
- under: s.under, over: s.over, count: s.count,
1142
- watchFound: s.watchFound, watchTotal: s.watchTotal,
1143
+ found: s.found, missed: s.missed, invented: s.invented, count: s.count,
1144
+ metrics: s.metrics, watch: s.watch,
1143
1145
  terms: res.terms || [], ms: res.ms,
1144
1146
  // What a file type was to this run -- "image", "text", "unmapped".
1145
1147
  type: item?.kind ?? null,
@@ -1153,7 +1155,7 @@ async function main() {
1153
1155
  });
1154
1156
  if (progress) {
1155
1157
  progress.write(JSON.stringify({
1156
- event: "item", id: kase.id, filename: kase.filename, n: rows.length,
1158
+ event: "item", id: kase.id, item: kase.item, n: rows.length,
1157
1159
  pass: s.pass, score: s.score, discarded: s.discarded,
1158
1160
  error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
1159
1161
  ms: res.ms,