evals-lab 0.1.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -332,7 +332,7 @@ const LIST_MODIFIERS = {
332
332
 
333
333
  const LIST_KIND = {
334
334
  label: "List", noun: "items", offers: ["items"],
335
- description: "Splits the reply into items (by commas, lines or a JSON array) that tests and later jobs read one by one.",
335
+ description: "Splits the reply into items (by commas, lines or a JSON array) that evals and later jobs read one by one.",
336
336
  settings: [
337
337
  { key: "parse", label: "Parse as", type: "select", choices: [
338
338
  { value: "csv", label: "CSV" }, { value: "lines", label: "One a line" }, { value: "json", label: "JSON array" }] },
@@ -3,13 +3,13 @@
3
3
  // read the text, ones that read it against what production replied, and
4
4
  // model-graded ones that ask a grader. Each is one METRICS entry the core
5
5
  // registers, as it registers kinds/list.ts; a plugin adds more the same way.
6
-
6
+
7
7
 
8
8
  const ok = (pass , reason , score = pass ? 1 : 0) => ({ pass, score, reason });
9
9
  const str = (v ) => (v == null ? "" : String(v));
10
10
  const num = (v ) => (typeof v === "number" ? v : str(v).trim() && !Number.isNaN(Number(v)) ? Number(v) : null);
11
- /** A list option: one entry a line. */
12
- const lines = (v ) => str(v).split("\n").map((l) => l.trim()).filter(Boolean);
11
+ /** A list option: one entry a line, or a list of them. */
12
+ const lines = (v ) => (Array.isArray(v) ? v.map(str) : str(v).split("\n")).map((l) => l.trim()).filter(Boolean);
13
13
  const short = (s ) => (s.length > 60 ? `${s.slice(0, 57)}…` : s);
14
14
  const json = (s ) => {
15
15
  try { return { value: JSON.parse(s) }; } catch (e) { return { error: e instanceof Error ? e.message : String(e) }; }
@@ -81,6 +81,45 @@ function distance(a , b ) {
81
81
  return d[b.length] ;
82
82
  }
83
83
 
84
+ // ---- matching what a reply holds ---------------------------------------------
85
+ // A kind that yields items (a list) is matched item by item, the way the lab
86
+ // matches a term (docs/datasets.md § Matching): a value's words in some item,
87
+ // in order and adjacent -- "cat" is in "tabby cat" and not in "cathedral" --
88
+ // and ignoring case where the metric's Ignore case is on, as it is by default.
89
+ // A reply the job threw away yields none. A reply read as text, or the API's
90
+ // own text (`of: "said"`), is matched as text.
91
+
92
+ /** The words of [s], lowercased unless [keepCase]: the lab's own reading of a term. */
93
+ const words = (s , keepCase = false) => (keepCase ? s : s.toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
94
+ const termIn = (items , term , ignoreCase = true) => {
95
+ const t = words(term, !ignoreCase);
96
+ if (!t.length) return false;
97
+ return items.some((item) => {
98
+ const w = words(item, !ignoreCase);
99
+ for (let i = 0; i + t.length <= w.length; i++) if (t.every((x, j) => w[i + j] === x)) return true;
100
+ return false;
101
+ });
102
+ };
103
+ /** Whether [input] is matched item by item. */
104
+ const byItem = (input , m ) => !input.plain && m.of !== "said";
105
+ /** Whether [input] holds [value], as [m] says to match it: in its items, or
106
+ in its text -- there, outside any of [except]. */
107
+ function holds(input , m , value , except = []) {
108
+ if (byItem(input, m)) {
109
+ const items = input.error ? [] : input.terms;
110
+ const ic = m.ignoreCase === true;
111
+ // An item holding an exception that holds the value does not count; the
112
+ // value anywhere else still does.
113
+ return items.some((item) => termIn([item], value, ic) && !except.some((e) => termIn([e], value, ic) && termIn([item], e, ic)));
114
+ }
115
+ const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
116
+ let text = fold(input.text);
117
+ for (const e of except) if (fold(e).includes(fold(value))) text = text.split(fold(e)).join(" ");
118
+ return text.includes(fold(value));
119
+ }
120
+ /** A requirement as Results names it: a term, or the group it was. */
121
+ const spoken = (g ) => (g.length === 1 ? g[0] : g);
122
+
84
123
  /** Production's reply, or the reason a metric that compares with it cannot. */
85
124
  const production = (input ) => {
86
125
  if (input.production == null) throw new Error("this item has no production reply to compare with");
@@ -107,8 +146,12 @@ const gradedPass = (v , threshold )
107
146
  };
108
147
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
109
148
 
149
+ // The families the Add metric picker lists the metrics under: what each reads.
150
+ const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
151
+
110
152
  const metrics = {
111
153
  equals: {
154
+ family: TEXT,
112
155
  label: "Equals",
113
156
  description: "Passes when the reply is exactly the value; as JSON, the same value with keys in any order.",
114
157
  // As JSON, a reply equals the value as a value -- keys in any order;
@@ -123,44 +166,53 @@ const metrics = {
123
166
  },
124
167
  },
125
168
  contains: {
169
+ family: TEXT,
126
170
  label: "Contains",
127
- description: "Passes when the reply contains the value.",
128
- options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
129
- defaults: () => ({ value: "", ignoreCase: false }),
171
+ description: "Passes when the reply contains the value; turned round, an exception excuses the value inside it.",
172
+ options: [{ key: "value", label: "Value", type: "text" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" },
173
+ { key: "except", label: "Except", type: "textarea" }],
174
+ defaults: () => ({ value: "", ignoreCase: true }),
130
175
  validate: (m, at, bad) => needs(m, ["value"], at, bad),
176
+ // Turned round (`not`), what it found is what the reply invented; a
177
+ // reading of items says so, for a run's totals.
131
178
  score: (input, m) => {
132
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
133
- const hit = fold(input.text).includes(fold(str(m.value)));
134
- return ok(hit, hit ? `has ${short(str(m.value))}` : `has no ${short(str(m.value))}`);
179
+ const v = str(m.value);
180
+ const hit = holds(input, m, v, lines(m.except));
181
+ const detail = !byItem(input, m) ? {} : m.not ? { invented: hit ? [v] : [] } : hit ? { found: [v], missed: [] } : { found: [], missed: [v] };
182
+ return { ...ok(hit, hit ? `has ${short(v)}` : `has no ${short(v)}`), ...detail };
135
183
  },
136
184
  },
137
185
  "contains-any": {
186
+ family: TEXT,
138
187
  label: "Contains any",
139
188
  description: "Passes when the reply contains at least one of the values, one a line.",
140
189
  options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
141
- defaults: () => ({ values: "", ignoreCase: false }),
190
+ defaults: () => ({ values: "", ignoreCase: true }),
142
191
  validate: (m, at, bad) => needs(m, ["values"], at, bad),
143
192
  score: (input, m) => {
144
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
145
- const hit = lines(m.values).find((v) => fold(input.text).includes(fold(v)));
146
- return ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${lines(m.values).join(", ")}`);
193
+ const all = lines(m.values);
194
+ const hit = all.find((v) => holds(input, m, v));
195
+ const detail = !byItem(input, m) ? {} : hit ? { found: [spoken(all)], missed: [] } : { found: [], missed: [spoken(all)] };
196
+ return { ...ok(!!hit, hit ? `has ${short(hit)}` : `has none of ${all.join(", ")}`), ...detail };
147
197
  },
148
198
  },
149
199
  "contains-all": {
200
+ family: TEXT,
150
201
  label: "Contains all",
151
202
  description: "Passes when the reply contains every value, one a line.",
152
203
  options: [{ key: "values", label: "Values", type: "textarea" }, { key: "ignoreCase", label: "Ignore case", type: "checkbox" }],
153
- defaults: () => ({ values: "", ignoreCase: false }),
204
+ defaults: () => ({ values: "", ignoreCase: true }),
154
205
  validate: (m, at, bad) => needs(m, ["values"], at, bad),
155
206
  score: (input, m) => {
156
- const fold = (s ) => (m.ignoreCase ? s.toLowerCase() : s);
157
207
  const all = lines(m.values);
158
- const missing = all.filter((v) => !fold(input.text).includes(fold(v)));
159
- return ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
160
- all.length ? (all.length - missing.length) / all.length : 1);
208
+ const missing = all.filter((v) => !holds(input, m, v));
209
+ const detail = !byItem(input, m) ? {} : { found: all.filter((v) => !missing.includes(v)), missed: missing };
210
+ return { ...ok(!missing.length, missing.length ? `has no ${missing.join(", ")}` : "has them all",
211
+ all.length ? (all.length - missing.length) / all.length : 1), ...detail };
161
212
  },
162
213
  },
163
214
  regex: {
215
+ family: TEXT,
164
216
  label: "Matches",
165
217
  description: "Passes when the reply matches the regular expression.",
166
218
  options: [{ key: "pattern", label: "Pattern", type: "text" }, { key: "flags", label: "Flags", type: "text" }],
@@ -172,6 +224,7 @@ const metrics = {
172
224
  },
173
225
  },
174
226
  "starts-with": {
227
+ family: TEXT,
175
228
  label: "Starts with",
176
229
  description: "Passes when the reply starts with the value.",
177
230
  options: [{ key: "value", label: "Value", type: "text" }],
@@ -183,6 +236,7 @@ const metrics = {
183
236
  },
184
237
  },
185
238
  "is-json": {
239
+ family: TEXT,
186
240
  label: "Is JSON",
187
241
  description: "Passes when the reply is JSON and, with a schema, holds to it.",
188
242
  options: [{ key: "schema", label: "Schema", type: "textarea" }],
@@ -197,6 +251,7 @@ const metrics = {
197
251
  },
198
252
  },
199
253
  "json-field": {
254
+ family: TEXT,
200
255
  label: "JSON field",
201
256
  description: "Reads one field of a JSON reply and compares it.",
202
257
  options: [
@@ -219,7 +274,21 @@ const metrics = {
219
274
  return ok(pass, `${str(m.path)} is ${short(text)}`);
220
275
  },
221
276
  },
277
+ // A case that expects its answer thrown away: the reply a job's Reject
278
+ // rules discarded is the finding.
279
+ discarded: {
280
+ family: RESULT,
281
+ label: "Discarded",
282
+ description: "Passes only when the job's rules discarded the reply.",
283
+ options: [],
284
+ defaults: () => ({}),
285
+ score: (input) => {
286
+ const thrown = !!input.error && input.error.startsWith("discarded: ");
287
+ return ok(thrown, thrown ? "discarded" : input.error ? input.error : `kept ${input.terms.length} items`);
288
+ },
289
+ },
222
290
  "item-count": {
291
+ family: RESULT,
223
292
  label: "Item count",
224
293
  description: "Passes when the number of items is within the bounds.",
225
294
  options: [{ key: "min", label: "At least", type: "number" }, { key: "max", label: "At most", type: "number" }],
@@ -231,6 +300,7 @@ const metrics = {
231
300
  },
232
301
  },
233
302
  "count-matches": {
303
+ family: RESULT,
234
304
  label: "Count matches",
235
305
  description: "Counts matches of the pattern: points, for the weighted mode.",
236
306
  counts: true,
@@ -246,6 +316,7 @@ const metrics = {
246
316
  },
247
317
  },
248
318
  latency: {
319
+ family: RESULT,
249
320
  label: "Latency",
250
321
  description: "Passes when the reply came back within the time set.",
251
322
  options: [{ key: "max", label: "At most (ms)", type: "number" }],
@@ -254,6 +325,7 @@ const metrics = {
254
325
  score: (input, m) => ok(input.ms <= (num(m.max) ?? 0), `${input.ms} ms`),
255
326
  },
256
327
  levenshtein: {
328
+ family: TEXT,
257
329
  label: "Near",
258
330
  description: "Passes when the reply is within a few edits of the value.",
259
331
  options: [{ key: "value", label: "Value", type: "textarea" }, { key: "max", label: "Edits at most", type: "number" }],
@@ -267,6 +339,7 @@ const metrics = {
267
339
  },
268
340
  // ---- against production ----------------------------------------------------
269
341
  "equals-production": {
342
+ family: PROD,
270
343
  label: "Same as production",
271
344
  description: "Passes when the reply is what production replied to the same item.",
272
345
  options: [],
@@ -279,6 +352,7 @@ const metrics = {
279
352
  },
280
353
  },
281
354
  "fields-equal-production": {
355
+ family: PROD,
282
356
  label: "Fields as production",
283
357
  description: "Passes when the fields listed are what production's reply had.",
284
358
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
@@ -295,6 +369,7 @@ const metrics = {
295
369
  },
296
370
  },
297
371
  "same-parse-outcome": {
372
+ family: PROD,
298
373
  label: "Parses as production",
299
374
  description: "Passes when the reply parses as JSON exactly when production's did.",
300
375
  options: [],
@@ -306,6 +381,7 @@ const metrics = {
306
381
  },
307
382
  // ---- model-graded -------------------------------------------------------------
308
383
  "llm-rubric": {
384
+ family: GRADED,
309
385
  label: "Rubric",
310
386
  description: "Asks the grader whether the reply meets the rubric.",
311
387
  graded: true,
@@ -317,6 +393,7 @@ const metrics = {
317
393
  num(m.threshold) ?? 0.5),
318
394
  },
319
395
  factuality: {
396
+ family: GRADED,
320
397
  label: "Factual",
321
398
  description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
322
399
  graded: true,
@@ -329,6 +406,7 @@ const metrics = {
329
406
  + `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
330
407
  },
331
408
  "judge-vs-production": {
409
+ family: GRADED,
332
410
  label: "Judged against production",
333
411
  description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
334
412
  graded: true,
package/lab/run-evals.js CHANGED
@@ -101,7 +101,7 @@ const USAGE = `Usage: node tools/prompt-lab/run-evals.js [options]
101
101
  job's tokenMappings. Default: none, so a prompt
102
102
  naming a token is refused, naming it.
103
103
  --pipeline <file> a run document (docs/pipeline-model.md) with one
104
- scenario and a graded test, graded case by case against
104
+ scenario and a graded eval, graded case by case against
105
105
  the set. Instead of --prompt, --model and --url. Each
106
106
  profile's key comes from $EVAL_API_KEY_<ID>, never the
107
107
  file; its content.files, when not empty, is the file
@@ -274,7 +274,7 @@ function readRun(file, dataset) {
274
274
  } catch (e) {
275
275
  broken(`${file}: ${e.message}`);
276
276
  }
277
- const named = isObject(doc) ? core.testsDataset(doc) : null;
277
+ const named = isObject(doc) ? core.evalsDataset(doc) : null;
278
278
  const bad = core.validatePipeline(doc, { datasets: dataset && named ? [named] : [] });
279
279
  if (bad.length) broken(`${file} cannot be run:\n ${bad.join("\n ")}`);
280
280
  return doc;
@@ -296,8 +296,8 @@ function readDataset(file) {
296
296
  broken(`${file}: ${e.message}`);
297
297
  }
298
298
  if (isObject(doc) && doc.format === "evals-lab/dataset") {
299
- if (![1, 2, 3, 4].includes(doc.version)) {
300
- broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 4`);
299
+ if (![1, 2, 3, 4, 5].includes(doc.version)) {
300
+ broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 5`);
301
301
  }
302
302
  doc = isObject(doc.dataset) ? doc.dataset.body : null;
303
303
  }
@@ -315,11 +315,12 @@ function readDataset(file) {
315
315
  return doc;
316
316
  }
317
317
 
318
- // The cases a graded run is scored against, by the file each names.
318
+ // The cases a graded run is scored against, by the item each names: exactly,
319
+ // as the Source names it.
319
320
  function gradedBy(dataset) {
320
321
  const out = new Map();
321
322
  for (const c of core.gradedSetFrom(dataset || {})) {
322
- if (!c.todo) out.set(c.filename, c);
323
+ if (!c.todo) out.set(c.item, c);
323
324
  }
324
325
  return out;
325
326
  }
@@ -453,21 +454,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
453
454
  // or, over Prompt only, the prompt -- and the stage reads it as it would a
454
455
  // model's.
455
456
  //
456
- // A call that asks for a whole request (an HTTP Request, #flows) builds it
457
- // from its step, the scenario's cell and the item's record, and is handed
458
- // only its words: the transport is where the three meet.
457
+ // A target step that asks for a whole request (an HTTP Request, #flows)
458
+ // builds it from the step, the job's flow step and the item's record, and is
459
+ // handed only its words: the transport is where the three meet.
460
+ //
461
+ // Any other step's reply, in a job that reads the flow's step through a Read
462
+ // as, is read the same way: put in the body the flow's API would have sent
463
+ // it in, then read as the flow reads it -- so a model asked in words and
464
+ // production are compared alike (docs/pipeline-model.md §16 › Targets).
459
465
  function callsFor(connections, links, text, plan, record) {
460
466
  return connections.map((c, k) => {
461
467
  const step = plan?.calls?.[k];
462
468
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
463
469
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
464
470
  }
465
- return core.CONNECTION_TYPES[core.typeOf(c)]?.local
471
+ const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
466
472
  ? async sent => core.localAnswer(c, text, sent)
467
473
  : (sent, url) => ask(links[k], c, sent, url);
474
+ if (!step?.readAs || !step?.step) return call;
475
+ return async (sent, url) => {
476
+ const r = await call(sent, url);
477
+ if (r.error || r.raw == null) return r;
478
+ try {
479
+ return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
480
+ } catch (e) {
481
+ return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
482
+ }
483
+ };
468
484
  });
469
485
  }
470
486
 
487
+ // What an expression token reads: the item's record, or a text item read as
488
+ // one, and the loop job 1's flow step sits in.
489
+ const tokenRecord = (run, record, text) => {
490
+ const r = record ?? (text != null ? textRecord(text) : null);
491
+ return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
492
+ };
493
+
494
+ // A prompt as a report restates it: an expression token reads an item's
495
+ // record, and a report has none to name, so its words stay as written.
496
+ const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
497
+
471
498
  /** A Setup profile the run carries, asked as a grader: its reply's text, or
472
499
  the failure as an error the metric reports. One transport per profile. */
473
500
  const graders = new WeakMap();
@@ -586,12 +613,12 @@ function readUtf8(file) {
586
613
  return fs.readFileSync(file).toString("utf8");
587
614
  }
588
615
 
589
- // Every test's reading of each scenario, keyed by the test's id, exactly as
590
- // the page reads it: a whole-run test's verdict, settled once every item is
591
- // in, and a per-item test's counts. [settled] is false for a run that
592
- // stopped short, whose whole-run tests have not settled.
593
- const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
594
- Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
616
+ // Every eval's reading of each scenario, keyed by the eval's id, exactly as
617
+ // the page reads it: a whole-run eval's verdict, settled once every item is
618
+ // in, and a per-item eval's counts. [settled] is false for a run that
619
+ // stopped short, whose whole-run evals have not settled.
620
+ const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
621
+ Object.fromEntries(core.scenarioEvals(run, i, items, settled).map(o => [o.id, {
595
622
  name: o.label, skipped: o.skipped,
596
623
  ...(o.whole
597
624
  ? { pass: o.verdict ? o.verdict.pass : null, detail: o.verdict?.detail ?? null }
@@ -600,7 +627,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
600
627
 
601
628
  // The run as the report restates it: each scenario's stages with the
602
629
  // connection each one asked.
603
- const scenariosOf = run => run.scenarios.map((sc, i) => {
630
+ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
604
631
  const { stages, connections } = core.stagesFor(run, i);
605
632
  return {
606
633
  n: i + 1, name: sc.name || null,
@@ -649,7 +676,7 @@ async function runSnapshot(o) {
649
676
  // Each scenario once: its stages, the jobs' token sets, and a transport
650
677
  // per stage to the profile that stage resolves to, keyed by that
651
678
  // profile's id.
652
- const plans = run.scenarios.map((_, i) => {
679
+ const plans = core.targetsOf(run).map((_, i) => {
653
680
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
654
681
  const links = connections.map(c => reach(c, core.keyVar(c.id)));
655
682
  return { stages, tokens, connections, links, calls, cells };
@@ -704,7 +731,7 @@ async function runSnapshot(o) {
704
731
  const textOf = item.kind === "text" && !item.bare
705
732
  ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
706
733
  const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
707
- { tokens: plan.tokens, text: textOf });
734
+ { tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
708
735
  // A case is found by its file's name, whatever kind of file it is: a
709
736
  // text item a dataset grades is graded like an image.
710
737
  const kase = item.name ? graded.get(item.name) ?? null : null;
@@ -728,7 +755,7 @@ async function runSnapshot(o) {
728
755
  verdict: cancelled ? "cancelled" : (unrun.length ? "incomplete" : "done"),
729
756
  run: {
730
757
  name: run.name ?? null,
731
- tests: run.tests,
758
+ evals: run.evals,
732
759
  verdicts: verdictsOf(run, results, !cancelled && !unrun.length),
733
760
  scenarios: scenariosOf(run),
734
761
  items: results.slice(0, total),
@@ -753,7 +780,7 @@ async function runSnapshot(o) {
753
780
  /** Re-score a run's stored results against the dataset --dataset hands over,
754
781
  * keeping the replies that were stored: what changes a verdict is the scorer
755
782
  * or the cases, never a new question to the model. The report is shaped like
756
- * a --run's, with the stored results as its items, and a whole-run test's
783
+ * a --run's, with the stored results as its items, and a whole-run eval's
757
784
  * verdict settled the same way. */
758
785
  async function runRescore(o){
759
786
  const dataset = readDataset(o.dataset);
@@ -775,7 +802,7 @@ async function runRescore(o){
775
802
  const graded = gradedBy(dataset);
776
803
 
777
804
  const only = o.only != null ? parseInt(o.only, 10) : null;
778
- // A test that reads every item (the Metrics) re-reads one with no case
805
+ // An eval that reads every item (the Metrics) re-reads one with no case
779
806
  // too, against the production reply the item kept.
780
807
  const items = await Promise.all(results.map(async (it, i) => {
781
808
  if (!isObject(it) || only != null && i !== only || !Array.isArray(it.scenarios)) return it;
@@ -793,7 +820,7 @@ async function runRescore(o){
793
820
  rescored: true,
794
821
  run: {
795
822
  name: run.name ?? null,
796
- tests: run.tests,
823
+ evals: run.evals,
797
824
  verdicts: verdictsOf(run, items, true),
798
825
  scenarios: scenariosOf(run),
799
826
  items,
@@ -822,16 +849,16 @@ function cliRun(o, dataset) {
822
849
  doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
823
850
  if (o.tokens) {
824
851
  try {
825
- doc.jobs[0] = core.withCall(doc.jobs[0], { tokenMappings: JSON.parse(fs.readFileSync(o.tokens, "utf8")) });
852
+ doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
826
853
  } catch (e) {
827
854
  broken(`${o.tokens}: ${e.message}`);
828
855
  }
829
856
  }
830
857
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
831
- doc.scenarios = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
832
- stages: [{ prompt: o.prompt ?? "" }] }];
833
- doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
834
- mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
858
+ doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
859
+ steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
860
+ doc.evals = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
861
+ mode: "all", threshold: null, grader: null, over: "item", metrics: [] }];
835
862
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
836
863
  return doc;
837
864
  }
@@ -881,7 +908,7 @@ async function main() {
881
908
  }
882
909
  if (o.run) {
883
910
  if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
884
- broken("--run names everything a run takes -- scenarios, files and text -- so it "
911
+ broken("--run names everything a run takes -- targets, files and text -- so it "
885
912
  + "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
886
913
  }
887
914
  const n = Number(o.timeout ?? REQUEST_CAP);
@@ -920,21 +947,23 @@ async function main() {
920
947
  }
921
948
  const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
922
949
  if (o.pipeline) {
923
- if (run.scenarios.length !== 1) {
924
- broken(`${o.pipeline} has ${run.scenarios.length} scenarios, and --pipeline grades one `
950
+ if (core.targetsOf(run).length !== 1) {
951
+ broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
925
952
  + "against the set -- --run runs them all.");
926
953
  }
927
- if (!core.testsDataset(run)) {
954
+ if (!core.evalsDataset(run)) {
928
955
  broken(`${o.pipeline} is not graded, and --pipeline grades a run case by case -- --run runs it.`);
929
956
  }
930
957
  if (core.contentOf(run).type !== "source") {
931
958
  broken(`${o.pipeline} is over inline text, and --pipeline grades the set's files -- --run runs it.`);
932
959
  }
933
960
  }
934
- // Case by case: each item against its case, as the case metric of the
935
- // test that names the dataset reads it (scoreCase).
961
+ // Case by case: each item against its case's metrics, as the eval that
962
+ // names the dataset reads them (core.readCase).
936
963
  const { stages, tokens, connections } = core.stagesFor(run, 0);
937
- const prompt = core.resolvePrompt(stages[0].text, core.tokenSet(tokens, 0));
964
+ const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
965
+ // A last stage that yields no items is matched as its text.
966
+ const plain = !core.OUTPUT_KINDS[stages.at(-1)?.kind ?? ""]?.terms;
938
967
 
939
968
  // The graded half, through the function the tab builds its own list with.
940
969
  const set = core.gradedSetFrom(dataset);
@@ -1024,11 +1053,11 @@ async function main() {
1024
1053
  + "--samples or --source names a directory.");
1025
1054
  const allowed = itemsOnly ?? graded;
1026
1055
  // $snapshotSet above: only a Source run sees it; --replies reads no file.
1027
- const inSnapshot = c => !snapshotSet || snapshotSet.has(c.filename);
1056
+ const inSnapshot = c => !snapshotSet || snapshotSet.has(c.item);
1028
1057
  // Only the run's own image items need ImageMagick: a text-only Source
1029
1058
  // -- the .txt, .csv case -- runs with nothing on PATH but the node
1030
1059
  // running this, which is the point of having text in the registry.
1031
- const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.filename)) === "image");
1060
+ const pictures = allowed.filter(c => inSnapshot(c) && HANDLERS.get(extOf(c.item)) === "image");
1032
1061
  if (pictures.length) {
1033
1062
  bin = imageMagick();
1034
1063
  if (!bin) broken("a run over images needs ImageMagick, to cap each "
@@ -1061,7 +1090,7 @@ async function main() {
1061
1090
  }
1062
1091
 
1063
1092
  // Only files the snapshot names are read from the Source.
1064
- const wanted = c => !snapshotSet || snapshotSet.has(c.filename);
1093
+ const wanted = c => !snapshotSet || snapshotSet.has(c.item);
1065
1094
  // The cancel marker is checked between items, so a cancel never loses the
1066
1095
  // reply in flight: the item that was being asked finishes, and the rest
1067
1096
  // are unrun rather than half-asked.
@@ -1085,18 +1114,18 @@ async function main() {
1085
1114
  calls = connections.map((c, i) => async () => ({ raw: said[i], ms: 0, conn: c.name }));
1086
1115
  } else {
1087
1116
  if (!wanted(kase)) {
1088
- unrun.push(`${kase.id}: ${kase.filename} is not in the run's file list`);
1117
+ unrun.push(`${kase.id}: ${kase.item} is not in the run's file list`);
1089
1118
  continue;
1090
1119
  }
1091
- const file = path.join(samples, kase.filename);
1120
+ const file = path.join(samples, kase.item);
1092
1121
  if (!fs.existsSync(file)) {
1093
- unrun.push(`${kase.id}: ${kase.filename} is not in ${samples}`);
1122
+ unrun.push(`${kase.id}: ${kase.item} is not in ${samples}`);
1094
1123
  continue;
1095
1124
  }
1096
1125
  try {
1097
1126
  item = itemContent(bin, file, connections[0]);
1098
1127
  } catch (e) {
1099
- unrun.push(`${kase.id}: ${kase.filename} could not be prepared — ${e.message}`);
1128
+ unrun.push(`${kase.id}: ${kase.item} could not be prepared — ${e.message}`);
1100
1129
  continue;
1101
1130
  }
1102
1131
  calls = callsFor(connections, links, item?.text ?? null);
@@ -1104,16 +1133,15 @@ async function main() {
1104
1133
 
1105
1134
  const res = await core.runPipeline(stages, item?.dataUrl ?? null, calls,
1106
1135
  { tokens, text: item?.text ?? null });
1107
- const s = core.scoreCase(kase, res);
1136
+ const s = core.readCase(kase, res, plain);
1108
1137
  // A file with no mapper ran anyway; the flag keeps the transcript honest
1109
1138
  // about what the model was actually shown. §4: stated, not hidden.
1110
1139
  if (item?.flags.length) res.transcript.unshift({ flag: item.flags.join(" ") });
1111
1140
  rows.push({
1112
- id: kase.id, filename: kase.filename, half: kase.half,
1141
+ id: kase.id, item: kase.item, half: kase.half,
1113
1142
  pass: s.pass, score: s.score, discarded: s.discarded, reasons: s.reasons,
1114
- found: s.found, missed: s.missed, invented: s.invented, unmet: s.unmet,
1115
- under: s.under, over: s.over, count: s.count,
1116
- watchFound: s.watchFound, watchTotal: s.watchTotal,
1143
+ found: s.found, missed: s.missed, invented: s.invented, count: s.count,
1144
+ metrics: s.metrics, watch: s.watch,
1117
1145
  terms: res.terms || [], ms: res.ms,
1118
1146
  // What a file type was to this run -- "image", "text", "unmapped".
1119
1147
  type: item?.kind ?? null,
@@ -1127,7 +1155,7 @@ async function main() {
1127
1155
  });
1128
1156
  if (progress) {
1129
1157
  progress.write(JSON.stringify({
1130
- event: "item", id: kase.id, filename: kase.filename, n: rows.length,
1158
+ event: "item", id: kase.id, item: kase.item, n: rows.length,
1131
1159
  pass: s.pass, score: s.score, discarded: s.discarded,
1132
1160
  error: s.discarded ? s.reasons[0]?.replace(/^Error - /, "") ?? null : null,
1133
1161
  ms: res.ms,
@@ -1170,7 +1198,7 @@ async function main() {
1170
1198
  stages: stages.map((s, i) => {
1171
1199
  const { id, ...connection } = connections[i];
1172
1200
  return { n: i + 1, kind: s.kind, withImage: s.withImage,
1173
- prompt: core.resolvePrompt(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1201
+ prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1174
1202
  }),
1175
1203
  } } : {}),
1176
1204
  // Which variable a key came from, never the key.