evals-lab 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -125,21 +125,26 @@ export function dropBy(rules , item )
125
125
  `itemCount`, and `share` of the items `items` matches. Every one given must
126
126
  hold; a rule that states none never fires. */
127
127
  export function rejectBy(rules , body , items ) {
128
- for (const r of rules) {
129
- const m = r.match || {};
130
- const conds = [];
131
- if (m.blank === true) conds.push(!body.trim());
132
- if (m.itemCount && typeof m.itemCount === "object") conds.push(inRange(m.itemCount )(items.length));
133
- if (m.share && typeof m.share === "object") {
134
- const test = matcher(m.items );
135
- // A share with nothing to measure cannot fire.
136
- conds.push(!!test && inRange(m.share )(items.length ? items.filter(test).length / items.length : 0));
137
- }
138
- if (conds.length && conds.every(Boolean)) return r.id;
139
- }
128
+ for (const r of rules) if (rejectHolds(r.match || {}, body, items)) return r.id;
140
129
  return null;
141
130
  }
142
131
 
132
+ /** Whether an answer whose text is [body] and items are [items] meets the
133
+ reject conditions [m]: `blank`, `itemCount` and the `share` of `items` the
134
+ items match. Every one given must hold; one that states none never fires.
135
+ The condition half of rejectBy, shared with the reject-reply filter. */
136
+ function rejectHolds(m , body , items ) {
137
+ const conds = [];
138
+ if (m.blank === true) conds.push(!body.trim());
139
+ if (m.itemCount && typeof m.itemCount === "object") conds.push(inRange(m.itemCount )(items.length));
140
+ if (m.share && typeof m.share === "object") {
141
+ const test = matcher(m.items );
142
+ // A share with nothing to measure cannot fire.
143
+ conds.push(!!test && inRange(m.share )(items.length ? items.filter(test).length / items.length : 0));
144
+ }
145
+ return conds.length > 0 && conds.every(Boolean);
146
+ }
147
+
143
148
  /** Why [rules] are not a rule list [scope] takes, one sentence each. */
144
149
  export function itemRulesProblems(rules , scope , at ) {
145
150
  if (!Array.isArray(rules)) return [`${at}: rules has to be a list`];
@@ -184,6 +189,226 @@ export function tryExamples(rule , scope )
184
189
  ];
185
190
  }
186
191
 
192
+ // ---- filters: the generic reply filter (issue #335) ----------------------------
193
+ //
194
+ // A Filter is the reusable unit #334 introduces: it supersedes the inline
195
+ // ItemRule a Drop items or Reject the answer modifier carries today. Where a
196
+ // rule was a bare matcher with examples, a filter has a base TYPE from a
197
+ // registry (not an enum) -- "drop-item" cleans each item of a list reply,
198
+ // "reject-reply" gates the whole answer -- generic conditions reusing the
199
+ // matcher above (ANDed), Options, and Tests that are its own checks. The two
200
+ // base types evaluate here, beside the matcher and rejectHolds they are built
201
+ // from; the Library-stored filter-set body, the pipeline link that carries
202
+ // one, and their version live in evals-core.ts, as DatasetBody and the
203
+ // eval-group link do. Nothing here replaces the Drop items / Reject modifiers:
204
+ // Runs wires a filter set into Responses and migrates the inline rules in a
205
+ // later phase (#339), so both base types and the ItemRule shape stand.
206
+
207
+ /** How a filter test splits its sample reply into items. "none" keeps the
208
+ reply whole, for a reject-reply filter that reads the text. */
209
+ export const FILTER_SEPARATORS = ["comma", "lines", "semicolon", "none"] ;
210
+
211
+
212
+ /** A sample reply and what the filter should leave of it: the editor splits
213
+ the sample by `separator`, applies the filter, and shows whether the result
214
+ reads as `expectedOutput`. A drop-item filter's expected output is the
215
+ items it keeps; a reject-reply filter's is blank when the reply should be
216
+ rejected, or left non-blank when it should pass. */
217
+
218
+
219
+
220
+
221
+
222
+
223
+ /** A filter's Options: keep going when it errors (rather than failing the
224
+ run), and how many passes it makes over the reply. */
225
+
226
+
227
+
228
+
229
+
230
+ /** A filter (issue #335): a base TYPE fixed at creation (a FILTER_TYPES id),
231
+ generic conditions in `match` (the matcher's keys for its type, ANDed),
232
+ Options and Tests. `id` is its name, as an ItemRule's was. */
233
+
234
+
235
+
236
+
237
+
238
+
239
+
240
+
241
+
242
+ /** A reply a filter works over: the text it came as, and the items parsed
243
+ from it (as the List kind parses a run's reply). */
244
+
245
+
246
+ /** What applying a filter or a set left: the items kept, what was dropped and
247
+ by which filter, and -- the first reject-reply filter that fired -- the
248
+ reply's rejection (null when none did). */
249
+
250
+
251
+ /** A base filter type (#334's decision: a registry, not an enum). A reader
252
+ asks the entry by `filter.type`; nothing branches on the id. */
253
+
254
+
255
+
256
+
257
+
258
+
259
+
260
+
261
+
262
+
263
+
264
+
265
+
266
+ /** The lab's base filter types. Generic code asks this registry by
267
+ `filter.type`; it never names "drop-item" or "reject-reply". */
268
+ export const FILTER_TYPES = {
269
+ "drop-item": {
270
+ label: "Drop item", scope: "item", conditions: MATCHER_KEYS,
271
+ apply(filter, reply) {
272
+ const test = matcher(filter.match);
273
+ if (!test) return { items: reply.items, dropped: [], reject: null };
274
+ const items = [], dropped = [];
275
+ for (const t of reply.items) {
276
+ if (test(t)) dropped.push({ item: t, by: filter.id });
277
+ else items.push(t);
278
+ }
279
+ return { items, dropped, reject: null };
280
+ },
281
+ },
282
+ "reject-reply": {
283
+ label: "Reject reply", scope: "answer", conditions: ["blank", "itemCount", "share", "items"],
284
+ apply(filter, reply) {
285
+ return { items: reply.items, dropped: [], reject: rejectHolds(filter.match, reply.body, reply.items) ? filter.id : null };
286
+ },
287
+ },
288
+ };
289
+
290
+ /** Apply one filter to [reply], honouring its Options: `repeat` passes over
291
+ the reply (a reject-reply filter stops at the first that fires), and
292
+ `ignoreFailures` leaves the reply as it was if the filter errors -- a
293
+ condition that cannot be read -- rather than failing the run. Pure. */
294
+ export function applyFilter(filter , reply ) {
295
+ const entry = FILTER_TYPES[filter.type];
296
+ if (!entry) return { items: reply.items, dropped: [], reject: null };
297
+ const passes = Number.isInteger(filter.options?.repeat) && filter.options.repeat > 0 ? filter.options.repeat : 1;
298
+ let items = reply.items;
299
+ const dropped = [];
300
+ let reject = null;
301
+ for (let pass = 0; pass < passes && reject == null; pass++) {
302
+ try {
303
+ const out = entry.apply(filter, { body: reply.body, items });
304
+ items = out.items;
305
+ dropped.push(...out.dropped);
306
+ reject = out.reject;
307
+ } catch (e) {
308
+ // Ignore failures: leave the reply as this pass found it; else the run
309
+ // records the fault.
310
+ if (!filter.options?.ignoreFailures) throw e;
311
+ break;
312
+ }
313
+ }
314
+ return { items, dropped, reject };
315
+ }
316
+
317
+ /** Apply a filter set's filters to [reply], in order: each drop-item filter
318
+ narrows the items the next reads, and the first reject-reply filter that
319
+ fires rejects the answer and stops the rest. Pure. */
320
+ export function applyFilterSet(filters , reply ) {
321
+ let items = reply.items;
322
+ const dropped = [];
323
+ for (const filter of filters) {
324
+ const out = applyFilter(filter, { body: reply.body, items });
325
+ items = out.items;
326
+ dropped.push(...out.dropped);
327
+ if (out.reject != null) return { items, dropped, reject: out.reject };
328
+ }
329
+ return { items, dropped, reject: null };
330
+ }
331
+
332
+ const SEPARATOR_RE = { comma: /,/, lines: /\n/, semicolon: /;/, none: null };
333
+
334
+ /** [sample] split into items by [separator]: each trimmed, none blank. "none"
335
+ keeps the reply whole as one item. */
336
+ export function splitSample(sample , separator ) {
337
+ const re = SEPARATOR_RE[separator];
338
+ const parts = re ? String(sample ?? "").split(re) : [String(sample ?? "")];
339
+ return parts.map(s => s.trim()).filter(Boolean);
340
+ }
341
+
342
+ /** Each of [filter]'s tests, tried against the filter alone: the sample split
343
+ by the test's separator, the filter applied, and the result compared with
344
+ the expected output. A drop-item test passes when the items kept read as
345
+ the expected items; a reject-reply test passes when the reply is rejected
346
+ exactly if the expected output is blank. Supersedes tryExamples. */
347
+ export function tryFilter(filter ) {
348
+ const scope = FILTER_TYPES[filter.type]?.scope ?? "item";
349
+ return (filter.tests || []).map(test => {
350
+ const out = applyFilter(filter, { body: test.sampleInput, items: splitSample(test.sampleInput, test.separator) });
351
+ if (scope === "answer") {
352
+ const rejected = out.reject != null;
353
+ return { test, ok: rejected === !test.expectedOutput.trim(), got: rejected ? "rejected" : out.items.join(", ") };
354
+ }
355
+ const want = splitSample(test.expectedOutput, test.separator);
356
+ const ok = out.items.length === want.length && out.items.every((t, i) => t === want[i]);
357
+ return { test, ok, got: out.items.join(", ") };
358
+ });
359
+ }
360
+
361
+ /** Why [filters] are not a filter list the lab reads, one sentence each: each
362
+ has a name, no name twice, a base type it has, whole-number repeat of 1 or
363
+ more, a boolean ignore-failures, and a matcher its type's scope takes
364
+ (itemRulesProblems). */
365
+ export function filterProblems(filters , at ) {
366
+ if (!Array.isArray(filters)) return [`${at}: filters has to be a list`];
367
+ const bad = [], seen = new Set ();
368
+ for (const f of filters ) {
369
+ const id = f && typeof f.id === "string" && f.id.trim() ? f.id : null;
370
+ if (!id) { bad.push(`${at}: a filter has no name`); continue; }
371
+ if (seen.has(id)) bad.push(`${at}: ${id} is named twice`);
372
+ seen.add(id);
373
+ const entry = typeof f.type === "string" ? FILTER_TYPES[f.type] : undefined;
374
+ if (!entry) { bad.push(`${at}: ${id} has no base type`); continue; }
375
+ const o = f.options;
376
+ if (!o || typeof o !== "object" || Array.isArray(o)) bad.push(`${at}: ${id} has no options`);
377
+ else {
378
+ if (!Number.isInteger(o.repeat) || o.repeat < 1) bad.push(`${at}: ${id}'s repeat has to be a whole number of 1 or more`);
379
+ if (typeof o.ignoreFailures !== "boolean") bad.push(`${at}: ${id}'s ignore failures has to be true or false`);
380
+ }
381
+ bad.push(...itemRulesProblems([{ id, match: f.match }], entry.scope, at));
382
+ }
383
+ return bad;
384
+ }
385
+
386
+ /** The filter an inline ItemRule becomes (issue #335): its name, why and
387
+ matcher kept as they are, its base [type] the one the modifier carrying it
388
+ fixed ("drop-item" for Drop items, "reject-reply" for Reject the answer),
389
+ its Options the defaults, and its examples its tests -- a caught example
390
+ becomes a sample the filter should empty (drop-item) or reject
391
+ (reject-reply), a kept example one it should leave. Pure: the match is
392
+ carried unchanged, so the filter cleans a reply exactly as the rule did.
393
+ The only reader of the old shape, for Runs to migrate its inline rules
394
+ (#339). */
395
+ export function filterFromItemRule(rule , type ) {
396
+ const scope = FILTER_TYPES[type]?.scope ?? "item";
397
+ const ex = rule.examples || {};
398
+ const caught = scope === "item" ? ex.drops || [] : ex.rejects || [];
399
+ const tests = [
400
+ ...caught.map((s) => ({ separator: "comma", sampleInput: s, expectedOutput: "" })),
401
+ ...(ex.keeps || []).map((s) => ({ separator: "comma", sampleInput: s, expectedOutput: s })),
402
+ ];
403
+ return {
404
+ id: rule.id, type,
405
+ ...(rule.why != null ? { why: rule.why } : {}),
406
+ match: rule.match ? { ...rule.match } : {},
407
+ options: { ignoreFailures: false, repeat: 1 },
408
+ tests,
409
+ };
410
+ }
411
+
187
412
  // ---- the modifiers -------------------------------------------------------------
188
413
 
189
414
  /** What an item's numbering looks like: "[1] suv", "(2) car", "3. dirt",
@@ -225,9 +450,28 @@ export function applyCase(item , mode ) {
225
450
  }
226
451
  }
227
452
 
453
+ /** The id of the synthetic modifier a linked or private filter set applies
454
+ under at run time (#339): it is never in a document or the Add menu -- a
455
+ job carries a `filters` step, which `stageModifiers` turns into one of
456
+ these where the step sits, so a set cleans and gates a list reply exactly
457
+ as the Drop items / Reject modifiers it replaced did, in the same order. */
458
+ export const FILTER_MODIFIER = "filterSet";
459
+
228
460
  const LIST_MODIFIERS = {
461
+ // A filter set's filters, applied where its link sits: the run's only reader
462
+ // of a resolved set, hidden from the Add menu (no options/defaults) because
463
+ // a set is linked in Responses, not added as a modifier.
464
+ [FILTER_MODIFIER]: {
465
+ accepts: ["list"], hidden: true,
466
+ apply(items, m, ctx) {
467
+ const out = applyFilterSet((m.filters ) || [], { body: ctx.body, items });
468
+ return out.reject != null
469
+ ? { value: out.items, dropped: out.dropped, reject: out.reject }
470
+ : { value: out.items, dropped: out.dropped };
471
+ },
472
+ },
229
473
  reject: {
230
- label: "Reject the answer if…", accepts: ["list"],
474
+ label: "Reject the answer if…", accepts: ["list"], hidden: true,
231
475
  description: "Throws the whole reply away when a rule below matches it.",
232
476
  options: [{ key: "rules", label: "Rules", type: "rules", scope: "answer" }],
233
477
  defaults: () => ({ type: "reject", rules: [] }),
@@ -248,7 +492,7 @@ const LIST_MODIFIERS = {
248
492
  apply: items => items.map(tidy).filter(Boolean),
249
493
  },
250
494
  drop: {
251
- label: "Drop items", accepts: ["list"],
495
+ label: "Drop items", accepts: ["list"], hidden: true,
252
496
  description: "Removes each item a rule below matches.",
253
497
  options: [{ key: "rules", label: "Rules", type: "rules", scope: "item" }],
254
498
  defaults: () => ({ type: "drop", rules: [] }),
@@ -135,16 +135,54 @@ function verdictOf(said )
135
135
  if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
136
136
  return read.value ;
137
137
  }
138
- const ask = async (ctx , prompt ) => {
138
+ /** The grader's verdict on [prompt], its reply held to [schema] where its API
139
+ can hold one. A reply that still cannot be read is asked for once more;
140
+ a second, or a grader that failed, is the grader's error -- said as one,
141
+ so it is not read as the reply failing. [ok] says whether a verdict read
142
+ is one this metric can use. */
143
+ async function graded(ctx , prompt , schema ,
144
+ ok = () => true) {
139
145
  if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
140
- return ctx.ask(prompt);
141
- };
146
+ let last = "";
147
+ for (let attempt = 0; attempt < 2; attempt++) {
148
+ let said ;
149
+ try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
150
+ try {
151
+ const v = verdictOf(said);
152
+ if (ok(v)) return v;
153
+ last = `the grader said ${short(JSON.stringify(v))}`;
154
+ } catch (e) { last = e instanceof Error ? e.message : String(e); }
155
+ }
156
+ throw new Error(`grader error: ${last}`);
157
+ }
142
158
  const gradedPass = (v , threshold ) => {
143
159
  const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
144
160
  const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
145
161
  return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
146
162
  };
163
+ /** The shape a pass-or-fail grader answers in, for its API to hold it to. */
164
+ const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
165
+ properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
147
166
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
167
+ /** What the grader is told of the task: what the target was asked, and the
168
+ guidance the metric adds, where there is either. */
169
+ const taskOf = (input , guidance ) =>
170
+ (input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
171
+ + (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
172
+
173
+ /** A grader's three verdicts on a reply set beside the recorded one; any of
174
+ them may be what a metric expects. */
175
+ const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
176
+ const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
177
+ /** What a judgement metric expects: the verdicts set, or -- set before
178
+ Expected was offered -- the one rule there was, worse fails. */
179
+ const expectedOf = (m ) =>
180
+ (Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
181
+ /** The expected verdicts as a reader says them: "better or same". */
182
+ const said = (values ) => {
183
+ const words = values.map((v) => v.toLowerCase());
184
+ return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
185
+ };
148
186
 
149
187
  // The families the Add metric picker lists the metrics under: what each reads.
150
188
  const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
@@ -342,6 +380,7 @@ const metrics = {
342
380
  // reply compared with the reply it is says nothing (docs/workflow-sources.md).
343
381
  "same-as-recorded": {
344
382
  family: RECORDED,
383
+ parsedOnly: true,
345
384
  label: "Same as recorded",
346
385
  description: "Passes when the reply is what production recorded for the same item.",
347
386
  options: [],
@@ -356,6 +395,7 @@ const metrics = {
356
395
  },
357
396
  "fields-equal-recorded": {
358
397
  family: RECORDED,
398
+ parsedOnly: true,
359
399
  label: "Fields as recorded",
360
400
  description: "Passes when the fields listed are what the recorded reply had.",
361
401
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
@@ -374,6 +414,7 @@ const metrics = {
374
414
  },
375
415
  "same-parse-as-recorded": {
376
416
  family: RECORDED,
417
+ parsedOnly: true,
377
418
  label: "Parses as recorded",
378
419
  description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
379
420
  options: [],
@@ -390,43 +431,58 @@ const metrics = {
390
431
  label: "Rubric",
391
432
  description: "Asks the grader whether the reply meets the rubric.",
392
433
  graded: true,
393
- options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
434
+ parsedOnly: true,
435
+ options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
394
436
  defaults: () => ({ rubric: "", threshold: 0.5 }),
395
437
  validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
396
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
397
- `You are grading an output against a rubric.\n\nRubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`)),
398
- num(m.threshold) ?? 0.5),
438
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
439
+ `You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
440
+ VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
399
441
  },
400
442
  factuality: {
401
443
  family: GRADED,
402
444
  label: "Factual",
403
445
  description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
404
446
  graded: true,
405
- options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
447
+ parsedOnly: true,
448
+ options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
406
449
  defaults: () => ({ reference: "", threshold: 0.5 }),
407
450
  // A blank reference is the recorded reply.
408
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
451
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
409
452
  `You are checking an output for factual consistency with a reference. Differences in wording or detail are `
410
- + `fine; a contradiction, or a fact the reference does not support, is not.\n\nReference:\n`
411
- + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
453
+ + `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
454
+ + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
412
455
  },
413
456
  "judge-vs-production": {
414
457
  family: GRADED,
415
458
  label: "Judged against recorded",
416
459
  description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
417
460
  graded: true,
418
- options: [{ key: "task", label: "Task", type: "textarea" }],
419
- defaults: () => ({ task: "" }),
461
+ parsedOnly: true,
462
+ expects: true,
463
+ options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
464
+ { key: "task", label: "Additional guidance (optional)", type: "textarea" }],
465
+ defaults: () => ({ expected: ["better", "same"], task: "" }),
466
+ validate: (m, at, bad) => {
467
+ const known = JUDGEMENTS.map((j) => j.value);
468
+ if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
469
+ || m.expected.some((v) => !known.includes(String(v))))) {
470
+ bad.push(`${at}: Expected needs at least one of ${said(known)}`);
471
+ }
472
+ },
420
473
  score: async (input, m, ctx) => {
421
- const v = verdictOf(await ask(ctx,
474
+ if (ctx.recordedTarget) return null;
475
+ const v = await graded(ctx,
422
476
  `Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
423
- + `RECORDED one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nRECORDED:\n${recorded(input)}\n\n`
477
+ + `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
424
478
  + `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
425
- + `"reason": "one sentence"}.`));
426
- const verdict = str(v.verdict).toLowerCase();
427
- if (!["better", "same", "worse"].includes(verdict)) throw new Error(`the grader said ${short(JSON.stringify(v))}`);
428
- return { pass: verdict !== "worse", score: verdict === "better" ? 1 : verdict === "same" ? 0.5 : 0,
429
- reason: `${verdict}: ${str(v.reason)}` };
479
+ + `"reason": "one sentence"}.`,
480
+ { type: "object", additionalProperties: false, required: ["verdict", "reason"],
481
+ properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
482
+ (r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
483
+ const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
484
+ return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
485
+ reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
430
486
  },
431
487
  },
432
488
  };
package/lab/run-evals.js CHANGED
@@ -127,6 +127,10 @@ const USAGE = `Usage: node run-evals.js [options]
127
127
  grades against each one its evals link (§17), so a run
128
128
  linking several groups is handed them here instead of the
129
129
  one --dataset. For --run and --rescore; not with --dataset.
130
+ --filter-sets <file> the filter-set bodies a --run kept, by "<id>@<n>" (§18):
131
+ a Library set a job's Responses links cleans and gates
132
+ its reply by. A private set carries its body in the
133
+ document, so only linked sets are here. For --run.
130
134
  --source <dir> the same, named the way a run names it: the directory a
131
135
  Source is stored at. --samples and --source are one
132
136
  flag by two names; both together are refused.
@@ -208,7 +212,7 @@ function broken(msg) {
208
212
 
209
213
  function parseArgs(argv) {
210
214
  const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
211
- groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
215
+ groups: "", filterSets: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
212
216
  timeout: null, progress: false, cancelFile: "",
213
217
  run: "", progressFile: "", resultsFile: "", from: null, only: null,
214
218
  rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
@@ -230,6 +234,7 @@ function parseArgs(argv) {
230
234
  case "--samples": o.samples = value(); break;
231
235
  case "--dataset": o.dataset = value(); break;
232
236
  case "--groups": o.groups = value(); break;
237
+ case "--filter-sets": o.filterSets = value(); break;
233
238
  case "--replies": o.replies = value(); break;
234
239
  case "--source": o.source = value(); break;
235
240
  case "--files": o.files = value(); break;
@@ -395,6 +400,30 @@ function gradingContext(o) {
395
400
  return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
396
401
  }
397
402
 
403
+ // The filter-set bodies a run kept, as --filter-sets hands them (§18): a JSON
404
+ // object of `<id>@<n>` to body -- a Library set's id and the version the run
405
+ // cleaned and gated with -- each read as today's (upgradeFilterSetBody). A
406
+ // private (`own`) set carries its body in the document, so only Library links
407
+ // are here. Returns a resolver a job's filters step reads a link's body by,
408
+ // mirroring the server's filter_set_ref / group_key.
409
+ function filterContext(o) {
410
+ if (!o.filterSets) return () => undefined;
411
+ let raw;
412
+ try { raw = JSON.parse(fs.readFileSync(o.filterSets, "utf8")); }
413
+ catch (e) { broken(`${o.filterSets}: ${e.message}`); }
414
+ if (!isObject(raw)) broken(`${o.filterSets}: the kept filter sets are a JSON object of "<id>@<n>": body`);
415
+ const bodies = new Map();
416
+ for (const [key, body] of Object.entries(raw)) {
417
+ const at = `${o.filterSets} [${key}]`;
418
+ const got = core.upgradeFilterSetBody(body);
419
+ const bad = core.filterSetBodyProblems(got, at);
420
+ if (bad.length) broken(bad[0]);
421
+ bodies.set(key, got);
422
+ }
423
+ const key = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
424
+ return (ref) => bodies.get(key(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined;
425
+ }
426
+
398
427
  // The cases a graded run is scored against, by the item each names: exactly,
399
428
  // as the Source names it. Built from the one group a single-group run names,
400
429
  // for an eval type of its own that reads a case (a plugin's); a `group` eval
@@ -605,8 +634,12 @@ function graderFor(run) {
605
634
  if (!c) return undefined;
606
635
  if (!made.has(ref.id)) {
607
636
  const link = reach(c, core.keyVar(ref.id, c.slug));
608
- made.set(ref.id, async prompt => {
609
- const r = await ask(link, { id: ref.id, ...c }, prompt, null);
637
+ made.set(ref.id, async (prompt, schema) => {
638
+ // Held to the metric's schema where the grader's API can hold a
639
+ // reply to one; a model or server that refuses the schema is asked
640
+ // again without it, the prompt alone saying the shape.
641
+ let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
642
+ if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
610
643
  if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
611
644
  return r.raw ?? "";
612
645
  });
@@ -660,13 +693,13 @@ async function send(to, c, step, cell, record) {
660
693
  }
661
694
  }
662
695
 
663
- async function ask(to, c, prompt, dataUrl) {
696
+ async function ask(to, c, prompt, dataUrl, schema = null) {
664
697
  const t0 = Date.now();
665
698
  const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
666
699
  try {
667
700
  // The connection's type builds the whole request: the URL, the headers
668
701
  // (Anthropic's key travels in x-api-key), and the body.
669
- const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
702
+ const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
670
703
  const r = await fetch(req.url, {
671
704
  method: "POST",
672
705
  headers: { "Content-Type": "application/json", ...req.headers },
@@ -896,9 +929,11 @@ async function runSnapshot(o) {
896
929
 
897
930
  // Each scenario once: its stages, the jobs' token sets, and a transport
898
931
  // per stage to the profile that stage resolves to, keyed by that
899
- // profile's id.
932
+ // profile's id. Its stages carry each linked or private filter set as a
933
+ // modifier that cleans and gates the reply where the link sits (#339).
934
+ const resolveFilters = filterContext(o);
900
935
  const plans = core.targetsOf(run).map((_, i) => {
901
- const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
936
+ const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i, resolveFilters);
902
937
  const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
903
938
  // The Recorded target replays the recorded reply, so a metric that compares
904
939
  // with the recorded reply has nothing to say of it (n/a).