evals-lab 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/README.md +1 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +1 -1
- package/lab/demo/pipelines/demo-2.json +1 -1
- package/lab/evals-core.mjs +428 -49
- package/lab/kinds/list.mjs +258 -14
- package/lab/metrics/builtin.mjs +76 -20
- package/lab/run-evals.js +42 -7
- package/lab/server.py +1200 -31
- package/lab/web/dist/assets/gallery-BpR9b4EM.js +3 -0
- package/lab/web/dist/assets/main-BMRmyJvo.css +1 -0
- package/lab/web/dist/assets/main-BfiFtaYW.js +19 -0
- package/lab/web/dist/assets/tokens-C67OCDd-.css +1 -0
- package/lab/web/dist/assets/tokens-DzZqM5IZ.js +59 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +15 -2
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +0 -3
- package/lab/web/dist/assets/main-BQL5j5oF.js +0 -20
- package/lab/web/dist/assets/main-Cza2gwQd.css +0 -1
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +0 -61
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +0 -1
package/lab/kinds/list.mjs
CHANGED
|
@@ -125,21 +125,26 @@ export function dropBy(rules , item )
|
|
|
125
125
|
`itemCount`, and `share` of the items `items` matches. Every one given must
|
|
126
126
|
hold; a rule that states none never fires. */
|
|
127
127
|
export function rejectBy(rules , body , items ) {
|
|
128
|
-
for (const r of rules) {
|
|
129
|
-
const m = r.match || {};
|
|
130
|
-
const conds = [];
|
|
131
|
-
if (m.blank === true) conds.push(!body.trim());
|
|
132
|
-
if (m.itemCount && typeof m.itemCount === "object") conds.push(inRange(m.itemCount )(items.length));
|
|
133
|
-
if (m.share && typeof m.share === "object") {
|
|
134
|
-
const test = matcher(m.items );
|
|
135
|
-
// A share with nothing to measure cannot fire.
|
|
136
|
-
conds.push(!!test && inRange(m.share )(items.length ? items.filter(test).length / items.length : 0));
|
|
137
|
-
}
|
|
138
|
-
if (conds.length && conds.every(Boolean)) return r.id;
|
|
139
|
-
}
|
|
128
|
+
for (const r of rules) if (rejectHolds(r.match || {}, body, items)) return r.id;
|
|
140
129
|
return null;
|
|
141
130
|
}
|
|
142
131
|
|
|
132
|
+
/** Whether an answer whose text is [body] and items are [items] meets the
|
|
133
|
+
reject conditions [m]: `blank`, `itemCount` and the `share` of `items` the
|
|
134
|
+
items match. Every one given must hold; one that states none never fires.
|
|
135
|
+
The condition half of rejectBy, shared with the reject-reply filter. */
|
|
136
|
+
function rejectHolds(m , body , items ) {
|
|
137
|
+
const conds = [];
|
|
138
|
+
if (m.blank === true) conds.push(!body.trim());
|
|
139
|
+
if (m.itemCount && typeof m.itemCount === "object") conds.push(inRange(m.itemCount )(items.length));
|
|
140
|
+
if (m.share && typeof m.share === "object") {
|
|
141
|
+
const test = matcher(m.items );
|
|
142
|
+
// A share with nothing to measure cannot fire.
|
|
143
|
+
conds.push(!!test && inRange(m.share )(items.length ? items.filter(test).length / items.length : 0));
|
|
144
|
+
}
|
|
145
|
+
return conds.length > 0 && conds.every(Boolean);
|
|
146
|
+
}
|
|
147
|
+
|
|
143
148
|
/** Why [rules] are not a rule list [scope] takes, one sentence each. */
|
|
144
149
|
export function itemRulesProblems(rules , scope , at ) {
|
|
145
150
|
if (!Array.isArray(rules)) return [`${at}: rules has to be a list`];
|
|
@@ -184,6 +189,226 @@ export function tryExamples(rule , scope )
|
|
|
184
189
|
];
|
|
185
190
|
}
|
|
186
191
|
|
|
192
|
+
// ---- filters: the generic reply filter (issue #335) ----------------------------
|
|
193
|
+
//
|
|
194
|
+
// A Filter is the reusable unit #334 introduces: it supersedes the inline
|
|
195
|
+
// ItemRule a Drop items or Reject the answer modifier carries today. Where a
|
|
196
|
+
// rule was a bare matcher with examples, a filter has a base TYPE from a
|
|
197
|
+
// registry (not an enum) -- "drop-item" cleans each item of a list reply,
|
|
198
|
+
// "reject-reply" gates the whole answer -- generic conditions reusing the
|
|
199
|
+
// matcher above (ANDed), Options, and Tests that are its own checks. The two
|
|
200
|
+
// base types evaluate here, beside the matcher and rejectHolds they are built
|
|
201
|
+
// from; the Library-stored filter-set body, the pipeline link that carries
|
|
202
|
+
// one, and their version live in evals-core.ts, as DatasetBody and the
|
|
203
|
+
// eval-group link do. Nothing here replaces the Drop items / Reject modifiers:
|
|
204
|
+
// Runs wires a filter set into Responses and migrates the inline rules in a
|
|
205
|
+
// later phase (#339), so both base types and the ItemRule shape stand.
|
|
206
|
+
|
|
207
|
+
/** How a filter test splits its sample reply into items. "none" keeps the
|
|
208
|
+
reply whole, for a reject-reply filter that reads the text. */
|
|
209
|
+
export const FILTER_SEPARATORS = ["comma", "lines", "semicolon", "none"] ;
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
/** A sample reply and what the filter should leave of it: the editor splits
|
|
213
|
+
the sample by `separator`, applies the filter, and shows whether the result
|
|
214
|
+
reads as `expectedOutput`. A drop-item filter's expected output is the
|
|
215
|
+
items it keeps; a reject-reply filter's is blank when the reply should be
|
|
216
|
+
rejected, or left non-blank when it should pass. */
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
/** A filter's Options: keep going when it errors (rather than failing the
|
|
224
|
+
run), and how many passes it makes over the reply. */
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
/** A filter (issue #335): a base TYPE fixed at creation (a FILTER_TYPES id),
|
|
231
|
+
generic conditions in `match` (the matcher's keys for its type, ANDed),
|
|
232
|
+
Options and Tests. `id` is its name, as an ItemRule's was. */
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
/** A reply a filter works over: the text it came as, and the items parsed
|
|
243
|
+
from it (as the List kind parses a run's reply). */
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
/** What applying a filter or a set left: the items kept, what was dropped and
|
|
247
|
+
by which filter, and -- the first reject-reply filter that fired -- the
|
|
248
|
+
reply's rejection (null when none did). */
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
/** A base filter type (#334's decision: a registry, not an enum). A reader
|
|
252
|
+
asks the entry by `filter.type`; nothing branches on the id. */
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
/** The lab's base filter types. Generic code asks this registry by
|
|
267
|
+
`filter.type`; it never names "drop-item" or "reject-reply". */
|
|
268
|
+
export const FILTER_TYPES = {
|
|
269
|
+
"drop-item": {
|
|
270
|
+
label: "Drop item", scope: "item", conditions: MATCHER_KEYS,
|
|
271
|
+
apply(filter, reply) {
|
|
272
|
+
const test = matcher(filter.match);
|
|
273
|
+
if (!test) return { items: reply.items, dropped: [], reject: null };
|
|
274
|
+
const items = [], dropped = [];
|
|
275
|
+
for (const t of reply.items) {
|
|
276
|
+
if (test(t)) dropped.push({ item: t, by: filter.id });
|
|
277
|
+
else items.push(t);
|
|
278
|
+
}
|
|
279
|
+
return { items, dropped, reject: null };
|
|
280
|
+
},
|
|
281
|
+
},
|
|
282
|
+
"reject-reply": {
|
|
283
|
+
label: "Reject reply", scope: "answer", conditions: ["blank", "itemCount", "share", "items"],
|
|
284
|
+
apply(filter, reply) {
|
|
285
|
+
return { items: reply.items, dropped: [], reject: rejectHolds(filter.match, reply.body, reply.items) ? filter.id : null };
|
|
286
|
+
},
|
|
287
|
+
},
|
|
288
|
+
};
|
|
289
|
+
|
|
290
|
+
/** Apply one filter to [reply], honouring its Options: `repeat` passes over
|
|
291
|
+
the reply (a reject-reply filter stops at the first that fires), and
|
|
292
|
+
`ignoreFailures` leaves the reply as it was if the filter errors -- a
|
|
293
|
+
condition that cannot be read -- rather than failing the run. Pure. */
|
|
294
|
+
export function applyFilter(filter , reply ) {
|
|
295
|
+
const entry = FILTER_TYPES[filter.type];
|
|
296
|
+
if (!entry) return { items: reply.items, dropped: [], reject: null };
|
|
297
|
+
const passes = Number.isInteger(filter.options?.repeat) && filter.options.repeat > 0 ? filter.options.repeat : 1;
|
|
298
|
+
let items = reply.items;
|
|
299
|
+
const dropped = [];
|
|
300
|
+
let reject = null;
|
|
301
|
+
for (let pass = 0; pass < passes && reject == null; pass++) {
|
|
302
|
+
try {
|
|
303
|
+
const out = entry.apply(filter, { body: reply.body, items });
|
|
304
|
+
items = out.items;
|
|
305
|
+
dropped.push(...out.dropped);
|
|
306
|
+
reject = out.reject;
|
|
307
|
+
} catch (e) {
|
|
308
|
+
// Ignore failures: leave the reply as this pass found it; else the run
|
|
309
|
+
// records the fault.
|
|
310
|
+
if (!filter.options?.ignoreFailures) throw e;
|
|
311
|
+
break;
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
return { items, dropped, reject };
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/** Apply a filter set's filters to [reply], in order: each drop-item filter
|
|
318
|
+
narrows the items the next reads, and the first reject-reply filter that
|
|
319
|
+
fires rejects the answer and stops the rest. Pure. */
|
|
320
|
+
export function applyFilterSet(filters , reply ) {
|
|
321
|
+
let items = reply.items;
|
|
322
|
+
const dropped = [];
|
|
323
|
+
for (const filter of filters) {
|
|
324
|
+
const out = applyFilter(filter, { body: reply.body, items });
|
|
325
|
+
items = out.items;
|
|
326
|
+
dropped.push(...out.dropped);
|
|
327
|
+
if (out.reject != null) return { items, dropped, reject: out.reject };
|
|
328
|
+
}
|
|
329
|
+
return { items, dropped, reject: null };
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
const SEPARATOR_RE = { comma: /,/, lines: /\n/, semicolon: /;/, none: null };
|
|
333
|
+
|
|
334
|
+
/** [sample] split into items by [separator]: each trimmed, none blank. "none"
|
|
335
|
+
keeps the reply whole as one item. */
|
|
336
|
+
export function splitSample(sample , separator ) {
|
|
337
|
+
const re = SEPARATOR_RE[separator];
|
|
338
|
+
const parts = re ? String(sample ?? "").split(re) : [String(sample ?? "")];
|
|
339
|
+
return parts.map(s => s.trim()).filter(Boolean);
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
/** Each of [filter]'s tests, tried against the filter alone: the sample split
|
|
343
|
+
by the test's separator, the filter applied, and the result compared with
|
|
344
|
+
the expected output. A drop-item test passes when the items kept read as
|
|
345
|
+
the expected items; a reject-reply test passes when the reply is rejected
|
|
346
|
+
exactly if the expected output is blank. Supersedes tryExamples. */
|
|
347
|
+
export function tryFilter(filter ) {
|
|
348
|
+
const scope = FILTER_TYPES[filter.type]?.scope ?? "item";
|
|
349
|
+
return (filter.tests || []).map(test => {
|
|
350
|
+
const out = applyFilter(filter, { body: test.sampleInput, items: splitSample(test.sampleInput, test.separator) });
|
|
351
|
+
if (scope === "answer") {
|
|
352
|
+
const rejected = out.reject != null;
|
|
353
|
+
return { test, ok: rejected === !test.expectedOutput.trim(), got: rejected ? "rejected" : out.items.join(", ") };
|
|
354
|
+
}
|
|
355
|
+
const want = splitSample(test.expectedOutput, test.separator);
|
|
356
|
+
const ok = out.items.length === want.length && out.items.every((t, i) => t === want[i]);
|
|
357
|
+
return { test, ok, got: out.items.join(", ") };
|
|
358
|
+
});
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
/** Why [filters] are not a filter list the lab reads, one sentence each: each
|
|
362
|
+
has a name, no name twice, a base type it has, whole-number repeat of 1 or
|
|
363
|
+
more, a boolean ignore-failures, and a matcher its type's scope takes
|
|
364
|
+
(itemRulesProblems). */
|
|
365
|
+
export function filterProblems(filters , at ) {
|
|
366
|
+
if (!Array.isArray(filters)) return [`${at}: filters has to be a list`];
|
|
367
|
+
const bad = [], seen = new Set ();
|
|
368
|
+
for (const f of filters ) {
|
|
369
|
+
const id = f && typeof f.id === "string" && f.id.trim() ? f.id : null;
|
|
370
|
+
if (!id) { bad.push(`${at}: a filter has no name`); continue; }
|
|
371
|
+
if (seen.has(id)) bad.push(`${at}: ${id} is named twice`);
|
|
372
|
+
seen.add(id);
|
|
373
|
+
const entry = typeof f.type === "string" ? FILTER_TYPES[f.type] : undefined;
|
|
374
|
+
if (!entry) { bad.push(`${at}: ${id} has no base type`); continue; }
|
|
375
|
+
const o = f.options;
|
|
376
|
+
if (!o || typeof o !== "object" || Array.isArray(o)) bad.push(`${at}: ${id} has no options`);
|
|
377
|
+
else {
|
|
378
|
+
if (!Number.isInteger(o.repeat) || o.repeat < 1) bad.push(`${at}: ${id}'s repeat has to be a whole number of 1 or more`);
|
|
379
|
+
if (typeof o.ignoreFailures !== "boolean") bad.push(`${at}: ${id}'s ignore failures has to be true or false`);
|
|
380
|
+
}
|
|
381
|
+
bad.push(...itemRulesProblems([{ id, match: f.match }], entry.scope, at));
|
|
382
|
+
}
|
|
383
|
+
return bad;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/** The filter an inline ItemRule becomes (issue #335): its name, why and
|
|
387
|
+
matcher kept as they are, its base [type] the one the modifier carrying it
|
|
388
|
+
fixed ("drop-item" for Drop items, "reject-reply" for Reject the answer),
|
|
389
|
+
its Options the defaults, and its examples its tests -- a caught example
|
|
390
|
+
becomes a sample the filter should empty (drop-item) or reject
|
|
391
|
+
(reject-reply), a kept example one it should leave. Pure: the match is
|
|
392
|
+
carried unchanged, so the filter cleans a reply exactly as the rule did.
|
|
393
|
+
The only reader of the old shape, for Runs to migrate its inline rules
|
|
394
|
+
(#339). */
|
|
395
|
+
export function filterFromItemRule(rule , type ) {
|
|
396
|
+
const scope = FILTER_TYPES[type]?.scope ?? "item";
|
|
397
|
+
const ex = rule.examples || {};
|
|
398
|
+
const caught = scope === "item" ? ex.drops || [] : ex.rejects || [];
|
|
399
|
+
const tests = [
|
|
400
|
+
...caught.map((s) => ({ separator: "comma", sampleInput: s, expectedOutput: "" })),
|
|
401
|
+
...(ex.keeps || []).map((s) => ({ separator: "comma", sampleInput: s, expectedOutput: s })),
|
|
402
|
+
];
|
|
403
|
+
return {
|
|
404
|
+
id: rule.id, type,
|
|
405
|
+
...(rule.why != null ? { why: rule.why } : {}),
|
|
406
|
+
match: rule.match ? { ...rule.match } : {},
|
|
407
|
+
options: { ignoreFailures: false, repeat: 1 },
|
|
408
|
+
tests,
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
|
|
187
412
|
// ---- the modifiers -------------------------------------------------------------
|
|
188
413
|
|
|
189
414
|
/** What an item's numbering looks like: "[1] suv", "(2) car", "3. dirt",
|
|
@@ -225,9 +450,28 @@ export function applyCase(item , mode ) {
|
|
|
225
450
|
}
|
|
226
451
|
}
|
|
227
452
|
|
|
453
|
+
/** The id of the synthetic modifier a linked or private filter set applies
|
|
454
|
+
under at run time (#339): it is never in a document or the Add menu -- a
|
|
455
|
+
job carries a `filters` step, which `stageModifiers` turns into one of
|
|
456
|
+
these where the step sits, so a set cleans and gates a list reply exactly
|
|
457
|
+
as the Drop items / Reject modifiers it replaced did, in the same order. */
|
|
458
|
+
export const FILTER_MODIFIER = "filterSet";
|
|
459
|
+
|
|
228
460
|
const LIST_MODIFIERS = {
|
|
461
|
+
// A filter set's filters, applied where its link sits: the run's only reader
|
|
462
|
+
// of a resolved set, hidden from the Add menu (no options/defaults) because
|
|
463
|
+
// a set is linked in Responses, not added as a modifier.
|
|
464
|
+
[FILTER_MODIFIER]: {
|
|
465
|
+
accepts: ["list"], hidden: true,
|
|
466
|
+
apply(items, m, ctx) {
|
|
467
|
+
const out = applyFilterSet((m.filters ) || [], { body: ctx.body, items });
|
|
468
|
+
return out.reject != null
|
|
469
|
+
? { value: out.items, dropped: out.dropped, reject: out.reject }
|
|
470
|
+
: { value: out.items, dropped: out.dropped };
|
|
471
|
+
},
|
|
472
|
+
},
|
|
229
473
|
reject: {
|
|
230
|
-
label: "Reject the answer if…", accepts: ["list"],
|
|
474
|
+
label: "Reject the answer if…", accepts: ["list"], hidden: true,
|
|
231
475
|
description: "Throws the whole reply away when a rule below matches it.",
|
|
232
476
|
options: [{ key: "rules", label: "Rules", type: "rules", scope: "answer" }],
|
|
233
477
|
defaults: () => ({ type: "reject", rules: [] }),
|
|
@@ -248,7 +492,7 @@ const LIST_MODIFIERS = {
|
|
|
248
492
|
apply: items => items.map(tidy).filter(Boolean),
|
|
249
493
|
},
|
|
250
494
|
drop: {
|
|
251
|
-
label: "Drop items", accepts: ["list"],
|
|
495
|
+
label: "Drop items", accepts: ["list"], hidden: true,
|
|
252
496
|
description: "Removes each item a rule below matches.",
|
|
253
497
|
options: [{ key: "rules", label: "Rules", type: "rules", scope: "item" }],
|
|
254
498
|
defaults: () => ({ type: "drop", rules: [] }),
|
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -135,16 +135,54 @@ function verdictOf(said )
|
|
|
135
135
|
if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
|
|
136
136
|
return read.value ;
|
|
137
137
|
}
|
|
138
|
-
|
|
138
|
+
/** The grader's verdict on [prompt], its reply held to [schema] where its API
|
|
139
|
+
can hold one. A reply that still cannot be read is asked for once more;
|
|
140
|
+
a second, or a grader that failed, is the grader's error -- said as one,
|
|
141
|
+
so it is not read as the reply failing. [ok] says whether a verdict read
|
|
142
|
+
is one this metric can use. */
|
|
143
|
+
async function graded(ctx , prompt , schema ,
|
|
144
|
+
ok = () => true) {
|
|
139
145
|
if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
|
|
140
|
-
|
|
141
|
-
|
|
146
|
+
let last = "";
|
|
147
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
148
|
+
let said ;
|
|
149
|
+
try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
|
|
150
|
+
try {
|
|
151
|
+
const v = verdictOf(said);
|
|
152
|
+
if (ok(v)) return v;
|
|
153
|
+
last = `the grader said ${short(JSON.stringify(v))}`;
|
|
154
|
+
} catch (e) { last = e instanceof Error ? e.message : String(e); }
|
|
155
|
+
}
|
|
156
|
+
throw new Error(`grader error: ${last}`);
|
|
157
|
+
}
|
|
142
158
|
const gradedPass = (v , threshold ) => {
|
|
143
159
|
const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
|
|
144
160
|
const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
|
|
145
161
|
return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
|
|
146
162
|
};
|
|
163
|
+
/** The shape a pass-or-fail grader answers in, for its API to hold it to. */
|
|
164
|
+
const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
|
|
165
|
+
properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
|
|
147
166
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
167
|
+
/** What the grader is told of the task: what the target was asked, and the
|
|
168
|
+
guidance the metric adds, where there is either. */
|
|
169
|
+
const taskOf = (input , guidance ) =>
|
|
170
|
+
(input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
|
|
171
|
+
+ (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
|
|
172
|
+
|
|
173
|
+
/** A grader's three verdicts on a reply set beside the recorded one; any of
|
|
174
|
+
them may be what a metric expects. */
|
|
175
|
+
const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
|
|
176
|
+
const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
|
|
177
|
+
/** What a judgement metric expects: the verdicts set, or -- set before
|
|
178
|
+
Expected was offered -- the one rule there was, worse fails. */
|
|
179
|
+
const expectedOf = (m ) =>
|
|
180
|
+
(Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
|
|
181
|
+
/** The expected verdicts as a reader says them: "better or same". */
|
|
182
|
+
const said = (values ) => {
|
|
183
|
+
const words = values.map((v) => v.toLowerCase());
|
|
184
|
+
return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
|
|
185
|
+
};
|
|
148
186
|
|
|
149
187
|
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
188
|
const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
|
|
@@ -342,6 +380,7 @@ const metrics = {
|
|
|
342
380
|
// reply compared with the reply it is says nothing (docs/workflow-sources.md).
|
|
343
381
|
"same-as-recorded": {
|
|
344
382
|
family: RECORDED,
|
|
383
|
+
parsedOnly: true,
|
|
345
384
|
label: "Same as recorded",
|
|
346
385
|
description: "Passes when the reply is what production recorded for the same item.",
|
|
347
386
|
options: [],
|
|
@@ -356,6 +395,7 @@ const metrics = {
|
|
|
356
395
|
},
|
|
357
396
|
"fields-equal-recorded": {
|
|
358
397
|
family: RECORDED,
|
|
398
|
+
parsedOnly: true,
|
|
359
399
|
label: "Fields as recorded",
|
|
360
400
|
description: "Passes when the fields listed are what the recorded reply had.",
|
|
361
401
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
@@ -374,6 +414,7 @@ const metrics = {
|
|
|
374
414
|
},
|
|
375
415
|
"same-parse-as-recorded": {
|
|
376
416
|
family: RECORDED,
|
|
417
|
+
parsedOnly: true,
|
|
377
418
|
label: "Parses as recorded",
|
|
378
419
|
description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
|
|
379
420
|
options: [],
|
|
@@ -390,43 +431,58 @@ const metrics = {
|
|
|
390
431
|
label: "Rubric",
|
|
391
432
|
description: "Asks the grader whether the reply meets the rubric.",
|
|
392
433
|
graded: true,
|
|
393
|
-
|
|
434
|
+
parsedOnly: true,
|
|
435
|
+
options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
394
436
|
defaults: () => ({ rubric: "", threshold: 0.5 }),
|
|
395
437
|
validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
|
|
396
|
-
score: async (input, m, ctx) => gradedPass(
|
|
397
|
-
`You are grading an output against a rubric.\n\
|
|
398
|
-
num(m.threshold) ?? 0.5),
|
|
438
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
439
|
+
`You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
|
|
440
|
+
VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
399
441
|
},
|
|
400
442
|
factuality: {
|
|
401
443
|
family: GRADED,
|
|
402
444
|
label: "Factual",
|
|
403
445
|
description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
|
|
404
446
|
graded: true,
|
|
405
|
-
|
|
447
|
+
parsedOnly: true,
|
|
448
|
+
options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
406
449
|
defaults: () => ({ reference: "", threshold: 0.5 }),
|
|
407
450
|
// A blank reference is the recorded reply.
|
|
408
|
-
score: async (input, m, ctx) => gradedPass(
|
|
451
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
409
452
|
`You are checking an output for factual consistency with a reference. Differences in wording or detail are `
|
|
410
|
-
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\
|
|
411
|
-
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}
|
|
453
|
+
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
|
|
454
|
+
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
412
455
|
},
|
|
413
456
|
"judge-vs-production": {
|
|
414
457
|
family: GRADED,
|
|
415
458
|
label: "Judged against recorded",
|
|
416
459
|
description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
|
|
417
460
|
graded: true,
|
|
418
|
-
|
|
419
|
-
|
|
461
|
+
parsedOnly: true,
|
|
462
|
+
expects: true,
|
|
463
|
+
options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
|
|
464
|
+
{ key: "task", label: "Additional guidance (optional)", type: "textarea" }],
|
|
465
|
+
defaults: () => ({ expected: ["better", "same"], task: "" }),
|
|
466
|
+
validate: (m, at, bad) => {
|
|
467
|
+
const known = JUDGEMENTS.map((j) => j.value);
|
|
468
|
+
if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
|
|
469
|
+
|| m.expected.some((v) => !known.includes(String(v))))) {
|
|
470
|
+
bad.push(`${at}: Expected needs at least one of ${said(known)}`);
|
|
471
|
+
}
|
|
472
|
+
},
|
|
420
473
|
score: async (input, m, ctx) => {
|
|
421
|
-
|
|
474
|
+
if (ctx.recordedTarget) return null;
|
|
475
|
+
const v = await graded(ctx,
|
|
422
476
|
`Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
|
|
423
|
-
+ `RECORDED one
|
|
477
|
+
+ `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
|
|
424
478
|
+ `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
|
|
425
|
-
+ `"reason": "one sentence"}
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
479
|
+
+ `"reason": "one sentence"}.`,
|
|
480
|
+
{ type: "object", additionalProperties: false, required: ["verdict", "reason"],
|
|
481
|
+
properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
|
|
482
|
+
(r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
|
|
483
|
+
const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
|
|
484
|
+
return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
|
|
485
|
+
reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
|
|
430
486
|
},
|
|
431
487
|
},
|
|
432
488
|
};
|
package/lab/run-evals.js
CHANGED
|
@@ -127,6 +127,10 @@ const USAGE = `Usage: node run-evals.js [options]
|
|
|
127
127
|
grades against each one its evals link (§17), so a run
|
|
128
128
|
linking several groups is handed them here instead of the
|
|
129
129
|
one --dataset. For --run and --rescore; not with --dataset.
|
|
130
|
+
--filter-sets <file> the filter-set bodies a --run kept, by "<id>@<n>" (§18):
|
|
131
|
+
a Library set a job's Responses links cleans and gates
|
|
132
|
+
its reply by. A private set carries its body in the
|
|
133
|
+
document, so only linked sets are here. For --run.
|
|
130
134
|
--source <dir> the same, named the way a run names it: the directory a
|
|
131
135
|
Source is stored at. --samples and --source are one
|
|
132
136
|
flag by two names; both together are refused.
|
|
@@ -208,7 +212,7 @@ function broken(msg) {
|
|
|
208
212
|
|
|
209
213
|
function parseArgs(argv) {
|
|
210
214
|
const o = { model: "", url: "", prompt: null, pipeline: "", samples: "", dataset: "",
|
|
211
|
-
groups: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
215
|
+
groups: "", filterSets: "", replies: "", json: "", source: "", files: "", item: "", resume: "",
|
|
212
216
|
timeout: null, progress: false, cancelFile: "",
|
|
213
217
|
run: "", progressFile: "", resultsFile: "", from: null, only: null,
|
|
214
218
|
rescore: false, tokens: "", plugins: "", minPass: null, junit: "" };
|
|
@@ -230,6 +234,7 @@ function parseArgs(argv) {
|
|
|
230
234
|
case "--samples": o.samples = value(); break;
|
|
231
235
|
case "--dataset": o.dataset = value(); break;
|
|
232
236
|
case "--groups": o.groups = value(); break;
|
|
237
|
+
case "--filter-sets": o.filterSets = value(); break;
|
|
233
238
|
case "--replies": o.replies = value(); break;
|
|
234
239
|
case "--source": o.source = value(); break;
|
|
235
240
|
case "--files": o.files = value(); break;
|
|
@@ -395,6 +400,30 @@ function gradingContext(o) {
|
|
|
395
400
|
return { single, bodies: null, primary: () => single, resolve: () => single ?? undefined };
|
|
396
401
|
}
|
|
397
402
|
|
|
403
|
+
// The filter-set bodies a run kept, as --filter-sets hands them (§18): a JSON
|
|
404
|
+
// object of `<id>@<n>` to body -- a Library set's id and the version the run
|
|
405
|
+
// cleaned and gated with -- each read as today's (upgradeFilterSetBody). A
|
|
406
|
+
// private (`own`) set carries its body in the document, so only Library links
|
|
407
|
+
// are here. Returns a resolver a job's filters step reads a link's body by,
|
|
408
|
+
// mirroring the server's filter_set_ref / group_key.
|
|
409
|
+
function filterContext(o) {
|
|
410
|
+
if (!o.filterSets) return () => undefined;
|
|
411
|
+
let raw;
|
|
412
|
+
try { raw = JSON.parse(fs.readFileSync(o.filterSets, "utf8")); }
|
|
413
|
+
catch (e) { broken(`${o.filterSets}: ${e.message}`); }
|
|
414
|
+
if (!isObject(raw)) broken(`${o.filterSets}: the kept filter sets are a JSON object of "<id>@<n>": body`);
|
|
415
|
+
const bodies = new Map();
|
|
416
|
+
for (const [key, body] of Object.entries(raw)) {
|
|
417
|
+
const at = `${o.filterSets} [${key}]`;
|
|
418
|
+
const got = core.upgradeFilterSetBody(body);
|
|
419
|
+
const bad = core.filterSetBodyProblems(got, at);
|
|
420
|
+
if (bad.length) broken(bad[0]);
|
|
421
|
+
bodies.set(key, got);
|
|
422
|
+
}
|
|
423
|
+
const key = (ref) => (ref && Number.isInteger(ref.n) ? `${ref.id}@${ref.n}` : `${ref && ref.id}`);
|
|
424
|
+
return (ref) => bodies.get(key(ref)) ?? bodies.get(String(ref && ref.id)) ?? undefined;
|
|
425
|
+
}
|
|
426
|
+
|
|
398
427
|
// The cases a graded run is scored against, by the item each names: exactly,
|
|
399
428
|
// as the Source names it. Built from the one group a single-group run names,
|
|
400
429
|
// for an eval type of its own that reads a case (a plugin's); a `group` eval
|
|
@@ -605,8 +634,12 @@ function graderFor(run) {
|
|
|
605
634
|
if (!c) return undefined;
|
|
606
635
|
if (!made.has(ref.id)) {
|
|
607
636
|
const link = reach(c, core.keyVar(ref.id, c.slug));
|
|
608
|
-
made.set(ref.id, async prompt => {
|
|
609
|
-
|
|
637
|
+
made.set(ref.id, async (prompt, schema) => {
|
|
638
|
+
// Held to the metric's schema where the grader's API can hold a
|
|
639
|
+
// reply to one; a model or server that refuses the schema is asked
|
|
640
|
+
// again without it, the prompt alone saying the shape.
|
|
641
|
+
let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
|
|
642
|
+
if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
610
643
|
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
611
644
|
return r.raw ?? "";
|
|
612
645
|
});
|
|
@@ -660,13 +693,13 @@ async function send(to, c, step, cell, record) {
|
|
|
660
693
|
}
|
|
661
694
|
}
|
|
662
695
|
|
|
663
|
-
async function ask(to, c, prompt, dataUrl) {
|
|
696
|
+
async function ask(to, c, prompt, dataUrl, schema = null) {
|
|
664
697
|
const t0 = Date.now();
|
|
665
698
|
const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
|
|
666
699
|
try {
|
|
667
700
|
// The connection's type builds the whole request: the URL, the headers
|
|
668
701
|
// (Anthropic's key travels in x-api-key), and the body.
|
|
669
|
-
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
|
|
702
|
+
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
|
|
670
703
|
const r = await fetch(req.url, {
|
|
671
704
|
method: "POST",
|
|
672
705
|
headers: { "Content-Type": "application/json", ...req.headers },
|
|
@@ -896,9 +929,11 @@ async function runSnapshot(o) {
|
|
|
896
929
|
|
|
897
930
|
// Each scenario once: its stages, the jobs' token sets, and a transport
|
|
898
931
|
// per stage to the profile that stage resolves to, keyed by that
|
|
899
|
-
// profile's id.
|
|
932
|
+
// profile's id. Its stages carry each linked or private filter set as a
|
|
933
|
+
// modifier that cleans and gates the reply where the link sits (#339).
|
|
934
|
+
const resolveFilters = filterContext(o);
|
|
900
935
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
901
|
-
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
936
|
+
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i, resolveFilters);
|
|
902
937
|
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
903
938
|
// The Recorded target replays the recorded reply, so a metric that compares
|
|
904
939
|
// with the recorded reply has nothing to say of it (n/a).
|