evals-lab 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +183 -34
- package/lab/metrics/builtin.mjs +76 -20
- package/lab/run-evals.js +8 -4
- package/lab/server.py +588 -12
- package/lab/web/dist/assets/gallery-SnUhXRBn.js +3 -0
- package/lab/web/dist/assets/main-Ca7o-nM0.css +1 -0
- package/lab/web/dist/assets/main-Dru4_P5G.js +20 -0
- package/lab/web/dist/assets/tokens-0az9gfTq.js +58 -0
- package/lab/web/dist/assets/tokens-CqWJKhOx.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +0 -3
- package/lab/web/dist/assets/main-BQL5j5oF.js +0 -20
- package/lab/web/dist/assets/main-Cza2gwQd.css +0 -1
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +0 -61
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +0 -1
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -135,16 +135,54 @@ function verdictOf(said )
|
|
|
135
135
|
if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
|
|
136
136
|
return read.value ;
|
|
137
137
|
}
|
|
138
|
-
|
|
138
|
+
/** The grader's verdict on [prompt], its reply held to [schema] where its API
|
|
139
|
+
can hold one. A reply that still cannot be read is asked for once more;
|
|
140
|
+
a second, or a grader that failed, is the grader's error -- said as one,
|
|
141
|
+
so it is not read as the reply failing. [ok] says whether a verdict read
|
|
142
|
+
is one this metric can use. */
|
|
143
|
+
async function graded(ctx , prompt , schema ,
|
|
144
|
+
ok = () => true) {
|
|
139
145
|
if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
|
|
140
|
-
|
|
141
|
-
|
|
146
|
+
let last = "";
|
|
147
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
148
|
+
let said ;
|
|
149
|
+
try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
|
|
150
|
+
try {
|
|
151
|
+
const v = verdictOf(said);
|
|
152
|
+
if (ok(v)) return v;
|
|
153
|
+
last = `the grader said ${short(JSON.stringify(v))}`;
|
|
154
|
+
} catch (e) { last = e instanceof Error ? e.message : String(e); }
|
|
155
|
+
}
|
|
156
|
+
throw new Error(`grader error: ${last}`);
|
|
157
|
+
}
|
|
142
158
|
const gradedPass = (v , threshold ) => {
|
|
143
159
|
const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
|
|
144
160
|
const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
|
|
145
161
|
return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
|
|
146
162
|
};
|
|
163
|
+
/** The shape a pass-or-fail grader answers in, for its API to hold it to. */
|
|
164
|
+
const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
|
|
165
|
+
properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
|
|
147
166
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
167
|
+
/** What the grader is told of the task: what the target was asked, and the
|
|
168
|
+
guidance the metric adds, where there is either. */
|
|
169
|
+
const taskOf = (input , guidance ) =>
|
|
170
|
+
(input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
|
|
171
|
+
+ (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
|
|
172
|
+
|
|
173
|
+
/** A grader's three verdicts on a reply set beside the recorded one; any of
|
|
174
|
+
them may be what a metric expects. */
|
|
175
|
+
const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
|
|
176
|
+
const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
|
|
177
|
+
/** What a judgement metric expects: the verdicts set, or -- set before
|
|
178
|
+
Expected was offered -- the one rule there was, worse fails. */
|
|
179
|
+
const expectedOf = (m ) =>
|
|
180
|
+
(Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
|
|
181
|
+
/** The expected verdicts as a reader says them: "better or same". */
|
|
182
|
+
const said = (values ) => {
|
|
183
|
+
const words = values.map((v) => v.toLowerCase());
|
|
184
|
+
return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
|
|
185
|
+
};
|
|
148
186
|
|
|
149
187
|
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
188
|
const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
|
|
@@ -342,6 +380,7 @@ const metrics = {
|
|
|
342
380
|
// reply compared with the reply it is says nothing (docs/workflow-sources.md).
|
|
343
381
|
"same-as-recorded": {
|
|
344
382
|
family: RECORDED,
|
|
383
|
+
parsedOnly: true,
|
|
345
384
|
label: "Same as recorded",
|
|
346
385
|
description: "Passes when the reply is what production recorded for the same item.",
|
|
347
386
|
options: [],
|
|
@@ -356,6 +395,7 @@ const metrics = {
|
|
|
356
395
|
},
|
|
357
396
|
"fields-equal-recorded": {
|
|
358
397
|
family: RECORDED,
|
|
398
|
+
parsedOnly: true,
|
|
359
399
|
label: "Fields as recorded",
|
|
360
400
|
description: "Passes when the fields listed are what the recorded reply had.",
|
|
361
401
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
@@ -374,6 +414,7 @@ const metrics = {
|
|
|
374
414
|
},
|
|
375
415
|
"same-parse-as-recorded": {
|
|
376
416
|
family: RECORDED,
|
|
417
|
+
parsedOnly: true,
|
|
377
418
|
label: "Parses as recorded",
|
|
378
419
|
description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
|
|
379
420
|
options: [],
|
|
@@ -390,43 +431,58 @@ const metrics = {
|
|
|
390
431
|
label: "Rubric",
|
|
391
432
|
description: "Asks the grader whether the reply meets the rubric.",
|
|
392
433
|
graded: true,
|
|
393
|
-
|
|
434
|
+
parsedOnly: true,
|
|
435
|
+
options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
394
436
|
defaults: () => ({ rubric: "", threshold: 0.5 }),
|
|
395
437
|
validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
|
|
396
|
-
score: async (input, m, ctx) => gradedPass(
|
|
397
|
-
`You are grading an output against a rubric.\n\
|
|
398
|
-
num(m.threshold) ?? 0.5),
|
|
438
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
439
|
+
`You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
|
|
440
|
+
VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
399
441
|
},
|
|
400
442
|
factuality: {
|
|
401
443
|
family: GRADED,
|
|
402
444
|
label: "Factual",
|
|
403
445
|
description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
|
|
404
446
|
graded: true,
|
|
405
|
-
|
|
447
|
+
parsedOnly: true,
|
|
448
|
+
options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
406
449
|
defaults: () => ({ reference: "", threshold: 0.5 }),
|
|
407
450
|
// A blank reference is the recorded reply.
|
|
408
|
-
score: async (input, m, ctx) => gradedPass(
|
|
451
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
409
452
|
`You are checking an output for factual consistency with a reference. Differences in wording or detail are `
|
|
410
|
-
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\
|
|
411
|
-
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}
|
|
453
|
+
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
|
|
454
|
+
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
412
455
|
},
|
|
413
456
|
"judge-vs-production": {
|
|
414
457
|
family: GRADED,
|
|
415
458
|
label: "Judged against recorded",
|
|
416
459
|
description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
|
|
417
460
|
graded: true,
|
|
418
|
-
|
|
419
|
-
|
|
461
|
+
parsedOnly: true,
|
|
462
|
+
expects: true,
|
|
463
|
+
options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
|
|
464
|
+
{ key: "task", label: "Additional guidance (optional)", type: "textarea" }],
|
|
465
|
+
defaults: () => ({ expected: ["better", "same"], task: "" }),
|
|
466
|
+
validate: (m, at, bad) => {
|
|
467
|
+
const known = JUDGEMENTS.map((j) => j.value);
|
|
468
|
+
if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
|
|
469
|
+
|| m.expected.some((v) => !known.includes(String(v))))) {
|
|
470
|
+
bad.push(`${at}: Expected needs at least one of ${said(known)}`);
|
|
471
|
+
}
|
|
472
|
+
},
|
|
420
473
|
score: async (input, m, ctx) => {
|
|
421
|
-
|
|
474
|
+
if (ctx.recordedTarget) return null;
|
|
475
|
+
const v = await graded(ctx,
|
|
422
476
|
`Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
|
|
423
|
-
+ `RECORDED one
|
|
477
|
+
+ `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
|
|
424
478
|
+ `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
|
|
425
|
-
+ `"reason": "one sentence"}
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
479
|
+
+ `"reason": "one sentence"}.`,
|
|
480
|
+
{ type: "object", additionalProperties: false, required: ["verdict", "reason"],
|
|
481
|
+
properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
|
|
482
|
+
(r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
|
|
483
|
+
const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
|
|
484
|
+
return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
|
|
485
|
+
reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
|
|
430
486
|
},
|
|
431
487
|
},
|
|
432
488
|
};
|
package/lab/run-evals.js
CHANGED
|
@@ -605,8 +605,12 @@ function graderFor(run) {
|
|
|
605
605
|
if (!c) return undefined;
|
|
606
606
|
if (!made.has(ref.id)) {
|
|
607
607
|
const link = reach(c, core.keyVar(ref.id, c.slug));
|
|
608
|
-
made.set(ref.id, async prompt => {
|
|
609
|
-
|
|
608
|
+
made.set(ref.id, async (prompt, schema) => {
|
|
609
|
+
// Held to the metric's schema where the grader's API can hold a
|
|
610
|
+
// reply to one; a model or server that refuses the schema is asked
|
|
611
|
+
// again without it, the prompt alone saying the shape.
|
|
612
|
+
let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
|
|
613
|
+
if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
610
614
|
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
611
615
|
return r.raw ?? "";
|
|
612
616
|
});
|
|
@@ -660,13 +664,13 @@ async function send(to, c, step, cell, record) {
|
|
|
660
664
|
}
|
|
661
665
|
}
|
|
662
666
|
|
|
663
|
-
async function ask(to, c, prompt, dataUrl) {
|
|
667
|
+
async function ask(to, c, prompt, dataUrl, schema = null) {
|
|
664
668
|
const t0 = Date.now();
|
|
665
669
|
const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
|
|
666
670
|
try {
|
|
667
671
|
// The connection's type builds the whole request: the URL, the headers
|
|
668
672
|
// (Anthropic's key travels in x-api-key), and the body.
|
|
669
|
-
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
|
|
673
|
+
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
|
|
670
674
|
const r = await fetch(req.url, {
|
|
671
675
|
method: "POST",
|
|
672
676
|
headers: { "Content-Type": "application/json", ...req.headers },
|