evals-lab 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +108 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +464 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/metrics/builtin.mjs +113 -52
- package/lab/run-evals.js +32 -8
- package/lab/server.py +1174 -179
- package/lab/web/dist/assets/gallery-SnUhXRBn.js +3 -0
- package/lab/web/dist/assets/main-Ca7o-nM0.css +1 -0
- package/lab/web/dist/assets/main-Dru4_P5G.js +20 -0
- package/lab/web/dist/assets/tokens-0az9gfTq.js +58 -0
- package/lab/web/dist/assets/tokens-CqWJKhOx.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +0 -3
- package/lab/web/dist/assets/main-DDoeU6hq.css +0 -1
- package/lab/web/dist/assets/main-nz6Q4jVm.js +0 -21
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +0 -59
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +0 -1
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -120,10 +120,10 @@ function holds(input , m , value , except
|
|
|
120
120
|
/** A requirement as Results names it: a term, or the group it was. */
|
|
121
121
|
const spoken = (g ) => (g.length === 1 ? g[0] : g);
|
|
122
122
|
|
|
123
|
-
/**
|
|
124
|
-
const
|
|
125
|
-
if (input.
|
|
126
|
-
return input.
|
|
123
|
+
/** The recorded reply, or the reason a metric that compares with it cannot. */
|
|
124
|
+
const recorded = (input ) => {
|
|
125
|
+
if (input.recorded == null) throw new Error("this item has no recorded reply to compare with");
|
|
126
|
+
return input.recorded;
|
|
127
127
|
};
|
|
128
128
|
|
|
129
129
|
// ---- model-graded -----------------------------------------------------------
|
|
@@ -135,19 +135,57 @@ function verdictOf(said )
|
|
|
135
135
|
if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
|
|
136
136
|
return read.value ;
|
|
137
137
|
}
|
|
138
|
-
|
|
138
|
+
/** The grader's verdict on [prompt], its reply held to [schema] where its API
|
|
139
|
+
can hold one. A reply that still cannot be read is asked for once more;
|
|
140
|
+
a second, or a grader that failed, is the grader's error -- said as one,
|
|
141
|
+
so it is not read as the reply failing. [ok] says whether a verdict read
|
|
142
|
+
is one this metric can use. */
|
|
143
|
+
async function graded(ctx , prompt , schema ,
|
|
144
|
+
ok = () => true) {
|
|
139
145
|
if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
|
|
140
|
-
|
|
141
|
-
|
|
146
|
+
let last = "";
|
|
147
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
148
|
+
let said ;
|
|
149
|
+
try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
|
|
150
|
+
try {
|
|
151
|
+
const v = verdictOf(said);
|
|
152
|
+
if (ok(v)) return v;
|
|
153
|
+
last = `the grader said ${short(JSON.stringify(v))}`;
|
|
154
|
+
} catch (e) { last = e instanceof Error ? e.message : String(e); }
|
|
155
|
+
}
|
|
156
|
+
throw new Error(`grader error: ${last}`);
|
|
157
|
+
}
|
|
142
158
|
const gradedPass = (v , threshold ) => {
|
|
143
159
|
const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
|
|
144
160
|
const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
|
|
145
161
|
return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
|
|
146
162
|
};
|
|
163
|
+
/** The shape a pass-or-fail grader answers in, for its API to hold it to. */
|
|
164
|
+
const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
|
|
165
|
+
properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
|
|
147
166
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
167
|
+
/** What the grader is told of the task: what the target was asked, and the
|
|
168
|
+
guidance the metric adds, where there is either. */
|
|
169
|
+
const taskOf = (input , guidance ) =>
|
|
170
|
+
(input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
|
|
171
|
+
+ (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
|
|
172
|
+
|
|
173
|
+
/** A grader's three verdicts on a reply set beside the recorded one; any of
|
|
174
|
+
them may be what a metric expects. */
|
|
175
|
+
const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
|
|
176
|
+
const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
|
|
177
|
+
/** What a judgement metric expects: the verdicts set, or -- set before
|
|
178
|
+
Expected was offered -- the one rule there was, worse fails. */
|
|
179
|
+
const expectedOf = (m ) =>
|
|
180
|
+
(Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
|
|
181
|
+
/** The expected verdicts as a reader says them: "better or same". */
|
|
182
|
+
const said = (values ) => {
|
|
183
|
+
const words = values.map((v) => v.toLowerCase());
|
|
184
|
+
return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
|
|
185
|
+
};
|
|
148
186
|
|
|
149
187
|
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
|
-
const TEXT = "The reply's text", RESULT = "The result",
|
|
188
|
+
const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
|
|
151
189
|
|
|
152
190
|
const metrics = {
|
|
153
191
|
equals: {
|
|
@@ -337,46 +375,54 @@ const metrics = {
|
|
|
337
375
|
1 - d / Math.max(1, Math.max(got.length, want.length)));
|
|
338
376
|
},
|
|
339
377
|
},
|
|
340
|
-
// ---- against
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
378
|
+
// ---- against the recorded reply --------------------------------------------
|
|
379
|
+
// n/a (null) when the target under test is the Recorded target itself: a
|
|
380
|
+
// reply compared with the reply it is says nothing (docs/workflow-sources.md).
|
|
381
|
+
"same-as-recorded": {
|
|
382
|
+
family: RECORDED,
|
|
383
|
+
parsedOnly: true,
|
|
384
|
+
label: "Same as recorded",
|
|
385
|
+
description: "Passes when the reply is what production recorded for the same item.",
|
|
345
386
|
options: [],
|
|
346
387
|
defaults: () => ({}),
|
|
347
|
-
score: (input) => {
|
|
348
|
-
|
|
388
|
+
score: (input, _m, ctx) => {
|
|
389
|
+
if (ctx.recordedTarget) return null;
|
|
390
|
+
const p = recorded(input).trim(), got = input.text.trim();
|
|
349
391
|
const a = json(got), b = json(p);
|
|
350
392
|
const equal = !a.error && !b.error ? same(a.value, b.value) : got === p;
|
|
351
|
-
return ok(equal, equal ? "as
|
|
393
|
+
return ok(equal, equal ? "as recorded" : `recorded ${short(p)}`);
|
|
352
394
|
},
|
|
353
395
|
},
|
|
354
|
-
"fields-equal-
|
|
355
|
-
family:
|
|
356
|
-
|
|
357
|
-
|
|
396
|
+
"fields-equal-recorded": {
|
|
397
|
+
family: RECORDED,
|
|
398
|
+
parsedOnly: true,
|
|
399
|
+
label: "Fields as recorded",
|
|
400
|
+
description: "Passes when the fields listed are what the recorded reply had.",
|
|
358
401
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
359
402
|
defaults: () => ({ fields: "" }),
|
|
360
403
|
validate: (m, at, bad) => needs(m, ["fields"], at, bad),
|
|
361
|
-
score: (input, m) => {
|
|
362
|
-
|
|
404
|
+
score: (input, m, ctx) => {
|
|
405
|
+
if (ctx.recordedTarget) return null;
|
|
406
|
+
const a = json(input.text), b = json(recorded(input));
|
|
363
407
|
if (a.error) return ok(false, `not JSON: ${a.error}`);
|
|
364
|
-
if (b.error) return ok(false, "
|
|
408
|
+
if (b.error) return ok(false, "the recorded reply was not JSON");
|
|
365
409
|
const fields = lines(m.fields);
|
|
366
410
|
const differ = fields.filter((f) => !same(atPath(a.value, f), atPath(b.value, f)));
|
|
367
|
-
return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from
|
|
411
|
+
return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from recorded` : "the fields match recorded",
|
|
368
412
|
fields.length ? (fields.length - differ.length) / fields.length : 1);
|
|
369
413
|
},
|
|
370
414
|
},
|
|
371
|
-
"same-parse-
|
|
372
|
-
family:
|
|
373
|
-
|
|
374
|
-
|
|
415
|
+
"same-parse-as-recorded": {
|
|
416
|
+
family: RECORDED,
|
|
417
|
+
parsedOnly: true,
|
|
418
|
+
label: "Parses as recorded",
|
|
419
|
+
description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
|
|
375
420
|
options: [],
|
|
376
421
|
defaults: () => ({}),
|
|
377
|
-
score: (input) => {
|
|
378
|
-
|
|
379
|
-
|
|
422
|
+
score: (input, _m, ctx) => {
|
|
423
|
+
if (ctx.recordedTarget) return null;
|
|
424
|
+
const mine = !json(input.text).error, theirs = !json(recorded(input)).error;
|
|
425
|
+
return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, recorded ${theirs ? "parsed" : "did not"}`);
|
|
380
426
|
},
|
|
381
427
|
},
|
|
382
428
|
// ---- model-graded -------------------------------------------------------------
|
|
@@ -385,43 +431,58 @@ const metrics = {
|
|
|
385
431
|
label: "Rubric",
|
|
386
432
|
description: "Asks the grader whether the reply meets the rubric.",
|
|
387
433
|
graded: true,
|
|
388
|
-
|
|
434
|
+
parsedOnly: true,
|
|
435
|
+
options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
389
436
|
defaults: () => ({ rubric: "", threshold: 0.5 }),
|
|
390
437
|
validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
|
|
391
|
-
score: async (input, m, ctx) => gradedPass(
|
|
392
|
-
`You are grading an output against a rubric.\n\
|
|
393
|
-
num(m.threshold) ?? 0.5),
|
|
438
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
439
|
+
`You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
|
|
440
|
+
VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
394
441
|
},
|
|
395
442
|
factuality: {
|
|
396
443
|
family: GRADED,
|
|
397
444
|
label: "Factual",
|
|
398
|
-
description: "Asks the grader whether the reply agrees with the reference, or
|
|
445
|
+
description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
|
|
399
446
|
graded: true,
|
|
400
|
-
|
|
447
|
+
parsedOnly: true,
|
|
448
|
+
options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
|
|
401
449
|
defaults: () => ({ reference: "", threshold: 0.5 }),
|
|
402
|
-
// A blank reference is
|
|
403
|
-
score: async (input, m, ctx) => gradedPass(
|
|
450
|
+
// A blank reference is the recorded reply.
|
|
451
|
+
score: async (input, m, ctx) => gradedPass(await graded(ctx,
|
|
404
452
|
`You are checking an output for factual consistency with a reference. Differences in wording or detail are `
|
|
405
|
-
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\
|
|
406
|
-
+ `${str(m.reference).trim() ||
|
|
453
|
+
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
|
|
454
|
+
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
|
|
407
455
|
},
|
|
408
456
|
"judge-vs-production": {
|
|
409
457
|
family: GRADED,
|
|
410
|
-
label: "Judged against
|
|
411
|
-
description: "Asks the grader whether the reply is better than, the same as or worse than
|
|
458
|
+
label: "Judged against recorded",
|
|
459
|
+
description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
|
|
412
460
|
graded: true,
|
|
413
|
-
|
|
414
|
-
|
|
461
|
+
parsedOnly: true,
|
|
462
|
+
expects: true,
|
|
463
|
+
options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
|
|
464
|
+
{ key: "task", label: "Additional guidance (optional)", type: "textarea" }],
|
|
465
|
+
defaults: () => ({ expected: ["better", "same"], task: "" }),
|
|
466
|
+
validate: (m, at, bad) => {
|
|
467
|
+
const known = JUDGEMENTS.map((j) => j.value);
|
|
468
|
+
if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
|
|
469
|
+
|| m.expected.some((v) => !known.includes(String(v))))) {
|
|
470
|
+
bad.push(`${at}: Expected needs at least one of ${said(known)}`);
|
|
471
|
+
}
|
|
472
|
+
},
|
|
415
473
|
score: async (input, m, ctx) => {
|
|
416
|
-
|
|
474
|
+
if (ctx.recordedTarget) return null;
|
|
475
|
+
const v = await graded(ctx,
|
|
417
476
|
`Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
|
|
418
|
-
+ `
|
|
477
|
+
+ `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
|
|
419
478
|
+ `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
|
|
420
|
-
+ `"reason": "one sentence"}
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
479
|
+
+ `"reason": "one sentence"}.`,
|
|
480
|
+
{ type: "object", additionalProperties: false, required: ["verdict", "reason"],
|
|
481
|
+
properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
|
|
482
|
+
(r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
|
|
483
|
+
const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
|
|
484
|
+
return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
|
|
485
|
+
reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
|
|
425
486
|
},
|
|
426
487
|
},
|
|
427
488
|
};
|
package/lab/run-evals.js
CHANGED
|
@@ -550,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
|
|
|
550
550
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
551
551
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
552
552
|
}
|
|
553
|
+
// The Recorded target replays each call's recorded reply, read as the flow
|
|
554
|
+
// reads one (like send, but sending nothing); it does not go through the
|
|
555
|
+
// model Read-as wrapper below -- the recorded body is read directly.
|
|
556
|
+
if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
|
|
557
|
+
return () => {
|
|
558
|
+
const r = record ?? textRecord(text);
|
|
559
|
+
const result = r && r.result;
|
|
560
|
+
if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
|
|
561
|
+
try {
|
|
562
|
+
const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
|
|
563
|
+
{ scope: r.scope || {}, now: r.at ?? null });
|
|
564
|
+
return { raw: read.raw, said: read.said, ms: 0, conn: null };
|
|
565
|
+
} catch (e) {
|
|
566
|
+
return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
|
|
567
|
+
}
|
|
568
|
+
};
|
|
569
|
+
}
|
|
553
570
|
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
554
|
-
? async sent => core.localAnswer(c, text, sent)
|
|
571
|
+
? async sent => core.localAnswer(c, text, sent, record)
|
|
555
572
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
556
573
|
if (!step?.readAs || !step?.step) return call;
|
|
557
574
|
return async (sent, url) => {
|
|
@@ -588,8 +605,12 @@ function graderFor(run) {
|
|
|
588
605
|
if (!c) return undefined;
|
|
589
606
|
if (!made.has(ref.id)) {
|
|
590
607
|
const link = reach(c, core.keyVar(ref.id, c.slug));
|
|
591
|
-
made.set(ref.id, async prompt => {
|
|
592
|
-
|
|
608
|
+
made.set(ref.id, async (prompt, schema) => {
|
|
609
|
+
// Held to the metric's schema where the grader's API can hold a
|
|
610
|
+
// reply to one; a model or server that refuses the schema is asked
|
|
611
|
+
// again without it, the prompt alone saying the shape.
|
|
612
|
+
let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
|
|
613
|
+
if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
|
|
593
614
|
if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
|
|
594
615
|
return r.raw ?? "";
|
|
595
616
|
});
|
|
@@ -643,13 +664,13 @@ async function send(to, c, step, cell, record) {
|
|
|
643
664
|
}
|
|
644
665
|
}
|
|
645
666
|
|
|
646
|
-
async function ask(to, c, prompt, dataUrl) {
|
|
667
|
+
async function ask(to, c, prompt, dataUrl, schema = null) {
|
|
647
668
|
const t0 = Date.now();
|
|
648
669
|
const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
|
|
649
670
|
try {
|
|
650
671
|
// The connection's type builds the whole request: the URL, the headers
|
|
651
672
|
// (Anthropic's key travels in x-api-key), and the body.
|
|
652
|
-
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
|
|
673
|
+
const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
|
|
653
674
|
const r = await fetch(req.url, {
|
|
654
675
|
method: "POST",
|
|
655
676
|
headers: { "Content-Type": "application/json", ...req.headers },
|
|
@@ -883,7 +904,10 @@ async function runSnapshot(o) {
|
|
|
883
904
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
884
905
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
885
906
|
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
886
|
-
|
|
907
|
+
// The Recorded target replays the recorded reply, so a metric that compares
|
|
908
|
+
// with the recorded reply has nothing to say of it (n/a).
|
|
909
|
+
const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
|
|
910
|
+
return { stages, tokens, connections, links, calls, cells, recordedTarget };
|
|
887
911
|
});
|
|
888
912
|
|
|
889
913
|
let results = [];
|
|
@@ -945,9 +969,9 @@ async function runSnapshot(o) {
|
|
|
945
969
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
946
970
|
// text item a dataset grades is graded like an image.
|
|
947
971
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
948
|
-
production = core.
|
|
972
|
+
production = core.recordedReplyOf(run, record);
|
|
949
973
|
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
950
|
-
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
|
|
974
|
+
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
|
|
951
975
|
}
|
|
952
976
|
// Production's reply to the item, where its record holds one: what a
|
|
953
977
|
// metric compares with, kept so a re-score reads it again.
|