evals-lab 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +285 -45
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +24 -4
- package/lab/server.py +586 -167
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +0 -3
- package/lab/web/dist/assets/main-DDoeU6hq.css +0 -1
- package/lab/web/dist/assets/main-nz6Q4jVm.js +0 -21
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +0 -59
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +0 -1
package/lab/metrics/builtin.mjs
CHANGED
|
@@ -120,10 +120,10 @@ function holds(input , m , value , except
|
|
|
120
120
|
/** A requirement as Results names it: a term, or the group it was. */
|
|
121
121
|
const spoken = (g ) => (g.length === 1 ? g[0] : g);
|
|
122
122
|
|
|
123
|
-
/**
|
|
124
|
-
const
|
|
125
|
-
if (input.
|
|
126
|
-
return input.
|
|
123
|
+
/** The recorded reply, or the reason a metric that compares with it cannot. */
|
|
124
|
+
const recorded = (input ) => {
|
|
125
|
+
if (input.recorded == null) throw new Error("this item has no recorded reply to compare with");
|
|
126
|
+
return input.recorded;
|
|
127
127
|
};
|
|
128
128
|
|
|
129
129
|
// ---- model-graded -----------------------------------------------------------
|
|
@@ -147,7 +147,7 @@ const gradedPass = (v , threshold )
|
|
|
147
147
|
const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
|
|
148
148
|
|
|
149
149
|
// The families the Add metric picker lists the metrics under: what each reads.
|
|
150
|
-
const TEXT = "The reply's text", RESULT = "The result",
|
|
150
|
+
const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
|
|
151
151
|
|
|
152
152
|
const metrics = {
|
|
153
153
|
equals: {
|
|
@@ -337,46 +337,51 @@ const metrics = {
|
|
|
337
337
|
1 - d / Math.max(1, Math.max(got.length, want.length)));
|
|
338
338
|
},
|
|
339
339
|
},
|
|
340
|
-
// ---- against
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
340
|
+
// ---- against the recorded reply --------------------------------------------
|
|
341
|
+
// n/a (null) when the target under test is the Recorded target itself: a
|
|
342
|
+
// reply compared with the reply it is says nothing (docs/workflow-sources.md).
|
|
343
|
+
"same-as-recorded": {
|
|
344
|
+
family: RECORDED,
|
|
345
|
+
label: "Same as recorded",
|
|
346
|
+
description: "Passes when the reply is what production recorded for the same item.",
|
|
345
347
|
options: [],
|
|
346
348
|
defaults: () => ({}),
|
|
347
|
-
score: (input) => {
|
|
348
|
-
|
|
349
|
+
score: (input, _m, ctx) => {
|
|
350
|
+
if (ctx.recordedTarget) return null;
|
|
351
|
+
const p = recorded(input).trim(), got = input.text.trim();
|
|
349
352
|
const a = json(got), b = json(p);
|
|
350
353
|
const equal = !a.error && !b.error ? same(a.value, b.value) : got === p;
|
|
351
|
-
return ok(equal, equal ? "as
|
|
354
|
+
return ok(equal, equal ? "as recorded" : `recorded ${short(p)}`);
|
|
352
355
|
},
|
|
353
356
|
},
|
|
354
|
-
"fields-equal-
|
|
355
|
-
family:
|
|
356
|
-
label: "Fields as
|
|
357
|
-
description: "Passes when the fields listed are what
|
|
357
|
+
"fields-equal-recorded": {
|
|
358
|
+
family: RECORDED,
|
|
359
|
+
label: "Fields as recorded",
|
|
360
|
+
description: "Passes when the fields listed are what the recorded reply had.",
|
|
358
361
|
options: [{ key: "fields", label: "Fields", type: "textarea" }],
|
|
359
362
|
defaults: () => ({ fields: "" }),
|
|
360
363
|
validate: (m, at, bad) => needs(m, ["fields"], at, bad),
|
|
361
|
-
score: (input, m) => {
|
|
362
|
-
|
|
364
|
+
score: (input, m, ctx) => {
|
|
365
|
+
if (ctx.recordedTarget) return null;
|
|
366
|
+
const a = json(input.text), b = json(recorded(input));
|
|
363
367
|
if (a.error) return ok(false, `not JSON: ${a.error}`);
|
|
364
|
-
if (b.error) return ok(false, "
|
|
368
|
+
if (b.error) return ok(false, "the recorded reply was not JSON");
|
|
365
369
|
const fields = lines(m.fields);
|
|
366
370
|
const differ = fields.filter((f) => !same(atPath(a.value, f), atPath(b.value, f)));
|
|
367
|
-
return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from
|
|
371
|
+
return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from recorded` : "the fields match recorded",
|
|
368
372
|
fields.length ? (fields.length - differ.length) / fields.length : 1);
|
|
369
373
|
},
|
|
370
374
|
},
|
|
371
|
-
"same-parse-
|
|
372
|
-
family:
|
|
373
|
-
label: "Parses as
|
|
374
|
-
description: "Passes when the reply parses as JSON exactly when
|
|
375
|
+
"same-parse-as-recorded": {
|
|
376
|
+
family: RECORDED,
|
|
377
|
+
label: "Parses as recorded",
|
|
378
|
+
description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
|
|
375
379
|
options: [],
|
|
376
380
|
defaults: () => ({}),
|
|
377
|
-
score: (input) => {
|
|
378
|
-
|
|
379
|
-
|
|
381
|
+
score: (input, _m, ctx) => {
|
|
382
|
+
if (ctx.recordedTarget) return null;
|
|
383
|
+
const mine = !json(input.text).error, theirs = !json(recorded(input)).error;
|
|
384
|
+
return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, recorded ${theirs ? "parsed" : "did not"}`);
|
|
380
385
|
},
|
|
381
386
|
},
|
|
382
387
|
// ---- model-graded -------------------------------------------------------------
|
|
@@ -395,27 +400,27 @@ const metrics = {
|
|
|
395
400
|
factuality: {
|
|
396
401
|
family: GRADED,
|
|
397
402
|
label: "Factual",
|
|
398
|
-
description: "Asks the grader whether the reply agrees with the reference, or
|
|
403
|
+
description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
|
|
399
404
|
graded: true,
|
|
400
405
|
options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
|
|
401
406
|
defaults: () => ({ reference: "", threshold: 0.5 }),
|
|
402
|
-
// A blank reference is
|
|
407
|
+
// A blank reference is the recorded reply.
|
|
403
408
|
score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
|
|
404
409
|
`You are checking an output for factual consistency with a reference. Differences in wording or detail are `
|
|
405
410
|
+ `fine; a contradiction, or a fact the reference does not support, is not.\n\nReference:\n`
|
|
406
|
-
+ `${str(m.reference).trim() ||
|
|
411
|
+
+ `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
|
|
407
412
|
},
|
|
408
413
|
"judge-vs-production": {
|
|
409
414
|
family: GRADED,
|
|
410
|
-
label: "Judged against
|
|
411
|
-
description: "Asks the grader whether the reply is better than, the same as or worse than
|
|
415
|
+
label: "Judged against recorded",
|
|
416
|
+
description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
|
|
412
417
|
graded: true,
|
|
413
418
|
options: [{ key: "task", label: "Task", type: "textarea" }],
|
|
414
419
|
defaults: () => ({ task: "" }),
|
|
415
420
|
score: async (input, m, ctx) => {
|
|
416
421
|
const v = verdictOf(await ask(ctx,
|
|
417
422
|
`Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
|
|
418
|
-
+ `
|
|
423
|
+
+ `RECORDED one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nRECORDED:\n${recorded(input)}\n\n`
|
|
419
424
|
+ `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
|
|
420
425
|
+ `"reason": "one sentence"}.`));
|
|
421
426
|
const verdict = str(v.verdict).toLowerCase();
|
package/lab/run-evals.js
CHANGED
|
@@ -550,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
|
|
|
550
550
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
551
551
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
552
552
|
}
|
|
553
|
+
// The Recorded target replays each call's recorded reply, read as the flow
|
|
554
|
+
// reads one (like send, but sending nothing); it does not go through the
|
|
555
|
+
// model Read-as wrapper below -- the recorded body is read directly.
|
|
556
|
+
if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
|
|
557
|
+
return () => {
|
|
558
|
+
const r = record ?? textRecord(text);
|
|
559
|
+
const result = r && r.result;
|
|
560
|
+
if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
|
|
561
|
+
try {
|
|
562
|
+
const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
|
|
563
|
+
{ scope: r.scope || {}, now: r.at ?? null });
|
|
564
|
+
return { raw: read.raw, said: read.said, ms: 0, conn: null };
|
|
565
|
+
} catch (e) {
|
|
566
|
+
return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
|
|
567
|
+
}
|
|
568
|
+
};
|
|
569
|
+
}
|
|
553
570
|
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
554
|
-
? async sent => core.localAnswer(c, text, sent)
|
|
571
|
+
? async sent => core.localAnswer(c, text, sent, record)
|
|
555
572
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
556
573
|
if (!step?.readAs || !step?.step) return call;
|
|
557
574
|
return async (sent, url) => {
|
|
@@ -883,7 +900,10 @@ async function runSnapshot(o) {
|
|
|
883
900
|
const plans = core.targetsOf(run).map((_, i) => {
|
|
884
901
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
885
902
|
const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
|
|
886
|
-
|
|
903
|
+
// The Recorded target replays the recorded reply, so a metric that compares
|
|
904
|
+
// with the recorded reply has nothing to say of it (n/a).
|
|
905
|
+
const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
|
|
906
|
+
return { stages, tokens, connections, links, calls, cells, recordedTarget };
|
|
887
907
|
});
|
|
888
908
|
|
|
889
909
|
let results = [];
|
|
@@ -945,9 +965,9 @@ async function runSnapshot(o) {
|
|
|
945
965
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
946
966
|
// text item a dataset grades is graded like an image.
|
|
947
967
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
948
|
-
production = core.
|
|
968
|
+
production = core.recordedReplyOf(run, record);
|
|
949
969
|
sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
|
|
950
|
-
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
|
|
970
|
+
{ production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
|
|
951
971
|
}
|
|
952
972
|
// Production's reply to the item, where its record holds one: what a
|
|
953
973
|
// metric compares with, kept so a re-score reads it again.
|