evals-lab 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -120,10 +120,10 @@ function holds(input , m , value , except
120
120
  /** A requirement as Results names it: a term, or the group it was. */
121
121
  const spoken = (g ) => (g.length === 1 ? g[0] : g);
122
122
 
123
- /** Production's reply, or the reason a metric that compares with it cannot. */
124
- const production = (input ) => {
125
- if (input.production == null) throw new Error("this item has no production reply to compare with");
126
- return input.production;
123
+ /** The recorded reply, or the reason a metric that compares with it cannot. */
124
+ const recorded = (input ) => {
125
+ if (input.recorded == null) throw new Error("this item has no recorded reply to compare with");
126
+ return input.recorded;
127
127
  };
128
128
 
129
129
  // ---- model-graded -----------------------------------------------------------
@@ -135,19 +135,57 @@ function verdictOf(said )
135
135
  if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
136
136
  return read.value ;
137
137
  }
138
- const ask = async (ctx , prompt ) => {
138
+ /** The grader's verdict on [prompt], its reply held to [schema] where its API
139
+ can hold one. A reply that still cannot be read is asked for once more;
140
+ a second, or a grader that failed, is the grader's error -- said as one,
141
+ so it is not read as the reply failing. [ok] says whether a verdict read
142
+ is one this metric can use. */
143
+ async function graded(ctx , prompt , schema ,
144
+ ok = () => true) {
139
145
  if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
140
- return ctx.ask(prompt);
141
- };
146
+ let last = "";
147
+ for (let attempt = 0; attempt < 2; attempt++) {
148
+ let said ;
149
+ try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
150
+ try {
151
+ const v = verdictOf(said);
152
+ if (ok(v)) return v;
153
+ last = `the grader said ${short(JSON.stringify(v))}`;
154
+ } catch (e) { last = e instanceof Error ? e.message : String(e); }
155
+ }
156
+ throw new Error(`grader error: ${last}`);
157
+ }
142
158
  const gradedPass = (v , threshold ) => {
143
159
  const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
144
160
  const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
145
161
  return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
146
162
  };
163
+ /** The shape a pass-or-fail grader answers in, for its API to hold it to. */
164
+ const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
165
+ properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
147
166
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
167
+ /** What the grader is told of the task: what the target was asked, and the
168
+ guidance the metric adds, where there is either. */
169
+ const taskOf = (input , guidance ) =>
170
+ (input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
171
+ + (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
172
+
173
+ /** A grader's three verdicts on a reply set beside the recorded one; any of
174
+ them may be what a metric expects. */
175
+ const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
176
+ const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
177
+ /** What a judgement metric expects: the verdicts set, or -- set before
178
+ Expected was offered -- the one rule there was, worse fails. */
179
+ const expectedOf = (m ) =>
180
+ (Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
181
+ /** The expected verdicts as a reader says them: "better or same". */
182
+ const said = (values ) => {
183
+ const words = values.map((v) => v.toLowerCase());
184
+ return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
185
+ };
148
186
 
149
187
  // The families the Add metric picker lists the metrics under: what each reads.
150
- const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
188
+ const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
151
189
 
152
190
  const metrics = {
153
191
  equals: {
@@ -337,46 +375,54 @@ const metrics = {
337
375
  1 - d / Math.max(1, Math.max(got.length, want.length)));
338
376
  },
339
377
  },
340
- // ---- against production ----------------------------------------------------
341
- "equals-production": {
342
- family: PROD,
343
- label: "Same as production",
344
- description: "Passes when the reply is what production replied to the same item.",
378
+ // ---- against the recorded reply --------------------------------------------
379
+ // n/a (null) when the target under test is the Recorded target itself: a
380
+ // reply compared with the reply it is says nothing (docs/workflow-sources.md).
381
+ "same-as-recorded": {
382
+ family: RECORDED,
383
+ parsedOnly: true,
384
+ label: "Same as recorded",
385
+ description: "Passes when the reply is what production recorded for the same item.",
345
386
  options: [],
346
387
  defaults: () => ({}),
347
- score: (input) => {
348
- const p = production(input).trim(), got = input.text.trim();
388
+ score: (input, _m, ctx) => {
389
+ if (ctx.recordedTarget) return null;
390
+ const p = recorded(input).trim(), got = input.text.trim();
349
391
  const a = json(got), b = json(p);
350
392
  const equal = !a.error && !b.error ? same(a.value, b.value) : got === p;
351
- return ok(equal, equal ? "as production replied" : `production replied ${short(p)}`);
393
+ return ok(equal, equal ? "as recorded" : `recorded ${short(p)}`);
352
394
  },
353
395
  },
354
- "fields-equal-production": {
355
- family: PROD,
356
- label: "Fields as production",
357
- description: "Passes when the fields listed are what production's reply had.",
396
+ "fields-equal-recorded": {
397
+ family: RECORDED,
398
+ parsedOnly: true,
399
+ label: "Fields as recorded",
400
+ description: "Passes when the fields listed are what the recorded reply had.",
358
401
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
359
402
  defaults: () => ({ fields: "" }),
360
403
  validate: (m, at, bad) => needs(m, ["fields"], at, bad),
361
- score: (input, m) => {
362
- const a = json(input.text), b = json(production(input));
404
+ score: (input, m, ctx) => {
405
+ if (ctx.recordedTarget) return null;
406
+ const a = json(input.text), b = json(recorded(input));
363
407
  if (a.error) return ok(false, `not JSON: ${a.error}`);
364
- if (b.error) return ok(false, "production's reply was not JSON");
408
+ if (b.error) return ok(false, "the recorded reply was not JSON");
365
409
  const fields = lines(m.fields);
366
410
  const differ = fields.filter((f) => !same(atPath(a.value, f), atPath(b.value, f)));
367
- return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from production` : "the fields match production",
411
+ return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from recorded` : "the fields match recorded",
368
412
  fields.length ? (fields.length - differ.length) / fields.length : 1);
369
413
  },
370
414
  },
371
- "same-parse-outcome": {
372
- family: PROD,
373
- label: "Parses as production",
374
- description: "Passes when the reply parses as JSON exactly when production's did.",
415
+ "same-parse-as-recorded": {
416
+ family: RECORDED,
417
+ parsedOnly: true,
418
+ label: "Parses as recorded",
419
+ description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
375
420
  options: [],
376
421
  defaults: () => ({}),
377
- score: (input) => {
378
- const mine = !json(input.text).error, theirs = !json(production(input)).error;
379
- return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, production's ${theirs ? "parsed" : "did not"}`);
422
+ score: (input, _m, ctx) => {
423
+ if (ctx.recordedTarget) return null;
424
+ const mine = !json(input.text).error, theirs = !json(recorded(input)).error;
425
+ return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, recorded ${theirs ? "parsed" : "did not"}`);
380
426
  },
381
427
  },
382
428
  // ---- model-graded -------------------------------------------------------------
@@ -385,43 +431,58 @@ const metrics = {
385
431
  label: "Rubric",
386
432
  description: "Asks the grader whether the reply meets the rubric.",
387
433
  graded: true,
388
- options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
434
+ parsedOnly: true,
435
+ options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
389
436
  defaults: () => ({ rubric: "", threshold: 0.5 }),
390
437
  validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
391
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
392
- `You are grading an output against a rubric.\n\nRubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`)),
393
- num(m.threshold) ?? 0.5),
438
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
439
+ `You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
440
+ VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
394
441
  },
395
442
  factuality: {
396
443
  family: GRADED,
397
444
  label: "Factual",
398
- description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
445
+ description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
399
446
  graded: true,
400
- options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
447
+ parsedOnly: true,
448
+ options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
401
449
  defaults: () => ({ reference: "", threshold: 0.5 }),
402
- // A blank reference is production's reply.
403
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
450
+ // A blank reference is the recorded reply.
451
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
404
452
  `You are checking an output for factual consistency with a reference. Differences in wording or detail are `
405
- + `fine; a contradiction, or a fact the reference does not support, is not.\n\nReference:\n`
406
- + `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
453
+ + `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
454
+ + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
407
455
  },
408
456
  "judge-vs-production": {
409
457
  family: GRADED,
410
- label: "Judged against production",
411
- description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
458
+ label: "Judged against recorded",
459
+ description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
412
460
  graded: true,
413
- options: [{ key: "task", label: "Task", type: "textarea" }],
414
- defaults: () => ({ task: "" }),
461
+ parsedOnly: true,
462
+ expects: true,
463
+ options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
464
+ { key: "task", label: "Additional guidance (optional)", type: "textarea" }],
465
+ defaults: () => ({ expected: ["better", "same"], task: "" }),
466
+ validate: (m, at, bad) => {
467
+ const known = JUDGEMENTS.map((j) => j.value);
468
+ if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
469
+ || m.expected.some((v) => !known.includes(String(v))))) {
470
+ bad.push(`${at}: Expected needs at least one of ${said(known)}`);
471
+ }
472
+ },
415
473
  score: async (input, m, ctx) => {
416
- const v = verdictOf(await ask(ctx,
474
+ if (ctx.recordedTarget) return null;
475
+ const v = await graded(ctx,
417
476
  `Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
418
- + `PRODUCTION one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nPRODUCTION:\n${production(input)}\n\n`
477
+ + `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
419
478
  + `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
420
- + `"reason": "one sentence"}.`));
421
- const verdict = str(v.verdict).toLowerCase();
422
- if (!["better", "same", "worse"].includes(verdict)) throw new Error(`the grader said ${short(JSON.stringify(v))}`);
423
- return { pass: verdict !== "worse", score: verdict === "better" ? 1 : verdict === "same" ? 0.5 : 0,
424
- reason: `${verdict}: ${str(v.reason)}` };
479
+ + `"reason": "one sentence"}.`,
480
+ { type: "object", additionalProperties: false, required: ["verdict", "reason"],
481
+ properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
482
+ (r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
483
+ const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
484
+ return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
485
+ reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
425
486
  },
426
487
  },
427
488
  };
package/lab/run-evals.js CHANGED
@@ -550,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
550
550
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
551
551
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
552
552
  }
553
+ // The Recorded target replays each call's recorded reply, read as the flow
554
+ // reads one (like send, but sending nothing); it does not go through the
555
+ // model Read-as wrapper below -- the recorded body is read directly.
556
+ if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
557
+ return () => {
558
+ const r = record ?? textRecord(text);
559
+ const result = r && r.result;
560
+ if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
561
+ try {
562
+ const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
563
+ { scope: r.scope || {}, now: r.at ?? null });
564
+ return { raw: read.raw, said: read.said, ms: 0, conn: null };
565
+ } catch (e) {
566
+ return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
567
+ }
568
+ };
569
+ }
553
570
  const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
554
- ? async sent => core.localAnswer(c, text, sent)
571
+ ? async sent => core.localAnswer(c, text, sent, record)
555
572
  : (sent, url) => ask(links[k], c, sent, url);
556
573
  if (!step?.readAs || !step?.step) return call;
557
574
  return async (sent, url) => {
@@ -588,8 +605,12 @@ function graderFor(run) {
588
605
  if (!c) return undefined;
589
606
  if (!made.has(ref.id)) {
590
607
  const link = reach(c, core.keyVar(ref.id, c.slug));
591
- made.set(ref.id, async prompt => {
592
- const r = await ask(link, { id: ref.id, ...c }, prompt, null);
608
+ made.set(ref.id, async (prompt, schema) => {
609
+ // Held to the metric's schema where the grader's API can hold a
610
+ // reply to one; a model or server that refuses the schema is asked
611
+ // again without it, the prompt alone saying the shape.
612
+ let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
613
+ if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
593
614
  if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
594
615
  return r.raw ?? "";
595
616
  });
@@ -643,13 +664,13 @@ async function send(to, c, step, cell, record) {
643
664
  }
644
665
  }
645
666
 
646
- async function ask(to, c, prompt, dataUrl) {
667
+ async function ask(to, c, prompt, dataUrl, schema = null) {
647
668
  const t0 = Date.now();
648
669
  const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
649
670
  try {
650
671
  // The connection's type builds the whole request: the URL, the headers
651
672
  // (Anthropic's key travels in x-api-key), and the body.
652
- const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
673
+ const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
653
674
  const r = await fetch(req.url, {
654
675
  method: "POST",
655
676
  headers: { "Content-Type": "application/json", ...req.headers },
@@ -883,7 +904,10 @@ async function runSnapshot(o) {
883
904
  const plans = core.targetsOf(run).map((_, i) => {
884
905
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
885
906
  const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
886
- return { stages, tokens, connections, links, calls, cells };
907
+ // The Recorded target replays the recorded reply, so a metric that compares
908
+ // with the recorded reply has nothing to say of it (n/a).
909
+ const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
910
+ return { stages, tokens, connections, links, calls, cells, recordedTarget };
887
911
  });
888
912
 
889
913
  let results = [];
@@ -945,9 +969,9 @@ async function runSnapshot(o) {
945
969
  // A case is found by its file's name, whatever kind of file it is: a
946
970
  // text item a dataset grades is graded like an image.
947
971
  const kase = item.name ? graded.get(item.name) ?? null : null;
948
- production = core.productionOf(run, record);
972
+ production = core.recordedReplyOf(run, record);
949
973
  sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
950
- { production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
974
+ { production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
951
975
  }
952
976
  // Production's reply to the item, where its record holds one: what a
953
977
  // metric compares with, kept so a re-score reads it again.