evals-lab 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -135,16 +135,54 @@ function verdictOf(said )
135
135
  if (read.error || !read.value || typeof read.value !== "object") throw new Error(`the grader's reply could not be read: ${short(said)}`);
136
136
  return read.value ;
137
137
  }
138
- const ask = async (ctx , prompt ) => {
138
+ /** The grader's verdict on [prompt], its reply held to [schema] where its API
139
+ can hold one. A reply that still cannot be read is asked for once more;
140
+ a second, or a grader that failed, is the grader's error -- said as one,
141
+ so it is not read as the reply failing. [ok] says whether a verdict read
142
+ is one this metric can use. */
143
+ async function graded(ctx , prompt , schema ,
144
+ ok = () => true) {
139
145
  if (!ctx.ask) throw new Error("a model-graded metric needs a grader, and this run has none");
140
- return ctx.ask(prompt);
141
- };
146
+ let last = "";
147
+ for (let attempt = 0; attempt < 2; attempt++) {
148
+ let said ;
149
+ try { said = await ctx.ask(prompt, schema); } catch (e) { throw new Error(`grader error: ${e instanceof Error ? e.message : String(e)}`); }
150
+ try {
151
+ const v = verdictOf(said);
152
+ if (ok(v)) return v;
153
+ last = `the grader said ${short(JSON.stringify(v))}`;
154
+ } catch (e) { last = e instanceof Error ? e.message : String(e); }
155
+ }
156
+ throw new Error(`grader error: ${last}`);
157
+ }
142
158
  const gradedPass = (v , threshold ) => {
143
159
  const score = typeof v.score === "number" ? Math.max(0, Math.min(1, v.score)) : v.pass === true ? 1 : 0;
144
160
  const pass = typeof v.pass === "boolean" ? v.pass && score >= threshold : score >= threshold;
145
161
  return { pass, score, reason: str(v.reason) || (pass ? "the grader passed it" : "the grader failed it") };
146
162
  };
163
+ /** The shape a pass-or-fail grader answers in, for its API to hold it to. */
164
+ const VERDICT_SCHEMA = { type: "object", additionalProperties: false, required: ["pass", "score", "reason"],
165
+ properties: { pass: { type: "boolean" }, score: { type: "number" }, reason: { type: "string" } } };
147
166
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
167
+ /** What the grader is told of the task: what the target was asked, and the
168
+ guidance the metric adds, where there is either. */
169
+ const taskOf = (input , guidance ) =>
170
+ (input.asked?.trim() ? `The task, as it was asked:\n${input.asked.trim()}\n\n` : "")
171
+ + (str(guidance).trim() ? `Guidance:\n${str(guidance).trim()}\n\n` : "");
172
+
173
+ /** A grader's three verdicts on a reply set beside the recorded one; any of
174
+ them may be what a metric expects. */
175
+ const JUDGEMENTS = [{ value: "better", label: "Better" }, { value: "same", label: "Same" }, { value: "worse", label: "Worse" }];
176
+ const JUDGEMENT_SCORE = { better: 1, same: 0.5, worse: 0 };
177
+ /** What a judgement metric expects: the verdicts set, or -- set before
178
+ Expected was offered -- the one rule there was, worse fails. */
179
+ const expectedOf = (m ) =>
180
+ (Array.isArray(m.expected) ? m.expected.map(String) : ["better", "same"]);
181
+ /** The expected verdicts as a reader says them: "better or same". */
182
+ const said = (values ) => {
183
+ const words = values.map((v) => v.toLowerCase());
184
+ return words.length < 2 ? words.join("") : `${words.slice(0, -1).join(", ")} or ${words.at(-1)}`;
185
+ };
148
186
 
149
187
  // The families the Add metric picker lists the metrics under: what each reads.
150
188
  const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
@@ -342,6 +380,7 @@ const metrics = {
342
380
  // reply compared with the reply it is says nothing (docs/workflow-sources.md).
343
381
  "same-as-recorded": {
344
382
  family: RECORDED,
383
+ parsedOnly: true,
345
384
  label: "Same as recorded",
346
385
  description: "Passes when the reply is what production recorded for the same item.",
347
386
  options: [],
@@ -356,6 +395,7 @@ const metrics = {
356
395
  },
357
396
  "fields-equal-recorded": {
358
397
  family: RECORDED,
398
+ parsedOnly: true,
359
399
  label: "Fields as recorded",
360
400
  description: "Passes when the fields listed are what the recorded reply had.",
361
401
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
@@ -374,6 +414,7 @@ const metrics = {
374
414
  },
375
415
  "same-parse-as-recorded": {
376
416
  family: RECORDED,
417
+ parsedOnly: true,
377
418
  label: "Parses as recorded",
378
419
  description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
379
420
  options: [],
@@ -390,43 +431,58 @@ const metrics = {
390
431
  label: "Rubric",
391
432
  description: "Asks the grader whether the reply meets the rubric.",
392
433
  graded: true,
393
- options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
434
+ parsedOnly: true,
435
+ options: [{ key: "rubric", label: "Rubric", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
394
436
  defaults: () => ({ rubric: "", threshold: 0.5 }),
395
437
  validate: (m, at, bad) => needs(m, ["rubric"], at, bad),
396
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
397
- `You are grading an output against a rubric.\n\nRubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`)),
398
- num(m.threshold) ?? 0.5),
438
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
439
+ `You are grading an output against a rubric.\n\n${taskOf(input, "")}Rubric:\n${str(m.rubric)}\n\nOutput:\n${input.text}\n\n${REPLY}`,
440
+ VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
399
441
  },
400
442
  factuality: {
401
443
  family: GRADED,
402
444
  label: "Factual",
403
445
  description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
404
446
  graded: true,
405
- options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
447
+ parsedOnly: true,
448
+ options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Expected: score at least", type: "number" }],
406
449
  defaults: () => ({ reference: "", threshold: 0.5 }),
407
450
  // A blank reference is the recorded reply.
408
- score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
451
+ score: async (input, m, ctx) => gradedPass(await graded(ctx,
409
452
  `You are checking an output for factual consistency with a reference. Differences in wording or detail are `
410
- + `fine; a contradiction, or a fact the reference does not support, is not.\n\nReference:\n`
411
- + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
453
+ + `fine; a contradiction, or a fact the reference does not support, is not.\n\n${taskOf(input, "")}Reference:\n`
454
+ + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`, VERDICT_SCHEMA), num(m.threshold) ?? 0.5),
412
455
  },
413
456
  "judge-vs-production": {
414
457
  family: GRADED,
415
458
  label: "Judged against recorded",
416
459
  description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
417
460
  graded: true,
418
- options: [{ key: "task", label: "Task", type: "textarea" }],
419
- defaults: () => ({ task: "" }),
461
+ parsedOnly: true,
462
+ expects: true,
463
+ options: [{ key: "expected", label: "Expected", type: "multi", choices: JUDGEMENTS },
464
+ { key: "task", label: "Additional guidance (optional)", type: "textarea" }],
465
+ defaults: () => ({ expected: ["better", "same"], task: "" }),
466
+ validate: (m, at, bad) => {
467
+ const known = JUDGEMENTS.map((j) => j.value);
468
+ if (m.expected !== undefined && (!Array.isArray(m.expected) || !m.expected.length
469
+ || m.expected.some((v) => !known.includes(String(v))))) {
470
+ bad.push(`${at}: Expected needs at least one of ${said(known)}`);
471
+ }
472
+ },
420
473
  score: async (input, m, ctx) => {
421
- const v = verdictOf(await ask(ctx,
474
+ if (ctx.recordedTarget) return null;
475
+ const v = await graded(ctx,
422
476
  `Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
423
- + `RECORDED one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nRECORDED:\n${recorded(input)}\n\n`
477
+ + `RECORDED one.\n\n${taskOf(input, m.task)}RECORDED:\n${recorded(input)}\n\n`
424
478
  + `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
425
- + `"reason": "one sentence"}.`));
426
- const verdict = str(v.verdict).toLowerCase();
427
- if (!["better", "same", "worse"].includes(verdict)) throw new Error(`the grader said ${short(JSON.stringify(v))}`);
428
- return { pass: verdict !== "worse", score: verdict === "better" ? 1 : verdict === "same" ? 0.5 : 0,
429
- reason: `${verdict}: ${str(v.reason)}` };
479
+ + `"reason": "one sentence"}.`,
480
+ { type: "object", additionalProperties: false, required: ["verdict", "reason"],
481
+ properties: { verdict: { type: "string", enum: JUDGEMENTS.map((j) => j.value) }, reason: { type: "string" } } },
482
+ (r) => str(r.verdict).toLowerCase() in JUDGEMENT_SCORE);
483
+ const verdict = str(v.verdict).toLowerCase(), expected = expectedOf(m);
484
+ return { pass: expected.includes(verdict), score: JUDGEMENT_SCORE[verdict] ,
485
+ reason: `${verdict} (expected ${said(expected)}): ${str(v.reason)}` };
430
486
  },
431
487
  },
432
488
  };
package/lab/run-evals.js CHANGED
@@ -605,8 +605,12 @@ function graderFor(run) {
605
605
  if (!c) return undefined;
606
606
  if (!made.has(ref.id)) {
607
607
  const link = reach(c, core.keyVar(ref.id, c.slug));
608
- made.set(ref.id, async prompt => {
609
- const r = await ask(link, { id: ref.id, ...c }, prompt, null);
608
+ made.set(ref.id, async (prompt, schema) => {
609
+ // Held to the metric's schema where the grader's API can hold a
610
+ // reply to one; a model or server that refuses the schema is asked
611
+ // again without it, the prompt alone saying the shape.
612
+ let r = await ask(link, { id: ref.id, ...c }, prompt, null, schema);
613
+ if (r.error && schema) r = await ask(link, { id: ref.id, ...c }, prompt, null);
610
614
  if (r.error) throw new Error(`the grader ${c.name || ref.id} failed: ${r.error}`);
611
615
  return r.raw ?? "";
612
616
  });
@@ -660,13 +664,13 @@ async function send(to, c, step, cell, record) {
660
664
  }
661
665
  }
662
666
 
663
- async function ask(to, c, prompt, dataUrl) {
667
+ async function ask(to, c, prompt, dataUrl, schema = null) {
664
668
  const t0 = Date.now();
665
669
  const said = more => ({ ms: Date.now() - t0, conn: to.name, ...more });
666
670
  try {
667
671
  // The connection's type builds the whole request: the URL, the headers
668
672
  // (Anthropic's key travels in x-api-key), and the body.
669
- const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key);
673
+ const req = core.connectionRequest(c, prompt, dataUrl, to.base, to.key, schema);
670
674
  const r = await fetch(req.url, {
671
675
  method: "POST",
672
676
  headers: { "Content-Type": "application/json", ...req.headers },