evals-lab 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -120,10 +120,10 @@ function holds(input , m , value , except
120
120
  /** A requirement as Results names it: a term, or the group it was. */
121
121
  const spoken = (g ) => (g.length === 1 ? g[0] : g);
122
122
 
123
- /** Production's reply, or the reason a metric that compares with it cannot. */
124
- const production = (input ) => {
125
- if (input.production == null) throw new Error("this item has no production reply to compare with");
126
- return input.production;
123
+ /** The recorded reply, or the reason a metric that compares with it cannot. */
124
+ const recorded = (input ) => {
125
+ if (input.recorded == null) throw new Error("this item has no recorded reply to compare with");
126
+ return input.recorded;
127
127
  };
128
128
 
129
129
  // ---- model-graded -----------------------------------------------------------
@@ -147,7 +147,7 @@ const gradedPass = (v , threshold )
147
147
  const REPLY = 'Reply with one JSON object and nothing else: {"pass": true or false, "score": a number from 0 to 1, "reason": "one sentence"}.';
148
148
 
149
149
  // The families the Add metric picker lists the metrics under: what each reads.
150
- const TEXT = "The reply's text", RESULT = "The result", PROD = "Production's reply", GRADED = "Model-graded";
150
+ const TEXT = "The reply's text", RESULT = "The result", RECORDED = "Recorded reply", GRADED = "Model-graded";
151
151
 
152
152
  const metrics = {
153
153
  equals: {
@@ -337,46 +337,51 @@ const metrics = {
337
337
  1 - d / Math.max(1, Math.max(got.length, want.length)));
338
338
  },
339
339
  },
340
- // ---- against production ----------------------------------------------------
341
- "equals-production": {
342
- family: PROD,
343
- label: "Same as production",
344
- description: "Passes when the reply is what production replied to the same item.",
340
+ // ---- against the recorded reply --------------------------------------------
341
+ // n/a (null) when the target under test is the Recorded target itself: a
342
+ // reply compared with the reply it is says nothing (docs/workflow-sources.md).
343
+ "same-as-recorded": {
344
+ family: RECORDED,
345
+ label: "Same as recorded",
346
+ description: "Passes when the reply is what production recorded for the same item.",
345
347
  options: [],
346
348
  defaults: () => ({}),
347
- score: (input) => {
348
- const p = production(input).trim(), got = input.text.trim();
349
+ score: (input, _m, ctx) => {
350
+ if (ctx.recordedTarget) return null;
351
+ const p = recorded(input).trim(), got = input.text.trim();
349
352
  const a = json(got), b = json(p);
350
353
  const equal = !a.error && !b.error ? same(a.value, b.value) : got === p;
351
- return ok(equal, equal ? "as production replied" : `production replied ${short(p)}`);
354
+ return ok(equal, equal ? "as recorded" : `recorded ${short(p)}`);
352
355
  },
353
356
  },
354
- "fields-equal-production": {
355
- family: PROD,
356
- label: "Fields as production",
357
- description: "Passes when the fields listed are what production's reply had.",
357
+ "fields-equal-recorded": {
358
+ family: RECORDED,
359
+ label: "Fields as recorded",
360
+ description: "Passes when the fields listed are what the recorded reply had.",
358
361
  options: [{ key: "fields", label: "Fields", type: "textarea" }],
359
362
  defaults: () => ({ fields: "" }),
360
363
  validate: (m, at, bad) => needs(m, ["fields"], at, bad),
361
- score: (input, m) => {
362
- const a = json(input.text), b = json(production(input));
364
+ score: (input, m, ctx) => {
365
+ if (ctx.recordedTarget) return null;
366
+ const a = json(input.text), b = json(recorded(input));
363
367
  if (a.error) return ok(false, `not JSON: ${a.error}`);
364
- if (b.error) return ok(false, "production's reply was not JSON");
368
+ if (b.error) return ok(false, "the recorded reply was not JSON");
365
369
  const fields = lines(m.fields);
366
370
  const differ = fields.filter((f) => !same(atPath(a.value, f), atPath(b.value, f)));
367
- return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from production` : "the fields match production",
371
+ return ok(!differ.length, differ.length ? `${differ.join(", ")} differ from recorded` : "the fields match recorded",
368
372
  fields.length ? (fields.length - differ.length) / fields.length : 1);
369
373
  },
370
374
  },
371
- "same-parse-outcome": {
372
- family: PROD,
373
- label: "Parses as production",
374
- description: "Passes when the reply parses as JSON exactly when production's did.",
375
+ "same-parse-as-recorded": {
376
+ family: RECORDED,
377
+ label: "Parses as recorded",
378
+ description: "Passes when the reply parses as JSON exactly when the recorded reply did.",
375
379
  options: [],
376
380
  defaults: () => ({}),
377
- score: (input) => {
378
- const mine = !json(input.text).error, theirs = !json(production(input)).error;
379
- return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, production's ${theirs ? "parsed" : "did not"}`);
381
+ score: (input, _m, ctx) => {
382
+ if (ctx.recordedTarget) return null;
383
+ const mine = !json(input.text).error, theirs = !json(recorded(input)).error;
384
+ return ok(mine === theirs, `${mine ? "parses" : "does not parse"}, recorded ${theirs ? "parsed" : "did not"}`);
380
385
  },
381
386
  },
382
387
  // ---- model-graded -------------------------------------------------------------
@@ -395,27 +400,27 @@ const metrics = {
395
400
  factuality: {
396
401
  family: GRADED,
397
402
  label: "Factual",
398
- description: "Asks the grader whether the reply agrees with the reference, or production's reply.",
403
+ description: "Asks the grader whether the reply agrees with the reference, or the recorded reply.",
399
404
  graded: true,
400
405
  options: [{ key: "reference", label: "Reference", type: "textarea" }, { key: "threshold", label: "Pass at", type: "number" }],
401
406
  defaults: () => ({ reference: "", threshold: 0.5 }),
402
- // A blank reference is production's reply.
407
+ // A blank reference is the recorded reply.
403
408
  score: async (input, m, ctx) => gradedPass(verdictOf(await ask(ctx,
404
409
  `You are checking an output for factual consistency with a reference. Differences in wording or detail are `
405
410
  + `fine; a contradiction, or a fact the reference does not support, is not.\n\nReference:\n`
406
- + `${str(m.reference).trim() || production(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
411
+ + `${str(m.reference).trim() || recorded(input)}\n\nOutput:\n${input.text}\n\n${REPLY}`)), num(m.threshold) ?? 0.5),
407
412
  },
408
413
  "judge-vs-production": {
409
414
  family: GRADED,
410
- label: "Judged against production",
411
- description: "Asks the grader whether the reply is better than, the same as or worse than production's.",
415
+ label: "Judged against recorded",
416
+ description: "Asks the grader whether the reply is better than, the same as or worse than the recorded one.",
412
417
  graded: true,
413
418
  options: [{ key: "task", label: "Task", type: "textarea" }],
414
419
  defaults: () => ({ task: "" }),
415
420
  score: async (input, m, ctx) => {
416
421
  const v = verdictOf(await ask(ctx,
417
422
  `Two outputs answer the same task. Say whether the NEW one is better than, the same as, or worse than the `
418
- + `PRODUCTION one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nPRODUCTION:\n${production(input)}\n\n`
423
+ + `RECORDED one.${str(m.task).trim() ? `\n\nTask:\n${str(m.task)}` : ""}\n\nRECORDED:\n${recorded(input)}\n\n`
419
424
  + `NEW:\n${input.text}\n\nReply with one JSON object and nothing else: {"verdict": "better" or "same" or "worse", `
420
425
  + `"reason": "one sentence"}.`));
421
426
  const verdict = str(v.verdict).toLowerCase();
package/lab/run-evals.js CHANGED
@@ -550,8 +550,25 @@ function callsFor(connections, links, text, plan, record) {
550
550
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
551
551
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
552
552
  }
553
+ // The Recorded target replays each call's recorded reply, read as the flow
554
+ // reads one (like send, but sending nothing); it does not go through the
555
+ // model Read-as wrapper below -- the recorded body is read directly.
556
+ if (core.CONNECTION_TYPES[core.typeOf(c)]?.replays) {
557
+ return () => {
558
+ const r = record ?? textRecord(text);
559
+ const result = r && r.result;
560
+ if (!result || typeof result.status !== "number") return { raw: "", said: "", ms: 0, conn: null };
561
+ try {
562
+ const read = core.httpReplyOf(step, plan.cells[k], result.body, result.status,
563
+ { scope: r.scope || {}, now: r.at ?? null });
564
+ return { raw: read.raw, said: read.said, ms: 0, conn: null };
565
+ } catch (e) {
566
+ return { error: `the recorded reply could not be read as the flow reads it -- ${e.message}`, ms: 0, conn: null };
567
+ }
568
+ };
569
+ }
553
570
  const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
554
- ? async sent => core.localAnswer(c, text, sent)
571
+ ? async sent => core.localAnswer(c, text, sent, record)
555
572
  : (sent, url) => ask(links[k], c, sent, url);
556
573
  if (!step?.readAs || !step?.step) return call;
557
574
  return async (sent, url) => {
@@ -883,7 +900,10 @@ async function runSnapshot(o) {
883
900
  const plans = core.targetsOf(run).map((_, i) => {
884
901
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
885
902
  const links = connections.map(c => reach(c, core.keyVar(c.id, c.slug)));
886
- return { stages, tokens, connections, links, calls, cells };
903
+ // The Recorded target replays the recorded reply, so a metric that compares
904
+ // with the recorded reply has nothing to say of it (n/a).
905
+ const recordedTarget = connections.some(c => core.CONNECTION_TYPES[core.typeOf(c)]?.replays);
906
+ return { stages, tokens, connections, links, calls, cells, recordedTarget };
887
907
  });
888
908
 
889
909
  let results = [];
@@ -945,9 +965,9 @@ async function runSnapshot(o) {
945
965
  // A case is found by its file's name, whatever kind of file it is: a
946
966
  // text item a dataset grades is graded like an image.
947
967
  const kase = item.name ? graded.get(item.name) ?? null : null;
948
- production = core.productionOf(run, record);
968
+ production = core.recordedReplyOf(run, record);
949
969
  sides.push({ res, scores: await core.itemScoresAsync(run, kase, res,
950
- { production, grader: graderFor(run), group: gctx.resolve, item: item.name }) });
970
+ { production, grader: graderFor(run), group: gctx.resolve, item: item.name, recordedTarget: plan.recordedTarget }) });
951
971
  }
952
972
  // Production's reply to the item, where its record holds one: what a
953
973
  // metric compares with, kept so a re-score reads it again.