evals-lab 0.1.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -453,21 +453,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
453
453
  // or, over Prompt only, the prompt -- and the stage reads it as it would a
454
454
  // model's.
455
455
  //
456
- // A call that asks for a whole request (an HTTP Request, #flows) builds it
457
- // from its step, the scenario's cell and the item's record, and is handed
458
- // only its words: the transport is where the three meet.
456
+ // A target step that asks for a whole request (an HTTP Request, #flows)
457
+ // builds it from the step, the job's flow step and the item's record, and is
458
+ // handed only its words: the transport is where the three meet.
459
+ //
460
+ // Any other step's reply, in a job that reads the flow's step through a Read
461
+ // as, is read the same way: put in the body the flow's API would have sent
462
+ // it in, then read as the flow reads it -- so a model asked in words and
463
+ // production are compared alike (docs/pipeline-model.md §16 › Targets).
459
464
  function callsFor(connections, links, text, plan, record) {
460
465
  return connections.map((c, k) => {
461
466
  const step = plan?.calls?.[k];
462
467
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
463
468
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
464
469
  }
465
- return core.CONNECTION_TYPES[core.typeOf(c)]?.local
470
+ const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
466
471
  ? async sent => core.localAnswer(c, text, sent)
467
472
  : (sent, url) => ask(links[k], c, sent, url);
473
+ if (!step?.readAs || !step?.step) return call;
474
+ return async (sent, url) => {
475
+ const r = await call(sent, url);
476
+ if (r.error || r.raw == null) return r;
477
+ try {
478
+ return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
479
+ } catch (e) {
480
+ return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
481
+ }
482
+ };
468
483
  });
469
484
  }
470
485
 
486
+ // What an expression token reads: the item's record, or a text item read as
487
+ // one, and the loop job 1's flow step sits in.
488
+ const tokenRecord = (run, record, text) => {
489
+ const r = record ?? (text != null ? textRecord(text) : null);
490
+ return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
491
+ };
492
+
493
+ // A prompt as a report restates it: an expression token reads an item's
494
+ // record, and a report has none to name, so its words stay as written.
495
+ const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
496
+
471
497
  /** A Setup profile the run carries, asked as a grader: its reply's text, or
472
498
  the failure as an error the metric reports. One transport per profile. */
473
499
  const graders = new WeakMap();
@@ -590,7 +616,7 @@ function readUtf8(file) {
590
616
  // the page reads it: a whole-run test's verdict, settled once every item is
591
617
  // in, and a per-item test's counts. [settled] is false for a run that
592
618
  // stopped short, whose whole-run tests have not settled.
593
- const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
619
+ const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
594
620
  Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
595
621
  name: o.label, skipped: o.skipped,
596
622
  ...(o.whole
@@ -600,7 +626,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
600
626
 
601
627
  // The run as the report restates it: each scenario's stages with the
602
628
  // connection each one asked.
603
- const scenariosOf = run => run.scenarios.map((sc, i) => {
629
+ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
604
630
  const { stages, connections } = core.stagesFor(run, i);
605
631
  return {
606
632
  n: i + 1, name: sc.name || null,
@@ -649,7 +675,7 @@ async function runSnapshot(o) {
649
675
  // Each scenario once: its stages, the jobs' token sets, and a transport
650
676
  // per stage to the profile that stage resolves to, keyed by that
651
677
  // profile's id.
652
- const plans = run.scenarios.map((_, i) => {
678
+ const plans = core.targetsOf(run).map((_, i) => {
653
679
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
654
680
  const links = connections.map(c => reach(c, core.keyVar(c.id)));
655
681
  return { stages, tokens, connections, links, calls, cells };
@@ -704,7 +730,7 @@ async function runSnapshot(o) {
704
730
  const textOf = item.kind === "text" && !item.bare
705
731
  ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
706
732
  const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
707
- { tokens: plan.tokens, text: textOf });
733
+ { tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
708
734
  // A case is found by its file's name, whatever kind of file it is: a
709
735
  // text item a dataset grades is graded like an image.
710
736
  const kase = item.name ? graded.get(item.name) ?? null : null;
@@ -822,14 +848,14 @@ function cliRun(o, dataset) {
822
848
  doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
823
849
  if (o.tokens) {
824
850
  try {
825
- doc.jobs[0] = core.withCall(doc.jobs[0], { tokenMappings: JSON.parse(fs.readFileSync(o.tokens, "utf8")) });
851
+ doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
826
852
  } catch (e) {
827
853
  broken(`${o.tokens}: ${e.message}`);
828
854
  }
829
855
  }
830
856
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
831
- doc.scenarios = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
832
- stages: [{ prompt: o.prompt ?? "" }] }];
857
+ doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
858
+ steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
833
859
  doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
834
860
  mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
835
861
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
@@ -881,7 +907,7 @@ async function main() {
881
907
  }
882
908
  if (o.run) {
883
909
  if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
884
- broken("--run names everything a run takes -- scenarios, files and text -- so it "
910
+ broken("--run names everything a run takes -- targets, files and text -- so it "
885
911
  + "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
886
912
  }
887
913
  const n = Number(o.timeout ?? REQUEST_CAP);
@@ -920,8 +946,8 @@ async function main() {
920
946
  }
921
947
  const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
922
948
  if (o.pipeline) {
923
- if (run.scenarios.length !== 1) {
924
- broken(`${o.pipeline} has ${run.scenarios.length} scenarios, and --pipeline grades one `
949
+ if (core.targetsOf(run).length !== 1) {
950
+ broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
925
951
  + "against the set -- --run runs them all.");
926
952
  }
927
953
  if (!core.testsDataset(run)) {
@@ -934,7 +960,7 @@ async function main() {
934
960
  // Case by case: each item against its case, as the case metric of the
935
961
  // test that names the dataset reads it (scoreCase).
936
962
  const { stages, tokens, connections } = core.stagesFor(run, 0);
937
- const prompt = core.resolvePrompt(stages[0].text, core.tokenSet(tokens, 0));
963
+ const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
938
964
 
939
965
  // The graded half, through the function the tab builds its own list with.
940
966
  const set = core.gradedSetFrom(dataset);
@@ -1170,7 +1196,7 @@ async function main() {
1170
1196
  stages: stages.map((s, i) => {
1171
1197
  const { id, ...connection } = connections[i];
1172
1198
  return { n: i + 1, kind: s.kind, withImage: s.withImage,
1173
- prompt: core.resolvePrompt(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1199
+ prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1174
1200
  }),
1175
1201
  } } : {}),
1176
1202
  // Which variable a key came from, never the key.