evals-lab 0.1.3 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -1
- package/bin/evals-lab.js +10 -2
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +34 -24
- package/lab/demo/pipelines/demo-2.json +34 -24
- package/lab/evals-core.mjs +701 -284
- package/lab/run-evals.js +42 -16
- package/lab/server.py +339 -80
- package/lab/web/dist/assets/gallery-DsetJSXv.js +3 -0
- package/lab/web/dist/assets/{main-DjQQums6.css → main-B-VtDGxC.css} +1 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +19 -0
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +51 -0
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-HJGRUBM-.js +0 -18
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/run-evals.js
CHANGED
|
@@ -453,21 +453,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
|
|
|
453
453
|
// or, over Prompt only, the prompt -- and the stage reads it as it would a
|
|
454
454
|
// model's.
|
|
455
455
|
//
|
|
456
|
-
// A
|
|
457
|
-
// from
|
|
458
|
-
// only its words: the transport is where the three meet.
|
|
456
|
+
// A target step that asks for a whole request (an HTTP Request, #flows)
|
|
457
|
+
// builds it from the step, the job's flow step and the item's record, and is
|
|
458
|
+
// handed only its words: the transport is where the three meet.
|
|
459
|
+
//
|
|
460
|
+
// Any other step's reply, in a job that reads the flow's step through a Read
|
|
461
|
+
// as, is read the same way: put in the body the flow's API would have sent
|
|
462
|
+
// it in, then read as the flow reads it -- so a model asked in words and
|
|
463
|
+
// production are compared alike (docs/pipeline-model.md §16 › Targets).
|
|
459
464
|
function callsFor(connections, links, text, plan, record) {
|
|
460
465
|
return connections.map((c, k) => {
|
|
461
466
|
const step = plan?.calls?.[k];
|
|
462
467
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
463
468
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
464
469
|
}
|
|
465
|
-
|
|
470
|
+
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
466
471
|
? async sent => core.localAnswer(c, text, sent)
|
|
467
472
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
473
|
+
if (!step?.readAs || !step?.step) return call;
|
|
474
|
+
return async (sent, url) => {
|
|
475
|
+
const r = await call(sent, url);
|
|
476
|
+
if (r.error || r.raw == null) return r;
|
|
477
|
+
try {
|
|
478
|
+
return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
|
|
479
|
+
} catch (e) {
|
|
480
|
+
return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
|
|
481
|
+
}
|
|
482
|
+
};
|
|
468
483
|
});
|
|
469
484
|
}
|
|
470
485
|
|
|
486
|
+
// What an expression token reads: the item's record, or a text item read as
|
|
487
|
+
// one, and the loop job 1's flow step sits in.
|
|
488
|
+
const tokenRecord = (run, record, text) => {
|
|
489
|
+
const r = record ?? (text != null ? textRecord(text) : null);
|
|
490
|
+
return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
|
|
491
|
+
};
|
|
492
|
+
|
|
493
|
+
// A prompt as a report restates it: an expression token reads an item's
|
|
494
|
+
// record, and a report has none to name, so its words stay as written.
|
|
495
|
+
const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
|
|
496
|
+
|
|
471
497
|
/** A Setup profile the run carries, asked as a grader: its reply's text, or
|
|
472
498
|
the failure as an error the metric reports. One transport per profile. */
|
|
473
499
|
const graders = new WeakMap();
|
|
@@ -590,7 +616,7 @@ function readUtf8(file) {
|
|
|
590
616
|
// the page reads it: a whole-run test's verdict, settled once every item is
|
|
591
617
|
// in, and a per-item test's counts. [settled] is false for a run that
|
|
592
618
|
// stopped short, whose whole-run tests have not settled.
|
|
593
|
-
const verdictsOf = (run, items, settled) => run.
|
|
619
|
+
const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
|
|
594
620
|
Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
|
|
595
621
|
name: o.label, skipped: o.skipped,
|
|
596
622
|
...(o.whole
|
|
@@ -600,7 +626,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
|
|
|
600
626
|
|
|
601
627
|
// The run as the report restates it: each scenario's stages with the
|
|
602
628
|
// connection each one asked.
|
|
603
|
-
const scenariosOf = run => run.
|
|
629
|
+
const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
604
630
|
const { stages, connections } = core.stagesFor(run, i);
|
|
605
631
|
return {
|
|
606
632
|
n: i + 1, name: sc.name || null,
|
|
@@ -649,7 +675,7 @@ async function runSnapshot(o) {
|
|
|
649
675
|
// Each scenario once: its stages, the jobs' token sets, and a transport
|
|
650
676
|
// per stage to the profile that stage resolves to, keyed by that
|
|
651
677
|
// profile's id.
|
|
652
|
-
const plans = run.
|
|
678
|
+
const plans = core.targetsOf(run).map((_, i) => {
|
|
653
679
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
654
680
|
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
655
681
|
return { stages, tokens, connections, links, calls, cells };
|
|
@@ -704,7 +730,7 @@ async function runSnapshot(o) {
|
|
|
704
730
|
const textOf = item.kind === "text" && !item.bare
|
|
705
731
|
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
706
732
|
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
707
|
-
{ tokens: plan.tokens, text: textOf });
|
|
733
|
+
{ tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
|
|
708
734
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
709
735
|
// text item a dataset grades is graded like an image.
|
|
710
736
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
@@ -822,14 +848,14 @@ function cliRun(o, dataset) {
|
|
|
822
848
|
doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
|
|
823
849
|
if (o.tokens) {
|
|
824
850
|
try {
|
|
825
|
-
doc.jobs[0] = core.
|
|
851
|
+
doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
|
|
826
852
|
} catch (e) {
|
|
827
853
|
broken(`${o.tokens}: ${e.message}`);
|
|
828
854
|
}
|
|
829
855
|
}
|
|
830
856
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
831
|
-
doc.
|
|
832
|
-
|
|
857
|
+
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
858
|
+
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
833
859
|
doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
|
|
834
860
|
mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
|
|
835
861
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
@@ -881,7 +907,7 @@ async function main() {
|
|
|
881
907
|
}
|
|
882
908
|
if (o.run) {
|
|
883
909
|
if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
|
|
884
|
-
broken("--run names everything a run takes --
|
|
910
|
+
broken("--run names everything a run takes -- targets, files and text -- so it "
|
|
885
911
|
+ "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
|
|
886
912
|
}
|
|
887
913
|
const n = Number(o.timeout ?? REQUEST_CAP);
|
|
@@ -920,8 +946,8 @@ async function main() {
|
|
|
920
946
|
}
|
|
921
947
|
const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
|
|
922
948
|
if (o.pipeline) {
|
|
923
|
-
if (run.
|
|
924
|
-
broken(`${o.pipeline} has ${run.
|
|
949
|
+
if (core.targetsOf(run).length !== 1) {
|
|
950
|
+
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
|
925
951
|
+ "against the set -- --run runs them all.");
|
|
926
952
|
}
|
|
927
953
|
if (!core.testsDataset(run)) {
|
|
@@ -934,7 +960,7 @@ async function main() {
|
|
|
934
960
|
// Case by case: each item against its case, as the case metric of the
|
|
935
961
|
// test that names the dataset reads it (scoreCase).
|
|
936
962
|
const { stages, tokens, connections } = core.stagesFor(run, 0);
|
|
937
|
-
const prompt =
|
|
963
|
+
const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
|
|
938
964
|
|
|
939
965
|
// The graded half, through the function the tab builds its own list with.
|
|
940
966
|
const set = core.gradedSetFrom(dataset);
|
|
@@ -1170,7 +1196,7 @@ async function main() {
|
|
|
1170
1196
|
stages: stages.map((s, i) => {
|
|
1171
1197
|
const { id, ...connection } = connections[i];
|
|
1172
1198
|
return { n: i + 1, kind: s.kind, withImage: s.withImage,
|
|
1173
|
-
prompt:
|
|
1199
|
+
prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
|
|
1174
1200
|
}),
|
|
1175
1201
|
} } : {}),
|
|
1176
1202
|
// Which variable a key came from, never the key.
|