evals-lab 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +17 -9
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +3 -7
- package/lab/demo/pipelines/demo-2.json +3 -7
- package/lab/evals-core.mjs +429 -434
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +35 -33
- package/lab/server.py +296 -82
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DsetJSXv.js +0 -3
- package/lab/web/dist/assets/main-B-VtDGxC.css +0 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +0 -19
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +0 -51
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
// WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
|
|
27
27
|
// script has not, so `prepare()` stays in the page and the script shells out
|
|
28
28
|
// to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
|
|
29
|
-
// the script writes JSON. Both read the same verdict out of
|
|
29
|
+
// the script writes JSON. Both read the same verdict out of the same Metrics.
|
|
30
30
|
//
|
|
31
31
|
// `runner-check.js` runs one set through two of the callers and asserts the
|
|
32
32
|
// verdicts and the totals are identical, because sharing a file is a claim
|
|
@@ -232,8 +232,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
232
232
|
|
|
233
233
|
|
|
234
234
|
|
|
235
|
-
/** What every
|
|
236
|
-
shown, an optional name ("
|
|
235
|
+
/** What every eval carries whatever its type: an id minted once and never
|
|
236
|
+
shown, an optional name ("Eval 2" when blank), and whether the evals after
|
|
237
237
|
it still read what it failed on (docs/pipeline-model.md §3). */
|
|
238
238
|
|
|
239
239
|
|
|
@@ -260,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
260
260
|
|
|
261
261
|
|
|
262
262
|
|
|
263
|
-
/** Checks of every reply (
|
|
263
|
+
/** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
|
|
264
264
|
a case's own for its item; all must pass, or weighted points reach the
|
|
265
265
|
threshold. A model-graded one asks the grader. */
|
|
266
266
|
|
|
@@ -275,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
275
275
|
|
|
276
276
|
|
|
277
277
|
|
|
278
|
-
/**
|
|
278
|
+
/** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
|
|
279
279
|
older document held (upgradePipeline reads it converted). */
|
|
280
280
|
|
|
281
281
|
|
|
282
|
-
/** A pipeline's
|
|
282
|
+
/** A pipeline's evals, in the order they read a run. */
|
|
283
283
|
|
|
284
284
|
|
|
285
285
|
|
|
@@ -482,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
482
482
|
|
|
483
483
|
// ---- grading ----
|
|
484
484
|
|
|
485
|
-
/** A graded case
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
485
|
+
/** A graded case (dataset body version 5): an Item and the Metrics its
|
|
486
|
+
reply is held to.
|
|
487
|
+
|
|
488
|
+
The item is named by `item`: a dataset joins on an item's name in the
|
|
489
|
+
Source it grades, exactly, and nothing about that is an image -- the
|
|
490
|
+
next dataset addresses a row of a spreadsheet or a line of a log by the
|
|
491
|
+
same key. Version 4 called it `filename`, and the lab's first app had a
|
|
492
|
+
name of its own for it; `upgradeDatasetBody` reads both as `item`.
|
|
493
|
+
|
|
494
|
+
A case's metrics are what a good answer is: each a metric as a
|
|
495
|
+
pipeline's Metrics eval holds one, added to the eval's own for this item
|
|
496
|
+
when an eval names the dataset. One with weight 0 is watched -- reported,
|
|
497
|
+
never scored. */
|
|
496
498
|
|
|
497
499
|
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
500
|
+
|
|
501
|
+
|
|
501
502
|
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
503
|
+
|
|
504
|
+
|
|
512
505
|
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
506
|
|
|
518
507
|
|
|
519
508
|
|
|
520
|
-
/** A graded set: cases.
|
|
509
|
+
/** A graded set: a dataset's body, or anything holding its cases. */
|
|
521
510
|
|
|
522
511
|
|
|
523
512
|
|
|
524
513
|
|
|
525
514
|
|
|
526
|
-
/** One row the
|
|
527
|
-
a case of a set validateEvals accepts, so it has its id and
|
|
515
|
+
/** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
|
|
516
|
+
a case of a set validateEvals accepts, so it has its id and item. */
|
|
528
517
|
|
|
529
518
|
|
|
530
|
-
|
|
519
|
+
|
|
531
520
|
|
|
532
521
|
|
|
533
522
|
|
|
534
523
|
/** A requirement as a score reports it: a term, or the group it was. */
|
|
535
524
|
|
|
536
525
|
|
|
537
|
-
/** How a graded case read a reply: the same shape both engines agree on. */
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
526
|
/** A run's totals, summed across cases. */
|
|
555
527
|
|
|
556
528
|
|
|
@@ -559,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
559
531
|
|
|
560
532
|
|
|
561
533
|
|
|
562
|
-
/** The whole-run verdict of
|
|
534
|
+
/** The whole-run verdict of an eval type, where it has one of its own. */
|
|
563
535
|
|
|
564
536
|
|
|
565
537
|
|
|
@@ -570,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
570
542
|
|
|
571
543
|
|
|
572
544
|
|
|
573
|
-
/** One thing
|
|
545
|
+
/** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
|
|
574
546
|
|
|
575
547
|
|
|
576
548
|
|
|
@@ -599,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
599
571
|
|
|
600
572
|
|
|
601
573
|
|
|
602
|
-
/** A per-item
|
|
574
|
+
/** A per-item eval's score as a stored row carries it. */
|
|
603
575
|
|
|
604
576
|
|
|
605
577
|
|
|
@@ -678,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
678
650
|
more entry, and no reader changes. */
|
|
679
651
|
|
|
680
652
|
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
|
|
681
656
|
|
|
682
657
|
|
|
683
658
|
|
|
@@ -698,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
698
673
|
|
|
699
674
|
|
|
700
675
|
|
|
701
|
-
/** What an earlier
|
|
676
|
+
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
702
677
|
|
|
703
678
|
|
|
704
679
|
|
|
705
680
|
|
|
706
|
-
/** One
|
|
681
|
+
/** One eval's reading of one scenario of a run (scenarioEvals). */
|
|
707
682
|
|
|
708
683
|
|
|
709
684
|
|
|
@@ -760,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
760
735
|
|
|
761
736
|
// ---- the registries ----
|
|
762
737
|
|
|
763
|
-
/** An option a modifier or
|
|
738
|
+
/** An option a modifier or an eval type exposes for editing. */
|
|
764
739
|
|
|
765
740
|
|
|
766
741
|
|
|
@@ -820,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
820
795
|
|
|
821
796
|
|
|
822
797
|
|
|
823
|
-
|
|
798
|
+
|
|
824
799
|
|
|
825
800
|
|
|
826
801
|
|
|
@@ -896,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
896
871
|
|
|
897
872
|
|
|
898
873
|
|
|
899
|
-
/**
|
|
874
|
+
/** An eval type: what it checks of a document, and how it scores a run. */
|
|
900
875
|
|
|
901
876
|
|
|
902
|
-
|
|
877
|
+
|
|
903
878
|
|
|
904
879
|
|
|
905
880
|
|
|
@@ -909,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
909
884
|
|
|
910
885
|
|
|
911
886
|
|
|
912
|
-
|
|
887
|
+
|
|
913
888
|
|
|
914
889
|
|
|
915
890
|
|
|
916
|
-
|
|
891
|
+
|
|
917
892
|
|
|
918
893
|
|
|
919
894
|
|
|
@@ -934,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
934
909
|
|
|
935
910
|
|
|
936
911
|
|
|
937
|
-
/** What
|
|
912
|
+
/** What an eval's `read` is handed besides the reply: production's reply to
|
|
938
913
|
the item, and a grader, where the run has them. */
|
|
939
914
|
|
|
940
915
|
|
|
@@ -1004,42 +979,122 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
1004
979
|
* submitted against, so nothing grades from a file.
|
|
1005
980
|
*/
|
|
1006
981
|
|
|
982
|
+
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
|
|
1007
988
|
|
|
1008
989
|
|
|
1009
990
|
|
|
1010
991
|
|
|
992
|
+
/** The dataset body's version: 6 marks itself; 5 named its Source; 4 and
|
|
993
|
+
earlier, neither. */
|
|
994
|
+
const DATASET_BODY_VERSION = 6 ;
|
|
995
|
+
|
|
996
|
+
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
997
|
+
reply read as a list, as well as one read as text. */
|
|
998
|
+
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
999
|
+
|
|
1011
1000
|
/**
|
|
1012
|
-
* [body] as this version of a dataset (
|
|
1001
|
+
* [body] as this version of a dataset (6), from any earlier one. Version 1
|
|
1013
1002
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1014
1003
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1015
1004
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
1016
1005
|
* job (docs/pipeline-model.md §13): they leave, and the terms a case
|
|
1017
1006
|
* watches for are its `watch`. Version 3 held the `prompt` a new scenario
|
|
1018
|
-
* started from, which the Prompt library holds now: it leaves.
|
|
1019
|
-
*
|
|
1020
|
-
*
|
|
1021
|
-
*
|
|
1007
|
+
* started from, which the Prompt library holds now: it leaves. Version 4's
|
|
1008
|
+
* case named its item `filename` and said what a good answer is in
|
|
1009
|
+
* expectations; each becomes the metric it is (`caseMetrics`), `why` is the
|
|
1010
|
+
* `note`, `traits` go, and the body names no Source yet. Version 5 matched
|
|
1011
|
+
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1012
|
+
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1013
|
+
* the body says its version. Every reader of a
|
|
1014
|
+
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1015
|
+
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1016
|
+
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
1017
|
+
* same answer, and server.py's upgrade_body is its twin. Anything else
|
|
1018
|
+
* comes back as it was.
|
|
1022
1019
|
*/
|
|
1023
1020
|
function upgradeDatasetBody (body ) {
|
|
1024
1021
|
if (!isObj(body)) return body;
|
|
1025
|
-
if (
|
|
1026
|
-
if (!("prompt" in body)) return body;
|
|
1027
|
-
const { prompt: _library, ...rest } = body ;
|
|
1028
|
-
return rest ;
|
|
1029
|
-
}
|
|
1022
|
+
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1030
1023
|
const b = body ;
|
|
1031
|
-
//
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
const out = {};
|
|
1036
|
-
for (const [k, v] of Object.entries(c)) out[RENAMED[k] ?? k] = v;
|
|
1037
|
-
return out;
|
|
1038
|
-
};
|
|
1024
|
+
// A body that names its Source, even as null, is version 5.
|
|
1025
|
+
if ("source" in b) {
|
|
1026
|
+
return { version: DATASET_BODY_VERSION, ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) };
|
|
1027
|
+
}
|
|
1039
1028
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1040
1029
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1041
1030
|
if (!raw) return body;
|
|
1042
|
-
return {
|
|
1031
|
+
return { version: DATASET_BODY_VERSION, source: null, cases: raw.map(caseOfV4).map(caseOfV5) };
|
|
1032
|
+
}
|
|
1033
|
+
|
|
1034
|
+
/** A version-5 case as a version-6 one: each Contains metric says Ignore
|
|
1035
|
+
case, as version 5 matched a list's items whatever it said. A reply read
|
|
1036
|
+
as text did mind its setting, but a case's metrics were written for a
|
|
1037
|
+
list -- each one converted from version 4, and each the case form made. */
|
|
1038
|
+
function caseOfV5(c ) {
|
|
1039
|
+
if (!isObj(c) || !Array.isArray(c.metrics)) return c;
|
|
1040
|
+
return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
|
|
1041
|
+
? { ...m, ignoreCase: true } : m)) };
|
|
1042
|
+
}
|
|
1043
|
+
|
|
1044
|
+
// vocab: the names older versions gave these fields
|
|
1045
|
+
const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
|
|
1046
|
+
/** What a version-4 case said, which its metrics say now. */
|
|
1047
|
+
const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
|
|
1048
|
+
|
|
1049
|
+
/** One case of any earlier version as a version-5 one: its id, item, todo
|
|
1050
|
+
and note, its expectations as metrics ahead of any metrics it held, and
|
|
1051
|
+
anything else it carried kept as it was. */
|
|
1052
|
+
function caseOfV4(c ) {
|
|
1053
|
+
if (!isObj(c)) return c;
|
|
1054
|
+
const was = {};
|
|
1055
|
+
for (const [k, v] of Object.entries(c)) {
|
|
1056
|
+
const key = CASE_RENAMED[k] ?? k;
|
|
1057
|
+
// A case naming its item both ways keeps the newer key's.
|
|
1058
|
+
if (!(key in was) || key === k) was[key] = v;
|
|
1059
|
+
}
|
|
1060
|
+
if (isStr(was.item)) delete was.filename;
|
|
1061
|
+
const out = {};
|
|
1062
|
+
if ("id" in was) out.id = was.id;
|
|
1063
|
+
out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
|
|
1064
|
+
out.todo = was.todo === true;
|
|
1065
|
+
out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
|
|
1066
|
+
out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
|
|
1067
|
+
for (const [k, v] of Object.entries(was)) {
|
|
1068
|
+
if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
|
|
1069
|
+
}
|
|
1070
|
+
return out;
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
/** A version-4 case's expectations as the metrics that say the same:
|
|
1074
|
+
`expect` is Contains all, each `anyOf` group a Contains any, each
|
|
1075
|
+
forbidden term a Contains turned round with the `allow` phrases that
|
|
1076
|
+
excuse it, the count bounds an Item count, `discarded` the Discarded
|
|
1077
|
+
metric, and each watched term a Contains any of weight 0 -- reported,
|
|
1078
|
+
never scored. The order is the one scoreCase read them in, so a reason
|
|
1079
|
+
reads in the order it did. */
|
|
1080
|
+
function caseMetrics(c ) {
|
|
1081
|
+
const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
|
|
1082
|
+
if (c.discarded === true) return [{ type: "discarded" }];
|
|
1083
|
+
const out = [];
|
|
1084
|
+
const expect = list(c.expect), allow = list(c.allow);
|
|
1085
|
+
if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
|
|
1086
|
+
for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
|
|
1087
|
+
if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
|
|
1088
|
+
}
|
|
1089
|
+
for (const t of list(c.forbid)) {
|
|
1090
|
+
// An exception excuses only the forbidden term inside it.
|
|
1091
|
+
const except = allow.filter(a => termIn([a], t));
|
|
1092
|
+
out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
|
|
1093
|
+
}
|
|
1094
|
+
const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
|
|
1095
|
+
if (min != null || max != null) out.push({ type: "item-count", min, max });
|
|
1096
|
+
for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
|
|
1097
|
+
return out;
|
|
1043
1098
|
}
|
|
1044
1099
|
|
|
1045
1100
|
/** The rules a version-1 or version-2 dataset body held, or null: what a
|
|
@@ -1055,6 +1110,9 @@ function datasetRules(body ) {
|
|
|
1055
1110
|
|
|
1056
1111
|
|
|
1057
1112
|
|
|
1113
|
+
|
|
1114
|
+
|
|
1115
|
+
|
|
1058
1116
|
|
|
1059
1117
|
|
|
1060
1118
|
|
|
@@ -1322,8 +1380,10 @@ function textPrompt(instruction , text ) {
|
|
|
1322
1380
|
}
|
|
1323
1381
|
|
|
1324
1382
|
// Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
|
|
1325
|
-
// work in, so that "New Zealand." and "new zealand" are the same two words
|
|
1326
|
-
|
|
1383
|
+
// work in, so that "New Zealand." and "new zealand" are the same two words --
|
|
1384
|
+
// unless [keepCase], for a metric whose Ignore case is off.
|
|
1385
|
+
const words = (s , keepCase = false) =>
|
|
1386
|
+
(keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
|
|
1327
1387
|
|
|
1328
1388
|
// The request Tagger builds, field for field -- when the target reads those
|
|
1329
1389
|
// fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
|
|
@@ -2039,11 +2099,12 @@ function budgetLabel(target ) {
|
|
|
2039
2099
|
// A term is present when its words appear in some item, in order and adjacent:
|
|
2040
2100
|
// "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
|
|
2041
2101
|
// looser scores "new zealand" against "zealandia" and flatters every run.
|
|
2042
|
-
|
|
2043
|
-
|
|
2102
|
+
// Letter case counts only where a metric's Ignore case is off.
|
|
2103
|
+
function termIn(items , term , ignoreCase = true) {
|
|
2104
|
+
const t = words(term, !ignoreCase);
|
|
2044
2105
|
if (!t.length) return false;
|
|
2045
2106
|
return items.some(item => {
|
|
2046
|
-
const w = words(item);
|
|
2107
|
+
const w = words(item, !ignoreCase);
|
|
2047
2108
|
for (let i = 0; i + t.length <= w.length; i++) {
|
|
2048
2109
|
if (t.every((x, j) => w[i + j] === x)) return true;
|
|
2049
2110
|
}
|
|
@@ -2070,153 +2131,55 @@ function forbiddenIn(items , term , allow
|
|
|
2070
2131
|
* check green. `evals-check.js` calls this directly.
|
|
2071
2132
|
*/
|
|
2072
2133
|
function gradedSetFrom(ev ) {
|
|
2073
|
-
return (ev.cases || []).map(c => ({ ...c,
|
|
2134
|
+
return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
|
|
2074
2135
|
}
|
|
2075
2136
|
|
|
2076
|
-
/**
|
|
2077
|
-
|
|
2078
|
-
|
|
2079
|
-
|
|
2080
|
-
|
|
2081
|
-
* case to an item -- the runner, the Datasets tab, a mapping against a Source
|
|
2082
|
-
* -- goes through here.
|
|
2083
|
-
*/
|
|
2084
|
-
function caseFile(kase ) { // vocab: the older spelling
|
|
2085
|
-
const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
|
|
2086
|
-
return typeof name === "string" ? name : "";
|
|
2137
|
+
/** The item a case grades: its `item`, or "" for none. The one reader, so
|
|
2138
|
+
everything that joins a case to an item -- the runner, the Library, a
|
|
2139
|
+
run's Results -- joins on the same key, exactly. */
|
|
2140
|
+
function caseItem(kase ) {
|
|
2141
|
+
return isStr(kase?.item) ? kase .item : "";
|
|
2087
2142
|
}
|
|
2088
2143
|
|
|
2089
|
-
/**
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
|
|
2093
|
-
|
|
2094
|
-
|
|
2095
|
-
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
|
|
2101
|
-
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
|
|
2110
|
-
else out[k] = v;
|
|
2111
|
-
}
|
|
2112
|
-
return out;
|
|
2113
|
-
}),
|
|
2114
|
-
};
|
|
2115
|
-
}
|
|
2144
|
+
/** One graded case's reading of a reply, now: its own metrics, scored as a
|
|
2145
|
+
Metrics eval scores them (all must pass; weight 0 is watched, never
|
|
2146
|
+
scored), with what they found, missed and invented carried up, and
|
|
2147
|
+
`watch` the readings of weight 0. What the page draws and the runner's
|
|
2148
|
+
case-by-case report writes. A model-graded metric needs a grader and the
|
|
2149
|
+
wait for one, so here it reads as not met: the runner's `--run` asks it. */
|
|
2150
|
+
|
|
2151
|
+
|
|
2152
|
+
|
|
2153
|
+
|
|
2154
|
+
|
|
2155
|
+
|
|
2156
|
+
|
|
2157
|
+
|
|
2158
|
+
|
|
2159
|
+
|
|
2160
|
+
|
|
2161
|
+
|
|
2162
|
+
|
|
2163
|
+
|
|
2116
2164
|
|
|
2117
|
-
|
|
2118
|
-
|
|
2119
|
-
|
|
2120
|
-
|
|
2121
|
-
|
|
2122
|
-
|
|
2123
|
-
|
|
2124
|
-
* were wanted has not done what was asked.
|
|
2125
|
-
*
|
|
2126
|
-
* Pure on purpose, and returning `reasons` as plain sentences rather than
|
|
2127
|
-
* markup. The dashboard was the first consumer; `run-evals.js` is the second
|
|
2128
|
-
* and CI the third, and neither can reach into a page for a rendered cell.
|
|
2129
|
-
* Issue #248 wants a failure written out as a task an agent can act on.
|
|
2130
|
-
*/
|
|
2131
|
-
function scoreCase(kase , res ) {
|
|
2132
|
-
// A case that expects its answer discarded passes on a discard and on
|
|
2133
|
-
// nothing else: the answer the job threw away is the finding.
|
|
2134
|
-
if (kase.discarded === true) {
|
|
2135
|
-
const thrown = !!res.error && res.error.startsWith("discarded: ");
|
|
2136
|
-
const n = (res.terms || []).length;
|
|
2137
|
-
return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
|
|
2138
|
-
unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
|
|
2139
|
-
reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
|
|
2140
|
-
}
|
|
2141
|
-
// A discarded answer is a failed item: whatever the pipeline produced
|
|
2142
|
-
// along the way does not count.
|
|
2143
|
-
const discarded = !!res.error;
|
|
2144
|
-
const terms = discarded ? [] : (res.terms || []);
|
|
2145
|
-
const expect = kase.expect || [], forbid = kase.forbid || [];
|
|
2146
|
-
|
|
2147
|
-
// `missed` stays populated for a discarded reply even though it reads as
|
|
2148
|
-
// vacuous, because the score depends on it: docs/datasets.md has such a reply
|
|
2149
|
-
// scoring zero against everything the entry asked for, groups included, and
|
|
2150
|
-
// `addToTally` gets there through `found + missed + invented`. Clear it and
|
|
2151
|
-
// a run that discarded every item would contribute nothing to the
|
|
2152
|
-
// total instead of contributing a nought, which flatters it.
|
|
2153
|
-
//
|
|
2154
|
-
// One member of a group is enough. Without this rule the set manufactures
|
|
2155
|
-
// failures out of synonyms.
|
|
2156
|
-
//
|
|
2157
|
-
// A requirement reads back as a term when it had one member and as the
|
|
2158
|
-
// group itself where a synonym list was allowed, so `missed` carries just
|
|
2159
|
-
// enough to state the reason: "FAIL: Missed dog" for a plain expect term,
|
|
2160
|
-
// "none of dog / puppy" for a group.
|
|
2161
|
-
//
|
|
2162
|
-
// Nothing satisfies a group when nothing was stored, so a discarded reply
|
|
2163
|
-
// misses every requirement rather than none -- the same cast that makes its
|
|
2164
|
-
// count bounds not breached by one. `unmet` is a finding only now that
|
|
2165
|
-
// `missed` names the unsatisfied groups itself, but the run report has
|
|
2166
|
-
// always carried it, so it stays.
|
|
2167
|
-
const met = (g ) => g.some(t => termIn(terms, t));
|
|
2168
|
-
const spoken = (g ) => g.length === 1 ? g[0] : g;
|
|
2169
|
-
const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
|
|
2170
|
-
const found = requirements.filter(met).map(spoken);
|
|
2171
|
-
const missed = requirements.filter(g => !met(g)).map(spoken);
|
|
2172
|
-
const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
|
|
2173
|
-
const unmet = discarded ? []
|
|
2174
|
-
: (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
|
|
2175
|
-
|
|
2176
|
-
const n = terms.length;
|
|
2177
|
-
// A null bound is not checked -- and neither is a bound on a reply that was
|
|
2178
|
-
// discarded. Nothing was counted out and found wanting there; the reply was
|
|
2179
|
-
// thrown away before it had a count, and saying "0 items, wanted at least 4"
|
|
2180
|
-
// states a second failure that never happened.
|
|
2181
|
-
const under = !discarded && kase.minCount != null && n < kase.minCount;
|
|
2182
|
-
const over = !discarded && kase.maxCount != null && n > kase.maxCount;
|
|
2183
|
-
|
|
2184
|
-
const denom = requirements.length + invented.length;
|
|
2185
|
-
// A case that passes is one whose whole expectation was met, not one that
|
|
2186
|
-
// scored well: an unsatisfied group is in `missed` alongside any term
|
|
2187
|
-
// missed, so pass needs nothing more than the terms already covered.
|
|
2188
|
-
const pass = !discarded && !missed.length && !invented.length && !under && !over;
|
|
2189
|
-
|
|
2190
|
-
// `found` and `missed` are groups, not terms, so the reasons word them one
|
|
2191
|
-
// by one: a bare term under one heading, a group on its own line.
|
|
2192
|
-
const reasons = [];
|
|
2193
|
-
if (discarded) {
|
|
2194
|
-
// Everything else would be derived from this one fact -- there are no
|
|
2195
|
-
// terms -- and would bury it. `scoreHtml` above suppresses `missed` for
|
|
2196
|
-
// the same reason: the reason that matters is already on the row.
|
|
2197
|
-
reasons.push(`Error - ${res.error}`);
|
|
2198
|
-
} else {
|
|
2199
|
-
const plain = missed.filter(t => typeof t === "string");
|
|
2200
|
-
if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
|
|
2201
|
-
for (const g of missed) {
|
|
2202
|
-
if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
|
|
2165
|
+
function readCase(kase , res , plain = false) {
|
|
2166
|
+
const input = metricInput(res , kase, { plain });
|
|
2167
|
+
const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
|
|
2168
|
+
const r = readMetric(m, input, {});
|
|
2169
|
+
if (r && typeof (r ).then === "function") {
|
|
2170
|
+
return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
|
|
2171
|
+
pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
|
|
2203
2172
|
}
|
|
2204
|
-
|
|
2205
|
-
|
|
2206
|
-
|
|
2207
|
-
|
|
2208
|
-
|
|
2209
|
-
|
|
2210
|
-
|
|
2211
|
-
|
|
2212
|
-
|
|
2213
|
-
|
|
2214
|
-
// beside an entry that asked for nothing -- a shape `evals-check.js` allows
|
|
2215
|
-
// even though every graded case now names an expectation.
|
|
2216
|
-
return { pass, score: denom ? found.length / denom : null, discarded,
|
|
2217
|
-
found, missed, invented, unmet, under, over, count: n,
|
|
2218
|
-
watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
|
|
2219
|
-
reasons };
|
|
2173
|
+
return r ? [r ] : [];
|
|
2174
|
+
});
|
|
2175
|
+
const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
|
|
2176
|
+
const counted = metrics.filter(r => r.weight !== 0);
|
|
2177
|
+
return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
|
|
2178
|
+
discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
|
|
2179
|
+
// A reply the job threw away fails on that one fact, which would
|
|
2180
|
+
// be buried under everything that derives from it.
|
|
2181
|
+
reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
|
|
2182
|
+
watch: metrics.filter(r => r.weight === 0) };
|
|
2220
2183
|
}
|
|
2221
2184
|
|
|
2222
2185
|
/**
|
|
@@ -2257,11 +2220,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
|
|
|
2257
2220
|
// docs/datasets.md: an item naming more terms should weigh more than
|
|
2258
2221
|
// one naming fewer, and averaging percentages lets the easiest case carry the
|
|
2259
2222
|
// score.
|
|
2260
|
-
function addToTally(t , s
|
|
2223
|
+
function addToTally(t , s ) {
|
|
2224
|
+
const found = s.found?.length ?? 0;
|
|
2261
2225
|
t.ran++;
|
|
2262
2226
|
if (s.pass) t.passed++;
|
|
2263
|
-
t.found +=
|
|
2264
|
-
t.of +=
|
|
2227
|
+
t.found += found;
|
|
2228
|
+
t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
|
|
2265
2229
|
}
|
|
2266
2230
|
|
|
2267
2231
|
// The headline percentage, in one place because it is a number and this file
|
|
@@ -2276,7 +2240,7 @@ function tallyPercent(t ) {
|
|
|
2276
2240
|
// A whole-run assertion with no graded set: All of / Any of / None of, a
|
|
2277
2241
|
// parse, a count, an exact reply and a length bound. It lives in the shared
|
|
2278
2242
|
// core because it is a pass/fail the tab, the worker and History all have to
|
|
2279
|
-
// agree on -- the same one-definition rule that keeps
|
|
2243
|
+
// agree on -- the same one-definition rule that keeps the case reader here.
|
|
2280
2244
|
function parseCount(raw , parse ) {
|
|
2281
2245
|
// An unparsed reply is one result, not no result. It used to return null,
|
|
2282
2246
|
// which made a Count of 1 fail as "null results" against a reply that
|
|
@@ -2666,10 +2630,10 @@ function applyModifiers (list , kind ,
|
|
|
2666
2630
|
// queue and run-evals.js alike, and every one of them reads it through the
|
|
2667
2631
|
// functions below rather than through a translation of its own.
|
|
2668
2632
|
//
|
|
2669
|
-
// The lab is generic, so what a reply is, what
|
|
2633
|
+
// The lab is generic, so what a reply is, what an eval scores and how a value
|
|
2670
2634
|
// is changed are registry entries. The ones here are the lab's own, and so
|
|
2671
2635
|
// are kinds/list.ts's -- the List kind and its modifiers. A dataset
|
|
2672
|
-
// registers nothing: it is data a graded
|
|
2636
|
+
// registers nothing: it is data a graded eval names by id, and a run carries.
|
|
2673
2637
|
|
|
2674
2638
|
// 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
|
|
2675
2639
|
// version 1 kept them as two maps (tokens: { values, blocks }).
|
|
@@ -2685,14 +2649,16 @@ function applyModifiers (list , kind ,
|
|
|
2685
2649
|
// 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
|
|
2686
2650
|
// 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
|
|
2687
2651
|
// Call (Prompt: the image flag and token mappings), Read Reply (the output).
|
|
2688
|
-
// 9:
|
|
2652
|
+
// 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
|
|
2689
2653
|
// they convert to (LEGACY_TESTS' toMetrics, proven equal by
|
|
2690
2654
|
// metrics-parity-check.js).
|
|
2691
2655
|
// 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
|
|
2692
2656
|
// Content, the flow step, Attach Image, Token mappings) and Responses (Read
|
|
2693
2657
|
// as, Response Format Validation, modifiers) -- and its call is gone: each
|
|
2694
2658
|
// scenario is a target, whose own step in each job is what it sends there.
|
|
2695
|
-
|
|
2659
|
+
// 11: `tests` are `evals`: the key renames and nothing in an eval changes,
|
|
2660
|
+
// so a stored result's scores, keyed by eval id, read as they did.
|
|
2661
|
+
const PIPELINE_VERSION = 12 ;
|
|
2696
2662
|
|
|
2697
2663
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2698
2664
|
// to know which registered it.
|
|
@@ -2700,7 +2666,7 @@ const STEP_TYPES = Object.create(null);
|
|
|
2700
2666
|
const CONTENT_TYPES = Object.create(null);
|
|
2701
2667
|
const OUTPUT_KINDS = Object.create(null);
|
|
2702
2668
|
const MODIFIERS = Object.create(null);
|
|
2703
|
-
const
|
|
2669
|
+
const EVAL_TYPES = Object.create(null);
|
|
2704
2670
|
const METRICS = Object.create(null);
|
|
2705
2671
|
const SOURCE_TYPES = Object.create(null);
|
|
2706
2672
|
|
|
@@ -2714,11 +2680,15 @@ let defaultOutputKind = "text";
|
|
|
2714
2680
|
const defaultKind = () => defaultOutputKind;
|
|
2715
2681
|
const kindOf = (st ) => st.kind ?? defaultOutputKind;
|
|
2716
2682
|
|
|
2683
|
+
/** A module's eval types, under either spelling (Kinds.testTypes). */
|
|
2684
|
+
const evalTypesOf = (k ) =>
|
|
2685
|
+
({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
|
|
2686
|
+
|
|
2717
2687
|
/** A module's entries into the registries, in one call. */
|
|
2718
2688
|
function registerKinds(k ) {
|
|
2719
2689
|
Object.assign(OUTPUT_KINDS, k.outputKinds || {});
|
|
2720
2690
|
Object.assign(MODIFIERS, k.modifiers || {});
|
|
2721
|
-
Object.assign(
|
|
2691
|
+
Object.assign(EVAL_TYPES, evalTypesOf(k));
|
|
2722
2692
|
Object.assign(SOURCE_TYPES, k.sourceTypes || {});
|
|
2723
2693
|
Object.assign(METRICS, k.metrics || {});
|
|
2724
2694
|
if (k.defaultKind) defaultOutputKind = k.defaultKind;
|
|
@@ -2749,7 +2719,7 @@ function pluginHost(pluginId ) {
|
|
|
2749
2719
|
registerKinds(k) {
|
|
2750
2720
|
taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
|
|
2751
2721
|
taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
|
|
2752
|
-
taken(
|
|
2722
|
+
taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
|
|
2753
2723
|
taken(METRICS, "metric", Object.keys(k.metrics || {}));
|
|
2754
2724
|
// The server keeps its own copy of the Source types, to refuse a row of
|
|
2755
2725
|
// one it does not know, and a plugin never reaches the server's code.
|
|
@@ -3015,11 +2985,9 @@ LEGACY_TESTS.single = {
|
|
|
3015
2985
|
},
|
|
3016
2986
|
};
|
|
3017
2987
|
|
|
3018
|
-
// A dataset's cases, scored item by item
|
|
3019
|
-
//
|
|
3020
|
-
//
|
|
3021
|
-
// yields, so it accepts every kind that yields any: plain text has none, and
|
|
3022
|
-
// would fail every case.
|
|
2988
|
+
// A dataset's cases, scored item by item. The eval references the dataset
|
|
2989
|
+
// the way a pipeline references a Source, and the run carries the body it
|
|
2990
|
+
// was submitted against.
|
|
3023
2991
|
LEGACY_TESTS.graded = {
|
|
3024
2992
|
label: "Graded set",
|
|
3025
2993
|
fields: ["type", "dataset"],
|
|
@@ -3027,16 +2995,15 @@ LEGACY_TESTS.graded = {
|
|
|
3027
2995
|
validate(t, ctx, bad){
|
|
3028
2996
|
const d = t.dataset;
|
|
3029
2997
|
if (!isRef(d) || (d.version != null && !isStr(d.version))) {
|
|
3030
|
-
return void bad.push("a graded
|
|
2998
|
+
return void bad.push("a graded eval has to name its dataset as { id, name }");
|
|
3031
2999
|
}
|
|
3032
3000
|
if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
|
|
3033
3001
|
},
|
|
3034
|
-
score: (t, kase, res) => scoreCase(kase, res),
|
|
3035
3002
|
};
|
|
3036
3003
|
|
|
3037
3004
|
// Metrics: checks of each reply, deterministic or model-graded, from the
|
|
3038
3005
|
// METRICS registry (metrics/builtin.ts registers the lab's own), with the
|
|
3039
|
-
//
|
|
3006
|
+
// eval's own list for every item and a case's `metrics` for its item. Scored
|
|
3040
3007
|
// all-must-pass -- every metric passes -- or weighted: points, each metric's
|
|
3041
3008
|
// score times its weight, against a threshold (#150's points, a negative
|
|
3042
3009
|
// weight taking them away).
|
|
@@ -3045,6 +3012,13 @@ LEGACY_TESTS.graded = {
|
|
|
3045
3012
|
const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
|
|
3046
3013
|
const SCORING_MODES = ["all", "weighted"];
|
|
3047
3014
|
|
|
3015
|
+
/** A metric's list option -- Contains all's values, Contains's exceptions --
|
|
3016
|
+
one entry a line, or a list of them. */
|
|
3017
|
+
function metricLines(v ) {
|
|
3018
|
+
const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
|
|
3019
|
+
return all.map(x => x.trim()).filter(Boolean);
|
|
3020
|
+
}
|
|
3021
|
+
|
|
3048
3022
|
/** What is wrong with a list of metrics, as sentences naming [at]. */
|
|
3049
3023
|
function metricsProblems(list , at , bad ) {
|
|
3050
3024
|
if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
|
|
@@ -3114,14 +3088,16 @@ function readMetric(m , input , ctx )
|
|
|
3114
3088
|
} catch (e) { return timed(failed(e)); }
|
|
3115
3089
|
}
|
|
3116
3090
|
|
|
3117
|
-
/**
|
|
3091
|
+
/** An eval's score from its metrics' readings, the way [mode] says: all must
|
|
3118
3092
|
pass, or weighted points against [threshold]. What a case metric found,
|
|
3119
3093
|
missed and invented is carried up, for a run's totals. Null for none. */
|
|
3120
3094
|
function scoreOf(metrics , mode = "all", threshold = null) {
|
|
3121
3095
|
if (!metrics.length) return null;
|
|
3122
|
-
|
|
3123
|
-
|
|
3124
|
-
|
|
3096
|
+
// A watched reading (weight 0) is reported, and counts toward nothing.
|
|
3097
|
+
const scored = metrics.filter(r => r.weight !== 0);
|
|
3098
|
+
const detail = scored.some(r => r.found || r.missed || r.invented) ? {
|
|
3099
|
+
found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
|
|
3100
|
+
invented: scored.flatMap(r => r.invented ?? []) } : {};
|
|
3125
3101
|
if (mode === "weighted") {
|
|
3126
3102
|
const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
|
|
3127
3103
|
return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
|
|
@@ -3164,7 +3140,7 @@ function readRun(list , run ) {
|
|
|
3164
3140
|
});
|
|
3165
3141
|
}
|
|
3166
3142
|
|
|
3167
|
-
|
|
3143
|
+
EVAL_TYPES.metrics = {
|
|
3168
3144
|
label: "Metrics",
|
|
3169
3145
|
description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
|
|
3170
3146
|
fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
|
|
@@ -3188,9 +3164,9 @@ TEST_TYPES.metrics = {
|
|
|
3188
3164
|
if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
|
|
3189
3165
|
if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
|
|
3190
3166
|
if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
|
|
3191
|
-
// The
|
|
3167
|
+
// The eval's own grader, or the lab's.
|
|
3192
3168
|
const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3193
|
-
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the
|
|
3169
|
+
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
|
|
3194
3170
|
// A grader is asked words, and needs a model to ask.
|
|
3195
3171
|
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3196
3172
|
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
@@ -3205,7 +3181,7 @@ TEST_TYPES.metrics = {
|
|
|
3205
3181
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3206
3182
|
.filter(Boolean).join(" ") })),
|
|
3207
3183
|
profiles: t => (isRef(t.grader) ? [t.grader] : []),
|
|
3208
|
-
// A run carries the lab's grader on
|
|
3184
|
+
// A run carries the lab's grader on an eval that names none and may ask one:
|
|
3209
3185
|
// a model-graded metric of its own, or a case's, over a dataset.
|
|
3210
3186
|
resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
|
|
3211
3187
|
? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
|
|
@@ -3220,44 +3196,28 @@ TEST_TYPES.metrics = {
|
|
|
3220
3196
|
ran: run.replies.length,
|
|
3221
3197
|
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3222
3198
|
},
|
|
3223
|
-
// Rule m{x} is the
|
|
3199
|
+
// Rule m{x} is the eval's own metric x, read in that place on every item.
|
|
3224
3200
|
// A score stored before Metrics -- a Graded set's, read as its conversion --
|
|
3225
3201
|
// has no readings of its own: its one metric's reading is the score's.
|
|
3226
3202
|
ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
|
|
3227
3203
|
: Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
|
|
3228
|
-
// Every item, with its case's own metrics where
|
|
3229
|
-
|
|
3204
|
+
// Every item, with its case's own metrics where the eval names the
|
|
3205
|
+
// dataset and the item has a case there.
|
|
3206
|
+
read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
|
|
3230
3207
|
metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
|
|
3231
3208
|
};
|
|
3232
3209
|
|
|
3233
3210
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3234
|
-
// What the
|
|
3235
|
-
//
|
|
3236
|
-
//
|
|
3237
|
-
// each is the core's own matcher.
|
|
3238
|
-
|
|
3239
|
-
// An item's case, as the Graded set scores it: its expectations, forbidden
|
|
3240
|
-
// terms and count bounds, with what it found, missed and invented kept.
|
|
3241
|
-
METRICS.case = {
|
|
3242
|
-
label: "Matches its case",
|
|
3243
|
-
description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
|
|
3244
|
-
perItem: true,
|
|
3245
|
-
needsTerms: true,
|
|
3246
|
-
options: [],
|
|
3247
|
-
defaults: () => ({}),
|
|
3248
|
-
score(input) {
|
|
3249
|
-
if (!input.kase) return null;
|
|
3250
|
-
const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
|
|
3251
|
-
return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
|
|
3252
|
-
found: s.found, missed: s.missed, invented: s.invented };
|
|
3253
|
-
},
|
|
3254
|
-
};
|
|
3211
|
+
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3212
|
+
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
3213
|
+
// rather than in metrics/builtin.ts because each is the core's own matcher.
|
|
3255
3214
|
|
|
3256
3215
|
// Items the reply holds -- all of them, or any -- matched as the lab matches
|
|
3257
3216
|
// a term; a kind that yields none is matched in the replies' text instead,
|
|
3258
3217
|
// ignoring case, as the Single Test did.
|
|
3259
3218
|
METRICS["has-items"] = {
|
|
3260
3219
|
label: "Has items",
|
|
3220
|
+
family: "The reply's text",
|
|
3261
3221
|
description: "Passes when the reply holds all, or any, of the items listed.",
|
|
3262
3222
|
options: [{ key: "values", label: "Items", type: "textarea" },
|
|
3263
3223
|
{ key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
|
|
@@ -3277,6 +3237,7 @@ METRICS["has-items"] = {
|
|
|
3277
3237
|
// A reply's length in characters, once trimmed.
|
|
3278
3238
|
METRICS.length = {
|
|
3279
3239
|
label: "Length",
|
|
3240
|
+
family: "The reply's text",
|
|
3280
3241
|
description: "Compares the reply's length in characters.",
|
|
3281
3242
|
options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
|
|
3282
3243
|
{ key: "n", label: "Characters", type: "number" }],
|
|
@@ -3292,6 +3253,7 @@ METRICS.length = {
|
|
|
3292
3253
|
// as the count says.
|
|
3293
3254
|
METRICS["parse-count"] = {
|
|
3294
3255
|
label: "Parses",
|
|
3256
|
+
family: "The result",
|
|
3295
3257
|
description: "Passes when the reply reads as the format chosen, holding the number of results set.",
|
|
3296
3258
|
// Unformatted is one result a reply, as the Single Test counted it.
|
|
3297
3259
|
options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
|
|
@@ -3310,12 +3272,13 @@ METRICS["parse-count"] = {
|
|
|
3310
3272
|
},
|
|
3311
3273
|
};
|
|
3312
3274
|
|
|
3313
|
-
/** What every
|
|
3275
|
+
/** What every eval carries, kept across a conversion. */
|
|
3314
3276
|
const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
|
|
3315
3277
|
|
|
3316
|
-
// A Graded set is a Metrics
|
|
3278
|
+
// A Graded set is a Metrics eval over the same dataset, with none of its own:
|
|
3279
|
+
// each case's metrics are what it scored (dataset-parity-check.js).
|
|
3317
3280
|
LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
|
|
3318
|
-
dataset: t.dataset, over: "item", metrics: [
|
|
3281
|
+
dataset: t.dataset, over: "item", metrics: [] });
|
|
3319
3282
|
|
|
3320
3283
|
// A Single Test is Metrics over the whole run: its lists over the run's items
|
|
3321
3284
|
// together, and exact, length and parse over each reply alone.
|
|
@@ -3333,7 +3296,7 @@ LEGACY_TESTS.single .toMetrics = t => {
|
|
|
3333
3296
|
return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
|
|
3334
3297
|
};
|
|
3335
3298
|
|
|
3336
|
-
/** The grader a Metrics
|
|
3299
|
+
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3337
3300
|
function graderCtx(t , more ) {
|
|
3338
3301
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3339
3302
|
return ask ? { ask } : {};
|
|
@@ -3343,12 +3306,14 @@ function graderCtx(t , more ) {
|
|
|
3343
3306
|
for one with none set: what History prints beside its name. */
|
|
3344
3307
|
function metricSummary(m ) {
|
|
3345
3308
|
const entry = METRICS[m.type];
|
|
3346
|
-
// Each option as it reads in the editor: a choice's label, a box
|
|
3347
|
-
//
|
|
3309
|
+
// Each option as it reads in the editor: a choice's label, a box by its
|
|
3310
|
+
// own label where it is not as a new metric has it (ticked, or "off"),
|
|
3311
|
+
// text by its first line.
|
|
3312
|
+
const fresh = entry?.defaults() ?? {};
|
|
3348
3313
|
const said = (entry?.options || []).map(o => {
|
|
3349
3314
|
const v = m[o.key];
|
|
3350
3315
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3351
|
-
if (o.type === "checkbox") return v ? o.label.toLowerCase() :
|
|
3316
|
+
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
3352
3317
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3353
3318
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3354
3319
|
}).filter(Boolean);
|
|
@@ -3531,7 +3496,7 @@ STEP_TYPES.readAs = {
|
|
|
3531
3496
|
// Response Format Validation: the kind a reply is read as.
|
|
3532
3497
|
STEP_TYPES.readReply = {
|
|
3533
3498
|
label: "Read Reply", slot: "responses", rank: 1,
|
|
3534
|
-
description: "How the reply is read before
|
|
3499
|
+
description: "How the reply is read before evals and later jobs see it.",
|
|
3535
3500
|
in: "text", out: step => step.out?.kind,
|
|
3536
3501
|
// Read inside the stage: runPipeline parses each reply as it comes back.
|
|
3537
3502
|
apply: "runPipeline",
|
|
@@ -3787,7 +3752,7 @@ STEP_TYPES.job = {
|
|
|
3787
3752
|
if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
|
|
3788
3753
|
const steps = c.steps;
|
|
3789
3754
|
// A job's steps are the entries with a stage; one without (a job, the
|
|
3790
|
-
//
|
|
3755
|
+
// evals) or of no type the lab has is not one.
|
|
3791
3756
|
const unknown = steps.find(st => !slotOf(st));
|
|
3792
3757
|
if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
|
|
3793
3758
|
if (steps.some(st => slotOf(st) === "target")) {
|
|
@@ -3808,54 +3773,55 @@ STEP_TYPES.job = {
|
|
|
3808
3773
|
}
|
|
3809
3774
|
},
|
|
3810
3775
|
};
|
|
3811
|
-
// What
|
|
3812
|
-
const
|
|
3776
|
+
// What an eval carries besides its type's own fields.
|
|
3777
|
+
const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
|
|
3813
3778
|
|
|
3814
|
-
/**
|
|
3815
|
-
const
|
|
3816
|
-
(doc.
|
|
3779
|
+
/** Eval [j]'s name, or the number it has always shown. */
|
|
3780
|
+
const evalLabel = (doc , j ) =>
|
|
3781
|
+
(doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
|
|
3817
3782
|
|
|
3818
|
-
/**
|
|
3783
|
+
/** An eval type that settles over the whole run rather than item by item. */
|
|
3819
3784
|
const isWholeRun = (type , t ) =>
|
|
3820
3785
|
type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
|
|
3821
3786
|
|
|
3822
3787
|
/**
|
|
3823
|
-
* A document's
|
|
3824
|
-
* the run document it was submitted with, so a run from before version
|
|
3825
|
-
*
|
|
3788
|
+
* A document's evals as a list, whatever version wrote it: a queue row keeps
|
|
3789
|
+
* the run document it was submitted with, so a run from before version 11
|
|
3790
|
+
* spells them `tests`, and one from before version 6 holds one or null, read
|
|
3791
|
+
* here as the upgrade reads it (testsList, evalsKey).
|
|
3826
3792
|
*/
|
|
3827
|
-
function
|
|
3828
|
-
const t = doc?.tests;
|
|
3793
|
+
function evalsOf(doc ) {
|
|
3794
|
+
const t = doc?.evals ?? doc?.tests;
|
|
3829
3795
|
if (Array.isArray(t)) return t ;
|
|
3830
3796
|
return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
|
|
3831
3797
|
}
|
|
3832
3798
|
|
|
3833
|
-
/** The dataset a document's
|
|
3799
|
+
/** The dataset a document's evals grade against, where one does: a run
|
|
3834
3800
|
grades against one (validatePipeline says so), so the first names it. */
|
|
3835
|
-
function
|
|
3836
|
-
const t =
|
|
3801
|
+
function evalsDataset(doc ) {
|
|
3802
|
+
const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
|
|
3837
3803
|
return t && "dataset" in t ? t.dataset : null;
|
|
3838
3804
|
}
|
|
3839
3805
|
|
|
3840
|
-
STEP_TYPES.
|
|
3806
|
+
STEP_TYPES.evals = {
|
|
3841
3807
|
in: "results", out: "verdict",
|
|
3842
3808
|
apply: "score",
|
|
3843
|
-
validate(
|
|
3844
|
-
if (!Array.isArray(
|
|
3809
|
+
validate(evals, ctx, bad, lastKind, doc){
|
|
3810
|
+
if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
|
|
3845
3811
|
const seen = new Set ();
|
|
3846
|
-
|
|
3847
|
-
const at =
|
|
3848
|
-
const type = isObj(t) &&
|
|
3812
|
+
evals.forEach((t , j ) => {
|
|
3813
|
+
const at = evalLabel(doc, j);
|
|
3814
|
+
const type = isObj(t) && EVAL_TYPES[t.type];
|
|
3849
3815
|
if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
|
|
3850
3816
|
const before = bad.length;
|
|
3851
3817
|
if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
|
|
3852
|
-
else if (seen.has(t.id)) bad.push(`${at} has the id of another
|
|
3818
|
+
else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
|
|
3853
3819
|
else seen.add(t.id);
|
|
3854
3820
|
if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
|
|
3855
3821
|
if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
|
|
3856
|
-
onlyFields(t, at, [...type.fields, ...
|
|
3822
|
+
onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
|
|
3857
3823
|
type.validate(t, ctx, bad);
|
|
3858
|
-
// Which kinds
|
|
3824
|
+
// Which kinds an eval scores is only worth saying of an eval that is whole.
|
|
3859
3825
|
if (bad.length > before) return;
|
|
3860
3826
|
const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
|
|
3861
3827
|
if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
|
|
@@ -3864,14 +3830,14 @@ STEP_TYPES.tests = {
|
|
|
3864
3830
|
}
|
|
3865
3831
|
});
|
|
3866
3832
|
// A run is handed one dataset's body to grade against (server-side-runs §4).
|
|
3867
|
-
const named = new Set(
|
|
3868
|
-
if (named.size > 1) bad.push("the
|
|
3833
|
+
const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
|
|
3834
|
+
if (named.size > 1) bad.push("the evals grade against one dataset at a time");
|
|
3869
3835
|
},
|
|
3870
3836
|
};
|
|
3871
3837
|
|
|
3872
3838
|
// ---- the document -------------------------------------------------------------
|
|
3873
3839
|
|
|
3874
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "
|
|
3840
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
|
|
3875
3841
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3876
3842
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3877
3843
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3888,7 +3854,7 @@ function versionProblem(doc ) {
|
|
|
3888
3854
|
/** What an upgrade from version 2 needs from outside the document: the rules
|
|
3889
3855
|
of the dataset a pipeline was graded against, which the retired
|
|
3890
3856
|
version-2 list kind read every reply under. Without them a job is upgraded with no
|
|
3891
|
-
rules, as a run with no graded
|
|
3857
|
+
rules, as a run with no graded eval parsed. */
|
|
3892
3858
|
|
|
3893
3859
|
|
|
3894
3860
|
|
|
@@ -3913,7 +3879,8 @@ function versionProblem(doc ) {
|
|
|
3913
3879
|
* as it was, for versionProblem to name. A copy: the caller's document is not
|
|
3914
3880
|
* touched. From version 5, its test -- or none -- becomes a list of one (or
|
|
3915
3881
|
* none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
|
|
3916
|
-
* renames and every job's `type` becomes `"job"`.
|
|
3882
|
+
* renames and every job's `type` becomes `"job"`. From version 10, its
|
|
3883
|
+
* `tests` are `evals` (evalsKey).
|
|
3917
3884
|
*/
|
|
3918
3885
|
/** Every id a current document must carry, minted for the ones [doc] lacks.
|
|
3919
3886
|
An id it already has is kept. */
|
|
@@ -3949,6 +3916,9 @@ function profileRefs(doc ) {
|
|
|
3949
3916
|
}
|
|
3950
3917
|
}
|
|
3951
3918
|
|
|
3919
|
+
/** [doc] with its profile references cut (profileRefs), for chaining. */
|
|
3920
|
+
const cutRefs = (doc ) => { profileRefs(doc); return doc; };
|
|
3921
|
+
|
|
3952
3922
|
/** Whether any profile reference in a pipeline holds more than { id, name }. */
|
|
3953
3923
|
function fatProfileRef(doc ) {
|
|
3954
3924
|
const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
|
|
@@ -3967,9 +3937,9 @@ function testsList(doc ) {
|
|
|
3967
3937
|
return doc;
|
|
3968
3938
|
}
|
|
3969
3939
|
|
|
3970
|
-
/** A new
|
|
3971
|
-
function
|
|
3972
|
-
const own =
|
|
3940
|
+
/** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
|
|
3941
|
+
function newEval(type , name = "", fields = {}) {
|
|
3942
|
+
const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
|
|
3973
3943
|
return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
|
|
3974
3944
|
}
|
|
3975
3945
|
|
|
@@ -4067,6 +4037,29 @@ function targetsFromScenarios(next ) {
|
|
|
4067
4037
|
return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
|
|
4068
4038
|
}
|
|
4069
4039
|
|
|
4040
|
+
/** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
|
|
4041
|
+
upgraded document reads the same, key for key, and each eval is as it
|
|
4042
|
+
was. Every version before 11 passes through here. */
|
|
4043
|
+
function evalsKey(next ) {
|
|
4044
|
+
next.version = PIPELINE_VERSION;
|
|
4045
|
+
// `tests` is the spelling of every version before 11.
|
|
4046
|
+
if (!("tests" in next)) return next;
|
|
4047
|
+
return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
|
|
4048
|
+
.map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
|
|
4049
|
+
}
|
|
4050
|
+
|
|
4051
|
+
/** Version 11 to 12: a Contains metric's Ignore case holds where the reply
|
|
4052
|
+
is matched item by item, as it does where it is matched as text, and it
|
|
4053
|
+
is kept as written. Until version 11's last hours (#199) every Contains
|
|
4054
|
+
metric matched the reply's text and minded its Ignore case, so what a
|
|
4055
|
+
stored metric says is what its author meant; only the item-by-item
|
|
4056
|
+
matching #199 added, case-blind for a few hours, read it otherwise.
|
|
4057
|
+
Every version before 12 ends here. */
|
|
4058
|
+
function caseAsWritten(next ) {
|
|
4059
|
+
next.version = PIPELINE_VERSION;
|
|
4060
|
+
return next;
|
|
4061
|
+
}
|
|
4062
|
+
|
|
4070
4063
|
function upgradePipeline (doc , ctx = {}) {
|
|
4071
4064
|
// A current document is read as it is, but for a profile reference the Runs
|
|
4072
4065
|
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
@@ -4077,11 +4070,29 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
4077
4070
|
out = clone(out) ;
|
|
4078
4071
|
profileRefs(out);
|
|
4079
4072
|
}
|
|
4080
|
-
return localSteps(out, ctx) ;
|
|
4073
|
+
return localSteps(withoutCaseMetric(out), ctx) ;
|
|
4081
4074
|
}
|
|
4082
|
-
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9].includes(doc.version )) return doc;
|
|
4083
|
-
|
|
4084
|
-
|
|
4075
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
|
|
4076
|
+
if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
|
|
4077
|
+
// Every version before 10 reads as version 9 first, then as 10, then as 11
|
|
4078
|
+
// and 12.
|
|
4079
|
+
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4080
|
+
// earlier ones are cut on their way through nineOf.
|
|
4081
|
+
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4082
|
+
return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
|
|
4083
|
+
}
|
|
4084
|
+
|
|
4085
|
+
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
4086
|
+
are metrics now (dataset body version 5), which an eval naming the
|
|
4087
|
+
dataset adds for each item, so the case's own metrics carry what that
|
|
4088
|
+
metric scored. Read so at every version, the current one included, as a
|
|
4089
|
+
document saved before it went still holds it. The same document where
|
|
4090
|
+
none does. */
|
|
4091
|
+
function withoutCaseMetric(doc ) {
|
|
4092
|
+
const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
|
|
4093
|
+
if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
|
|
4094
|
+
return { ...doc, evals: doc.evals.map((t ) => (has(t)
|
|
4095
|
+
? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
|
|
4085
4096
|
}
|
|
4086
4097
|
|
|
4087
4098
|
/** [doc] with each step asked of a profile whose type a target step stands
|
|
@@ -4165,7 +4176,7 @@ function tokenMappingsFromV1(set ) {
|
|
|
4165
4176
|
function blankPipeline(opts = {}) {
|
|
4166
4177
|
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
4167
4178
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
4168
|
-
jobs: [job], targets: [],
|
|
4179
|
+
jobs: [job], targets: [], evals: [] };
|
|
4169
4180
|
}
|
|
4170
4181
|
|
|
4171
4182
|
/**
|
|
@@ -4303,7 +4314,7 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4303
4314
|
: `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
|
|
4304
4315
|
}
|
|
4305
4316
|
});
|
|
4306
|
-
STEP_TYPES.
|
|
4317
|
+
STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
4307
4318
|
if (bad.length) return bad;
|
|
4308
4319
|
|
|
4309
4320
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
@@ -4400,9 +4411,9 @@ function profileIds(doc )
|
|
|
4400
4411
|
if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
4401
4412
|
}
|
|
4402
4413
|
}
|
|
4403
|
-
// Then the ones
|
|
4404
|
-
for (const t of
|
|
4405
|
-
for (const ref of
|
|
4414
|
+
// Then the ones an eval asks (a grader), so a run carries them too.
|
|
4415
|
+
for (const t of evalsOf(doc)) {
|
|
4416
|
+
for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
4406
4417
|
}
|
|
4407
4418
|
return ids;
|
|
4408
4419
|
}
|
|
@@ -4415,9 +4426,9 @@ function profileIds(doc )
|
|
|
4415
4426
|
function resolvePipeline(doc ,
|
|
4416
4427
|
ctx = {}) {
|
|
4417
4428
|
const run = clone(doc) ;
|
|
4418
|
-
// What the lab supplies
|
|
4429
|
+
// What the lab supplies an eval -- its grader -- before the profiles it
|
|
4419
4430
|
// asks are carried.
|
|
4420
|
-
run.
|
|
4431
|
+
run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
|
|
4421
4432
|
run.profiles = {};
|
|
4422
4433
|
for (const id of profileIds(run)) {
|
|
4423
4434
|
const p = ctx.profiles?.(id);
|
|
@@ -4426,7 +4437,7 @@ function resolvePipeline(doc ,
|
|
|
4426
4437
|
const content = contentOf(run) ;
|
|
4427
4438
|
if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
|
|
4428
4439
|
if (ctx.datasetVersion) {
|
|
4429
|
-
for (const t of run.
|
|
4440
|
+
for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
|
|
4430
4441
|
}
|
|
4431
4442
|
if (isStr(ctx.comment)) run.comment = ctx.comment;
|
|
4432
4443
|
return run;
|
|
@@ -4440,7 +4451,7 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4440
4451
|
delete doc.plugins;
|
|
4441
4452
|
const content = contentOf(doc) ;
|
|
4442
4453
|
if (content) { delete content.files; delete content.revs; }
|
|
4443
|
-
for (const t of doc.
|
|
4454
|
+
for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
|
|
4444
4455
|
return doc;
|
|
4445
4456
|
}
|
|
4446
4457
|
|
|
@@ -4498,7 +4509,7 @@ function importPipeline(input , ctx = {}) {
|
|
|
4498
4509
|
if (st?.profile) st.profile = remap(st.profile, "profile");
|
|
4499
4510
|
}
|
|
4500
4511
|
}
|
|
4501
|
-
for (const t of Array.isArray(next.
|
|
4512
|
+
for (const t of Array.isArray(next.evals) ? next.evals : []) {
|
|
4502
4513
|
if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
|
|
4503
4514
|
}
|
|
4504
4515
|
// Its references are this lab's now, so a step asking this lab's Echo
|
|
@@ -4530,7 +4541,7 @@ function mintIds(doc ) {
|
|
|
4530
4541
|
for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
|
|
4531
4542
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4532
4543
|
}
|
|
4533
|
-
for (const t of Array.isArray(doc.
|
|
4544
|
+
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4534
4545
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4535
4546
|
}
|
|
4536
4547
|
return doc;
|
|
@@ -4608,14 +4619,14 @@ function modifierSummary(m ) {
|
|
|
4608
4619
|
return said.length ? said.join(", ") : "on";
|
|
4609
4620
|
}
|
|
4610
4621
|
|
|
4611
|
-
// ---- the
|
|
4612
|
-
//
|
|
4622
|
+
// ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
|
|
4623
|
+
// Evals only read a run: none changes what a later one sees, and none stops
|
|
4613
4624
|
// the model being sent the next item. What order changes is Continue on
|
|
4614
|
-
// failure. A per-item
|
|
4625
|
+
// failure. A per-item eval that fails and does not continue stops the evals
|
|
4615
4626
|
// after it for that item alone -- they read Skipped there, and a whole-run
|
|
4616
|
-
//
|
|
4627
|
+
// eval after it pools the items it did not stop. A whole-run eval settles
|
|
4617
4628
|
// once every item is in, and one that fails then and does not continue
|
|
4618
|
-
// leaves every
|
|
4629
|
+
// leaves every eval after it Skipped.
|
|
4619
4630
|
|
|
4620
4631
|
const SKIPPED = Object.freeze({ skipped: true });
|
|
4621
4632
|
const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
@@ -4623,16 +4634,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
|
4623
4634
|
const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
|
|
4624
4635
|
|
|
4625
4636
|
/**
|
|
4626
|
-
* One reply's scores under a run's per-item
|
|
4627
|
-
*
|
|
4637
|
+
* One reply's scores under a run's per-item evals, by eval id, in order: a
|
|
4638
|
+
* eval with no case to score leaves no entry, and one after a failure that
|
|
4628
4639
|
* does not continue reads Skipped. Null where nothing was scored.
|
|
4629
4640
|
*/
|
|
4630
|
-
function itemScores(run , kase , res )
|
|
4641
|
+
function itemScores(run , kase , res ) {
|
|
4631
4642
|
if (!kase) return null;
|
|
4632
|
-
const out
|
|
4643
|
+
const out = {};
|
|
4633
4644
|
let stopped = false;
|
|
4634
|
-
for (const t of
|
|
4635
|
-
const type =
|
|
4645
|
+
for (const t of evalsOf(run)) {
|
|
4646
|
+
const type = EVAL_TYPES[t.type];
|
|
4636
4647
|
if (!type?.score) continue;
|
|
4637
4648
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4638
4649
|
const s = type.score(t, kase, res);
|
|
@@ -4650,19 +4661,19 @@ function productionOf(run , record )
|
|
|
4650
4661
|
}
|
|
4651
4662
|
|
|
4652
4663
|
/**
|
|
4653
|
-
* itemScores for the runner, which can wait:
|
|
4664
|
+
* itemScores for the runner, which can wait: an eval that reads every item
|
|
4654
4665
|
* (`read`: the Metrics, which may ask a grader) scores one with no case too.
|
|
4655
4666
|
*/
|
|
4656
4667
|
async function itemScoresAsync(run , kase , res ,
|
|
4657
4668
|
more = {}) {
|
|
4658
4669
|
const out = {};
|
|
4659
4670
|
let stopped = false;
|
|
4660
|
-
for (const t of
|
|
4661
|
-
const type =
|
|
4671
|
+
for (const t of evalsOf(run)) {
|
|
4672
|
+
const type = EVAL_TYPES[t.type];
|
|
4662
4673
|
if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
|
|
4663
4674
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4664
|
-
//
|
|
4665
|
-
//
|
|
4675
|
+
// An eval with nothing to read on this item leaves no entry, as a graded
|
|
4676
|
+
// eval does on an item with no case.
|
|
4666
4677
|
const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
|
|
4667
4678
|
: type.score (t, kase , res);
|
|
4668
4679
|
if (!s) continue;
|
|
@@ -4673,22 +4684,22 @@ async function itemScoresAsync(run , kase ,
|
|
|
4673
4684
|
}
|
|
4674
4685
|
|
|
4675
4686
|
/**
|
|
4676
|
-
* Every
|
|
4687
|
+
* Every eval's reading of scenario [i] of a run, in the run's order, from
|
|
4677
4688
|
* the items so far. [settled] says every item is in: only then has a
|
|
4678
|
-
* whole-run
|
|
4689
|
+
* whole-run eval settled, so only then does its failure skip the evals
|
|
4679
4690
|
* after it.
|
|
4680
4691
|
*/
|
|
4681
|
-
function
|
|
4692
|
+
function scenarioEvals(run , i , items ,
|
|
4682
4693
|
settled = true) {
|
|
4683
4694
|
const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
|
|
4684
|
-
// Which items a per-item
|
|
4695
|
+
// Which items a per-item eval has stopped so far, for the evals after it.
|
|
4685
4696
|
const stopped = cells.map(() => false);
|
|
4686
4697
|
let skipRest = false;
|
|
4687
|
-
const
|
|
4688
|
-
return
|
|
4689
|
-
const type =
|
|
4698
|
+
const evals = evalsOf(run);
|
|
4699
|
+
return evals.map((t, j) => {
|
|
4700
|
+
const type = EVAL_TYPES[t.type];
|
|
4690
4701
|
const whole = isWholeRun(type, t);
|
|
4691
|
-
const base = { id: t.id, label:
|
|
4702
|
+
const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
|
|
4692
4703
|
verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
|
|
4693
4704
|
ran: 0, passed: 0, skippedItems: 0 };
|
|
4694
4705
|
if (skipRest) {
|
|
@@ -4715,7 +4726,7 @@ function scenarioTests(run , i , items
|
|
|
4715
4726
|
});
|
|
4716
4727
|
}
|
|
4717
4728
|
|
|
4718
|
-
/** A scenario's pass or fail over every
|
|
4729
|
+
/** A scenario's pass or fail over every eval that read it: null where none
|
|
4719
4730
|
has anything to say yet. */
|
|
4720
4731
|
function scenarioPasses(outcomes ) {
|
|
4721
4732
|
let said = false;
|
|
@@ -4756,95 +4767,77 @@ function validateEvals(ev , files
|
|
|
4756
4767
|
return ["the graded set has to be a JSON object with a `cases` list"];
|
|
4757
4768
|
}
|
|
4758
4769
|
if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
|
|
4770
|
+
if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
|
|
4759
4771
|
if (bad.length) return bad;
|
|
4760
4772
|
|
|
4761
4773
|
const graded = ev.cases || [];
|
|
4762
4774
|
// Identity first: every rule below reports which case is at fault, so a
|
|
4763
4775
|
// case with no usable id makes the rest of the report unreadable.
|
|
4764
|
-
const seen = new
|
|
4776
|
+
const seen = new Set ();
|
|
4765
4777
|
for (const c of graded) {
|
|
4766
|
-
{
|
|
4767
|
-
|
|
4768
|
-
|
|
4769
|
-
continue;
|
|
4770
|
-
}
|
|
4771
|
-
if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
|
|
4772
|
-
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4773
|
-
else seen.set(c.id, "cases");
|
|
4774
|
-
if (!caseFile(c).trim()) {
|
|
4775
|
-
bad.push(`${c.id || "a case"} names no file`);
|
|
4776
|
-
}
|
|
4777
|
-
for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
|
|
4778
|
-
if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
|
|
4779
|
-
}
|
|
4780
|
-
for (const key of ["minCount", "maxCount"]) {
|
|
4781
|
-
if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
|
|
4782
|
-
}
|
|
4783
|
-
if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
|
|
4784
|
-
// A case's own metrics, which a Metrics test adds to its own for this item.
|
|
4785
|
-
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4778
|
+
if (!isObj(c)) {
|
|
4779
|
+
bad.push("cases holds something that is not a case");
|
|
4780
|
+
continue;
|
|
4786
4781
|
}
|
|
4782
|
+
if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
|
|
4783
|
+
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4784
|
+
else seen.add(c.id);
|
|
4785
|
+
if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
|
|
4786
|
+
if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
|
|
4787
|
+
if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
|
|
4788
|
+
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4787
4789
|
}
|
|
4788
4790
|
if (bad.length) return bad;
|
|
4789
4791
|
|
|
4790
|
-
//
|
|
4791
|
-
// the case is unpassable however well the model
|
|
4792
|
+
// What a case says, read from its metrics: a term the scorer cannot match
|
|
4793
|
+
// can never be produced, so the case is unpassable however well the model
|
|
4794
|
+
// answers -- and the same goes for a bound no reply can meet.
|
|
4792
4795
|
const matchable = (term ) => termIn([String(term).trim()], term);
|
|
4793
|
-
|
|
4796
|
+
const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
|
|
4794
4797
|
for (const c of graded) {
|
|
4795
4798
|
if (c.todo) continue;
|
|
4796
4799
|
const say = (m ) => bad.push(`${c.id}: ${m}`);
|
|
4797
|
-
const
|
|
4798
|
-
|
|
4799
|
-
if (
|
|
4800
|
+
const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
|
|
4801
|
+
if (!scored.length) { say("graded but states nothing to expect"); continue; }
|
|
4802
|
+
if (scored.some(m => m.type === "discarded" && !m.not)) {
|
|
4800
4803
|
// What is discarded holds nothing, so there is nothing else to expect.
|
|
4801
|
-
if (
|
|
4802
|
-
say("expects its answer discarded, and states something the answer should hold too");
|
|
4803
|
-
}
|
|
4804
|
+
if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
|
|
4804
4805
|
continue;
|
|
4805
4806
|
}
|
|
4806
|
-
const
|
|
4807
|
-
|
|
4808
|
-
for (const t of
|
|
4809
|
-
|
|
4810
|
-
|
|
4811
|
-
|
|
4812
|
-
for (const
|
|
4813
|
-
}
|
|
4814
|
-
// An exception excuses only the forbidden terms inside it, so one that
|
|
4815
|
-
// holds none of them changes nothing and reads as if it did.
|
|
4816
|
-
for (const a of c.allow || []) {
|
|
4817
|
-
if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
|
|
4807
|
+
const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
|
|
4808
|
+
const forbids = scored.filter(m => m.not && m.type === "contains");
|
|
4809
|
+
for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
|
|
4810
|
+
// An exception excuses only the forbidden term inside it, so one that
|
|
4811
|
+
// holds it not changes nothing and reads as if it did.
|
|
4812
|
+
for (const m of forbids) {
|
|
4813
|
+
for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
|
|
4818
4814
|
}
|
|
4819
4815
|
// A term on both lists cannot be produced and cannot be withheld.
|
|
4820
|
-
|
|
4821
|
-
|
|
4822
|
-
|
|
4823
|
-
|
|
4824
|
-
|
|
4825
|
-
|
|
4826
|
-
|
|
4827
|
-
|
|
4828
|
-
|
|
4829
|
-
|
|
4830
|
-
|
|
4831
|
-
|
|
4832
|
-
|
|
4833
|
-
const things = expect.length + anyOf.length;
|
|
4834
|
-
if (hi != null && things > hi) {
|
|
4835
|
-
say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
|
|
4816
|
+
const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
|
|
4817
|
+
for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
|
|
4818
|
+
for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
|
|
4819
|
+
const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
|
|
4820
|
+
if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
|
|
4821
|
+
// A ceiling below one says no answer is acceptable, which is a case
|
|
4822
|
+
// that can never pass rather than a strict one.
|
|
4823
|
+
if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
|
|
4824
|
+
// A case that names more distinct things than the reply may carry, in
|
|
4825
|
+
// the scorer's own counting of a thing (#486: each term Contains all
|
|
4826
|
+
// wants is one, each Contains any is one), can never pass.
|
|
4827
|
+
const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
|
|
4828
|
+
if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
|
|
4836
4829
|
}
|
|
4837
4830
|
}
|
|
4838
4831
|
|
|
4839
|
-
// One
|
|
4840
|
-
//
|
|
4832
|
+
// One item, one case. An item here twice is two cases of it, graded
|
|
4833
|
+
// separately, and both would be listed.
|
|
4841
4834
|
const where = new Map ();
|
|
4842
4835
|
for (const c of graded) {
|
|
4843
|
-
const name =
|
|
4836
|
+
const name = caseItem(c);
|
|
4844
4837
|
const counted = where.get(name);
|
|
4845
4838
|
if (counted) {
|
|
4846
|
-
bad.push(`${name} is graded twice — one
|
|
4847
|
-
+ `two
|
|
4839
|
+
bad.push(`${name} is graded twice — one item, `
|
|
4840
|
+
+ `two cases. Grade it once.`);
|
|
4848
4841
|
} else where.set(name, true);
|
|
4849
4842
|
}
|
|
4850
4843
|
const gradedAt = (name ) => where.has(name);
|
|
@@ -4865,7 +4858,7 @@ function validateEvals(ev , files
|
|
|
4865
4858
|
for (const f of shown) {
|
|
4866
4859
|
if (!gradedAt(f)) {
|
|
4867
4860
|
bad.push(`${f} is in the Source and this dataset does not grade it, `
|
|
4868
|
-
+ `so the
|
|
4861
|
+
+ `so the Cases view does not list it`);
|
|
4869
4862
|
}
|
|
4870
4863
|
}
|
|
4871
4864
|
return bad;
|
|
@@ -4887,7 +4880,9 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4887
4880
|
const want = Number(n), warn = [];
|
|
4888
4881
|
for (const c of ev.cases || []) {
|
|
4889
4882
|
if (!c || c.todo) continue;
|
|
4890
|
-
const things = (c.
|
|
4883
|
+
const things = (Array.isArray(c.metrics) ? c.metrics : [])
|
|
4884
|
+
.filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
|
|
4885
|
+
.reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
|
|
4891
4886
|
if (things > want) {
|
|
4892
4887
|
warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
|
|
4893
4888
|
}
|
|
@@ -4906,7 +4901,7 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4906
4901
|
* `evals-check.js` asserts the round trip against the real file.
|
|
4907
4902
|
*/
|
|
4908
4903
|
function evalsJson(ev ) {
|
|
4909
|
-
return JSON.stringify(
|
|
4904
|
+
return JSON.stringify(ev, null, 2) + "\n";
|
|
4910
4905
|
}
|
|
4911
4906
|
|
|
4912
4907
|
|
|
@@ -4920,14 +4915,14 @@ registerMetrics({ registerKinds });
|
|
|
4920
4915
|
export {
|
|
4921
4916
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
4922
4917
|
loopReplyError, preparedSize,
|
|
4923
|
-
termIn, forbiddenIn,
|
|
4918
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4924
4919
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
4925
4920
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
4926
4921
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
4927
4922
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
4928
4923
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
4929
4924
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
4930
|
-
PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS,
|
|
4925
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
|
|
4931
4926
|
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
4932
4927
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
4933
4928
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
@@ -4935,9 +4930,9 @@ CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWord
|
|
|
4935
4930
|
contentOf, withContent, replyOf,
|
|
4936
4931
|
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
4937
4932
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
4938
|
-
blankPipeline, upgradePipeline, fatProfileRef, newId,
|
|
4939
|
-
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores,
|
|
4940
|
-
|
|
4933
|
+
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
4934
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
4935
|
+
evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
|
|
4941
4936
|
pipelineToYaml, importPipeline, yamlToPipeline,
|
|
4942
4937
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
4943
4938
|
};
|