evals-lab 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +95 -0
- package/README.md +17 -9
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +3 -7
- package/lab/demo/pipelines/demo-2.json +3 -7
- package/lab/evals-core.mjs +510 -444
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +35 -33
- package/lab/server.py +442 -115
- package/lab/web/dist/assets/gallery-B-7oyY37.js +3 -0
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/main-C_b7QoTv.css +1 -0
- package/lab/web/dist/assets/main-DyDG-V9N.js +21 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DsetJSXv.js +0 -3
- package/lab/web/dist/assets/main-B-VtDGxC.css +0 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +0 -19
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +0 -51
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
// WHAT IS NOT HERE. Preparing an image: the tab has a canvas and the
|
|
27
27
|
// script has not, so `prepare()` stays in the page and the script shells out
|
|
28
28
|
// to ImageMagick with the same arithmetic. Rendering: the tab writes HTML and
|
|
29
|
-
// the script writes JSON. Both read the same verdict out of
|
|
29
|
+
// the script writes JSON. Both read the same verdict out of the same Metrics.
|
|
30
30
|
//
|
|
31
31
|
// `runner-check.js` runs one set through two of the callers and asserts the
|
|
32
32
|
// verdicts and the totals are identical, because sharing a file is a claim
|
|
@@ -232,8 +232,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
232
232
|
|
|
233
233
|
|
|
234
234
|
|
|
235
|
-
/** What every
|
|
236
|
-
shown, an optional name ("
|
|
235
|
+
/** What every eval carries whatever its type: an id minted once and never
|
|
236
|
+
shown, an optional name ("Eval 2" when blank), and whether the evals after
|
|
237
237
|
it still read what it failed on (docs/pipeline-model.md §3). */
|
|
238
238
|
|
|
239
239
|
|
|
@@ -260,7 +260,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
260
260
|
|
|
261
261
|
|
|
262
262
|
|
|
263
|
-
/** Checks of every reply (
|
|
263
|
+
/** Checks of every reply (EVAL_TYPES.metrics): its own for every item, and
|
|
264
264
|
a case's own for its item; all must pass, or weighted points reach the
|
|
265
265
|
threshold. A model-graded one asks the grader. */
|
|
266
266
|
|
|
@@ -275,11 +275,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
275
275
|
|
|
276
276
|
|
|
277
277
|
|
|
278
|
-
/**
|
|
278
|
+
/** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
|
|
279
279
|
older document held (upgradePipeline reads it converted). */
|
|
280
280
|
|
|
281
281
|
|
|
282
|
-
/** A pipeline's
|
|
282
|
+
/** A pipeline's evals, in the order they read a run. */
|
|
283
283
|
|
|
284
284
|
|
|
285
285
|
|
|
@@ -482,75 +482,47 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
482
482
|
|
|
483
483
|
// ---- grading ----
|
|
484
484
|
|
|
485
|
-
/** A graded case
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
485
|
+
/** A graded case (dataset body version 5): an Item and the Metrics its
|
|
486
|
+
reply is held to.
|
|
487
|
+
|
|
488
|
+
The item is named by `item`: a dataset joins on an item's name in the
|
|
489
|
+
Source it grades, exactly, and nothing about that is an image -- the
|
|
490
|
+
next dataset addresses a row of a spreadsheet or a line of a log by the
|
|
491
|
+
same key. Version 4 called it `filename`, and the lab's first app had a
|
|
492
|
+
name of its own for it; `upgradeDatasetBody` reads both as `item`.
|
|
493
|
+
|
|
494
|
+
A case's metrics are what a good answer is: each a metric as a
|
|
495
|
+
pipeline's Metrics eval holds one, added to the eval's own for this item
|
|
496
|
+
when an eval names the dataset. One with weight 0 is watched -- reported,
|
|
497
|
+
never scored. */
|
|
496
498
|
|
|
497
499
|
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
500
|
+
|
|
501
|
+
|
|
501
502
|
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
503
|
+
|
|
504
|
+
|
|
512
505
|
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
506
|
|
|
518
507
|
|
|
519
508
|
|
|
520
|
-
/** A graded set: cases.
|
|
509
|
+
/** A graded set: a dataset's body, or anything holding its cases. */
|
|
521
510
|
|
|
522
511
|
|
|
523
512
|
|
|
524
513
|
|
|
525
514
|
|
|
526
|
-
/** One row the
|
|
527
|
-
a case of a set validateEvals accepts, so it has its id and
|
|
515
|
+
/** One row of the Library's Dataset group's cases table, as gradedSetFrom builds it:
|
|
516
|
+
a case of a set validateEvals accepts, so it has its id and item. */
|
|
528
517
|
|
|
529
518
|
|
|
530
|
-
|
|
519
|
+
|
|
531
520
|
|
|
532
521
|
|
|
533
522
|
|
|
534
523
|
/** A requirement as a score reports it: a term, or the group it was. */
|
|
535
524
|
|
|
536
525
|
|
|
537
|
-
/** How a graded case read a reply: the same shape both engines agree on. */
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
526
|
/** A run's totals, summed across cases. */
|
|
555
527
|
|
|
556
528
|
|
|
@@ -559,7 +531,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
559
531
|
|
|
560
532
|
|
|
561
533
|
|
|
562
|
-
/** The whole-run verdict of
|
|
534
|
+
/** The whole-run verdict of an eval type, where it has one of its own. */
|
|
563
535
|
|
|
564
536
|
|
|
565
537
|
|
|
@@ -570,7 +542,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
570
542
|
|
|
571
543
|
|
|
572
544
|
|
|
573
|
-
/** One thing
|
|
545
|
+
/** One thing an eval asserts, as its type names it: "Exact" "outdoor". */
|
|
574
546
|
|
|
575
547
|
|
|
576
548
|
|
|
@@ -599,7 +571,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
599
571
|
|
|
600
572
|
|
|
601
573
|
|
|
602
|
-
/** A per-item
|
|
574
|
+
/** A per-item eval's score as a stored row carries it. */
|
|
603
575
|
|
|
604
576
|
|
|
605
577
|
|
|
@@ -678,6 +650,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
678
650
|
more entry, and no reader changes. */
|
|
679
651
|
|
|
680
652
|
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
|
|
681
656
|
|
|
682
657
|
|
|
683
658
|
|
|
@@ -698,12 +673,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
698
673
|
|
|
699
674
|
|
|
700
675
|
|
|
701
|
-
/** What an earlier
|
|
676
|
+
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
702
677
|
|
|
703
678
|
|
|
704
679
|
|
|
705
680
|
|
|
706
|
-
/** One
|
|
681
|
+
/** One eval's reading of one scenario of a run (scenarioEvals). */
|
|
707
682
|
|
|
708
683
|
|
|
709
684
|
|
|
@@ -760,7 +735,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
760
735
|
|
|
761
736
|
// ---- the registries ----
|
|
762
737
|
|
|
763
|
-
/** An option a modifier or
|
|
738
|
+
/** An option a modifier or an eval type exposes for editing. */
|
|
764
739
|
|
|
765
740
|
|
|
766
741
|
|
|
@@ -820,7 +795,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
820
795
|
|
|
821
796
|
|
|
822
797
|
|
|
823
|
-
|
|
798
|
+
|
|
824
799
|
|
|
825
800
|
|
|
826
801
|
|
|
@@ -896,10 +871,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
896
871
|
|
|
897
872
|
|
|
898
873
|
|
|
899
|
-
/**
|
|
874
|
+
/** An eval type: what it checks of a document, and how it scores a run. */
|
|
900
875
|
|
|
901
876
|
|
|
902
|
-
|
|
877
|
+
|
|
903
878
|
|
|
904
879
|
|
|
905
880
|
|
|
@@ -909,11 +884,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
909
884
|
|
|
910
885
|
|
|
911
886
|
|
|
912
|
-
|
|
887
|
+
|
|
913
888
|
|
|
914
889
|
|
|
915
890
|
|
|
916
|
-
|
|
891
|
+
|
|
917
892
|
|
|
918
893
|
|
|
919
894
|
|
|
@@ -934,7 +909,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
934
909
|
|
|
935
910
|
|
|
936
911
|
|
|
937
|
-
/** What
|
|
912
|
+
/** What an eval's `read` is handed besides the reply: production's reply to
|
|
938
913
|
the item, and a grader, where the run has them. */
|
|
939
914
|
|
|
940
915
|
|
|
@@ -1004,42 +979,151 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
1004
979
|
* submitted against, so nothing grades from a file.
|
|
1005
980
|
*/
|
|
1006
981
|
|
|
1007
|
-
|
|
982
|
+
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
|
|
997
|
+
|
|
1008
998
|
|
|
1009
999
|
|
|
1010
1000
|
|
|
1001
|
+
/** An eval group's scoring (docs/pipeline-model.md §17). */
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
/** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
|
|
1008
|
+
named its Source; 4 and earlier, neither. */
|
|
1009
|
+
const DATASET_BODY_VERSION = 7 ;
|
|
1010
|
+
|
|
1011
|
+
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
1012
|
+
reply read as a list, as well as one read as text. */
|
|
1013
|
+
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
1014
|
+
|
|
1011
1015
|
/**
|
|
1012
|
-
* [body] as this version of a dataset (
|
|
1016
|
+
* [body] as this version of a dataset (7), from any earlier one. Version 1
|
|
1013
1017
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1014
1018
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1015
1019
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
1016
1020
|
* job (docs/pipeline-model.md §13): they leave, and the terms a case
|
|
1017
1021
|
* watches for are its `watch`. Version 3 held the `prompt` a new scenario
|
|
1018
|
-
* started from, which the Prompt library holds now: it leaves.
|
|
1019
|
-
*
|
|
1020
|
-
*
|
|
1021
|
-
*
|
|
1022
|
+
* started from, which the Prompt library holds now: it leaves. Version 4's
|
|
1023
|
+
* case named its item `filename` and said what a good answer is in
|
|
1024
|
+
* expectations; each becomes the metric it is (`caseMetrics`), `why` is the
|
|
1025
|
+
* `note`, `traits` go, and the body names no Source yet. Version 5 matched
|
|
1026
|
+
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1027
|
+
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1028
|
+
* the body says its version. Version 6 is a group's cases alone: version 7
|
|
1029
|
+
* adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
|
|
1030
|
+
* stored row is read that way rather than rewritten. Every reader of a
|
|
1031
|
+
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1032
|
+
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1033
|
+
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
1034
|
+
* same answer, and server.py's upgrade_body is its twin. Anything else
|
|
1035
|
+
* comes back as it was.
|
|
1022
1036
|
*/
|
|
1023
1037
|
function upgradeDatasetBody (body ) {
|
|
1024
1038
|
if (!isObj(body)) return body;
|
|
1025
|
-
if (
|
|
1026
|
-
if (!("prompt" in body)) return body;
|
|
1027
|
-
const { prompt: _library, ...rest } = body ;
|
|
1028
|
-
return rest ;
|
|
1029
|
-
}
|
|
1039
|
+
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1030
1040
|
const b = body ;
|
|
1031
|
-
//
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
return out;
|
|
1038
|
-
};
|
|
1041
|
+
// A body saying any other version is one this lab does not read.
|
|
1042
|
+
if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
|
|
1043
|
+
// A body that names its Source, even as null, is version 5.
|
|
1044
|
+
if ("source" in b) {
|
|
1045
|
+
return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
|
|
1046
|
+
}
|
|
1039
1047
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1040
1048
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1041
1049
|
if (!raw) return body;
|
|
1042
|
-
return {
|
|
1050
|
+
return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
/** A version-6 body as a version-7 eval group: scored All, the lab's
|
|
1054
|
+
grader, and no metrics of its own for every item or the whole run --
|
|
1055
|
+
what a Metrics eval naming the dataset with none of its own graded, which
|
|
1056
|
+
is what every Graded set converted to. */
|
|
1057
|
+
function groupOfV6(b ) {
|
|
1058
|
+
const { version: _v, ...rest } = b;
|
|
1059
|
+
return { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null, every: [], run: [],
|
|
1060
|
+
...rest } ;
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1063
|
+
/** A version-5 case as a version-6 one: each Contains metric says Ignore
|
|
1064
|
+
case, as version 5 matched a list's items whatever it said. A reply read
|
|
1065
|
+
as text did mind its setting, but a case's metrics were written for a
|
|
1066
|
+
list -- each one converted from version 4, and each the case form made. */
|
|
1067
|
+
function caseOfV5(c ) {
|
|
1068
|
+
if (!isObj(c) || !Array.isArray(c.metrics)) return c;
|
|
1069
|
+
return { ...c, metrics: c.metrics.map((m ) => (isObj(m) && CASE_FOLDING.includes(m.type) && m.ignoreCase !== true
|
|
1070
|
+
? { ...m, ignoreCase: true } : m)) };
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
// vocab: the names older versions gave these fields
|
|
1074
|
+
const CASE_RENAMED = { minTags: "minCount", maxTags: "maxCount", textInImage: "watch", photo: "filename" }; // vocab: as above
|
|
1075
|
+
/** What a version-4 case said, which its metrics say now. */
|
|
1076
|
+
const CASE_V4 = ["filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded"];
|
|
1077
|
+
|
|
1078
|
+
/** One case of any earlier version as a version-5 one: its id, item, todo
|
|
1079
|
+
and note, its expectations as metrics ahead of any metrics it held, and
|
|
1080
|
+
anything else it carried kept as it was. */
|
|
1081
|
+
function caseOfV4(c ) {
|
|
1082
|
+
if (!isObj(c)) return c;
|
|
1083
|
+
const was = {};
|
|
1084
|
+
for (const [k, v] of Object.entries(c)) {
|
|
1085
|
+
const key = CASE_RENAMED[k] ?? k;
|
|
1086
|
+
// A case naming its item both ways keeps the newer key's.
|
|
1087
|
+
if (!(key in was) || key === k) was[key] = v;
|
|
1088
|
+
}
|
|
1089
|
+
if (isStr(was.item)) delete was.filename;
|
|
1090
|
+
const out = {};
|
|
1091
|
+
if ("id" in was) out.id = was.id;
|
|
1092
|
+
out.item = isStr(was.item) ? was.item : isStr(was.filename) ? was.filename : "";
|
|
1093
|
+
out.todo = was.todo === true;
|
|
1094
|
+
out.note = isStr(was.note) ? was.note : isStr(was.why) ? was.why : "";
|
|
1095
|
+
out.metrics = [...caseMetrics(was), ...(Array.isArray(was.metrics) ? was.metrics : [])];
|
|
1096
|
+
for (const [k, v] of Object.entries(was)) {
|
|
1097
|
+
if (!(k in out) && !CASE_V4.includes(k)) out[k] = v;
|
|
1098
|
+
}
|
|
1099
|
+
return out;
|
|
1100
|
+
}
|
|
1101
|
+
|
|
1102
|
+
/** A version-4 case's expectations as the metrics that say the same:
|
|
1103
|
+
`expect` is Contains all, each `anyOf` group a Contains any, each
|
|
1104
|
+
forbidden term a Contains turned round with the `allow` phrases that
|
|
1105
|
+
excuse it, the count bounds an Item count, `discarded` the Discarded
|
|
1106
|
+
metric, and each watched term a Contains any of weight 0 -- reported,
|
|
1107
|
+
never scored. The order is the one scoreCase read them in, so a reason
|
|
1108
|
+
reads in the order it did. */
|
|
1109
|
+
function caseMetrics(c ) {
|
|
1110
|
+
const list = (v ) => (Array.isArray(v) ? v.filter(isStr) : []);
|
|
1111
|
+
if (c.discarded === true) return [{ type: "discarded" }];
|
|
1112
|
+
const out = [];
|
|
1113
|
+
const expect = list(c.expect), allow = list(c.allow);
|
|
1114
|
+
if (expect.length) out.push({ type: "contains-all", values: expect.join("\n") });
|
|
1115
|
+
for (const g of Array.isArray(c.anyOf) ? c.anyOf : []) {
|
|
1116
|
+
if (list(g).length) out.push({ type: "contains-any", values: list(g).join("\n") });
|
|
1117
|
+
}
|
|
1118
|
+
for (const t of list(c.forbid)) {
|
|
1119
|
+
// An exception excuses only the forbidden term inside it.
|
|
1120
|
+
const except = allow.filter(a => termIn([a], t));
|
|
1121
|
+
out.push({ type: "contains", value: t, not: true, ...(except.length ? { except: except.join("\n") } : {}) });
|
|
1122
|
+
}
|
|
1123
|
+
const min = Number.isInteger(c.minCount) ? c.minCount : null, max = Number.isInteger(c.maxCount) ? c.maxCount : null;
|
|
1124
|
+
if (min != null || max != null) out.push({ type: "item-count", min, max });
|
|
1125
|
+
for (const t of list(c.watch)) out.push({ type: "contains-any", values: t, weight: 0 });
|
|
1126
|
+
return out;
|
|
1043
1127
|
}
|
|
1044
1128
|
|
|
1045
1129
|
/** The rules a version-1 or version-2 dataset body held, or null: what a
|
|
@@ -1055,6 +1139,9 @@ function datasetRules(body ) {
|
|
|
1055
1139
|
|
|
1056
1140
|
|
|
1057
1141
|
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
|
|
1058
1145
|
|
|
1059
1146
|
|
|
1060
1147
|
|
|
@@ -1322,8 +1409,10 @@ function textPrompt(instruction , text ) {
|
|
|
1322
1409
|
}
|
|
1323
1410
|
|
|
1324
1411
|
// Mirrors TagRules' WORD. The unit the echo guard and the eval matcher both
|
|
1325
|
-
// work in, so that "New Zealand." and "new zealand" are the same two words
|
|
1326
|
-
|
|
1412
|
+
// work in, so that "New Zealand." and "new zealand" are the same two words --
|
|
1413
|
+
// unless [keepCase], for a metric whose Ignore case is off.
|
|
1414
|
+
const words = (s , keepCase = false) =>
|
|
1415
|
+
(keepCase ? String(s||"") : String(s||"").toLowerCase()).match(/[\p{L}\p{N}]+/gu) || [];
|
|
1327
1416
|
|
|
1328
1417
|
// The request Tagger builds, field for field -- when the target reads those
|
|
1329
1418
|
// fields. A llama.cpp server does; the hosted OpenAI-shaped providers do
|
|
@@ -2039,11 +2128,12 @@ function budgetLabel(target ) {
|
|
|
2039
2128
|
// A term is present when its words appear in some item, in order and adjacent:
|
|
2040
2129
|
// "cat" is in "cat" and in "tabby cat", and is not in "cathedral". Anything
|
|
2041
2130
|
// looser scores "new zealand" against "zealandia" and flatters every run.
|
|
2042
|
-
|
|
2043
|
-
|
|
2131
|
+
// Letter case counts only where a metric's Ignore case is off.
|
|
2132
|
+
function termIn(items , term , ignoreCase = true) {
|
|
2133
|
+
const t = words(term, !ignoreCase);
|
|
2044
2134
|
if (!t.length) return false;
|
|
2045
2135
|
return items.some(item => {
|
|
2046
|
-
const w = words(item);
|
|
2136
|
+
const w = words(item, !ignoreCase);
|
|
2047
2137
|
for (let i = 0; i + t.length <= w.length; i++) {
|
|
2048
2138
|
if (t.every((x, j) => w[i + j] === x)) return true;
|
|
2049
2139
|
}
|
|
@@ -2070,153 +2160,55 @@ function forbiddenIn(items , term , allow
|
|
|
2070
2160
|
* check green. `evals-check.js` calls this directly.
|
|
2071
2161
|
*/
|
|
2072
2162
|
function gradedSetFrom(ev ) {
|
|
2073
|
-
return (ev.cases || []).map(c => ({ ...c,
|
|
2163
|
+
return (ev.cases || []).map(c => ({ ...c, item: caseItem(c), half: "cases" }) );
|
|
2074
2164
|
}
|
|
2075
2165
|
|
|
2076
|
-
/**
|
|
2077
|
-
|
|
2078
|
-
|
|
2079
|
-
|
|
2080
|
-
|
|
2081
|
-
* case to an item -- the runner, the Datasets tab, a mapping against a Source
|
|
2082
|
-
* -- goes through here.
|
|
2083
|
-
*/
|
|
2084
|
-
function caseFile(kase ) { // vocab: the older spelling
|
|
2085
|
-
const name = kase?.filename ?? kase?.photo; // vocab: the older spelling
|
|
2086
|
-
return typeof name === "string" ? name : "";
|
|
2166
|
+
/** The item a case grades: its `item`, or "" for none. The one reader, so
|
|
2167
|
+
everything that joins a case to an item -- the runner, the Library, a
|
|
2168
|
+
run's Results -- joins on the same key, exactly. */
|
|
2169
|
+
function caseItem(kase ) {
|
|
2170
|
+
return isStr(kase?.item) ? kase .item : "";
|
|
2087
2171
|
}
|
|
2088
2172
|
|
|
2089
|
-
/**
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
|
|
2093
|
-
|
|
2094
|
-
|
|
2095
|
-
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
|
|
2101
|
-
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
if (k === OLD) { if (!("filename" in (c ))) out["filename"] = v; }
|
|
2110
|
-
else out[k] = v;
|
|
2111
|
-
}
|
|
2112
|
-
return out;
|
|
2113
|
-
}),
|
|
2114
|
-
};
|
|
2115
|
-
}
|
|
2173
|
+
/** One graded case's reading of a reply, now: its own metrics, scored as a
|
|
2174
|
+
Metrics eval scores them (all must pass; weight 0 is watched, never
|
|
2175
|
+
scored), with what they found, missed and invented carried up, and
|
|
2176
|
+
`watch` the readings of weight 0. What the page draws and the runner's
|
|
2177
|
+
case-by-case report writes. A model-graded metric needs a grader and the
|
|
2178
|
+
wait for one, so here it reads as not met: the runner's `--run` asks it. */
|
|
2179
|
+
|
|
2180
|
+
|
|
2181
|
+
|
|
2182
|
+
|
|
2183
|
+
|
|
2184
|
+
|
|
2185
|
+
|
|
2186
|
+
|
|
2187
|
+
|
|
2188
|
+
|
|
2189
|
+
|
|
2190
|
+
|
|
2191
|
+
|
|
2192
|
+
|
|
2116
2193
|
|
|
2117
|
-
|
|
2118
|
-
|
|
2119
|
-
|
|
2120
|
-
|
|
2121
|
-
|
|
2122
|
-
|
|
2123
|
-
|
|
2124
|
-
* were wanted has not done what was asked.
|
|
2125
|
-
*
|
|
2126
|
-
* Pure on purpose, and returning `reasons` as plain sentences rather than
|
|
2127
|
-
* markup. The dashboard was the first consumer; `run-evals.js` is the second
|
|
2128
|
-
* and CI the third, and neither can reach into a page for a rendered cell.
|
|
2129
|
-
* Issue #248 wants a failure written out as a task an agent can act on.
|
|
2130
|
-
*/
|
|
2131
|
-
function scoreCase(kase , res ) {
|
|
2132
|
-
// A case that expects its answer discarded passes on a discard and on
|
|
2133
|
-
// nothing else: the answer the job threw away is the finding.
|
|
2134
|
-
if (kase.discarded === true) {
|
|
2135
|
-
const thrown = !!res.error && res.error.startsWith("discarded: ");
|
|
2136
|
-
const n = (res.terms || []).length;
|
|
2137
|
-
return { pass: thrown, score: thrown ? 1 : 0, discarded: !!res.error, found: [], missed: [], invented: [],
|
|
2138
|
-
unmet: [], under: false, over: false, count: res.error ? 0 : n, watchFound: [], watchTotal: 0,
|
|
2139
|
-
reasons: thrown ? [] : [res.error ? `Error - ${res.error}` : `FAIL: kept ${n} items, and the case expects the answer discarded`] };
|
|
2140
|
-
}
|
|
2141
|
-
// A discarded answer is a failed item: whatever the pipeline produced
|
|
2142
|
-
// along the way does not count.
|
|
2143
|
-
const discarded = !!res.error;
|
|
2144
|
-
const terms = discarded ? [] : (res.terms || []);
|
|
2145
|
-
const expect = kase.expect || [], forbid = kase.forbid || [];
|
|
2146
|
-
|
|
2147
|
-
// `missed` stays populated for a discarded reply even though it reads as
|
|
2148
|
-
// vacuous, because the score depends on it: docs/datasets.md has such a reply
|
|
2149
|
-
// scoring zero against everything the entry asked for, groups included, and
|
|
2150
|
-
// `addToTally` gets there through `found + missed + invented`. Clear it and
|
|
2151
|
-
// a run that discarded every item would contribute nothing to the
|
|
2152
|
-
// total instead of contributing a nought, which flatters it.
|
|
2153
|
-
//
|
|
2154
|
-
// One member of a group is enough. Without this rule the set manufactures
|
|
2155
|
-
// failures out of synonyms.
|
|
2156
|
-
//
|
|
2157
|
-
// A requirement reads back as a term when it had one member and as the
|
|
2158
|
-
// group itself where a synonym list was allowed, so `missed` carries just
|
|
2159
|
-
// enough to state the reason: "FAIL: Missed dog" for a plain expect term,
|
|
2160
|
-
// "none of dog / puppy" for a group.
|
|
2161
|
-
//
|
|
2162
|
-
// Nothing satisfies a group when nothing was stored, so a discarded reply
|
|
2163
|
-
// misses every requirement rather than none -- the same cast that makes its
|
|
2164
|
-
// count bounds not breached by one. `unmet` is a finding only now that
|
|
2165
|
-
// `missed` names the unsatisfied groups itself, but the run report has
|
|
2166
|
-
// always carried it, so it stays.
|
|
2167
|
-
const met = (g ) => g.some(t => termIn(terms, t));
|
|
2168
|
-
const spoken = (g ) => g.length === 1 ? g[0] : g;
|
|
2169
|
-
const requirements = [...expect.map(t => [t]), ...(kase.anyOf || [])];
|
|
2170
|
-
const found = requirements.filter(met).map(spoken);
|
|
2171
|
-
const missed = requirements.filter(g => !met(g)).map(spoken);
|
|
2172
|
-
const invented = forbid.filter(t => forbiddenIn(terms, t, kase.allow));
|
|
2173
|
-
const unmet = discarded ? []
|
|
2174
|
-
: (kase.anyOf || []).filter(g => !g.some(t => termIn(terms, t)));
|
|
2175
|
-
|
|
2176
|
-
const n = terms.length;
|
|
2177
|
-
// A null bound is not checked -- and neither is a bound on a reply that was
|
|
2178
|
-
// discarded. Nothing was counted out and found wanting there; the reply was
|
|
2179
|
-
// thrown away before it had a count, and saying "0 items, wanted at least 4"
|
|
2180
|
-
// states a second failure that never happened.
|
|
2181
|
-
const under = !discarded && kase.minCount != null && n < kase.minCount;
|
|
2182
|
-
const over = !discarded && kase.maxCount != null && n > kase.maxCount;
|
|
2183
|
-
|
|
2184
|
-
const denom = requirements.length + invented.length;
|
|
2185
|
-
// A case that passes is one whose whole expectation was met, not one that
|
|
2186
|
-
// scored well: an unsatisfied group is in `missed` alongside any term
|
|
2187
|
-
// missed, so pass needs nothing more than the terms already covered.
|
|
2188
|
-
const pass = !discarded && !missed.length && !invented.length && !under && !over;
|
|
2189
|
-
|
|
2190
|
-
// `found` and `missed` are groups, not terms, so the reasons word them one
|
|
2191
|
-
// by one: a bare term under one heading, a group on its own line.
|
|
2192
|
-
const reasons = [];
|
|
2193
|
-
if (discarded) {
|
|
2194
|
-
// Everything else would be derived from this one fact -- there are no
|
|
2195
|
-
// terms -- and would bury it. `scoreHtml` above suppresses `missed` for
|
|
2196
|
-
// the same reason: the reason that matters is already on the row.
|
|
2197
|
-
reasons.push(`Error - ${res.error}`);
|
|
2198
|
-
} else {
|
|
2199
|
-
const plain = missed.filter(t => typeof t === "string");
|
|
2200
|
-
if (plain.length) reasons.push(`FAIL: Missed ${plain.join(", ")}`);
|
|
2201
|
-
for (const g of missed) {
|
|
2202
|
-
if (Array.isArray(g)) reasons.push(`none of ${g.join(" / ")}`);
|
|
2194
|
+
function readCase(kase , res , plain = false) {
|
|
2195
|
+
const input = metricInput(res , kase, { plain });
|
|
2196
|
+
const metrics = (Array.isArray(kase.metrics) ? kase.metrics : []).flatMap(m => {
|
|
2197
|
+
const r = readMetric(m, input, {});
|
|
2198
|
+
if (r && typeof (r ).then === "function") {
|
|
2199
|
+
return [{ type: m.type, label: METRICS[m.type]?.label ?? m.type, weight: typeof m.weight === "number" ? m.weight : 1,
|
|
2200
|
+
pass: false, score: 0, reason: "a model-graded metric needs a grader" } ];
|
|
2203
2201
|
}
|
|
2204
|
-
|
|
2205
|
-
|
|
2206
|
-
|
|
2207
|
-
|
|
2208
|
-
|
|
2209
|
-
|
|
2210
|
-
|
|
2211
|
-
|
|
2212
|
-
|
|
2213
|
-
|
|
2214
|
-
// beside an entry that asked for nothing -- a shape `evals-check.js` allows
|
|
2215
|
-
// even though every graded case now names an expectation.
|
|
2216
|
-
return { pass, score: denom ? found.length / denom : null, discarded,
|
|
2217
|
-
found, missed, invented, unmet, under, over, count: n,
|
|
2218
|
-
watchFound: watch.filter(t => termIn(terms, t)), watchTotal: watch.length,
|
|
2219
|
-
reasons };
|
|
2202
|
+
return r ? [r ] : [];
|
|
2203
|
+
});
|
|
2204
|
+
const s = scoreOf(metrics) ?? { pass: true, score: null, metrics };
|
|
2205
|
+
const counted = metrics.filter(r => r.weight !== 0);
|
|
2206
|
+
return { ...s, metrics, found: s.found ?? [], missed: s.missed ?? [], invented: s.invented ?? [],
|
|
2207
|
+
discarded: !!res.error, count: res.error ? 0 : (res.terms || []).length,
|
|
2208
|
+
// A reply the job threw away fails on that one fact, which would
|
|
2209
|
+
// be buried under everything that derives from it.
|
|
2210
|
+
reasons: res.error && !s.pass ? [`Error - ${res.error}`] : counted.filter(r => !r.pass).map(r => `${r.label}: ${r.reason}`),
|
|
2211
|
+
watch: metrics.filter(r => r.weight === 0) };
|
|
2220
2212
|
}
|
|
2221
2213
|
|
|
2222
2214
|
/**
|
|
@@ -2257,11 +2249,12 @@ const emptyTally = () => ({ ran: 0, passed: 0, found: 0, of: 0 });
|
|
|
2257
2249
|
// docs/datasets.md: an item naming more terms should weigh more than
|
|
2258
2250
|
// one naming fewer, and averaging percentages lets the easiest case carry the
|
|
2259
2251
|
// score.
|
|
2260
|
-
function addToTally(t , s
|
|
2252
|
+
function addToTally(t , s ) {
|
|
2253
|
+
const found = s.found?.length ?? 0;
|
|
2261
2254
|
t.ran++;
|
|
2262
2255
|
if (s.pass) t.passed++;
|
|
2263
|
-
t.found +=
|
|
2264
|
-
t.of +=
|
|
2256
|
+
t.found += found;
|
|
2257
|
+
t.of += found + (s.missed?.length ?? 0) + (s.invented?.length ?? 0);
|
|
2265
2258
|
}
|
|
2266
2259
|
|
|
2267
2260
|
// The headline percentage, in one place because it is a number and this file
|
|
@@ -2276,7 +2269,7 @@ function tallyPercent(t ) {
|
|
|
2276
2269
|
// A whole-run assertion with no graded set: All of / Any of / None of, a
|
|
2277
2270
|
// parse, a count, an exact reply and a length bound. It lives in the shared
|
|
2278
2271
|
// core because it is a pass/fail the tab, the worker and History all have to
|
|
2279
|
-
// agree on -- the same one-definition rule that keeps
|
|
2272
|
+
// agree on -- the same one-definition rule that keeps the case reader here.
|
|
2280
2273
|
function parseCount(raw , parse ) {
|
|
2281
2274
|
// An unparsed reply is one result, not no result. It used to return null,
|
|
2282
2275
|
// which made a Count of 1 fail as "null results" against a reply that
|
|
@@ -2666,10 +2659,10 @@ function applyModifiers (list , kind ,
|
|
|
2666
2659
|
// queue and run-evals.js alike, and every one of them reads it through the
|
|
2667
2660
|
// functions below rather than through a translation of its own.
|
|
2668
2661
|
//
|
|
2669
|
-
// The lab is generic, so what a reply is, what
|
|
2662
|
+
// The lab is generic, so what a reply is, what an eval scores and how a value
|
|
2670
2663
|
// is changed are registry entries. The ones here are the lab's own, and so
|
|
2671
2664
|
// are kinds/list.ts's -- the List kind and its modifiers. A dataset
|
|
2672
|
-
// registers nothing: it is data a graded
|
|
2665
|
+
// registers nothing: it is data a graded eval names by id, and a run carries.
|
|
2673
2666
|
|
|
2674
2667
|
// 2: a job's mappings are a list of { name, type, … } (tokenMappings), where
|
|
2675
2668
|
// version 1 kept them as two maps (tokens: { values, blocks }).
|
|
@@ -2685,14 +2678,16 @@ function applyModifiers (list , kind ,
|
|
|
2685
2678
|
// 7: `chains` are `jobs`: the field renames and each job's `type` is "job".
|
|
2686
2679
|
// 8: a job is its steps -- job 1's Attach Content (the pipeline's content), a
|
|
2687
2680
|
// Call (Prompt: the image flag and token mappings), Read Reply (the output).
|
|
2688
|
-
// 9:
|
|
2681
|
+
// 9: an eval is Metrics. A Single Test and a Graded set are read as the Metrics
|
|
2689
2682
|
// they convert to (LEGACY_TESTS' toMetrics, proven equal by
|
|
2690
2683
|
// metrics-parity-check.js).
|
|
2691
2684
|
// 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
|
|
2692
2685
|
// Content, the flow step, Attach Image, Token mappings) and Responses (Read
|
|
2693
2686
|
// as, Response Format Validation, modifiers) -- and its call is gone: each
|
|
2694
2687
|
// scenario is a target, whose own step in each job is what it sends there.
|
|
2695
|
-
|
|
2688
|
+
// 11: `tests` are `evals`: the key renames and nothing in an eval changes,
|
|
2689
|
+
// so a stored result's scores, keyed by eval id, read as they did.
|
|
2690
|
+
const PIPELINE_VERSION = 12 ;
|
|
2696
2691
|
|
|
2697
2692
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2698
2693
|
// to know which registered it.
|
|
@@ -2700,7 +2695,7 @@ const STEP_TYPES = Object.create(null);
|
|
|
2700
2695
|
const CONTENT_TYPES = Object.create(null);
|
|
2701
2696
|
const OUTPUT_KINDS = Object.create(null);
|
|
2702
2697
|
const MODIFIERS = Object.create(null);
|
|
2703
|
-
const
|
|
2698
|
+
const EVAL_TYPES = Object.create(null);
|
|
2704
2699
|
const METRICS = Object.create(null);
|
|
2705
2700
|
const SOURCE_TYPES = Object.create(null);
|
|
2706
2701
|
|
|
@@ -2714,11 +2709,15 @@ let defaultOutputKind = "text";
|
|
|
2714
2709
|
const defaultKind = () => defaultOutputKind;
|
|
2715
2710
|
const kindOf = (st ) => st.kind ?? defaultOutputKind;
|
|
2716
2711
|
|
|
2712
|
+
/** A module's eval types, under either spelling (Kinds.testTypes). */
|
|
2713
|
+
const evalTypesOf = (k ) =>
|
|
2714
|
+
({ ...(k.testTypes || {}), ...(k.evalTypes || {}) });
|
|
2715
|
+
|
|
2717
2716
|
/** A module's entries into the registries, in one call. */
|
|
2718
2717
|
function registerKinds(k ) {
|
|
2719
2718
|
Object.assign(OUTPUT_KINDS, k.outputKinds || {});
|
|
2720
2719
|
Object.assign(MODIFIERS, k.modifiers || {});
|
|
2721
|
-
Object.assign(
|
|
2720
|
+
Object.assign(EVAL_TYPES, evalTypesOf(k));
|
|
2722
2721
|
Object.assign(SOURCE_TYPES, k.sourceTypes || {});
|
|
2723
2722
|
Object.assign(METRICS, k.metrics || {});
|
|
2724
2723
|
if (k.defaultKind) defaultOutputKind = k.defaultKind;
|
|
@@ -2749,7 +2748,7 @@ function pluginHost(pluginId ) {
|
|
|
2749
2748
|
registerKinds(k) {
|
|
2750
2749
|
taken(OUTPUT_KINDS, "output kind", Object.keys(k.outputKinds || {}));
|
|
2751
2750
|
taken(MODIFIERS, "modifier", Object.keys(k.modifiers || {}));
|
|
2752
|
-
taken(
|
|
2751
|
+
taken(EVAL_TYPES, "eval type", Object.keys(evalTypesOf(k)));
|
|
2753
2752
|
taken(METRICS, "metric", Object.keys(k.metrics || {}));
|
|
2754
2753
|
// The server keeps its own copy of the Source types, to refuse a row of
|
|
2755
2754
|
// one it does not know, and a plugin never reaches the server's code.
|
|
@@ -3015,11 +3014,9 @@ LEGACY_TESTS.single = {
|
|
|
3015
3014
|
},
|
|
3016
3015
|
};
|
|
3017
3016
|
|
|
3018
|
-
// A dataset's cases, scored item by item
|
|
3019
|
-
//
|
|
3020
|
-
//
|
|
3021
|
-
// yields, so it accepts every kind that yields any: plain text has none, and
|
|
3022
|
-
// would fail every case.
|
|
3017
|
+
// A dataset's cases, scored item by item. The eval references the dataset
|
|
3018
|
+
// the way a pipeline references a Source, and the run carries the body it
|
|
3019
|
+
// was submitted against.
|
|
3023
3020
|
LEGACY_TESTS.graded = {
|
|
3024
3021
|
label: "Graded set",
|
|
3025
3022
|
fields: ["type", "dataset"],
|
|
@@ -3027,16 +3024,15 @@ LEGACY_TESTS.graded = {
|
|
|
3027
3024
|
validate(t, ctx, bad){
|
|
3028
3025
|
const d = t.dataset;
|
|
3029
3026
|
if (!isRef(d) || (d.version != null && !isStr(d.version))) {
|
|
3030
|
-
return void bad.push("a graded
|
|
3027
|
+
return void bad.push("a graded eval has to name its dataset as { id, name }");
|
|
3031
3028
|
}
|
|
3032
3029
|
if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
|
|
3033
3030
|
},
|
|
3034
|
-
score: (t, kase, res) => scoreCase(kase, res),
|
|
3035
3031
|
};
|
|
3036
3032
|
|
|
3037
3033
|
// Metrics: checks of each reply, deterministic or model-graded, from the
|
|
3038
3034
|
// METRICS registry (metrics/builtin.ts registers the lab's own), with the
|
|
3039
|
-
//
|
|
3035
|
+
// eval's own list for every item and a case's `metrics` for its item. Scored
|
|
3040
3036
|
// all-must-pass -- every metric passes -- or weighted: points, each metric's
|
|
3041
3037
|
// score times its weight, against a threshold (#150's points, a negative
|
|
3042
3038
|
// weight taking them away).
|
|
@@ -3045,6 +3041,13 @@ LEGACY_TESTS.graded = {
|
|
|
3045
3041
|
const METRIC_FIELDS = ["type", "not", "weight", "metric", "of", "every"];
|
|
3046
3042
|
const SCORING_MODES = ["all", "weighted"];
|
|
3047
3043
|
|
|
3044
|
+
/** A metric's list option -- Contains all's values, Contains's exceptions --
|
|
3045
|
+
one entry a line, or a list of them. */
|
|
3046
|
+
function metricLines(v ) {
|
|
3047
|
+
const all = Array.isArray(v) ? v.map(x => String(x ?? "")) : String(v ?? "").split("\n");
|
|
3048
|
+
return all.map(x => x.trim()).filter(Boolean);
|
|
3049
|
+
}
|
|
3050
|
+
|
|
3048
3051
|
/** What is wrong with a list of metrics, as sentences naming [at]. */
|
|
3049
3052
|
function metricsProblems(list , at , bad ) {
|
|
3050
3053
|
if (!Array.isArray(list)) return void bad.push(`${at}: metrics has to be a list`);
|
|
@@ -3114,14 +3117,16 @@ function readMetric(m , input , ctx )
|
|
|
3114
3117
|
} catch (e) { return timed(failed(e)); }
|
|
3115
3118
|
}
|
|
3116
3119
|
|
|
3117
|
-
/**
|
|
3120
|
+
/** An eval's score from its metrics' readings, the way [mode] says: all must
|
|
3118
3121
|
pass, or weighted points against [threshold]. What a case metric found,
|
|
3119
3122
|
missed and invented is carried up, for a run's totals. Null for none. */
|
|
3120
3123
|
function scoreOf(metrics , mode = "all", threshold = null) {
|
|
3121
3124
|
if (!metrics.length) return null;
|
|
3122
|
-
|
|
3123
|
-
|
|
3124
|
-
|
|
3125
|
+
// A watched reading (weight 0) is reported, and counts toward nothing.
|
|
3126
|
+
const scored = metrics.filter(r => r.weight !== 0);
|
|
3127
|
+
const detail = scored.some(r => r.found || r.missed || r.invented) ? {
|
|
3128
|
+
found: scored.flatMap(r => r.found ?? []), missed: scored.flatMap(r => r.missed ?? []),
|
|
3129
|
+
invented: scored.flatMap(r => r.invented ?? []) } : {};
|
|
3125
3130
|
if (mode === "weighted") {
|
|
3126
3131
|
const points = metrics.reduce((sum, r) => sum + r.weight * r.score, 0);
|
|
3127
3132
|
return { pass: points >= (threshold ?? 0), score: points, points: true, metrics, ...detail };
|
|
@@ -3164,7 +3169,7 @@ function readRun(list , run ) {
|
|
|
3164
3169
|
});
|
|
3165
3170
|
}
|
|
3166
3171
|
|
|
3167
|
-
|
|
3172
|
+
EVAL_TYPES.metrics = {
|
|
3168
3173
|
label: "Metrics",
|
|
3169
3174
|
description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
|
|
3170
3175
|
fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
|
|
@@ -3188,9 +3193,9 @@ TEST_TYPES.metrics = {
|
|
|
3188
3193
|
if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
|
|
3189
3194
|
if (t.threshold != null && typeof t.threshold !== "number") bad.push("the Metrics' threshold has to be a number");
|
|
3190
3195
|
if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
|
|
3191
|
-
// The
|
|
3196
|
+
// The eval's own grader, or the lab's.
|
|
3192
3197
|
const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3193
|
-
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the
|
|
3198
|
+
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
|
|
3194
3199
|
// A grader is asked words, and needs a model to ask.
|
|
3195
3200
|
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3196
3201
|
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
@@ -3205,59 +3210,35 @@ TEST_TYPES.metrics = {
|
|
|
3205
3210
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3206
3211
|
.filter(Boolean).join(" ") })),
|
|
3207
3212
|
profiles: t => (isRef(t.grader) ? [t.grader] : []),
|
|
3208
|
-
// A run carries the lab's grader on
|
|
3213
|
+
// A run carries the lab's grader on an eval that names none and may ask one:
|
|
3209
3214
|
// a model-graded metric of its own, or a case's, over a dataset.
|
|
3210
3215
|
resolve: (t, ctx) => (!isRef(t.grader) && isRef(ctx.grader) && (gradedIn(t.metrics) || isRef(t.dataset))
|
|
3211
3216
|
? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
|
|
3212
3217
|
wholeRun: t => t.over === "run",
|
|
3213
3218
|
// Over the whole run: every metric over the replies together, or each alone.
|
|
3214
|
-
verdict(t, ress, kind)
|
|
3215
|
-
|
|
3216
|
-
const metrics = readRun(t.metrics || [], run);
|
|
3217
|
-
const s = scoreOf(metrics, t.mode, t.threshold);
|
|
3218
|
-
const off = metrics.filter(r => !r.pass);
|
|
3219
|
-
return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
|
|
3220
|
-
ran: run.replies.length,
|
|
3221
|
-
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3222
|
-
},
|
|
3223
|
-
// Rule m{x} is the test's own metric x, read in that place on every item.
|
|
3219
|
+
verdict: (t, ress, kind) => wholeRunVerdict(t.metrics || [], ress, kind, t.mode, t.threshold),
|
|
3220
|
+
// Rule m{x} is the eval's own metric x, read in that place on every item.
|
|
3224
3221
|
// A score stored before Metrics -- a Graded set's, read as its conversion --
|
|
3225
3222
|
// has no readings of its own: its one metric's reading is the score's.
|
|
3226
3223
|
ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
|
|
3227
3224
|
: Array.isArray(t?.metrics) && t.metrics.length === 1 ? score.pass : null),
|
|
3228
|
-
// Every item, with its case's own metrics where
|
|
3229
|
-
|
|
3225
|
+
// Every item, with its case's own metrics where the eval names the
|
|
3226
|
+
// dataset and the item has a case there.
|
|
3227
|
+
read: async (t, kase, res, more) => (await readMetrics([...(t.metrics || []), ...((isRef(t.dataset) && kase?.metrics) || [])],
|
|
3230
3228
|
metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
|
|
3231
3229
|
};
|
|
3232
3230
|
|
|
3233
3231
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3234
|
-
// What the
|
|
3235
|
-
//
|
|
3236
|
-
//
|
|
3237
|
-
// each is the core's own matcher.
|
|
3238
|
-
|
|
3239
|
-
// An item's case, as the Graded set scores it: its expectations, forbidden
|
|
3240
|
-
// terms and count bounds, with what it found, missed and invented kept.
|
|
3241
|
-
METRICS.case = {
|
|
3242
|
-
label: "Matches its case",
|
|
3243
|
-
description: "Scores the reply against the item's case in the dataset: the items it expects, forbids and how many.",
|
|
3244
|
-
perItem: true,
|
|
3245
|
-
needsTerms: true,
|
|
3246
|
-
options: [],
|
|
3247
|
-
defaults: () => ({}),
|
|
3248
|
-
score(input) {
|
|
3249
|
-
if (!input.kase) return null;
|
|
3250
|
-
const s = scoreCase(input.kase, { ...(input.error ? { error: input.error } : {}), terms: input.terms });
|
|
3251
|
-
return { pass: s.pass, score: s.score ?? (s.pass ? 1 : 0), reason: s.reasons.join(" · ") || "all matched",
|
|
3252
|
-
found: s.found, missed: s.missed, invented: s.invented };
|
|
3253
|
-
},
|
|
3254
|
-
};
|
|
3232
|
+
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3233
|
+
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
3234
|
+
// rather than in metrics/builtin.ts because each is the core's own matcher.
|
|
3255
3235
|
|
|
3256
3236
|
// Items the reply holds -- all of them, or any -- matched as the lab matches
|
|
3257
3237
|
// a term; a kind that yields none is matched in the replies' text instead,
|
|
3258
3238
|
// ignoring case, as the Single Test did.
|
|
3259
3239
|
METRICS["has-items"] = {
|
|
3260
3240
|
label: "Has items",
|
|
3241
|
+
family: "The reply's text",
|
|
3261
3242
|
description: "Passes when the reply holds all, or any, of the items listed.",
|
|
3262
3243
|
options: [{ key: "values", label: "Items", type: "textarea" },
|
|
3263
3244
|
{ key: "need", label: "Needs", type: "select", choices: [{ value: "all", label: "all" }, { value: "any", label: "any" }] }],
|
|
@@ -3277,6 +3258,7 @@ METRICS["has-items"] = {
|
|
|
3277
3258
|
// A reply's length in characters, once trimmed.
|
|
3278
3259
|
METRICS.length = {
|
|
3279
3260
|
label: "Length",
|
|
3261
|
+
family: "The reply's text",
|
|
3280
3262
|
description: "Compares the reply's length in characters.",
|
|
3281
3263
|
options: [{ key: "op", label: "Is", type: "select", choices: LENGTH_OPS.map(([value, label]) => ({ value, label })) },
|
|
3282
3264
|
{ key: "n", label: "Characters", type: "number" }],
|
|
@@ -3292,6 +3274,7 @@ METRICS.length = {
|
|
|
3292
3274
|
// as the count says.
|
|
3293
3275
|
METRICS["parse-count"] = {
|
|
3294
3276
|
label: "Parses",
|
|
3277
|
+
family: "The result",
|
|
3295
3278
|
description: "Passes when the reply reads as the format chosen, holding the number of results set.",
|
|
3296
3279
|
// Unformatted is one result a reply, as the Single Test counted it.
|
|
3297
3280
|
options: [{ key: "parse", label: "As", type: "select", choices: [{ value: "", label: "Unformatted" }, { value: "csv", label: "csv" },
|
|
@@ -3310,12 +3293,13 @@ METRICS["parse-count"] = {
|
|
|
3310
3293
|
},
|
|
3311
3294
|
};
|
|
3312
3295
|
|
|
3313
|
-
/** What every
|
|
3296
|
+
/** What every eval carries, kept across a conversion. */
|
|
3314
3297
|
const common = (t ) => ({ id: t.id, name: t.name ?? "", continueOnFailure: t.continueOnFailure !== false });
|
|
3315
3298
|
|
|
3316
|
-
// A Graded set is a Metrics
|
|
3299
|
+
// A Graded set is a Metrics eval over the same dataset, with none of its own:
|
|
3300
|
+
// each case's metrics are what it scored (dataset-parity-check.js).
|
|
3317
3301
|
LEGACY_TESTS.graded .toMetrics = t => ({ ...common(t), type: "metrics", mode: "all", threshold: null, grader: null,
|
|
3318
|
-
dataset: t.dataset, over: "item", metrics: [
|
|
3302
|
+
dataset: t.dataset, over: "item", metrics: [] });
|
|
3319
3303
|
|
|
3320
3304
|
// A Single Test is Metrics over the whole run: its lists over the run's items
|
|
3321
3305
|
// together, and exact, length and parse over each reply alone.
|
|
@@ -3333,7 +3317,39 @@ LEGACY_TESTS.single .toMetrics = t => {
|
|
|
3333
3317
|
return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
|
|
3334
3318
|
};
|
|
3335
3319
|
|
|
3336
|
-
/**
|
|
3320
|
+
/** [list] read over a whole run's replies so far, scored the way [mode]
|
|
3321
|
+
says: a Metrics eval's verdict over the run, and an eval group's Whole run. */
|
|
3322
|
+
function wholeRunVerdict(list , ress , kind ,
|
|
3323
|
+
mode = "all", threshold = null) {
|
|
3324
|
+
const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
|
|
3325
|
+
const metrics = readRun(list, run);
|
|
3326
|
+
const s = scoreOf(metrics, mode, threshold);
|
|
3327
|
+
const off = metrics.filter(r => !r.pass);
|
|
3328
|
+
return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
|
|
3329
|
+
ran: run.replies.length,
|
|
3330
|
+
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3331
|
+
}
|
|
3332
|
+
|
|
3333
|
+
/**
|
|
3334
|
+
* An eval group's verdict on one item (docs/pipeline-model.md §17): its
|
|
3335
|
+
* Every item metrics and [kase]'s own -- the item's case in the group, or
|
|
3336
|
+
* null where it has none -- read of the reply under the group's scoring.
|
|
3337
|
+
* The reading a Metrics eval naming the group as its dataset gives, which
|
|
3338
|
+
* groups-check.js holds it to. Null where no metric had anything to read.
|
|
3339
|
+
*/
|
|
3340
|
+
function readGroup(group , kase , res , more = {}) {
|
|
3341
|
+
return readMetrics([...group.every, ...(kase?.metrics ?? [])], metricInput(res, kase, more), graderCtx(group, more),
|
|
3342
|
+
group.scoring.mode, group.scoring.threshold);
|
|
3343
|
+
}
|
|
3344
|
+
|
|
3345
|
+
/** An eval group's Whole run verdict, from the replies so far: its `run`
|
|
3346
|
+
metrics under its scoring. Null for a group with no Whole run metrics,
|
|
3347
|
+
which has nothing to say of a run. */
|
|
3348
|
+
function readGroupRun(group , ress , kind = null) {
|
|
3349
|
+
return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
|
|
3350
|
+
}
|
|
3351
|
+
|
|
3352
|
+
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3337
3353
|
function graderCtx(t , more ) {
|
|
3338
3354
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3339
3355
|
return ask ? { ask } : {};
|
|
@@ -3343,12 +3359,14 @@ function graderCtx(t , more ) {
|
|
|
3343
3359
|
for one with none set: what History prints beside its name. */
|
|
3344
3360
|
function metricSummary(m ) {
|
|
3345
3361
|
const entry = METRICS[m.type];
|
|
3346
|
-
// Each option as it reads in the editor: a choice's label, a box
|
|
3347
|
-
//
|
|
3362
|
+
// Each option as it reads in the editor: a choice's label, a box by its
|
|
3363
|
+
// own label where it is not as a new metric has it (ticked, or "off"),
|
|
3364
|
+
// text by its first line.
|
|
3365
|
+
const fresh = entry?.defaults() ?? {};
|
|
3348
3366
|
const said = (entry?.options || []).map(o => {
|
|
3349
3367
|
const v = m[o.key];
|
|
3350
3368
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3351
|
-
if (o.type === "checkbox") return v ? o.label.toLowerCase() :
|
|
3369
|
+
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
3352
3370
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3353
3371
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3354
3372
|
}).filter(Boolean);
|
|
@@ -3531,7 +3549,7 @@ STEP_TYPES.readAs = {
|
|
|
3531
3549
|
// Response Format Validation: the kind a reply is read as.
|
|
3532
3550
|
STEP_TYPES.readReply = {
|
|
3533
3551
|
label: "Read Reply", slot: "responses", rank: 1,
|
|
3534
|
-
description: "How the reply is read before
|
|
3552
|
+
description: "How the reply is read before evals and later jobs see it.",
|
|
3535
3553
|
in: "text", out: step => step.out?.kind,
|
|
3536
3554
|
// Read inside the stage: runPipeline parses each reply as it comes back.
|
|
3537
3555
|
apply: "runPipeline",
|
|
@@ -3787,7 +3805,7 @@ STEP_TYPES.job = {
|
|
|
3787
3805
|
if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
|
|
3788
3806
|
const steps = c.steps;
|
|
3789
3807
|
// A job's steps are the entries with a stage; one without (a job, the
|
|
3790
|
-
//
|
|
3808
|
+
// evals) or of no type the lab has is not one.
|
|
3791
3809
|
const unknown = steps.find(st => !slotOf(st));
|
|
3792
3810
|
if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
|
|
3793
3811
|
if (steps.some(st => slotOf(st) === "target")) {
|
|
@@ -3808,54 +3826,55 @@ STEP_TYPES.job = {
|
|
|
3808
3826
|
}
|
|
3809
3827
|
},
|
|
3810
3828
|
};
|
|
3811
|
-
// What
|
|
3812
|
-
const
|
|
3829
|
+
// What an eval carries besides its type's own fields.
|
|
3830
|
+
const EVAL_FIELDS = ["id", "name", "continueOnFailure"];
|
|
3813
3831
|
|
|
3814
|
-
/**
|
|
3815
|
-
const
|
|
3816
|
-
(doc.
|
|
3832
|
+
/** Eval [j]'s name, or the number it has always shown. */
|
|
3833
|
+
const evalLabel = (doc , j ) =>
|
|
3834
|
+
(doc.evals?.[j]?.name || "").trim() || `Eval ${j + 1}`;
|
|
3817
3835
|
|
|
3818
|
-
/**
|
|
3836
|
+
/** An eval type that settles over the whole run rather than item by item. */
|
|
3819
3837
|
const isWholeRun = (type , t ) =>
|
|
3820
3838
|
type?.wholeRun && t ? type.wholeRun(t) : !!type?.verdict && !type?.score;
|
|
3821
3839
|
|
|
3822
3840
|
/**
|
|
3823
|
-
* A document's
|
|
3824
|
-
* the run document it was submitted with, so a run from before version
|
|
3825
|
-
*
|
|
3841
|
+
* A document's evals as a list, whatever version wrote it: a queue row keeps
|
|
3842
|
+
* the run document it was submitted with, so a run from before version 11
|
|
3843
|
+
* spells them `tests`, and one from before version 6 holds one or null, read
|
|
3844
|
+
* here as the upgrade reads it (testsList, evalsKey).
|
|
3826
3845
|
*/
|
|
3827
|
-
function
|
|
3828
|
-
const t = doc?.tests;
|
|
3846
|
+
function evalsOf(doc ) {
|
|
3847
|
+
const t = doc?.evals ?? doc?.tests;
|
|
3829
3848
|
if (Array.isArray(t)) return t ;
|
|
3830
3849
|
return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
|
|
3831
3850
|
}
|
|
3832
3851
|
|
|
3833
|
-
/** The dataset a document's
|
|
3852
|
+
/** The dataset a document's evals grade against, where one does: a run
|
|
3834
3853
|
grades against one (validatePipeline says so), so the first names it. */
|
|
3835
|
-
function
|
|
3836
|
-
const t =
|
|
3854
|
+
function evalsDataset(doc ) {
|
|
3855
|
+
const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
|
|
3837
3856
|
return t && "dataset" in t ? t.dataset : null;
|
|
3838
3857
|
}
|
|
3839
3858
|
|
|
3840
|
-
STEP_TYPES.
|
|
3859
|
+
STEP_TYPES.evals = {
|
|
3841
3860
|
in: "results", out: "verdict",
|
|
3842
3861
|
apply: "score",
|
|
3843
|
-
validate(
|
|
3844
|
-
if (!Array.isArray(
|
|
3862
|
+
validate(evals, ctx, bad, lastKind, doc){
|
|
3863
|
+
if (!Array.isArray(evals)) return void bad.push("evals has to be a list, empty for an unscored run");
|
|
3845
3864
|
const seen = new Set ();
|
|
3846
|
-
|
|
3847
|
-
const at =
|
|
3848
|
-
const type = isObj(t) &&
|
|
3865
|
+
evals.forEach((t , j ) => {
|
|
3866
|
+
const at = evalLabel(doc, j);
|
|
3867
|
+
const type = isObj(t) && EVAL_TYPES[t.type];
|
|
3849
3868
|
if (!type) return void bad.push(`${at} is of type "${t?.type}", which is not one this lab has`);
|
|
3850
3869
|
const before = bad.length;
|
|
3851
3870
|
if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
|
|
3852
|
-
else if (seen.has(t.id)) bad.push(`${at} has the id of another
|
|
3871
|
+
else if (seen.has(t.id)) bad.push(`${at} has the id of another eval`);
|
|
3853
3872
|
else seen.add(t.id);
|
|
3854
3873
|
if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
|
|
3855
3874
|
if (typeof t.continueOnFailure !== "boolean") bad.push(`${at}: continueOnFailure has to be true or false`);
|
|
3856
|
-
onlyFields(t, at, [...type.fields, ...
|
|
3875
|
+
onlyFields(t, at, [...type.fields, ...EVAL_FIELDS], bad);
|
|
3857
3876
|
type.validate(t, ctx, bad);
|
|
3858
|
-
// Which kinds
|
|
3877
|
+
// Which kinds an eval scores is only worth saying of an eval that is whole.
|
|
3859
3878
|
if (bad.length > before) return;
|
|
3860
3879
|
const accepts = typeof type.accepts === "function" ? type.accepts(t) : type.accepts;
|
|
3861
3880
|
if (accepts && lastKind && OUTPUT_KINDS[lastKind] && !accepts.includes(lastKind)) {
|
|
@@ -3864,14 +3883,14 @@ STEP_TYPES.tests = {
|
|
|
3864
3883
|
}
|
|
3865
3884
|
});
|
|
3866
3885
|
// A run is handed one dataset's body to grade against (server-side-runs §4).
|
|
3867
|
-
const named = new Set(
|
|
3868
|
-
if (named.size > 1) bad.push("the
|
|
3886
|
+
const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
|
|
3887
|
+
if (named.size > 1) bad.push("the evals grade against one dataset at a time");
|
|
3869
3888
|
},
|
|
3870
3889
|
};
|
|
3871
3890
|
|
|
3872
3891
|
// ---- the document -------------------------------------------------------------
|
|
3873
3892
|
|
|
3874
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "
|
|
3893
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
|
|
3875
3894
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3876
3895
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3877
3896
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3888,7 +3907,7 @@ function versionProblem(doc ) {
|
|
|
3888
3907
|
/** What an upgrade from version 2 needs from outside the document: the rules
|
|
3889
3908
|
of the dataset a pipeline was graded against, which the retired
|
|
3890
3909
|
version-2 list kind read every reply under. Without them a job is upgraded with no
|
|
3891
|
-
rules, as a run with no graded
|
|
3910
|
+
rules, as a run with no graded eval parsed. */
|
|
3892
3911
|
|
|
3893
3912
|
|
|
3894
3913
|
|
|
@@ -3913,7 +3932,8 @@ function versionProblem(doc ) {
|
|
|
3913
3932
|
* as it was, for versionProblem to name. A copy: the caller's document is not
|
|
3914
3933
|
* touched. From version 5, its test -- or none -- becomes a list of one (or
|
|
3915
3934
|
* none), the test's id `t1`. From version 6, `chains` are `jobs`: the field
|
|
3916
|
-
* renames and every job's `type` becomes `"job"`.
|
|
3935
|
+
* renames and every job's `type` becomes `"job"`. From version 10, its
|
|
3936
|
+
* `tests` are `evals` (evalsKey).
|
|
3917
3937
|
*/
|
|
3918
3938
|
/** Every id a current document must carry, minted for the ones [doc] lacks.
|
|
3919
3939
|
An id it already has is kept. */
|
|
@@ -3949,6 +3969,9 @@ function profileRefs(doc ) {
|
|
|
3949
3969
|
}
|
|
3950
3970
|
}
|
|
3951
3971
|
|
|
3972
|
+
/** [doc] with its profile references cut (profileRefs), for chaining. */
|
|
3973
|
+
const cutRefs = (doc ) => { profileRefs(doc); return doc; };
|
|
3974
|
+
|
|
3952
3975
|
/** Whether any profile reference in a pipeline holds more than { id, name }. */
|
|
3953
3976
|
function fatProfileRef(doc ) {
|
|
3954
3977
|
const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
|
|
@@ -3967,9 +3990,9 @@ function testsList(doc ) {
|
|
|
3967
3990
|
return doc;
|
|
3968
3991
|
}
|
|
3969
3992
|
|
|
3970
|
-
/** A new
|
|
3971
|
-
function
|
|
3972
|
-
const own =
|
|
3993
|
+
/** A new eval of [type] at the lab's defaults, named [name], continuing on failure. */
|
|
3994
|
+
function newEval(type , name = "", fields = {}) {
|
|
3995
|
+
const own = EVAL_TYPES[type]?.defaults?.() ?? { type };
|
|
3973
3996
|
return { ...own, ...fields, type, id: newId(), name, continueOnFailure: true } ;
|
|
3974
3997
|
}
|
|
3975
3998
|
|
|
@@ -4067,6 +4090,29 @@ function targetsFromScenarios(next ) {
|
|
|
4067
4090
|
return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
|
|
4068
4091
|
}
|
|
4069
4092
|
|
|
4093
|
+
/** Version 10 to 11: `tests` are `evals` -- the key renames in place, so an
|
|
4094
|
+
upgraded document reads the same, key for key, and each eval is as it
|
|
4095
|
+
was. Every version before 11 passes through here. */
|
|
4096
|
+
function evalsKey(next ) {
|
|
4097
|
+
next.version = PIPELINE_VERSION;
|
|
4098
|
+
// `tests` is the spelling of every version before 11.
|
|
4099
|
+
if (!("tests" in next)) return next;
|
|
4100
|
+
return Object.fromEntries(Object.entries(next).filter(([k]) => k !== "evals")
|
|
4101
|
+
.map(([k, v]) => (k === "tests" ? ["evals", v] : [k, v])));
|
|
4102
|
+
}
|
|
4103
|
+
|
|
4104
|
+
/** Version 11 to 12: a Contains metric's Ignore case holds where the reply
|
|
4105
|
+
is matched item by item, as it does where it is matched as text, and it
|
|
4106
|
+
is kept as written. Until version 11's last hours (#199) every Contains
|
|
4107
|
+
metric matched the reply's text and minded its Ignore case, so what a
|
|
4108
|
+
stored metric says is what its author meant; only the item-by-item
|
|
4109
|
+
matching #199 added, case-blind for a few hours, read it otherwise.
|
|
4110
|
+
Every version before 12 ends here. */
|
|
4111
|
+
function caseAsWritten(next ) {
|
|
4112
|
+
next.version = PIPELINE_VERSION;
|
|
4113
|
+
return next;
|
|
4114
|
+
}
|
|
4115
|
+
|
|
4070
4116
|
function upgradePipeline (doc , ctx = {}) {
|
|
4071
4117
|
// A current document is read as it is, but for a profile reference the Runs
|
|
4072
4118
|
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
@@ -4077,11 +4123,29 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
4077
4123
|
out = clone(out) ;
|
|
4078
4124
|
profileRefs(out);
|
|
4079
4125
|
}
|
|
4080
|
-
return localSteps(out, ctx) ;
|
|
4126
|
+
return localSteps(withoutCaseMetric(out), ctx) ;
|
|
4081
4127
|
}
|
|
4082
|
-
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9].includes(doc.version )) return doc;
|
|
4083
|
-
|
|
4084
|
-
|
|
4128
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
|
|
4129
|
+
if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
|
|
4130
|
+
// Every version before 10 reads as version 9 first, then as 10, then as 11
|
|
4131
|
+
// and 12.
|
|
4132
|
+
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4133
|
+
// earlier ones are cut on their way through nineOf.
|
|
4134
|
+
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4135
|
+
return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
|
|
4136
|
+
}
|
|
4137
|
+
|
|
4138
|
+
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
4139
|
+
are metrics now (dataset body version 5), which an eval naming the
|
|
4140
|
+
dataset adds for each item, so the case's own metrics carry what that
|
|
4141
|
+
metric scored. Read so at every version, the current one included, as a
|
|
4142
|
+
document saved before it went still holds it. The same document where
|
|
4143
|
+
none does. */
|
|
4144
|
+
function withoutCaseMetric(doc ) {
|
|
4145
|
+
const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
|
|
4146
|
+
if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
|
|
4147
|
+
return { ...doc, evals: doc.evals.map((t ) => (has(t)
|
|
4148
|
+
? { ...t, metrics: t.metrics.filter((m ) => !(isObj(m) && m.type === "case")) } : t)) };
|
|
4085
4149
|
}
|
|
4086
4150
|
|
|
4087
4151
|
/** [doc] with each step asked of a profile whose type a target step stands
|
|
@@ -4165,7 +4229,7 @@ function tokenMappingsFromV1(set ) {
|
|
|
4165
4229
|
function blankPipeline(opts = {}) {
|
|
4166
4230
|
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
4167
4231
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
4168
|
-
jobs: [job], targets: [],
|
|
4232
|
+
jobs: [job], targets: [], evals: [] };
|
|
4169
4233
|
}
|
|
4170
4234
|
|
|
4171
4235
|
/**
|
|
@@ -4303,7 +4367,7 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4303
4367
|
: `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
|
|
4304
4368
|
}
|
|
4305
4369
|
});
|
|
4306
|
-
STEP_TYPES.
|
|
4370
|
+
STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
4307
4371
|
if (bad.length) return bad;
|
|
4308
4372
|
|
|
4309
4373
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
@@ -4400,9 +4464,9 @@ function profileIds(doc )
|
|
|
4400
4464
|
if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
4401
4465
|
}
|
|
4402
4466
|
}
|
|
4403
|
-
// Then the ones
|
|
4404
|
-
for (const t of
|
|
4405
|
-
for (const ref of
|
|
4467
|
+
// Then the ones an eval asks (a grader), so a run carries them too.
|
|
4468
|
+
for (const t of evalsOf(doc)) {
|
|
4469
|
+
for (const ref of EVAL_TYPES[t.type]?.profiles?.(t) ?? []) if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
4406
4470
|
}
|
|
4407
4471
|
return ids;
|
|
4408
4472
|
}
|
|
@@ -4415,9 +4479,9 @@ function profileIds(doc )
|
|
|
4415
4479
|
function resolvePipeline(doc ,
|
|
4416
4480
|
ctx = {}) {
|
|
4417
4481
|
const run = clone(doc) ;
|
|
4418
|
-
// What the lab supplies
|
|
4482
|
+
// What the lab supplies an eval -- its grader -- before the profiles it
|
|
4419
4483
|
// asks are carried.
|
|
4420
|
-
run.
|
|
4484
|
+
run.evals = (Array.isArray(run.evals) ? run.evals : []).map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, ctx) ?? t);
|
|
4421
4485
|
run.profiles = {};
|
|
4422
4486
|
for (const id of profileIds(run)) {
|
|
4423
4487
|
const p = ctx.profiles?.(id);
|
|
@@ -4426,7 +4490,7 @@ function resolvePipeline(doc ,
|
|
|
4426
4490
|
const content = contentOf(run) ;
|
|
4427
4491
|
if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
|
|
4428
4492
|
if (ctx.datasetVersion) {
|
|
4429
|
-
for (const t of run.
|
|
4493
|
+
for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
|
|
4430
4494
|
}
|
|
4431
4495
|
if (isStr(ctx.comment)) run.comment = ctx.comment;
|
|
4432
4496
|
return run;
|
|
@@ -4440,7 +4504,7 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4440
4504
|
delete doc.plugins;
|
|
4441
4505
|
const content = contentOf(doc) ;
|
|
4442
4506
|
if (content) { delete content.files; delete content.revs; }
|
|
4443
|
-
for (const t of doc.
|
|
4507
|
+
for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
|
|
4444
4508
|
return doc;
|
|
4445
4509
|
}
|
|
4446
4510
|
|
|
@@ -4498,7 +4562,7 @@ function importPipeline(input , ctx = {}) {
|
|
|
4498
4562
|
if (st?.profile) st.profile = remap(st.profile, "profile");
|
|
4499
4563
|
}
|
|
4500
4564
|
}
|
|
4501
|
-
for (const t of Array.isArray(next.
|
|
4565
|
+
for (const t of Array.isArray(next.evals) ? next.evals : []) {
|
|
4502
4566
|
if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
|
|
4503
4567
|
}
|
|
4504
4568
|
// Its references are this lab's now, so a step asking this lab's Echo
|
|
@@ -4530,7 +4594,7 @@ function mintIds(doc ) {
|
|
|
4530
4594
|
for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
|
|
4531
4595
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4532
4596
|
}
|
|
4533
|
-
for (const t of Array.isArray(doc.
|
|
4597
|
+
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4534
4598
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4535
4599
|
}
|
|
4536
4600
|
return doc;
|
|
@@ -4608,14 +4672,14 @@ function modifierSummary(m ) {
|
|
|
4608
4672
|
return said.length ? said.join(", ") : "on";
|
|
4609
4673
|
}
|
|
4610
4674
|
|
|
4611
|
-
// ---- the
|
|
4612
|
-
//
|
|
4675
|
+
// ---- the evals, in order (docs/pipeline-model.md §3) ---------------------------
|
|
4676
|
+
// Evals only read a run: none changes what a later one sees, and none stops
|
|
4613
4677
|
// the model being sent the next item. What order changes is Continue on
|
|
4614
|
-
// failure. A per-item
|
|
4678
|
+
// failure. A per-item eval that fails and does not continue stops the evals
|
|
4615
4679
|
// after it for that item alone -- they read Skipped there, and a whole-run
|
|
4616
|
-
//
|
|
4680
|
+
// eval after it pools the items it did not stop. A whole-run eval settles
|
|
4617
4681
|
// once every item is in, and one that fails then and does not continue
|
|
4618
|
-
// leaves every
|
|
4682
|
+
// leaves every eval after it Skipped.
|
|
4619
4683
|
|
|
4620
4684
|
const SKIPPED = Object.freeze({ skipped: true });
|
|
4621
4685
|
const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
@@ -4623,16 +4687,16 @@ const isSkipped = (s ) => isObj(s) && s.skipped === true;
|
|
|
4623
4687
|
const failedScore = (s ) => !!s && !isSkipped(s) && !s.pass;
|
|
4624
4688
|
|
|
4625
4689
|
/**
|
|
4626
|
-
* One reply's scores under a run's per-item
|
|
4627
|
-
*
|
|
4690
|
+
* One reply's scores under a run's per-item evals, by eval id, in order: a
|
|
4691
|
+
* eval with no case to score leaves no entry, and one after a failure that
|
|
4628
4692
|
* does not continue reads Skipped. Null where nothing was scored.
|
|
4629
4693
|
*/
|
|
4630
|
-
function itemScores(run , kase , res )
|
|
4694
|
+
function itemScores(run , kase , res ) {
|
|
4631
4695
|
if (!kase) return null;
|
|
4632
|
-
const out
|
|
4696
|
+
const out = {};
|
|
4633
4697
|
let stopped = false;
|
|
4634
|
-
for (const t of
|
|
4635
|
-
const type =
|
|
4698
|
+
for (const t of evalsOf(run)) {
|
|
4699
|
+
const type = EVAL_TYPES[t.type];
|
|
4636
4700
|
if (!type?.score) continue;
|
|
4637
4701
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4638
4702
|
const s = type.score(t, kase, res);
|
|
@@ -4650,19 +4714,19 @@ function productionOf(run , record )
|
|
|
4650
4714
|
}
|
|
4651
4715
|
|
|
4652
4716
|
/**
|
|
4653
|
-
* itemScores for the runner, which can wait:
|
|
4717
|
+
* itemScores for the runner, which can wait: an eval that reads every item
|
|
4654
4718
|
* (`read`: the Metrics, which may ask a grader) scores one with no case too.
|
|
4655
4719
|
*/
|
|
4656
4720
|
async function itemScoresAsync(run , kase , res ,
|
|
4657
4721
|
more = {}) {
|
|
4658
4722
|
const out = {};
|
|
4659
4723
|
let stopped = false;
|
|
4660
|
-
for (const t of
|
|
4661
|
-
const type =
|
|
4724
|
+
for (const t of evalsOf(run)) {
|
|
4725
|
+
const type = EVAL_TYPES[t.type];
|
|
4662
4726
|
if (isWholeRun(type, t) || (!type?.read && !(type?.score && kase))) continue;
|
|
4663
4727
|
if (stopped) { out[t.id] = SKIPPED; continue; }
|
|
4664
|
-
//
|
|
4665
|
-
//
|
|
4728
|
+
// An eval with nothing to read on this item leaves no entry, as a graded
|
|
4729
|
+
// eval does on an item with no case.
|
|
4666
4730
|
const s = type.read ? await type.read(t, kase ?? null, res, { plain: !OUTPUT_KINDS[lastKind(run) ?? ""]?.terms, ...more })
|
|
4667
4731
|
: type.score (t, kase , res);
|
|
4668
4732
|
if (!s) continue;
|
|
@@ -4673,22 +4737,22 @@ async function itemScoresAsync(run , kase ,
|
|
|
4673
4737
|
}
|
|
4674
4738
|
|
|
4675
4739
|
/**
|
|
4676
|
-
* Every
|
|
4740
|
+
* Every eval's reading of scenario [i] of a run, in the run's order, from
|
|
4677
4741
|
* the items so far. [settled] says every item is in: only then has a
|
|
4678
|
-
* whole-run
|
|
4742
|
+
* whole-run eval settled, so only then does its failure skip the evals
|
|
4679
4743
|
* after it.
|
|
4680
4744
|
*/
|
|
4681
|
-
function
|
|
4745
|
+
function scenarioEvals(run , i , items ,
|
|
4682
4746
|
settled = true) {
|
|
4683
4747
|
const cells = (items || []).map(it => (it && !it.unrun ? it.scenarios?.[i] : undefined));
|
|
4684
|
-
// Which items a per-item
|
|
4748
|
+
// Which items a per-item eval has stopped so far, for the evals after it.
|
|
4685
4749
|
const stopped = cells.map(() => false);
|
|
4686
4750
|
let skipRest = false;
|
|
4687
|
-
const
|
|
4688
|
-
return
|
|
4689
|
-
const type =
|
|
4751
|
+
const evals = evalsOf(run);
|
|
4752
|
+
return evals.map((t, j) => {
|
|
4753
|
+
const type = EVAL_TYPES[t.type];
|
|
4690
4754
|
const whole = isWholeRun(type, t);
|
|
4691
|
-
const base = { id: t.id, label:
|
|
4755
|
+
const base = { id: t.id, label: evalLabel({ evals }, j), whole, skipped: skipRest,
|
|
4692
4756
|
verdict: null , rules: type?.rules?.(t) ?? [], items: cells.map(() => null) ,
|
|
4693
4757
|
ran: 0, passed: 0, skippedItems: 0 };
|
|
4694
4758
|
if (skipRest) {
|
|
@@ -4715,7 +4779,7 @@ function scenarioTests(run , i , items
|
|
|
4715
4779
|
});
|
|
4716
4780
|
}
|
|
4717
4781
|
|
|
4718
|
-
/** A scenario's pass or fail over every
|
|
4782
|
+
/** A scenario's pass or fail over every eval that read it: null where none
|
|
4719
4783
|
has anything to say yet. */
|
|
4720
4784
|
function scenarioPasses(outcomes ) {
|
|
4721
4785
|
let said = false;
|
|
@@ -4756,95 +4820,95 @@ function validateEvals(ev , files
|
|
|
4756
4820
|
return ["the graded set has to be a JSON object with a `cases` list"];
|
|
4757
4821
|
}
|
|
4758
4822
|
if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
|
|
4823
|
+
if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
|
|
4824
|
+
// An eval group's own sections (version 7): what a Metrics eval held.
|
|
4825
|
+
if (ev.scoring != null) {
|
|
4826
|
+
const sc = ev.scoring;
|
|
4827
|
+
if (!isObj(sc) || !SCORING_MODES.includes(sc.mode)) bad.push(`a group is scored ${SCORING_MODES.join(" or ")}`);
|
|
4828
|
+
else if (sc.mode === "weighted" && typeof sc.threshold !== "number") bad.push("a group scored in points needs Pass at: the points an item has to reach");
|
|
4829
|
+
else if (sc.threshold != null && typeof sc.threshold !== "number") bad.push("a group's Pass at has to be a number");
|
|
4830
|
+
}
|
|
4831
|
+
if (ev.grader != null && !isRef(ev.grader)) bad.push("a group names its grader as { id, name }, or null");
|
|
4832
|
+
if (ev.every != null) metricsProblems(ev.every, "Every item", bad);
|
|
4833
|
+
if (ev.run != null) {
|
|
4834
|
+
metricsProblems(ev.run, "Whole run", bad);
|
|
4835
|
+
// Read once over every reply: nothing that reads one item, or asks a grader.
|
|
4836
|
+
for (const m of Array.isArray(ev.run) ? ev.run : []) {
|
|
4837
|
+
const entry = isObj(m) ? METRICS[m.type] : undefined;
|
|
4838
|
+
if (entry?.graded) bad.push(`Whole run: ${entry.label} is model-graded, so it cannot read a whole run`);
|
|
4839
|
+
else if (entry?.perItem) bad.push(`Whole run: ${entry.label} reads one item at a time, so it cannot read a whole run`);
|
|
4840
|
+
}
|
|
4841
|
+
}
|
|
4759
4842
|
if (bad.length) return bad;
|
|
4760
4843
|
|
|
4761
4844
|
const graded = ev.cases || [];
|
|
4762
4845
|
// Identity first: every rule below reports which case is at fault, so a
|
|
4763
4846
|
// case with no usable id makes the rest of the report unreadable.
|
|
4764
|
-
const seen = new
|
|
4847
|
+
const seen = new Set ();
|
|
4765
4848
|
for (const c of graded) {
|
|
4766
|
-
{
|
|
4767
|
-
|
|
4768
|
-
|
|
4769
|
-
continue;
|
|
4770
|
-
}
|
|
4771
|
-
if (typeof c.id !== "string" || !c.id.trim()) bad.push("a case has no id");
|
|
4772
|
-
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4773
|
-
else seen.set(c.id, "cases");
|
|
4774
|
-
if (!caseFile(c).trim()) {
|
|
4775
|
-
bad.push(`${c.id || "a case"} names no file`);
|
|
4776
|
-
}
|
|
4777
|
-
for (const key of ["expect", "forbid", "allow", "anyOf", "watch", "traits"]) {
|
|
4778
|
-
if (c[key] != null && !Array.isArray(c[key])) bad.push(`${c.id}: ${key} has to be a list`);
|
|
4779
|
-
}
|
|
4780
|
-
for (const key of ["minCount", "maxCount"]) {
|
|
4781
|
-
if (c[key] != null && !Number.isInteger(c[key])) bad.push(`${c.id}: ${key} has to be a whole number`);
|
|
4782
|
-
}
|
|
4783
|
-
if (c.discarded != null && c.discarded !== true) bad.push(`${c.id}: discarded is true or left out`);
|
|
4784
|
-
// A case's own metrics, which a Metrics test adds to its own for this item.
|
|
4785
|
-
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4849
|
+
if (!isObj(c)) {
|
|
4850
|
+
bad.push("cases holds something that is not a case");
|
|
4851
|
+
continue;
|
|
4786
4852
|
}
|
|
4853
|
+
if (!isStr(c.id) || !c.id.trim()) bad.push("a case has no id");
|
|
4854
|
+
else if (seen.has(c.id)) bad.push(`${c.id} is graded twice`);
|
|
4855
|
+
else seen.add(c.id);
|
|
4856
|
+
if (!caseItem(c).trim()) bad.push(`${c.id || "a case"} names no item`);
|
|
4857
|
+
if (c.todo != null && typeof c.todo !== "boolean") bad.push(`${c.id}: todo is true or false`);
|
|
4858
|
+
if (c.note != null && !isStr(c.note)) bad.push(`${c.id}: note has to be text`);
|
|
4859
|
+
if (c.metrics != null) metricsProblems(c.metrics, String(c.id), bad);
|
|
4787
4860
|
}
|
|
4788
4861
|
if (bad.length) return bad;
|
|
4789
4862
|
|
|
4790
|
-
//
|
|
4791
|
-
// the case is unpassable however well the model
|
|
4863
|
+
// What a case says, read from its metrics: a term the scorer cannot match
|
|
4864
|
+
// can never be produced, so the case is unpassable however well the model
|
|
4865
|
+
// answers -- and the same goes for a bound no reply can meet.
|
|
4792
4866
|
const matchable = (term ) => termIn([String(term).trim()], term);
|
|
4793
|
-
|
|
4867
|
+
const values = (m ) => metricLines(m.type === "contains" ? m.value : m.values);
|
|
4794
4868
|
for (const c of graded) {
|
|
4795
4869
|
if (c.todo) continue;
|
|
4796
4870
|
const say = (m ) => bad.push(`${c.id}: ${m}`);
|
|
4797
|
-
const
|
|
4798
|
-
|
|
4799
|
-
if (
|
|
4871
|
+
const scored = (c.metrics ?? []).filter(m => m.weight !== 0);
|
|
4872
|
+
if (!scored.length) { say("graded but states nothing to expect"); continue; }
|
|
4873
|
+
if (scored.some(m => m.type === "discarded" && !m.not)) {
|
|
4800
4874
|
// What is discarded holds nothing, so there is nothing else to expect.
|
|
4801
|
-
if (
|
|
4802
|
-
say("expects its answer discarded, and states something the answer should hold too");
|
|
4803
|
-
}
|
|
4875
|
+
if (scored.length > 1) say("expects its answer discarded, and states something the answer should hold too");
|
|
4804
4876
|
continue;
|
|
4805
4877
|
}
|
|
4806
|
-
const
|
|
4807
|
-
|
|
4808
|
-
for (const t of
|
|
4809
|
-
|
|
4810
|
-
|
|
4811
|
-
|
|
4812
|
-
for (const
|
|
4813
|
-
}
|
|
4814
|
-
// An exception excuses only the forbidden terms inside it, so one that
|
|
4815
|
-
// holds none of them changes nothing and reads as if it did.
|
|
4816
|
-
for (const a of c.allow || []) {
|
|
4817
|
-
if (!forbid.some((t ) => termIn([a], t))) say(`allows ${a}, which holds nothing it forbids`);
|
|
4878
|
+
const wants = scored.filter(m => !m.not && (m.type === "contains-all" || m.type === "contains-any"));
|
|
4879
|
+
const forbids = scored.filter(m => m.not && m.type === "contains");
|
|
4880
|
+
for (const m of wants) for (const t of values(m)) if (!matchable(t)) say(`expects ${t}, which its own scorer cannot match`);
|
|
4881
|
+
// An exception excuses only the forbidden term inside it, so one that
|
|
4882
|
+
// holds it not changes nothing and reads as if it did.
|
|
4883
|
+
for (const m of forbids) {
|
|
4884
|
+
for (const e of metricLines(m.except)) if (!termIn([e], String(m.value ?? ""), m.ignoreCase === true)) say(`allows ${e}, which holds nothing it forbids`);
|
|
4818
4885
|
}
|
|
4819
4886
|
// A term on both lists cannot be produced and cannot be withheld.
|
|
4820
|
-
|
|
4821
|
-
|
|
4822
|
-
|
|
4823
|
-
|
|
4824
|
-
|
|
4825
|
-
|
|
4826
|
-
|
|
4827
|
-
|
|
4828
|
-
|
|
4829
|
-
|
|
4830
|
-
|
|
4831
|
-
|
|
4832
|
-
|
|
4833
|
-
const things = expect.length + anyOf.length;
|
|
4834
|
-
if (hi != null && things > hi) {
|
|
4835
|
-
say(`asks for ${things} things and maxCount ${hi} admits ${hi}`);
|
|
4887
|
+
const forbidden = new Set(forbids.map(m => String(m.value ?? "")));
|
|
4888
|
+
for (const m of wants) for (const t of values(m)) if (forbidden.has(t)) say(`${t} is both expected and forbidden`);
|
|
4889
|
+
for (const m of scored.filter(m => m.type === "item-count" && !m.not)) {
|
|
4890
|
+
const lo = m.min == null || m.min === "" ? null : Number(m.min), hi = m.max == null || m.max === "" ? null : Number(m.max);
|
|
4891
|
+
if (lo != null && hi != null && lo > hi) say(`at least ${lo} is above at most ${hi}`);
|
|
4892
|
+
// A ceiling below one says no answer is acceptable, which is a case
|
|
4893
|
+
// that can never pass rather than a strict one.
|
|
4894
|
+
if (hi != null && hi < 1) say(`at most ${hi} leaves no answer that could pass`);
|
|
4895
|
+
// A case that names more distinct things than the reply may carry, in
|
|
4896
|
+
// the scorer's own counting of a thing (#486: each term Contains all
|
|
4897
|
+
// wants is one, each Contains any is one), can never pass.
|
|
4898
|
+
const things = wants.reduce((n, w) => n + (w.type === "contains-all" ? values(w).length : 1), 0);
|
|
4899
|
+
if (hi != null && things > hi) say(`asks for ${things} things and at most ${hi} admits ${hi}`);
|
|
4836
4900
|
}
|
|
4837
4901
|
}
|
|
4838
4902
|
|
|
4839
|
-
// One
|
|
4840
|
-
//
|
|
4903
|
+
// One item, one case. An item here twice is two cases of it, graded
|
|
4904
|
+
// separately, and both would be listed.
|
|
4841
4905
|
const where = new Map ();
|
|
4842
4906
|
for (const c of graded) {
|
|
4843
|
-
const name =
|
|
4907
|
+
const name = caseItem(c);
|
|
4844
4908
|
const counted = where.get(name);
|
|
4845
4909
|
if (counted) {
|
|
4846
|
-
bad.push(`${name} is graded twice — one
|
|
4847
|
-
+ `two
|
|
4910
|
+
bad.push(`${name} is graded twice — one item, `
|
|
4911
|
+
+ `two cases. Grade it once.`);
|
|
4848
4912
|
} else where.set(name, true);
|
|
4849
4913
|
}
|
|
4850
4914
|
const gradedAt = (name ) => where.has(name);
|
|
@@ -4865,7 +4929,7 @@ function validateEvals(ev , files
|
|
|
4865
4929
|
for (const f of shown) {
|
|
4866
4930
|
if (!gradedAt(f)) {
|
|
4867
4931
|
bad.push(`${f} is in the Source and this dataset does not grade it, `
|
|
4868
|
-
+ `so the
|
|
4932
|
+
+ `so the Cases view does not list it`);
|
|
4869
4933
|
}
|
|
4870
4934
|
}
|
|
4871
4935
|
return bad;
|
|
@@ -4887,7 +4951,9 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4887
4951
|
const want = Number(n), warn = [];
|
|
4888
4952
|
for (const c of ev.cases || []) {
|
|
4889
4953
|
if (!c || c.todo) continue;
|
|
4890
|
-
const things = (c.
|
|
4954
|
+
const things = (Array.isArray(c.metrics) ? c.metrics : [])
|
|
4955
|
+
.filter(m => !m.not && m.weight !== 0 && (m.type === "contains-all" || m.type === "contains-any"))
|
|
4956
|
+
.reduce((k, m) => k + (m.type === "contains-all" ? metricLines(m.values).length : 1), 0);
|
|
4891
4957
|
if (things > want) {
|
|
4892
4958
|
warn.push(`${c.id}: asks for ${things} things and the prompt asks for up to ${want}`);
|
|
4893
4959
|
}
|
|
@@ -4906,7 +4972,7 @@ function evalsWarnings(ev , prompt ) {
|
|
|
4906
4972
|
* `evals-check.js` asserts the round trip against the real file.
|
|
4907
4973
|
*/
|
|
4908
4974
|
function evalsJson(ev ) {
|
|
4909
|
-
return JSON.stringify(
|
|
4975
|
+
return JSON.stringify(ev, null, 2) + "\n";
|
|
4910
4976
|
}
|
|
4911
4977
|
|
|
4912
4978
|
|
|
@@ -4920,14 +4986,14 @@ registerMetrics({ registerKinds });
|
|
|
4920
4986
|
export {
|
|
4921
4987
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
4922
4988
|
loopReplyError, preparedSize,
|
|
4923
|
-
termIn, forbiddenIn,
|
|
4989
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4924
4990
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
4925
4991
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
4926
4992
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
4927
4993
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
4928
4994
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
4929
4995
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
4930
|
-
PIPELINE_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS,
|
|
4996
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
|
|
4931
4997
|
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
4932
4998
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
4933
4999
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
@@ -4935,9 +5001,9 @@ CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWord
|
|
|
4935
5001
|
contentOf, withContent, replyOf,
|
|
4936
5002
|
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
4937
5003
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
4938
|
-
blankPipeline, upgradePipeline, fatProfileRef, newId,
|
|
4939
|
-
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores,
|
|
4940
|
-
|
|
5004
|
+
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5005
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
5006
|
+
evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
|
|
4941
5007
|
pipelineToYaml, importPipeline, yamlToPipeline,
|
|
4942
5008
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
4943
5009
|
};
|