evals-lab 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +77 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +353 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/kinds/list.mjs +2 -1
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +220 -83
- package/lab/server.py +617 -176
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
import yaml from "./js-yaml.mjs";
|
|
36
36
|
|
|
37
37
|
import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
|
|
38
|
+
import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
|
|
39
|
+
import { flowApi, } from "./flows/flowApi.mjs";
|
|
38
40
|
import registerMetrics from "./metrics/builtin.mjs";
|
|
39
41
|
import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
|
|
40
42
|
|
|
@@ -543,6 +545,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
543
545
|
|
|
544
546
|
|
|
545
547
|
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
|
|
546
554
|
|
|
547
555
|
|
|
548
556
|
|
|
@@ -646,9 +654,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
646
654
|
|
|
647
655
|
|
|
648
656
|
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
|
|
652
660
|
|
|
653
661
|
|
|
654
662
|
|
|
@@ -662,6 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
662
670
|
|
|
663
671
|
|
|
664
672
|
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
|
|
665
677
|
|
|
666
678
|
|
|
667
679
|
/** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
|
|
@@ -775,16 +787,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
775
787
|
|
|
776
788
|
// ---- the registries ----
|
|
777
789
|
|
|
778
|
-
/** An option a modifier or an eval type exposes for editing.
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
790
|
+
/** An option a modifier or an eval type exposes for editing. `hint` says
|
|
791
|
+
what it does in one short line, under its control, where the label
|
|
792
|
+
cannot: a switch's effect, not why it exists. */
|
|
793
|
+
|
|
794
|
+
|
|
795
|
+
|
|
782
796
|
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
|
|
786
800
|
|
|
787
|
-
|
|
801
|
+
|
|
788
802
|
|
|
789
803
|
/** What an output kind made of a reply. */
|
|
790
804
|
|
|
@@ -878,11 +892,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
878
892
|
|
|
879
893
|
|
|
880
894
|
/** A kind of Source: what one of its items is, and what the page and the
|
|
881
|
-
runner do with it. A reader asks the entry; it never branches on an id.
|
|
895
|
+
runner do with it. A reader asks the entry; it never branches on an id.
|
|
896
|
+
The row's `type` is the kind; a workflow kind's platform -- which flow
|
|
897
|
+
engine a row speaks to -- is `config.platform`, a WORKFLOW_PLATFORMS id. */
|
|
882
898
|
|
|
883
899
|
|
|
884
900
|
|
|
885
901
|
|
|
902
|
+
|
|
903
|
+
|
|
886
904
|
|
|
887
905
|
|
|
888
906
|
|
|
@@ -891,10 +909,68 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
891
909
|
|
|
892
910
|
|
|
893
911
|
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
|
|
894
916
|
|
|
895
917
|
|
|
918
|
+
|
|
919
|
+
|
|
920
|
+
|
|
896
921
|
|
|
897
922
|
|
|
923
|
+
/** What a workflow of one platform -- a flow engine -- answers: the pieces
|
|
924
|
+
that were Power-Automate-specific, lifted behind the platform id so
|
|
925
|
+
generic code asks the entry and never names a platform. Power Automate is
|
|
926
|
+
the one built; a second (Logic App, n8n) is one more entry. */
|
|
927
|
+
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
/** A platform's management API reader (flowApi.ts for Power Automate); its
|
|
963
|
+
shape is the platform's own, so generic code holds it opaquely. */
|
|
964
|
+
|
|
965
|
+
|
|
966
|
+
const WORKFLOW_PLATFORMS = Object.create(null);
|
|
967
|
+
/** The platform a workflow Source speaks to, from its `config.platform`;
|
|
968
|
+
undefined for a row that is not a workflow or names an unknown platform. */
|
|
969
|
+
const platformOf = (src ) => {
|
|
970
|
+
const id = src?.config?.platform;
|
|
971
|
+
return isStr(id) ? WORKFLOW_PLATFORMS[id] : undefined;
|
|
972
|
+
};
|
|
973
|
+
|
|
898
974
|
|
|
899
975
|
|
|
900
976
|
|
|
@@ -959,6 +1035,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
959
1035
|
the item, and a grader, where the run has them. */
|
|
960
1036
|
|
|
961
1037
|
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
|
|
962
1041
|
|
|
963
1042
|
|
|
964
1043
|
|
|
@@ -966,6 +1045,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
966
1045
|
|
|
967
1046
|
|
|
968
1047
|
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
|
|
1051
|
+
|
|
969
1052
|
|
|
970
1053
|
|
|
971
1054
|
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
@@ -1053,16 +1136,43 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
1053
1136
|
|
|
1054
1137
|
|
|
1055
1138
|
|
|
1056
|
-
/** The dataset body's version:
|
|
1057
|
-
|
|
1058
|
-
|
|
1139
|
+
/** The dataset body's version: 8 reads the recorded-reply metric ids under
|
|
1140
|
+
their new names (#299); 7 is an eval group (§17); 6 marks itself; 5 named
|
|
1141
|
+
its Source; 4 and earlier, neither. */
|
|
1142
|
+
const DATASET_BODY_VERSION = 8 ;
|
|
1143
|
+
|
|
1144
|
+
/** The recorded-reply metric type ids renamed at version 8 (#299): the family
|
|
1145
|
+
read "production" before the Recorded target gave it a home. The ids are
|
|
1146
|
+
stored tokens inside eval group bodies and a pipeline's private group, so
|
|
1147
|
+
they are mapped wherever a body is read rather than rewritten in place. */
|
|
1148
|
+
const RECORDED_IDS = {
|
|
1149
|
+
"equals-production": "same-as-recorded",
|
|
1150
|
+
"fields-equal-production": "fields-equal-recorded",
|
|
1151
|
+
"same-parse-outcome": "same-parse-as-recorded",
|
|
1152
|
+
};
|
|
1153
|
+
const renameMetric = (m ) =>
|
|
1154
|
+
isObj(m) && isStr(m.type) && RECORDED_IDS[m.type] ? { ...m, type: RECORDED_IDS[m.type] } : m;
|
|
1155
|
+
const renameMetrics = (list ) =>
|
|
1156
|
+
Array.isArray(list) ? list.map(renameMetric) : list;
|
|
1157
|
+
/** A version-7 body (or a freshly made version-8 one) with its recorded-reply
|
|
1158
|
+
metric ids read under their version-8 names, at version 8. */
|
|
1159
|
+
function recordedIdsV7(b ) {
|
|
1160
|
+
return {
|
|
1161
|
+
...b,
|
|
1162
|
+
version: DATASET_BODY_VERSION,
|
|
1163
|
+
...(Array.isArray(b.every) ? { every: renameMetrics(b.every) } : {}),
|
|
1164
|
+
...(Array.isArray(b.run) ? { run: renameMetrics(b.run) } : {}),
|
|
1165
|
+
...(Array.isArray(b.cases) ? { cases: b.cases.map((c ) =>
|
|
1166
|
+
isObj(c) && Array.isArray(c.metrics) ? { ...c, metrics: renameMetrics(c.metrics) } : c) } : {}),
|
|
1167
|
+
} ;
|
|
1168
|
+
}
|
|
1059
1169
|
|
|
1060
1170
|
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
1061
1171
|
reply read as a list, as well as one read as text. */
|
|
1062
1172
|
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
1063
1173
|
|
|
1064
1174
|
/**
|
|
1065
|
-
* [body] as this version of a dataset (
|
|
1175
|
+
* [body] as this version of a dataset (8), from any earlier one. Version 1
|
|
1066
1176
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1067
1177
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1068
1178
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
@@ -1075,8 +1185,12 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
|
1075
1185
|
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1076
1186
|
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1077
1187
|
* the body says its version. Version 6 is a group's cases alone: version 7
|
|
1078
|
-
* adds its scoring, grader, Every item and Whole run (`groupOfV6`)
|
|
1079
|
-
*
|
|
1188
|
+
* adds its scoring, grader, Every item and Whole run (`groupOfV6`). Version 8
|
|
1189
|
+
* reads the recorded-reply metric ids (`equals-production` and its two
|
|
1190
|
+
* siblings) under their new names (`RECORDED_IDS`, #299), in a group's own
|
|
1191
|
+
* metrics and its cases' -- a pipeline's private group is a group body, so
|
|
1192
|
+
* this covers it too. A stored row is read that way rather than rewritten.
|
|
1193
|
+
* Every reader of a
|
|
1080
1194
|
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1081
1195
|
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1082
1196
|
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
@@ -1087,19 +1201,25 @@ function upgradeDatasetBody (body ) {
|
|
|
1087
1201
|
if (!isObj(body)) return body;
|
|
1088
1202
|
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1089
1203
|
const b = body ;
|
|
1090
|
-
|
|
1091
|
-
|
|
1204
|
+
if ("version" in b) {
|
|
1205
|
+
// Version 7 is an eval group whose recorded-reply metrics are read under
|
|
1206
|
+
// their version-8 ids; version 6 is its cases alone. Any other version is
|
|
1207
|
+
// one this lab does not read.
|
|
1208
|
+
if (b.version === 7) return recordedIdsV7(b);
|
|
1209
|
+
if (b.version === 6) return recordedIdsV7(groupOfV6(b));
|
|
1210
|
+
return body;
|
|
1211
|
+
}
|
|
1092
1212
|
// A body that names its Source, even as null, is version 5.
|
|
1093
1213
|
if ("source" in b) {
|
|
1094
|
-
return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
|
|
1214
|
+
return recordedIdsV7(groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) }));
|
|
1095
1215
|
}
|
|
1096
1216
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1097
1217
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1098
1218
|
if (!raw) return body;
|
|
1099
|
-
return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
|
|
1219
|
+
return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
|
|
1100
1220
|
}
|
|
1101
1221
|
|
|
1102
|
-
/** A version-6 body as a version-
|
|
1222
|
+
/** A version-6 body as a version-8 eval group: scored All, the lab's
|
|
1103
1223
|
grader, and no metrics of its own for every item or the whole run --
|
|
1104
1224
|
what a Metrics eval naming the dataset with none of its own graded, which
|
|
1105
1225
|
is what every Graded set converted to. */
|
|
@@ -1243,7 +1363,12 @@ function datasetRules(body ) {
|
|
|
1243
1363
|
|
|
1244
1364
|
|
|
1245
1365
|
|
|
1246
|
-
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
|
|
1371
|
+
|
|
1247
1372
|
|
|
1248
1373
|
|
|
1249
1374
|
|
|
@@ -1614,12 +1739,33 @@ CONNECTION_TYPES.echo = {
|
|
|
1614
1739
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1615
1740
|
};
|
|
1616
1741
|
|
|
1742
|
+
/** A call's recorded result as transport-level text -- what a model asked the
|
|
1743
|
+
same way would have come back with, before the flow reads it. The runner's
|
|
1744
|
+
replay branch reads it as the flow reads a reply (`httpReplyOf`); this is
|
|
1745
|
+
the fallback where there is no Read as to read it through. */
|
|
1746
|
+
function recordedRaw(record ) {
|
|
1747
|
+
const result = isObj(record) && isObj(record.result) ? record.result : null;
|
|
1748
|
+
if (!result) return "";
|
|
1749
|
+
return isStr(result.body) ? result.body : JSON.stringify(result.body ?? "");
|
|
1750
|
+
}
|
|
1751
|
+
|
|
1752
|
+
// Recorded (#299): a target that replays each call's recorded reply, sending
|
|
1753
|
+
// nothing and holding no key or model. The runner reads the recorded result as
|
|
1754
|
+
// the flow reads a reply; `local` is the no-Read-as fallback.
|
|
1755
|
+
CONNECTION_TYPES.recorded = {
|
|
1756
|
+
id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
|
|
1757
|
+
description: "Replays the reply production recorded for each call. Sends nothing.",
|
|
1758
|
+
local: (_item, _sent, record) => recordedRaw(record),
|
|
1759
|
+
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1760
|
+
};
|
|
1761
|
+
|
|
1617
1762
|
/** A local type's answer to one stage, as a model's would come back. An item
|
|
1618
1763
|
with no text of its own (Prompt only) is answered from the prompt, and
|
|
1619
1764
|
that reply is read against nothing -- the prompt is the reply, not an
|
|
1620
|
-
instruction it could repeat.
|
|
1621
|
-
|
|
1622
|
-
|
|
1765
|
+
instruction it could repeat. A replaying type (Recorded) is handed the
|
|
1766
|
+
item's record to answer from. */
|
|
1767
|
+
function localAnswer(conn , text , sent , record ) {
|
|
1768
|
+
const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent, record);
|
|
1623
1769
|
return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
|
|
1624
1770
|
}
|
|
1625
1771
|
|
|
@@ -2991,24 +3137,54 @@ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
|
|
|
2991
3137
|
SOURCE_TYPES.files = {
|
|
2992
3138
|
label: "File Library",
|
|
2993
3139
|
description: "A folder of images and text files you upload.",
|
|
3140
|
+
group: "Files",
|
|
2994
3141
|
noun: "file",
|
|
2995
3142
|
uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
|
|
2996
3143
|
sendsImage: true,
|
|
3144
|
+
hasActions: false,
|
|
2997
3145
|
itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
|
|
2998
3146
|
};
|
|
2999
3147
|
|
|
3000
|
-
// A
|
|
3001
|
-
//
|
|
3002
|
-
// the
|
|
3003
|
-
//
|
|
3004
|
-
//
|
|
3005
|
-
|
|
3006
|
-
|
|
3007
|
-
|
|
3148
|
+
// A workflow Source: a cloud flow, read through a platform (WORKFLOW_PLATFORMS,
|
|
3149
|
+
// below) its row names in `config.platform`. Its items are records -- one call
|
|
3150
|
+
// of one of the flow's steps each: what the step's template read, the request
|
|
3151
|
+
// production sent and what came back (flows/record.ts) -- and it keeps the
|
|
3152
|
+
// flow's definition beside them. The kind is generic; the platform owns what
|
|
3153
|
+
// is engine-specific, and the type's shown label is the platform's.
|
|
3154
|
+
SOURCE_TYPES.workflow = {
|
|
3155
|
+
label: "Workflow",
|
|
3156
|
+
description: "A cloud flow, with records of its steps' calls from its run history.",
|
|
3157
|
+
group: "Workflows",
|
|
3008
3158
|
noun: "record",
|
|
3009
3159
|
uploads: [".json"],
|
|
3010
3160
|
sendsImage: false,
|
|
3161
|
+
hasActions: true,
|
|
3011
3162
|
itemText: () => false,
|
|
3163
|
+
platforms: ["power-automate"], // platform: the WORKFLOW_PLATFORMS a workflow offers
|
|
3164
|
+
};
|
|
3165
|
+
|
|
3166
|
+
// Power Automate (docs/power-automate.md): the one workflow platform built.
|
|
3167
|
+
// Its pieces were the lab's Power-Automate-specific code -- flows/record.ts,
|
|
3168
|
+
// flows/wdl.ts and flowApi.ts -- gathered here so generic code calls the
|
|
3169
|
+
// entry, never a platform id.
|
|
3170
|
+
WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapter's own id
|
|
3171
|
+
id: "power-automate", // platform: its id, the row's config.platform
|
|
3172
|
+
label: "Power Automate flow",
|
|
3173
|
+
nouns: { workflow: "flow", action: "action" },
|
|
3174
|
+
signIn: "microsoft",
|
|
3175
|
+
uploads: { ".json": "application/json; charset=utf-8" },
|
|
3176
|
+
keepsDefinition: true,
|
|
3177
|
+
api: token => flowApi(token),
|
|
3178
|
+
stepsOf: flowStepsOf,
|
|
3179
|
+
actionsOf: actionsByName,
|
|
3180
|
+
readerFor,
|
|
3181
|
+
evaluate,
|
|
3182
|
+
asText,
|
|
3183
|
+
readsOf,
|
|
3184
|
+
// Copy request… writes the body back as a flow's HTTP action holds it: JSON
|
|
3185
|
+
// whose strings keep Power Automate's own `@{…}` expressions untouched, so a
|
|
3186
|
+
// rebuilt body round-trips into the flow (docs/power-automate.md).
|
|
3187
|
+
render: body => JSON.stringify(body, null, 2),
|
|
3012
3188
|
};
|
|
3013
3189
|
|
|
3014
3190
|
CONTENT_TYPES.source = {
|
|
@@ -3188,7 +3364,11 @@ function metricInput(res , kase , more )
|
|
|
3188
3364
|
const last = res.transcript?.at(-1);
|
|
3189
3365
|
const text = last?.got ?? res.raw ?? "";
|
|
3190
3366
|
return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
|
|
3191
|
-
|
|
3367
|
+
// A case that has been given its own expected reply ("Take B as
|
|
3368
|
+
// expected") is compared with that; otherwise the Source's recorded
|
|
3369
|
+
// reply for the item.
|
|
3370
|
+
error: res.error ?? null, ms: res.ms ?? 0,
|
|
3371
|
+
recorded: typeof kase?.recorded === "string" ? kase.recorded : (more.production ?? null), kase };
|
|
3192
3372
|
}
|
|
3193
3373
|
|
|
3194
3374
|
/** A whole run's replies as one input: every reply's text together, and
|
|
@@ -3202,7 +3382,7 @@ function runInput(ress , plain )
|
|
|
3202
3382
|
terms.push(...(r.terms || []));
|
|
3203
3383
|
ms += r.ms ?? 0;
|
|
3204
3384
|
}
|
|
3205
|
-
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms,
|
|
3385
|
+
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
|
|
3206
3386
|
}
|
|
3207
3387
|
|
|
3208
3388
|
/** One metric's reading of [input]: `of`, `not` and a failure all applied,
|
|
@@ -3487,10 +3667,21 @@ EVAL_TYPES.group = {
|
|
|
3487
3667
|
read: async (t, kase, res, more) => {
|
|
3488
3668
|
const group = groupOf(t, more);
|
|
3489
3669
|
const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
|
|
3490
|
-
|
|
3670
|
+
const ref = casesRef(t);
|
|
3671
|
+
// The item's case in this eval's own group: resolved from the group whose
|
|
3672
|
+
// cases it reads where the runner names the item (grading several groups
|
|
3673
|
+
// at once, --groups), else the case handed in (a single group).
|
|
3674
|
+
const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
|
|
3675
|
+
return readGroup({ ...group, grader } , ref ? own : null, res, more);
|
|
3491
3676
|
},
|
|
3492
3677
|
};
|
|
3493
3678
|
|
|
3679
|
+
/** The case for an item in a group's body, by the name its file has, or null:
|
|
3680
|
+
a non-todo case whose item matches, as a Library group reads one. */
|
|
3681
|
+
function caseIn(body , item ) {
|
|
3682
|
+
return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
|
|
3683
|
+
}
|
|
3684
|
+
|
|
3494
3685
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3495
3686
|
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3496
3687
|
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
@@ -3612,10 +3803,38 @@ function readGroupRun(group , ress
|
|
|
3612
3803
|
return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
|
|
3613
3804
|
}
|
|
3614
3805
|
|
|
3806
|
+
/**
|
|
3807
|
+
* [body] with the cases named in [replies] expecting the reply given for
|
|
3808
|
+
* their item -- Results' "Take B as expected" (docs/workflow-sources.md §
|
|
3809
|
+
* Calls are graded cases). Each such case keeps everything else and gains
|
|
3810
|
+
* `recorded` (the reply the recorded-reply metrics compare against, read in
|
|
3811
|
+
* place of the Source's) and `todo: false`, so a later run grades the item
|
|
3812
|
+
* against that reply. A selected item the group has no case for, or one whose
|
|
3813
|
+
* reply is missing (Target B errored, or is the Recorded target), is left
|
|
3814
|
+
* untouched and named in `missing`; `taken` names the items whose expected was
|
|
3815
|
+
* written. Pure: the same body and replies give the same body, so Undo is the
|
|
3816
|
+
* body as it was.
|
|
3817
|
+
*/
|
|
3818
|
+
function takeAsExpected(body , replies )
|
|
3819
|
+
{
|
|
3820
|
+
const byItem = new Map ();
|
|
3821
|
+
body.cases.forEach((c, i) => { const item = caseItem(c); if (item) byItem.set(item, i); });
|
|
3822
|
+
const cases = body.cases.slice();
|
|
3823
|
+
const taken = [], missing = [];
|
|
3824
|
+
for (const item of Object.keys(replies)) {
|
|
3825
|
+
const reply = replies[item];
|
|
3826
|
+
const at = byItem.get(item);
|
|
3827
|
+
if (at == null || typeof reply !== "string") { missing.push(item); continue; }
|
|
3828
|
+
cases[at] = { ...cases[at], recorded: reply, todo: false };
|
|
3829
|
+
taken.push(item);
|
|
3830
|
+
}
|
|
3831
|
+
return { body: { ...body, cases }, taken, missing };
|
|
3832
|
+
}
|
|
3833
|
+
|
|
3615
3834
|
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3616
3835
|
function graderCtx(t , more ) {
|
|
3617
3836
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3618
|
-
return ask ? { ask } : {};
|
|
3837
|
+
return { ...(ask ? { ask } : {}), ...(more.recordedTarget ? { recordedTarget: true } : {}) };
|
|
3619
3838
|
}
|
|
3620
3839
|
|
|
3621
3840
|
/** A metric's options in a few words -- "watering", "^\\{" -- or nothing
|
|
@@ -3791,9 +4010,30 @@ STEP_TYPES.echo = {
|
|
|
3791
4010
|
},
|
|
3792
4011
|
};
|
|
3793
4012
|
|
|
3794
|
-
|
|
4013
|
+
// Recorded (#299): a target that replays each call's recorded reply, read as
|
|
4014
|
+
// the flow reads one. No profile, no prompt -- the record is the reply.
|
|
4015
|
+
STEP_TYPES.recorded = {
|
|
4016
|
+
label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
|
|
4017
|
+
description: "Replays the reply production recorded for each call. Sends nothing.",
|
|
4018
|
+
firstJobOnly: "a call's recorded reply is job 1's",
|
|
4019
|
+
in: "item", out: "text",
|
|
4020
|
+
apply: "runPipeline",
|
|
4021
|
+
// `prompt` only so the step satisfies the target-step shape; it is never
|
|
4022
|
+
// sent (the record is the reply), and the prompt rule exempts it below.
|
|
4023
|
+
fields: ["type", "prompt"],
|
|
4024
|
+
validate(){},
|
|
4025
|
+
};
|
|
4026
|
+
|
|
4027
|
+
/** What a local target step is answered as, by the connection type it stands
|
|
4028
|
+
in for (`replaces`): Echo's own connection, or Recorded's -- each with no
|
|
3795
4029
|
address, key or model. */
|
|
3796
4030
|
const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
|
|
4031
|
+
const RECORDED_CONNECTION = { name: "Recorded", url: "", model: "", type: "recorded" };
|
|
4032
|
+
const LOCAL_CONNECTIONS =
|
|
4033
|
+
{ echo: ECHO_CONNECTION, recorded: RECORDED_CONNECTION };
|
|
4034
|
+
/** The connection a local target step stands in for, by its entry. */
|
|
4035
|
+
const localConnOf = (entry ) =>
|
|
4036
|
+
LOCAL_CONNECTIONS[entry?.replaces ?? ""] ?? ECHO_CONNECTION;
|
|
3797
4037
|
|
|
3798
4038
|
// Responses: how a job's reply is read, in order -- possibly not at all.
|
|
3799
4039
|
|
|
@@ -3901,9 +4141,15 @@ function targetsOf(doc
|
|
|
3901
4141
|
return Array.isArray(doc?.targets) ? doc .targets : [];
|
|
3902
4142
|
}
|
|
3903
4143
|
|
|
3904
|
-
/** Target [i]'s
|
|
4144
|
+
/** Target [i]'s letter, A for the first: the same in every job, so a
|
|
4145
|
+
Target reads as one column through them. A number past Z. */
|
|
4146
|
+
function targetLetter(i ) {
|
|
4147
|
+
return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
|
|
4148
|
+
}
|
|
4149
|
+
|
|
4150
|
+
/** Target [i]'s name, or its letter's: "Target A". */
|
|
3905
4151
|
function targetLabel(doc , i ) {
|
|
3906
|
-
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i
|
|
4152
|
+
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
|
|
3907
4153
|
}
|
|
3908
4154
|
|
|
3909
4155
|
/** What target [i] sends in job [k]: its step there. */
|
|
@@ -3927,6 +4173,16 @@ function targetProfileOf(doc
|
|
|
3927
4173
|
return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
|
|
3928
4174
|
}
|
|
3929
4175
|
|
|
4176
|
+
/** Whether target [i] is a Recorded target: it replays each call's recorded
|
|
4177
|
+
reply rather than sending anything (its job-1 step answers locally with a
|
|
4178
|
+
connection that `replays`). A recorded-reply metric compares the recorded
|
|
4179
|
+
reply with itself there, so it says nothing (n/a). Asked through the
|
|
4180
|
+
registry, never a step id. */
|
|
4181
|
+
function recordedTargetOf(doc , i ) {
|
|
4182
|
+
const entry = targetEntryOf(doc, i, 0);
|
|
4183
|
+
return !!(entry?.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays);
|
|
4184
|
+
}
|
|
4185
|
+
|
|
3930
4186
|
/** [doc] with target [i]'s step in job [k] changed: [change] merged into
|
|
3931
4187
|
it, or the step [change] makes of it. */
|
|
3932
4188
|
function withTarget (doc , i , k ,
|
|
@@ -4150,11 +4406,8 @@ STEP_TYPES.evals = {
|
|
|
4150
4406
|
+ `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
|
|
4151
4407
|
}
|
|
4152
4408
|
});
|
|
4153
|
-
// The worker
|
|
4154
|
-
//
|
|
4155
|
-
// group (#233).
|
|
4156
|
-
const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
|
|
4157
|
-
if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
|
|
4409
|
+
// The worker reads a body per group now (#233), so the evals may grade
|
|
4410
|
+
// against several Library groups at once; the queue keeps each body.
|
|
4158
4411
|
},
|
|
4159
4412
|
};
|
|
4160
4413
|
|
|
@@ -4621,11 +4874,16 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4621
4874
|
// what it is read against. Over Prompt only the prompt is the reply.
|
|
4622
4875
|
// What target [i] asks in job [k]: Echo's own connection for a step
|
|
4623
4876
|
// answered here, else its profile as the lookups hold it.
|
|
4624
|
-
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
4877
|
+
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k))
|
|
4625
4878
|
: c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
|
|
4626
4879
|
const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
|
|
4627
4880
|
for (let k = 0; k < n; k++) {
|
|
4628
4881
|
const i = targets.findIndex((_, i) => {
|
|
4882
|
+
const entry = targetEntryOf(doc, i, k);
|
|
4883
|
+
// A step with no prompt of its own, or a replaying target (Recorded,
|
|
4884
|
+
// which answers from the record), is never asked for one.
|
|
4885
|
+
if (!entry?.fields?.includes("prompt")) return false;
|
|
4886
|
+
if (entry.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays) return false;
|
|
4629
4887
|
const words = targetStepOf(doc, i, k)?.prompt;
|
|
4630
4888
|
return (!isStr(words) || !words.trim())
|
|
4631
4889
|
&& !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
|
|
@@ -4638,12 +4896,14 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4638
4896
|
const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
|
|
4639
4897
|
.find(n => !/\.(?:txt|md|csv)$/i.test(n));
|
|
4640
4898
|
targets.forEach((_, i) => {
|
|
4641
|
-
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
4899
|
+
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k)) : lookup(targetProfileOf(doc, i, k) )));
|
|
4642
4900
|
conns.forEach((p , k ) => {
|
|
4643
4901
|
if (!local(p)) return;
|
|
4644
|
-
const
|
|
4645
|
-
|
|
4646
|
-
|
|
4902
|
+
const entry = CONNECTION_TYPES[typeOf(p )] ;
|
|
4903
|
+
// A replaying target (Recorded) answers from the record, not the item
|
|
4904
|
+
// text, so the text rules do not apply -- only the job-1 one does.
|
|
4905
|
+
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${entry.label} answers only job 1${entry.replays ? "" : ", with the item's own text"}`);
|
|
4906
|
+
else if (nonText && !entry.replays) bad.push(`${targetLabel(doc, i)}: ${entry.label} answers each text item with its own text, and ${nonText} is not text`);
|
|
4647
4907
|
});
|
|
4648
4908
|
// A connection answers what its target's step asks: words, or a whole request.
|
|
4649
4909
|
conns.forEach((p , k ) => {
|
|
@@ -5241,7 +5501,7 @@ function stagesFor(run , i )
|
|
|
5241
5501
|
if (targetEntryOf(run, i, k)?.local) {
|
|
5242
5502
|
const id = targetProfileOf(run, i, k)?.id;
|
|
5243
5503
|
const held = id != null ? run.profiles?.[id] : undefined;
|
|
5244
|
-
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...
|
|
5504
|
+
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...localConnOf(targetEntryOf(run, i, k)) };
|
|
5245
5505
|
}
|
|
5246
5506
|
const id = targetProfileOf(run, i, k) .id;
|
|
5247
5507
|
return { id, ...(run.profiles?.[id] || {}) };
|
|
@@ -5256,7 +5516,7 @@ function stagesFor(run , i )
|
|
|
5256
5516
|
function scenarioProfile(run , i ) {
|
|
5257
5517
|
const id = targetsOf(run)[i]?.profile?.id;
|
|
5258
5518
|
// A target that asks nothing ran as Echo.
|
|
5259
|
-
if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name:
|
|
5519
|
+
if (id == null && targetEntryOf(run, i, 0)?.local) { const lc = localConnOf(targetEntryOf(run, i, 0)); return { id: "", name: lc.name, settings: { ...lc } }; }
|
|
5260
5520
|
const conn = id != null ? run.profiles?.[id] : null;
|
|
5261
5521
|
return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
|
|
5262
5522
|
}
|
|
@@ -5317,9 +5577,9 @@ function itemScores(run , kase , res
|
|
|
5317
5577
|
return Object.keys(out).length ? out : null;
|
|
5318
5578
|
}
|
|
5319
5579
|
|
|
5320
|
-
/** What production
|
|
5321
|
-
call says, from the item's record, or nobody does. */
|
|
5322
|
-
function
|
|
5580
|
+
/** What production recorded as the reply to an item, for a run's metrics: its
|
|
5581
|
+
last job's call says, from the item's record, or nobody does. */
|
|
5582
|
+
function recordedReplyOf(run , record ) {
|
|
5323
5583
|
const last = run.jobs.at(-1), flow = flowStepOf(last);
|
|
5324
5584
|
return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
|
|
5325
5585
|
}
|
|
@@ -5390,22 +5650,40 @@ function scenarioEvals(run , i , items
|
|
|
5390
5650
|
});
|
|
5391
5651
|
}
|
|
5392
5652
|
|
|
5653
|
+
/** One eval group's verdict over a scenario (§17): a whole-run group's own
|
|
5654
|
+
verdict, or a per-item group's "every item it read passed"; null where it
|
|
5655
|
+
was skipped, or had nothing to read yet. */
|
|
5656
|
+
function groupPass(o ) {
|
|
5657
|
+
if (o.skipped) return null;
|
|
5658
|
+
if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
|
|
5659
|
+
return o.ran ? o.passed === o.ran : null;
|
|
5660
|
+
}
|
|
5661
|
+
|
|
5393
5662
|
/** A scenario's pass or fail over every eval that read it: null where none
|
|
5394
5663
|
has anything to say yet. */
|
|
5395
5664
|
function scenarioPasses(outcomes ) {
|
|
5396
|
-
|
|
5397
|
-
|
|
5398
|
-
|
|
5399
|
-
|
|
5400
|
-
|
|
5401
|
-
|
|
5402
|
-
|
|
5403
|
-
|
|
5404
|
-
|
|
5405
|
-
|
|
5406
|
-
|
|
5407
|
-
|
|
5408
|
-
return
|
|
5665
|
+
const passes = outcomes.map(groupPass);
|
|
5666
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5667
|
+
return !passes.includes(false);
|
|
5668
|
+
}
|
|
5669
|
+
|
|
5670
|
+
/** A scenario's overall verdict under a run's pass rule (§17): each linked
|
|
5671
|
+
group's verdict folded together the way `pass` says -- every group passes,
|
|
5672
|
+
or at least a number of them. Null where no group has a verdict yet, so a
|
|
5673
|
+
run with no evals, or one still grading, reads as it does without a rule. */
|
|
5674
|
+
function overallVerdict(outcomes , pass ) {
|
|
5675
|
+
const passes = outcomes.map(groupPass);
|
|
5676
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5677
|
+
return passVerdict(pass, passes);
|
|
5678
|
+
}
|
|
5679
|
+
|
|
5680
|
+
/** Whether a run passes a Target under its overall pass rule (§17): `all`
|
|
5681
|
+
passes when no linked group fails, `atLeast` when at least `count` pass.
|
|
5682
|
+
[passes] is each group's pass (true), fail (false), or neither (null: it
|
|
5683
|
+
graded nothing, or an earlier group skipped it). */
|
|
5684
|
+
function passVerdict(pass , passes ) {
|
|
5685
|
+
if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
|
|
5686
|
+
return !passes.includes(false);
|
|
5409
5687
|
}
|
|
5410
5688
|
|
|
5411
5689
|
/**
|
|
@@ -5597,23 +5875,23 @@ registerMetrics({ registerKinds });
|
|
|
5597
5875
|
export {
|
|
5598
5876
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
5599
5877
|
loopReplyError, preparedSize,
|
|
5600
|
-
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
5878
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, takeAsExpected, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
5601
5879
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
5602
5880
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
5603
5881
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
5604
5882
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
5605
5883
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
5606
5884
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
5607
|
-
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync,
|
|
5608
|
-
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
5885
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
|
|
5886
|
+
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
|
|
5609
5887
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
5610
5888
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5611
5889
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5612
5890
|
contentOf, withContent, replyOf,
|
|
5613
|
-
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5891
|
+
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
|
|
5614
5892
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5615
5893
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5616
|
-
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
5894
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|
|
5617
5895
|
evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
|
|
5618
5896
|
pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
|
|
5619
5897
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|