evals-lab 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +285 -45
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +24 -4
- package/lab/server.py +586 -167
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +0 -3
- package/lab/web/dist/assets/main-DDoeU6hq.css +0 -1
- package/lab/web/dist/assets/main-nz6Q4jVm.js +0 -21
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +0 -59
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,46 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.7.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- **Copy your data directory before upgrading** (see "User data" in the
|
|
13
|
+
README). An eval group 0.7.0 saves is dataset version 8, which 0.6.x
|
|
14
|
+
cannot read. To go back, reinstall 0.6.0 and restore the copy.
|
|
15
|
+
- On its first start, 0.7.0 moves everything the lab keeps into one
|
|
16
|
+
workspace, Default. Nothing on the page changes.
|
|
17
|
+
- Stored eval groups are not rewritten on upgrade: a version-7 group reads
|
|
18
|
+
as version 8, and is saved as version 8 at its next edit.
|
|
19
|
+
- Export JSON writes version 8, which 0.6.x refuses to import. Import reads
|
|
20
|
+
versions 1 to 8.
|
|
21
|
+
- The metrics that compare a reply with production's are now the Recorded
|
|
22
|
+
reply metrics. Groups that use them grade as before.
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- Test… on a Power Automate Source's action makes an eval group from the
|
|
27
|
+
Source's calls, graded against the replies production recorded, and a
|
|
28
|
+
pipeline to run it, then opens it in Runs.
|
|
29
|
+
- The Recorded target replays each call's recorded reply and sends nothing,
|
|
30
|
+
so a run compares an edited request with what production did.
|
|
31
|
+
- Adding a Power Automate Source imports the calls of its last five runs in
|
|
32
|
+
the background, while the tab stays open.
|
|
33
|
+
- Results of a run with a Recorded target: a table of each check with a
|
|
34
|
+
column per Target (n/a where a check cannot apply), a call that opens both
|
|
35
|
+
replies side by side, and Take B as expected, which makes the selected
|
|
36
|
+
calls expect Target B's reply.
|
|
37
|
+
- An HTTP Request target shows its words with the item's fields as chips.
|
|
38
|
+
Select prompt… and Save to library… use the Prompt library, and Copy
|
|
39
|
+
request… writes the request back for Power Automate, or as JSON.
|
|
40
|
+
|
|
41
|
+
### Changed
|
|
42
|
+
|
|
43
|
+
- A Power Automate Source's records are its Calls, with Add call.
|
|
44
|
+
- Targets sit side by side on a wide screen. A job's Content and Responses
|
|
45
|
+
fold to one line until opened.
|
|
46
|
+
- A row with one action shows it as a button; two or more are in its ⋯ menu.
|
|
47
|
+
|
|
8
48
|
## 0.6.0
|
|
9
49
|
|
|
10
50
|
### Added
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.7.0 (2026.10.06-456)
|
package/lab/evals-core.mjs
CHANGED
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
import yaml from "./js-yaml.mjs";
|
|
36
36
|
|
|
37
37
|
import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
|
|
38
|
+
import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
|
|
39
|
+
import { flowApi, } from "./flows/flowApi.mjs";
|
|
38
40
|
import registerMetrics from "./metrics/builtin.mjs";
|
|
39
41
|
import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
|
|
40
42
|
|
|
@@ -543,6 +545,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
543
545
|
|
|
544
546
|
|
|
545
547
|
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
|
|
546
554
|
|
|
547
555
|
|
|
548
556
|
|
|
@@ -646,9 +654,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
646
654
|
|
|
647
655
|
|
|
648
656
|
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
|
|
652
660
|
|
|
653
661
|
|
|
654
662
|
|
|
@@ -662,6 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
662
670
|
|
|
663
671
|
|
|
664
672
|
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
|
|
665
677
|
|
|
666
678
|
|
|
667
679
|
/** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
|
|
@@ -880,11 +892,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
880
892
|
|
|
881
893
|
|
|
882
894
|
/** A kind of Source: what one of its items is, and what the page and the
|
|
883
|
-
runner do with it. A reader asks the entry; it never branches on an id.
|
|
895
|
+
runner do with it. A reader asks the entry; it never branches on an id.
|
|
896
|
+
The row's `type` is the kind; a workflow kind's platform -- which flow
|
|
897
|
+
engine a row speaks to -- is `config.platform`, a WORKFLOW_PLATFORMS id. */
|
|
884
898
|
|
|
885
899
|
|
|
886
900
|
|
|
887
901
|
|
|
902
|
+
|
|
903
|
+
|
|
888
904
|
|
|
889
905
|
|
|
890
906
|
|
|
@@ -893,10 +909,68 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
893
909
|
|
|
894
910
|
|
|
895
911
|
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
|
|
896
916
|
|
|
897
917
|
|
|
918
|
+
|
|
919
|
+
|
|
920
|
+
|
|
898
921
|
|
|
899
922
|
|
|
923
|
+
/** What a workflow of one platform -- a flow engine -- answers: the pieces
|
|
924
|
+
that were Power-Automate-specific, lifted behind the platform id so
|
|
925
|
+
generic code asks the entry and never names a platform. Power Automate is
|
|
926
|
+
the one built; a second (Logic App, n8n) is one more entry. */
|
|
927
|
+
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
/** A platform's management API reader (flowApi.ts for Power Automate); its
|
|
963
|
+
shape is the platform's own, so generic code holds it opaquely. */
|
|
964
|
+
|
|
965
|
+
|
|
966
|
+
const WORKFLOW_PLATFORMS = Object.create(null);
|
|
967
|
+
/** The platform a workflow Source speaks to, from its `config.platform`;
|
|
968
|
+
undefined for a row that is not a workflow or names an unknown platform. */
|
|
969
|
+
const platformOf = (src ) => {
|
|
970
|
+
const id = src?.config?.platform;
|
|
971
|
+
return isStr(id) ? WORKFLOW_PLATFORMS[id] : undefined;
|
|
972
|
+
};
|
|
973
|
+
|
|
900
974
|
|
|
901
975
|
|
|
902
976
|
|
|
@@ -961,6 +1035,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
961
1035
|
the item, and a grader, where the run has them. */
|
|
962
1036
|
|
|
963
1037
|
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
|
|
964
1041
|
|
|
965
1042
|
|
|
966
1043
|
|
|
@@ -1059,16 +1136,43 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
1059
1136
|
|
|
1060
1137
|
|
|
1061
1138
|
|
|
1062
|
-
/** The dataset body's version:
|
|
1063
|
-
|
|
1064
|
-
|
|
1139
|
+
/** The dataset body's version: 8 reads the recorded-reply metric ids under
|
|
1140
|
+
their new names (#299); 7 is an eval group (§17); 6 marks itself; 5 named
|
|
1141
|
+
its Source; 4 and earlier, neither. */
|
|
1142
|
+
const DATASET_BODY_VERSION = 8 ;
|
|
1143
|
+
|
|
1144
|
+
/** The recorded-reply metric type ids renamed at version 8 (#299): the family
|
|
1145
|
+
read "production" before the Recorded target gave it a home. The ids are
|
|
1146
|
+
stored tokens inside eval group bodies and a pipeline's private group, so
|
|
1147
|
+
they are mapped wherever a body is read rather than rewritten in place. */
|
|
1148
|
+
const RECORDED_IDS = {
|
|
1149
|
+
"equals-production": "same-as-recorded",
|
|
1150
|
+
"fields-equal-production": "fields-equal-recorded",
|
|
1151
|
+
"same-parse-outcome": "same-parse-as-recorded",
|
|
1152
|
+
};
|
|
1153
|
+
const renameMetric = (m ) =>
|
|
1154
|
+
isObj(m) && isStr(m.type) && RECORDED_IDS[m.type] ? { ...m, type: RECORDED_IDS[m.type] } : m;
|
|
1155
|
+
const renameMetrics = (list ) =>
|
|
1156
|
+
Array.isArray(list) ? list.map(renameMetric) : list;
|
|
1157
|
+
/** A version-7 body (or a freshly made version-8 one) with its recorded-reply
|
|
1158
|
+
metric ids read under their version-8 names, at version 8. */
|
|
1159
|
+
function recordedIdsV7(b ) {
|
|
1160
|
+
return {
|
|
1161
|
+
...b,
|
|
1162
|
+
version: DATASET_BODY_VERSION,
|
|
1163
|
+
...(Array.isArray(b.every) ? { every: renameMetrics(b.every) } : {}),
|
|
1164
|
+
...(Array.isArray(b.run) ? { run: renameMetrics(b.run) } : {}),
|
|
1165
|
+
...(Array.isArray(b.cases) ? { cases: b.cases.map((c ) =>
|
|
1166
|
+
isObj(c) && Array.isArray(c.metrics) ? { ...c, metrics: renameMetrics(c.metrics) } : c) } : {}),
|
|
1167
|
+
} ;
|
|
1168
|
+
}
|
|
1065
1169
|
|
|
1066
1170
|
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
1067
1171
|
reply read as a list, as well as one read as text. */
|
|
1068
1172
|
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
1069
1173
|
|
|
1070
1174
|
/**
|
|
1071
|
-
* [body] as this version of a dataset (
|
|
1175
|
+
* [body] as this version of a dataset (8), from any earlier one. Version 1
|
|
1072
1176
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1073
1177
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1074
1178
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
@@ -1081,8 +1185,12 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
|
1081
1185
|
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1082
1186
|
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1083
1187
|
* the body says its version. Version 6 is a group's cases alone: version 7
|
|
1084
|
-
* adds its scoring, grader, Every item and Whole run (`groupOfV6`)
|
|
1085
|
-
*
|
|
1188
|
+
* adds its scoring, grader, Every item and Whole run (`groupOfV6`). Version 8
|
|
1189
|
+
* reads the recorded-reply metric ids (`equals-production` and its two
|
|
1190
|
+
* siblings) under their new names (`RECORDED_IDS`, #299), in a group's own
|
|
1191
|
+
* metrics and its cases' -- a pipeline's private group is a group body, so
|
|
1192
|
+
* this covers it too. A stored row is read that way rather than rewritten.
|
|
1193
|
+
* Every reader of a
|
|
1086
1194
|
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1087
1195
|
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1088
1196
|
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
@@ -1093,19 +1201,25 @@ function upgradeDatasetBody (body ) {
|
|
|
1093
1201
|
if (!isObj(body)) return body;
|
|
1094
1202
|
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1095
1203
|
const b = body ;
|
|
1096
|
-
|
|
1097
|
-
|
|
1204
|
+
if ("version" in b) {
|
|
1205
|
+
// Version 7 is an eval group whose recorded-reply metrics are read under
|
|
1206
|
+
// their version-8 ids; version 6 is its cases alone. Any other version is
|
|
1207
|
+
// one this lab does not read.
|
|
1208
|
+
if (b.version === 7) return recordedIdsV7(b);
|
|
1209
|
+
if (b.version === 6) return recordedIdsV7(groupOfV6(b));
|
|
1210
|
+
return body;
|
|
1211
|
+
}
|
|
1098
1212
|
// A body that names its Source, even as null, is version 5.
|
|
1099
1213
|
if ("source" in b) {
|
|
1100
|
-
return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
|
|
1214
|
+
return recordedIdsV7(groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) }));
|
|
1101
1215
|
}
|
|
1102
1216
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1103
1217
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1104
1218
|
if (!raw) return body;
|
|
1105
|
-
return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
|
|
1219
|
+
return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
|
|
1106
1220
|
}
|
|
1107
1221
|
|
|
1108
|
-
/** A version-6 body as a version-
|
|
1222
|
+
/** A version-6 body as a version-8 eval group: scored All, the lab's
|
|
1109
1223
|
grader, and no metrics of its own for every item or the whole run --
|
|
1110
1224
|
what a Metrics eval naming the dataset with none of its own graded, which
|
|
1111
1225
|
is what every Graded set converted to. */
|
|
@@ -1249,7 +1363,12 @@ function datasetRules(body ) {
|
|
|
1249
1363
|
|
|
1250
1364
|
|
|
1251
1365
|
|
|
1252
|
-
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
|
|
1371
|
+
|
|
1253
1372
|
|
|
1254
1373
|
|
|
1255
1374
|
|
|
@@ -1620,12 +1739,33 @@ CONNECTION_TYPES.echo = {
|
|
|
1620
1739
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1621
1740
|
};
|
|
1622
1741
|
|
|
1742
|
+
/** A call's recorded result as transport-level text -- what a model asked the
|
|
1743
|
+
same way would have come back with, before the flow reads it. The runner's
|
|
1744
|
+
replay branch reads it as the flow reads a reply (`httpReplyOf`); this is
|
|
1745
|
+
the fallback where there is no Read as to read it through. */
|
|
1746
|
+
function recordedRaw(record ) {
|
|
1747
|
+
const result = isObj(record) && isObj(record.result) ? record.result : null;
|
|
1748
|
+
if (!result) return "";
|
|
1749
|
+
return isStr(result.body) ? result.body : JSON.stringify(result.body ?? "");
|
|
1750
|
+
}
|
|
1751
|
+
|
|
1752
|
+
// Recorded (#299): a target that replays each call's recorded reply, sending
|
|
1753
|
+
// nothing and holding no key or model. The runner reads the recorded result as
|
|
1754
|
+
// the flow reads a reply; `local` is the no-Read-as fallback.
|
|
1755
|
+
CONNECTION_TYPES.recorded = {
|
|
1756
|
+
id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
|
|
1757
|
+
description: "Replays the reply production recorded for each call. Sends nothing.",
|
|
1758
|
+
local: (_item, _sent, record) => recordedRaw(record),
|
|
1759
|
+
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1760
|
+
};
|
|
1761
|
+
|
|
1623
1762
|
/** A local type's answer to one stage, as a model's would come back. An item
|
|
1624
1763
|
with no text of its own (Prompt only) is answered from the prompt, and
|
|
1625
1764
|
that reply is read against nothing -- the prompt is the reply, not an
|
|
1626
|
-
instruction it could repeat.
|
|
1627
|
-
|
|
1628
|
-
|
|
1765
|
+
instruction it could repeat. A replaying type (Recorded) is handed the
|
|
1766
|
+
item's record to answer from. */
|
|
1767
|
+
function localAnswer(conn , text , sent , record ) {
|
|
1768
|
+
const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent, record);
|
|
1629
1769
|
return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
|
|
1630
1770
|
}
|
|
1631
1771
|
|
|
@@ -2997,24 +3137,54 @@ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
|
|
|
2997
3137
|
SOURCE_TYPES.files = {
|
|
2998
3138
|
label: "File Library",
|
|
2999
3139
|
description: "A folder of images and text files you upload.",
|
|
3140
|
+
group: "Files",
|
|
3000
3141
|
noun: "file",
|
|
3001
3142
|
uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
|
|
3002
3143
|
sendsImage: true,
|
|
3144
|
+
hasActions: false,
|
|
3003
3145
|
itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
|
|
3004
3146
|
};
|
|
3005
3147
|
|
|
3006
|
-
// A
|
|
3007
|
-
//
|
|
3008
|
-
// the
|
|
3009
|
-
//
|
|
3010
|
-
//
|
|
3011
|
-
|
|
3012
|
-
|
|
3013
|
-
|
|
3148
|
+
// A workflow Source: a cloud flow, read through a platform (WORKFLOW_PLATFORMS,
|
|
3149
|
+
// below) its row names in `config.platform`. Its items are records -- one call
|
|
3150
|
+
// of one of the flow's steps each: what the step's template read, the request
|
|
3151
|
+
// production sent and what came back (flows/record.ts) -- and it keeps the
|
|
3152
|
+
// flow's definition beside them. The kind is generic; the platform owns what
|
|
3153
|
+
// is engine-specific, and the type's shown label is the platform's.
|
|
3154
|
+
SOURCE_TYPES.workflow = {
|
|
3155
|
+
label: "Workflow",
|
|
3156
|
+
description: "A cloud flow, with records of its steps' calls from its run history.",
|
|
3157
|
+
group: "Workflows",
|
|
3014
3158
|
noun: "record",
|
|
3015
3159
|
uploads: [".json"],
|
|
3016
3160
|
sendsImage: false,
|
|
3161
|
+
hasActions: true,
|
|
3017
3162
|
itemText: () => false,
|
|
3163
|
+
platforms: ["power-automate"], // platform: the WORKFLOW_PLATFORMS a workflow offers
|
|
3164
|
+
};
|
|
3165
|
+
|
|
3166
|
+
// Power Automate (docs/power-automate.md): the one workflow platform built.
|
|
3167
|
+
// Its pieces were the lab's Power-Automate-specific code -- flows/record.ts,
|
|
3168
|
+
// flows/wdl.ts and flowApi.ts -- gathered here so generic code calls the
|
|
3169
|
+
// entry, never a platform id.
|
|
3170
|
+
WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapter's own id
|
|
3171
|
+
id: "power-automate", // platform: its id, the row's config.platform
|
|
3172
|
+
label: "Power Automate flow",
|
|
3173
|
+
nouns: { workflow: "flow", action: "action" },
|
|
3174
|
+
signIn: "microsoft",
|
|
3175
|
+
uploads: { ".json": "application/json; charset=utf-8" },
|
|
3176
|
+
keepsDefinition: true,
|
|
3177
|
+
api: token => flowApi(token),
|
|
3178
|
+
stepsOf: flowStepsOf,
|
|
3179
|
+
actionsOf: actionsByName,
|
|
3180
|
+
readerFor,
|
|
3181
|
+
evaluate,
|
|
3182
|
+
asText,
|
|
3183
|
+
readsOf,
|
|
3184
|
+
// Copy request… writes the body back as a flow's HTTP action holds it: JSON
|
|
3185
|
+
// whose strings keep Power Automate's own `@{…}` expressions untouched, so a
|
|
3186
|
+
// rebuilt body round-trips into the flow (docs/power-automate.md).
|
|
3187
|
+
render: body => JSON.stringify(body, null, 2),
|
|
3018
3188
|
};
|
|
3019
3189
|
|
|
3020
3190
|
CONTENT_TYPES.source = {
|
|
@@ -3194,7 +3364,11 @@ function metricInput(res , kase , more )
|
|
|
3194
3364
|
const last = res.transcript?.at(-1);
|
|
3195
3365
|
const text = last?.got ?? res.raw ?? "";
|
|
3196
3366
|
return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
|
|
3197
|
-
|
|
3367
|
+
// A case that has been given its own expected reply ("Take B as
|
|
3368
|
+
// expected") is compared with that; otherwise the Source's recorded
|
|
3369
|
+
// reply for the item.
|
|
3370
|
+
error: res.error ?? null, ms: res.ms ?? 0,
|
|
3371
|
+
recorded: typeof kase?.recorded === "string" ? kase.recorded : (more.production ?? null), kase };
|
|
3198
3372
|
}
|
|
3199
3373
|
|
|
3200
3374
|
/** A whole run's replies as one input: every reply's text together, and
|
|
@@ -3208,7 +3382,7 @@ function runInput(ress , plain )
|
|
|
3208
3382
|
terms.push(...(r.terms || []));
|
|
3209
3383
|
ms += r.ms ?? 0;
|
|
3210
3384
|
}
|
|
3211
|
-
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms,
|
|
3385
|
+
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
|
|
3212
3386
|
}
|
|
3213
3387
|
|
|
3214
3388
|
/** One metric's reading of [input]: `of`, `not` and a failure all applied,
|
|
@@ -3629,10 +3803,38 @@ function readGroupRun(group , ress
|
|
|
3629
3803
|
return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
|
|
3630
3804
|
}
|
|
3631
3805
|
|
|
3806
|
+
/**
|
|
3807
|
+
* [body] with the cases named in [replies] expecting the reply given for
|
|
3808
|
+
* their item -- Results' "Take B as expected" (docs/workflow-sources.md §
|
|
3809
|
+
* Calls are graded cases). Each such case keeps everything else and gains
|
|
3810
|
+
* `recorded` (the reply the recorded-reply metrics compare against, read in
|
|
3811
|
+
* place of the Source's) and `todo: false`, so a later run grades the item
|
|
3812
|
+
* against that reply. A selected item the group has no case for, or one whose
|
|
3813
|
+
* reply is missing (Target B errored, or is the Recorded target), is left
|
|
3814
|
+
* untouched and named in `missing`; `taken` names the items whose expected was
|
|
3815
|
+
* written. Pure: the same body and replies give the same body, so Undo is the
|
|
3816
|
+
* body as it was.
|
|
3817
|
+
*/
|
|
3818
|
+
function takeAsExpected(body , replies )
|
|
3819
|
+
{
|
|
3820
|
+
const byItem = new Map ();
|
|
3821
|
+
body.cases.forEach((c, i) => { const item = caseItem(c); if (item) byItem.set(item, i); });
|
|
3822
|
+
const cases = body.cases.slice();
|
|
3823
|
+
const taken = [], missing = [];
|
|
3824
|
+
for (const item of Object.keys(replies)) {
|
|
3825
|
+
const reply = replies[item];
|
|
3826
|
+
const at = byItem.get(item);
|
|
3827
|
+
if (at == null || typeof reply !== "string") { missing.push(item); continue; }
|
|
3828
|
+
cases[at] = { ...cases[at], recorded: reply, todo: false };
|
|
3829
|
+
taken.push(item);
|
|
3830
|
+
}
|
|
3831
|
+
return { body: { ...body, cases }, taken, missing };
|
|
3832
|
+
}
|
|
3833
|
+
|
|
3632
3834
|
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3633
3835
|
function graderCtx(t , more ) {
|
|
3634
3836
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3635
|
-
return ask ? { ask } : {};
|
|
3837
|
+
return { ...(ask ? { ask } : {}), ...(more.recordedTarget ? { recordedTarget: true } : {}) };
|
|
3636
3838
|
}
|
|
3637
3839
|
|
|
3638
3840
|
/** A metric's options in a few words -- "watering", "^\\{" -- or nothing
|
|
@@ -3808,9 +4010,30 @@ STEP_TYPES.echo = {
|
|
|
3808
4010
|
},
|
|
3809
4011
|
};
|
|
3810
4012
|
|
|
3811
|
-
|
|
4013
|
+
// Recorded (#299): a target that replays each call's recorded reply, read as
|
|
4014
|
+
// the flow reads one. No profile, no prompt -- the record is the reply.
|
|
4015
|
+
STEP_TYPES.recorded = {
|
|
4016
|
+
label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
|
|
4017
|
+
description: "Replays the reply production recorded for each call. Sends nothing.",
|
|
4018
|
+
firstJobOnly: "a call's recorded reply is job 1's",
|
|
4019
|
+
in: "item", out: "text",
|
|
4020
|
+
apply: "runPipeline",
|
|
4021
|
+
// `prompt` only so the step satisfies the target-step shape; it is never
|
|
4022
|
+
// sent (the record is the reply), and the prompt rule exempts it below.
|
|
4023
|
+
fields: ["type", "prompt"],
|
|
4024
|
+
validate(){},
|
|
4025
|
+
};
|
|
4026
|
+
|
|
4027
|
+
/** What a local target step is answered as, by the connection type it stands
|
|
4028
|
+
in for (`replaces`): Echo's own connection, or Recorded's -- each with no
|
|
3812
4029
|
address, key or model. */
|
|
3813
4030
|
const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
|
|
4031
|
+
const RECORDED_CONNECTION = { name: "Recorded", url: "", model: "", type: "recorded" };
|
|
4032
|
+
const LOCAL_CONNECTIONS =
|
|
4033
|
+
{ echo: ECHO_CONNECTION, recorded: RECORDED_CONNECTION };
|
|
4034
|
+
/** The connection a local target step stands in for, by its entry. */
|
|
4035
|
+
const localConnOf = (entry ) =>
|
|
4036
|
+
LOCAL_CONNECTIONS[entry?.replaces ?? ""] ?? ECHO_CONNECTION;
|
|
3814
4037
|
|
|
3815
4038
|
// Responses: how a job's reply is read, in order -- possibly not at all.
|
|
3816
4039
|
|
|
@@ -3950,6 +4173,16 @@ function targetProfileOf(doc
|
|
|
3950
4173
|
return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
|
|
3951
4174
|
}
|
|
3952
4175
|
|
|
4176
|
+
/** Whether target [i] is a Recorded target: it replays each call's recorded
|
|
4177
|
+
reply rather than sending anything (its job-1 step answers locally with a
|
|
4178
|
+
connection that `replays`). A recorded-reply metric compares the recorded
|
|
4179
|
+
reply with itself there, so it says nothing (n/a). Asked through the
|
|
4180
|
+
registry, never a step id. */
|
|
4181
|
+
function recordedTargetOf(doc , i ) {
|
|
4182
|
+
const entry = targetEntryOf(doc, i, 0);
|
|
4183
|
+
return !!(entry?.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays);
|
|
4184
|
+
}
|
|
4185
|
+
|
|
3953
4186
|
/** [doc] with target [i]'s step in job [k] changed: [change] merged into
|
|
3954
4187
|
it, or the step [change] makes of it. */
|
|
3955
4188
|
function withTarget (doc , i , k ,
|
|
@@ -4641,11 +4874,16 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4641
4874
|
// what it is read against. Over Prompt only the prompt is the reply.
|
|
4642
4875
|
// What target [i] asks in job [k]: Echo's own connection for a step
|
|
4643
4876
|
// answered here, else its profile as the lookups hold it.
|
|
4644
|
-
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
4877
|
+
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k))
|
|
4645
4878
|
: c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
|
|
4646
4879
|
const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
|
|
4647
4880
|
for (let k = 0; k < n; k++) {
|
|
4648
4881
|
const i = targets.findIndex((_, i) => {
|
|
4882
|
+
const entry = targetEntryOf(doc, i, k);
|
|
4883
|
+
// A step with no prompt of its own, or a replaying target (Recorded,
|
|
4884
|
+
// which answers from the record), is never asked for one.
|
|
4885
|
+
if (!entry?.fields?.includes("prompt")) return false;
|
|
4886
|
+
if (entry.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays) return false;
|
|
4649
4887
|
const words = targetStepOf(doc, i, k)?.prompt;
|
|
4650
4888
|
return (!isStr(words) || !words.trim())
|
|
4651
4889
|
&& !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
|
|
@@ -4658,12 +4896,14 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4658
4896
|
const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
|
|
4659
4897
|
.find(n => !/\.(?:txt|md|csv)$/i.test(n));
|
|
4660
4898
|
targets.forEach((_, i) => {
|
|
4661
|
-
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
4899
|
+
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k)) : lookup(targetProfileOf(doc, i, k) )));
|
|
4662
4900
|
conns.forEach((p , k ) => {
|
|
4663
4901
|
if (!local(p)) return;
|
|
4664
|
-
const
|
|
4665
|
-
|
|
4666
|
-
|
|
4902
|
+
const entry = CONNECTION_TYPES[typeOf(p )] ;
|
|
4903
|
+
// A replaying target (Recorded) answers from the record, not the item
|
|
4904
|
+
// text, so the text rules do not apply -- only the job-1 one does.
|
|
4905
|
+
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${entry.label} answers only job 1${entry.replays ? "" : ", with the item's own text"}`);
|
|
4906
|
+
else if (nonText && !entry.replays) bad.push(`${targetLabel(doc, i)}: ${entry.label} answers each text item with its own text, and ${nonText} is not text`);
|
|
4667
4907
|
});
|
|
4668
4908
|
// A connection answers what its target's step asks: words, or a whole request.
|
|
4669
4909
|
conns.forEach((p , k ) => {
|
|
@@ -5261,7 +5501,7 @@ function stagesFor(run , i )
|
|
|
5261
5501
|
if (targetEntryOf(run, i, k)?.local) {
|
|
5262
5502
|
const id = targetProfileOf(run, i, k)?.id;
|
|
5263
5503
|
const held = id != null ? run.profiles?.[id] : undefined;
|
|
5264
|
-
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...
|
|
5504
|
+
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...localConnOf(targetEntryOf(run, i, k)) };
|
|
5265
5505
|
}
|
|
5266
5506
|
const id = targetProfileOf(run, i, k) .id;
|
|
5267
5507
|
return { id, ...(run.profiles?.[id] || {}) };
|
|
@@ -5276,7 +5516,7 @@ function stagesFor(run , i )
|
|
|
5276
5516
|
function scenarioProfile(run , i ) {
|
|
5277
5517
|
const id = targetsOf(run)[i]?.profile?.id;
|
|
5278
5518
|
// A target that asks nothing ran as Echo.
|
|
5279
|
-
if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name:
|
|
5519
|
+
if (id == null && targetEntryOf(run, i, 0)?.local) { const lc = localConnOf(targetEntryOf(run, i, 0)); return { id: "", name: lc.name, settings: { ...lc } }; }
|
|
5280
5520
|
const conn = id != null ? run.profiles?.[id] : null;
|
|
5281
5521
|
return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
|
|
5282
5522
|
}
|
|
@@ -5337,9 +5577,9 @@ function itemScores(run , kase , res
|
|
|
5337
5577
|
return Object.keys(out).length ? out : null;
|
|
5338
5578
|
}
|
|
5339
5579
|
|
|
5340
|
-
/** What production
|
|
5341
|
-
call says, from the item's record, or nobody does. */
|
|
5342
|
-
function
|
|
5580
|
+
/** What production recorded as the reply to an item, for a run's metrics: its
|
|
5581
|
+
last job's call says, from the item's record, or nobody does. */
|
|
5582
|
+
function recordedReplyOf(run , record ) {
|
|
5343
5583
|
const last = run.jobs.at(-1), flow = flowStepOf(last);
|
|
5344
5584
|
return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
|
|
5345
5585
|
}
|
|
@@ -5635,20 +5875,20 @@ registerMetrics({ registerKinds });
|
|
|
5635
5875
|
export {
|
|
5636
5876
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
5637
5877
|
loopReplyError, preparedSize,
|
|
5638
|
-
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
5878
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, takeAsExpected, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
5639
5879
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
5640
5880
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
5641
5881
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
5642
5882
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
5643
5883
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
5644
5884
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
5645
|
-
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync,
|
|
5646
|
-
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
5885
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
|
|
5886
|
+
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
|
|
5647
5887
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
5648
5888
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5649
5889
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5650
5890
|
contentOf, withContent, replyOf,
|
|
5651
|
-
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5891
|
+
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
|
|
5652
5892
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5653
5893
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5654
5894
|
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|