evals-lab 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +108 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +464 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/metrics/builtin.mjs +113 -52
- package/lab/run-evals.js +32 -8
- package/lab/server.py +1174 -179
- package/lab/web/dist/assets/gallery-SnUhXRBn.js +3 -0
- package/lab/web/dist/assets/main-Ca7o-nM0.css +1 -0
- package/lab/web/dist/assets/main-Dru4_P5G.js +20 -0
- package/lab/web/dist/assets/tokens-0az9gfTq.js +58 -0
- package/lab/web/dist/assets/tokens-CqWJKhOx.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +0 -3
- package/lab/web/dist/assets/main-DDoeU6hq.css +0 -1
- package/lab/web/dist/assets/main-nz6Q4jVm.js +0 -21
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +0 -59
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
import yaml from "./js-yaml.mjs";
|
|
36
36
|
|
|
37
37
|
import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
|
|
38
|
+
import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
|
|
39
|
+
import { flowApi, } from "./flows/flowApi.mjs";
|
|
38
40
|
import registerMetrics from "./metrics/builtin.mjs";
|
|
39
41
|
import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
|
|
40
42
|
|
|
@@ -543,6 +545,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
543
545
|
|
|
544
546
|
|
|
545
547
|
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
|
|
546
554
|
|
|
547
555
|
|
|
548
556
|
|
|
@@ -646,9 +654,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
646
654
|
|
|
647
655
|
|
|
648
656
|
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
|
|
652
663
|
|
|
653
664
|
|
|
654
665
|
|
|
@@ -659,9 +670,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
659
670
|
|
|
660
671
|
/** What a metric needs besides the reply: a model to grade with. */
|
|
661
672
|
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
|
|
665
681
|
|
|
666
682
|
|
|
667
683
|
/** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
|
|
@@ -711,6 +727,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
711
727
|
|
|
712
728
|
|
|
713
729
|
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
|
|
714
737
|
|
|
715
738
|
|
|
716
739
|
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
@@ -785,6 +808,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
785
808
|
|
|
786
809
|
|
|
787
810
|
|
|
811
|
+
|
|
812
|
+
|
|
813
|
+
|
|
788
814
|
|
|
789
815
|
|
|
790
816
|
|
|
@@ -869,9 +895,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
869
895
|
|
|
870
896
|
|
|
871
897
|
|
|
898
|
+
|
|
872
899
|
|
|
873
|
-
|
|
874
|
-
|
|
900
|
+
|
|
875
901
|
|
|
876
902
|
|
|
877
903
|
/** Lookups, and whether it is a run document being validated. */
|
|
@@ -880,11 +906,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
880
906
|
|
|
881
907
|
|
|
882
908
|
/** A kind of Source: what one of its items is, and what the page and the
|
|
883
|
-
runner do with it. A reader asks the entry; it never branches on an id.
|
|
909
|
+
runner do with it. A reader asks the entry; it never branches on an id.
|
|
910
|
+
The row's `type` is the kind; a workflow kind's platform -- which flow
|
|
911
|
+
engine a row speaks to -- is `config.platform`, a WORKFLOW_PLATFORMS id. */
|
|
884
912
|
|
|
885
913
|
|
|
886
914
|
|
|
887
915
|
|
|
916
|
+
|
|
917
|
+
|
|
888
918
|
|
|
889
919
|
|
|
890
920
|
|
|
@@ -893,10 +923,68 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
893
923
|
|
|
894
924
|
|
|
895
925
|
|
|
926
|
+
|
|
927
|
+
|
|
928
|
+
|
|
929
|
+
|
|
896
930
|
|
|
897
931
|
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
|
|
898
935
|
|
|
899
936
|
|
|
937
|
+
/** What a workflow of one platform -- a flow engine -- answers: the pieces
|
|
938
|
+
that were Power-Automate-specific, lifted behind the platform id so
|
|
939
|
+
generic code asks the entry and never names a platform. Power Automate is
|
|
940
|
+
the one built; a second (Logic App, n8n) is one more entry. */
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
|
|
973
|
+
|
|
974
|
+
|
|
975
|
+
|
|
976
|
+
/** A platform's management API reader (flowApi.ts for Power Automate); its
|
|
977
|
+
shape is the platform's own, so generic code holds it opaquely. */
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
const WORKFLOW_PLATFORMS = Object.create(null);
|
|
981
|
+
/** The platform a workflow Source speaks to, from its `config.platform`;
|
|
982
|
+
undefined for a row that is not a workflow or names an unknown platform. */
|
|
983
|
+
const platformOf = (src ) => {
|
|
984
|
+
const id = src?.config?.platform;
|
|
985
|
+
return isStr(id) ? WORKFLOW_PLATFORMS[id] : undefined;
|
|
986
|
+
};
|
|
987
|
+
|
|
900
988
|
|
|
901
989
|
|
|
902
990
|
|
|
@@ -961,10 +1049,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
961
1049
|
the item, and a grader, where the run has them. */
|
|
962
1050
|
|
|
963
1051
|
|
|
1052
|
+
|
|
1053
|
+
|
|
1054
|
+
|
|
964
1055
|
|
|
965
1056
|
|
|
966
1057
|
|
|
967
|
-
|
|
1058
|
+
|
|
968
1059
|
|
|
969
1060
|
|
|
970
1061
|
|
|
@@ -1059,16 +1150,43 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
1059
1150
|
|
|
1060
1151
|
|
|
1061
1152
|
|
|
1062
|
-
/** The dataset body's version:
|
|
1063
|
-
|
|
1064
|
-
|
|
1153
|
+
/** The dataset body's version: 8 reads the recorded-reply metric ids under
|
|
1154
|
+
their new names (#299); 7 is an eval group (§17); 6 marks itself; 5 named
|
|
1155
|
+
its Source; 4 and earlier, neither. */
|
|
1156
|
+
const DATASET_BODY_VERSION = 8 ;
|
|
1157
|
+
|
|
1158
|
+
/** The recorded-reply metric type ids renamed at version 8 (#299): the family
|
|
1159
|
+
read "production" before the Recorded target gave it a home. The ids are
|
|
1160
|
+
stored tokens inside eval group bodies and a pipeline's private group, so
|
|
1161
|
+
they are mapped wherever a body is read rather than rewritten in place. */
|
|
1162
|
+
const RECORDED_IDS = {
|
|
1163
|
+
"equals-production": "same-as-recorded",
|
|
1164
|
+
"fields-equal-production": "fields-equal-recorded",
|
|
1165
|
+
"same-parse-outcome": "same-parse-as-recorded",
|
|
1166
|
+
};
|
|
1167
|
+
const renameMetric = (m ) =>
|
|
1168
|
+
isObj(m) && isStr(m.type) && RECORDED_IDS[m.type] ? { ...m, type: RECORDED_IDS[m.type] } : m;
|
|
1169
|
+
const renameMetrics = (list ) =>
|
|
1170
|
+
Array.isArray(list) ? list.map(renameMetric) : list;
|
|
1171
|
+
/** A version-7 body (or a freshly made version-8 one) with its recorded-reply
|
|
1172
|
+
metric ids read under their version-8 names, at version 8. */
|
|
1173
|
+
function recordedIdsV7(b ) {
|
|
1174
|
+
return {
|
|
1175
|
+
...b,
|
|
1176
|
+
version: DATASET_BODY_VERSION,
|
|
1177
|
+
...(Array.isArray(b.every) ? { every: renameMetrics(b.every) } : {}),
|
|
1178
|
+
...(Array.isArray(b.run) ? { run: renameMetrics(b.run) } : {}),
|
|
1179
|
+
...(Array.isArray(b.cases) ? { cases: b.cases.map((c ) =>
|
|
1180
|
+
isObj(c) && Array.isArray(c.metrics) ? { ...c, metrics: renameMetrics(c.metrics) } : c) } : {}),
|
|
1181
|
+
} ;
|
|
1182
|
+
}
|
|
1065
1183
|
|
|
1066
1184
|
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
1067
1185
|
reply read as a list, as well as one read as text. */
|
|
1068
1186
|
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
1069
1187
|
|
|
1070
1188
|
/**
|
|
1071
|
-
* [body] as this version of a dataset (
|
|
1189
|
+
* [body] as this version of a dataset (8), from any earlier one. Version 1
|
|
1072
1190
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1073
1191
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1074
1192
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
@@ -1081,8 +1199,12 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
|
1081
1199
|
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1082
1200
|
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1083
1201
|
* the body says its version. Version 6 is a group's cases alone: version 7
|
|
1084
|
-
* adds its scoring, grader, Every item and Whole run (`groupOfV6`)
|
|
1085
|
-
*
|
|
1202
|
+
* adds its scoring, grader, Every item and Whole run (`groupOfV6`). Version 8
|
|
1203
|
+
* reads the recorded-reply metric ids (`equals-production` and its two
|
|
1204
|
+
* siblings) under their new names (`RECORDED_IDS`, #299), in a group's own
|
|
1205
|
+
* metrics and its cases' -- a pipeline's private group is a group body, so
|
|
1206
|
+
* this covers it too. A stored row is read that way rather than rewritten.
|
|
1207
|
+
* Every reader of a
|
|
1086
1208
|
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1087
1209
|
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1088
1210
|
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
@@ -1093,19 +1215,25 @@ function upgradeDatasetBody (body ) {
|
|
|
1093
1215
|
if (!isObj(body)) return body;
|
|
1094
1216
|
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1095
1217
|
const b = body ;
|
|
1096
|
-
|
|
1097
|
-
|
|
1218
|
+
if ("version" in b) {
|
|
1219
|
+
// Version 7 is an eval group whose recorded-reply metrics are read under
|
|
1220
|
+
// their version-8 ids; version 6 is its cases alone. Any other version is
|
|
1221
|
+
// one this lab does not read.
|
|
1222
|
+
if (b.version === 7) return recordedIdsV7(b);
|
|
1223
|
+
if (b.version === 6) return recordedIdsV7(groupOfV6(b));
|
|
1224
|
+
return body;
|
|
1225
|
+
}
|
|
1098
1226
|
// A body that names its Source, even as null, is version 5.
|
|
1099
1227
|
if ("source" in b) {
|
|
1100
|
-
return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
|
|
1228
|
+
return recordedIdsV7(groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) }));
|
|
1101
1229
|
}
|
|
1102
1230
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1103
1231
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1104
1232
|
if (!raw) return body;
|
|
1105
|
-
return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
|
|
1233
|
+
return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
|
|
1106
1234
|
}
|
|
1107
1235
|
|
|
1108
|
-
/** A version-6 body as a version-
|
|
1236
|
+
/** A version-6 body as a version-8 eval group: scored All, the lab's
|
|
1109
1237
|
grader, and no metrics of its own for every item or the whole run --
|
|
1110
1238
|
what a Metrics eval naming the dataset with none of its own graded, which
|
|
1111
1239
|
is what every Graded set converted to. */
|
|
@@ -1209,6 +1337,10 @@ function datasetRules(body ) {
|
|
|
1209
1337
|
|
|
1210
1338
|
|
|
1211
1339
|
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
|
|
1343
|
+
|
|
1212
1344
|
|
|
1213
1345
|
|
|
1214
1346
|
/** An HTTP request as a connection type builds it, for the relay to send. */
|
|
@@ -1232,6 +1364,10 @@ function datasetRules(body ) {
|
|
|
1232
1364
|
|
|
1233
1365
|
|
|
1234
1366
|
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
|
|
1235
1371
|
|
|
1236
1372
|
|
|
1237
1373
|
|
|
@@ -1240,8 +1376,9 @@ function datasetRules(body ) {
|
|
|
1240
1376
|
|
|
1241
1377
|
|
|
1242
1378
|
|
|
1243
|
-
|
|
1244
|
-
|
|
1379
|
+
|
|
1380
|
+
|
|
1381
|
+
|
|
1245
1382
|
|
|
1246
1383
|
|
|
1247
1384
|
|
|
@@ -1249,7 +1386,12 @@ function datasetRules(body ) {
|
|
|
1249
1386
|
|
|
1250
1387
|
|
|
1251
1388
|
|
|
1252
|
-
|
|
1389
|
+
|
|
1390
|
+
|
|
1391
|
+
|
|
1392
|
+
|
|
1393
|
+
|
|
1394
|
+
|
|
1253
1395
|
|
|
1254
1396
|
|
|
1255
1397
|
|
|
@@ -1589,6 +1731,8 @@ function ollamaType(id , label , cloud ) {
|
|
|
1589
1731
|
body: { model: conn.model, messages: [message], options, stream: false },
|
|
1590
1732
|
};
|
|
1591
1733
|
},
|
|
1734
|
+
// Ollama's own: the schema is the format.
|
|
1735
|
+
structured: (body, schema) => ({ ...body, format: schema }),
|
|
1592
1736
|
parseReply(j) {
|
|
1593
1737
|
const r = j ;
|
|
1594
1738
|
return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
|
|
@@ -1620,28 +1764,55 @@ CONNECTION_TYPES.echo = {
|
|
|
1620
1764
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1621
1765
|
};
|
|
1622
1766
|
|
|
1767
|
+
/** A call's recorded result as transport-level text -- what a model asked the
|
|
1768
|
+
same way would have come back with, before the flow reads it. The runner's
|
|
1769
|
+
replay branch reads it as the flow reads a reply (`httpReplyOf`); this is
|
|
1770
|
+
the fallback where there is no Read as to read it through. */
|
|
1771
|
+
function recordedRaw(record ) {
|
|
1772
|
+
const result = isObj(record) && isObj(record.result) ? record.result : null;
|
|
1773
|
+
if (!result) return "";
|
|
1774
|
+
return isStr(result.body) ? result.body : JSON.stringify(result.body ?? "");
|
|
1775
|
+
}
|
|
1776
|
+
|
|
1777
|
+
// Recorded (#299): a target that replays each call's recorded reply, sending
|
|
1778
|
+
// nothing and holding no key or model. The runner reads the recorded result as
|
|
1779
|
+
// the flow reads a reply; `local` is the no-Read-as fallback.
|
|
1780
|
+
CONNECTION_TYPES.recorded = {
|
|
1781
|
+
id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
|
|
1782
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
1783
|
+
local: (_item, _sent, record) => recordedRaw(record),
|
|
1784
|
+
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1785
|
+
};
|
|
1786
|
+
|
|
1623
1787
|
/** A local type's answer to one stage, as a model's would come back. An item
|
|
1624
1788
|
with no text of its own (Prompt only) is answered from the prompt, and
|
|
1625
1789
|
that reply is read against nothing -- the prompt is the reply, not an
|
|
1626
|
-
instruction it could repeat.
|
|
1627
|
-
|
|
1628
|
-
|
|
1790
|
+
instruction it could repeat. A replaying type (Recorded) is handed the
|
|
1791
|
+
item's record to answer from. */
|
|
1792
|
+
function localAnswer(conn , text , sent , record ) {
|
|
1793
|
+
const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent, record);
|
|
1629
1794
|
return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
|
|
1630
1795
|
}
|
|
1631
1796
|
|
|
1797
|
+
/** A chat completions body asking for a reply held to [schema]: OpenAI's
|
|
1798
|
+
response_format, which llama-server and most compatible servers take too. */
|
|
1799
|
+
const chatSchema = (body , schema ) =>
|
|
1800
|
+
({ ...body, response_format: { type: "json_schema", json_schema: { name: "reply", strict: true, schema } } });
|
|
1801
|
+
|
|
1632
1802
|
CONNECTION_TYPES["openai-compatible"] = {
|
|
1633
1803
|
id: "openai-compatible", label: "OpenAI-compatible",
|
|
1634
1804
|
description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
|
|
1635
1805
|
settings: [
|
|
1636
|
-
|
|
1806
|
+
// A reasoning model refuses a temperature.
|
|
1807
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1808
|
+
appliesTo: (conn) => !/^(gpt-5|o\d)/i.test(String(conn.model ?? "")) },
|
|
1637
1809
|
{ key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1638
1810
|
],
|
|
1639
|
-
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1640
|
-
//
|
|
1641
|
-
//
|
|
1811
|
+
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1812
|
+
// max_completion_tokens stays with this type, as the hosted providers
|
|
1813
|
+
// reject the fields a llama.cpp server takes.
|
|
1642
1814
|
request(conn, prompt, dataUrl, base, key) {
|
|
1643
1815
|
const hosted = isHostedUrl(base);
|
|
1644
|
-
const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
|
|
1645
1816
|
const temp = asNumber(conn.temperature);
|
|
1646
1817
|
const predict = asNumber(conn.nPredict);
|
|
1647
1818
|
return {
|
|
@@ -1649,7 +1820,7 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1649
1820
|
headers: bearer(key),
|
|
1650
1821
|
body: {
|
|
1651
1822
|
model: conn.model,
|
|
1652
|
-
...(
|
|
1823
|
+
...(temp == null ? {} : { temperature: temp }),
|
|
1653
1824
|
...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
|
|
1654
1825
|
messages: [{ role: "user", content: [
|
|
1655
1826
|
{ type: "text", text: prompt },
|
|
@@ -1658,15 +1829,23 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1658
1829
|
},
|
|
1659
1830
|
};
|
|
1660
1831
|
},
|
|
1832
|
+
structured: chatSchema,
|
|
1661
1833
|
parseReply: chatReply,
|
|
1662
1834
|
listModels, parseModels: idsFrom,
|
|
1663
1835
|
};
|
|
1664
1836
|
|
|
1837
|
+
/** The Claude models that refuse a sampling setting: Opus from 4.7, and
|
|
1838
|
+
every Opus, Sonnet, Fable and Mythos from 5. Older ones (Opus 4.6, Sonnet
|
|
1839
|
+
4.6, Haiku 4.5) take one. */
|
|
1840
|
+
const CLAUDE_FIXED_SAMPLING = /claude-(opus-4-[7-9]|(opus|sonnet|fable|mythos)-[5-9])/i;
|
|
1841
|
+
|
|
1665
1842
|
CONNECTION_TYPES.anthropic = {
|
|
1666
1843
|
id: "anthropic", label: "Anthropic",
|
|
1667
1844
|
description: "Claude, through Anthropic's Messages API.",
|
|
1845
|
+
url: "https://api.anthropic.com",
|
|
1668
1846
|
settings: [
|
|
1669
|
-
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: ""
|
|
1847
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1848
|
+
appliesTo: (conn) => !CLAUDE_FIXED_SAMPLING.test(String(conn.model ?? "")) },
|
|
1670
1849
|
{ key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1671
1850
|
],
|
|
1672
1851
|
// The Messages API: the x-api-key and anthropic-version headers, base64
|
|
@@ -1691,6 +1870,9 @@ CONNECTION_TYPES.anthropic = {
|
|
|
1691
1870
|
},
|
|
1692
1871
|
};
|
|
1693
1872
|
},
|
|
1873
|
+
// The Messages API's structured outputs. A model that cannot hold one
|
|
1874
|
+
// refuses the request, and the grader asks again without it.
|
|
1875
|
+
structured: (body, schema) => ({ ...body, output_config: { ...(body.output_config ), format: { type: "json_schema", schema } } }),
|
|
1694
1876
|
parseReply(j) {
|
|
1695
1877
|
const r = j ;
|
|
1696
1878
|
const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
|
|
@@ -1763,6 +1945,8 @@ CONNECTION_TYPES["llama.cpp"] = {
|
|
|
1763
1945
|
},
|
|
1764
1946
|
};
|
|
1765
1947
|
},
|
|
1948
|
+
// llama-server takes OpenAI's response_format, a schema and all.
|
|
1949
|
+
structured: chatSchema,
|
|
1766
1950
|
parseReply: chatReply,
|
|
1767
1951
|
listModels, parseModels: idsFrom,
|
|
1768
1952
|
// A reply that arrived is still a failure when this profile asked for the
|
|
@@ -1958,12 +2142,14 @@ function extraHeaders(raw ) {
|
|
|
1958
2142
|
}
|
|
1959
2143
|
return out;
|
|
1960
2144
|
}
|
|
1961
|
-
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2145
|
+
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2146
|
+
path. A step's path is the whole path production called, so an address
|
|
2147
|
+
entered with one (`…/v1/messages`) would send it twice. */
|
|
1962
2148
|
function originBase(raw ) {
|
|
1963
|
-
let s = String(raw || "").trim()
|
|
2149
|
+
let s = String(raw || "").trim();
|
|
1964
2150
|
if (!s) return "";
|
|
1965
2151
|
if (!/^https?:\/\//.test(s)) s = "https://" + s;
|
|
1966
|
-
return s;
|
|
2152
|
+
try { return new URL(s).origin; } catch { return s.replace(/\/+$/, ""); }
|
|
1967
2153
|
}
|
|
1968
2154
|
const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
|
|
1969
2155
|
CONNECTION_TYPES.http = {
|
|
@@ -2090,9 +2276,14 @@ function splitOllama(p ) {
|
|
|
2090
2276
|
* connection carries its type's settings under `options`, so they are merged
|
|
2091
2277
|
* up before the type reads them.
|
|
2092
2278
|
*/
|
|
2093
|
-
function connectionRequest(conn , prompt , dataUrl , base , key
|
|
2279
|
+
function connectionRequest(conn , prompt , dataUrl , base , key ,
|
|
2280
|
+
schema ) {
|
|
2094
2281
|
const flat = { ...conn, ...(conn?.options || {}) };
|
|
2095
|
-
|
|
2282
|
+
const type = typeEntry(flat);
|
|
2283
|
+
// A setting the model refuses is not sent, whatever the profile holds.
|
|
2284
|
+
for (const s of type.settings) if (s.appliesTo && !s.appliesTo(flat)) delete flat[s.key];
|
|
2285
|
+
const req = type.request(flat, prompt, dataUrl, base, key);
|
|
2286
|
+
return schema && type.structured ? { ...req, body: type.structured(req.body , schema) } : req;
|
|
2096
2287
|
}
|
|
2097
2288
|
|
|
2098
2289
|
/** Where [conn]'s requests hang off: its type's own reading of its address,
|
|
@@ -2794,6 +2985,15 @@ function registerKinds(k ) {
|
|
|
2794
2985
|
|
|
2795
2986
|
|
|
2796
2987
|
|
|
2988
|
+
|
|
2989
|
+
|
|
2990
|
+
|
|
2991
|
+
|
|
2992
|
+
|
|
2993
|
+
|
|
2994
|
+
|
|
2995
|
+
|
|
2996
|
+
|
|
2797
2997
|
|
|
2798
2998
|
|
|
2799
2999
|
function pluginHost(pluginId ) {
|
|
@@ -2821,6 +3021,14 @@ function pluginHost(pluginId ) {
|
|
|
2821
3021
|
taken(CONNECTION_TYPES, "connection type", [t.id]);
|
|
2822
3022
|
CONNECTION_TYPES[t.id] = t;
|
|
2823
3023
|
},
|
|
3024
|
+
registerWorkflowPlatform(t) {
|
|
3025
|
+
taken(WORKFLOW_PLATFORMS, "workflow platform", [t.id]);
|
|
3026
|
+
WORKFLOW_PLATFORMS[t.id] = t;
|
|
3027
|
+
},
|
|
3028
|
+
registerWizard(w) {
|
|
3029
|
+
taken(WIZARDS, "wizard", [w.id]);
|
|
3030
|
+
WIZARDS[w.id] = w;
|
|
3031
|
+
},
|
|
2824
3032
|
};
|
|
2825
3033
|
}
|
|
2826
3034
|
|
|
@@ -2918,6 +3126,13 @@ function withSlugs (list
|
|
|
2918
3126
|
/** Whether [s] is a slug as a profile may hold one. */
|
|
2919
3127
|
const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
|
|
2920
3128
|
|
|
3129
|
+
/** Where [p] connects: its address, or blank, its type's own -- an Anthropic
|
|
3130
|
+
profile left blank reaches Anthropic. Blank for a type with none means the
|
|
3131
|
+
lab's own Ollama, which the runner and the relay fill in. */
|
|
3132
|
+
function addressOf(p ) {
|
|
3133
|
+
return String(p.url ?? "").trim() || CONNECTION_TYPES[typeOf(p )]?.url || "";
|
|
3134
|
+
}
|
|
3135
|
+
|
|
2921
3136
|
/** A stored profile as a run carries it: its request settings, no key. */
|
|
2922
3137
|
function connectionOf(p ) {
|
|
2923
3138
|
const type = typeOf(p);
|
|
@@ -2926,7 +3141,7 @@ function connectionOf(p ) {
|
|
|
2926
3141
|
// temperature is a common field, carried at the top like px/format/quality;
|
|
2927
3142
|
// the options bag holds the type's own settings only.
|
|
2928
3143
|
for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
|
|
2929
|
-
const out = { name: p.name || "unnamed", url: p
|
|
3144
|
+
const out = { name: p.name || "unnamed", url: addressOf(p), model: p.model || "", type };
|
|
2930
3145
|
if (isSlug(p.slug)) out.slug = p.slug;
|
|
2931
3146
|
for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
|
|
2932
3147
|
if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
|
|
@@ -2997,26 +3212,120 @@ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
|
|
|
2997
3212
|
SOURCE_TYPES.files = {
|
|
2998
3213
|
label: "File Library",
|
|
2999
3214
|
description: "A folder of images and text files you upload.",
|
|
3215
|
+
group: "Files",
|
|
3000
3216
|
noun: "file",
|
|
3001
3217
|
uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
|
|
3002
3218
|
sendsImage: true,
|
|
3219
|
+
hasActions: false,
|
|
3003
3220
|
itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
|
|
3004
3221
|
};
|
|
3005
3222
|
|
|
3006
|
-
// A
|
|
3007
|
-
//
|
|
3008
|
-
// the
|
|
3009
|
-
//
|
|
3010
|
-
//
|
|
3011
|
-
|
|
3012
|
-
|
|
3013
|
-
|
|
3223
|
+
// A workflow Source: a cloud flow, read through a platform (WORKFLOW_PLATFORMS,
|
|
3224
|
+
// below) its row names in `config.platform`. Its items are records -- one call
|
|
3225
|
+
// of one of the flow's steps each: what the step's template read, the request
|
|
3226
|
+
// production sent and what came back (flows/record.ts) -- and it keeps the
|
|
3227
|
+
// flow's definition beside them. The kind is generic; the platform owns what
|
|
3228
|
+
// is engine-specific, and the type's shown label is the platform's.
|
|
3229
|
+
SOURCE_TYPES.workflow = {
|
|
3230
|
+
label: "Workflow",
|
|
3231
|
+
description: "A cloud flow, with records of its steps' calls from its run history.",
|
|
3232
|
+
group: "Workflows",
|
|
3014
3233
|
noun: "record",
|
|
3015
3234
|
uploads: [".json"],
|
|
3016
3235
|
sendsImage: false,
|
|
3236
|
+
hasActions: true,
|
|
3017
3237
|
itemText: () => false,
|
|
3238
|
+
platforms: ["power-automate"], // platform: the WORKFLOW_PLATFORMS a workflow offers
|
|
3239
|
+
};
|
|
3240
|
+
|
|
3241
|
+
// Power Automate (docs/power-automate.md): the one workflow platform built.
|
|
3242
|
+
// Its pieces were the lab's Power-Automate-specific code -- flows/record.ts,
|
|
3243
|
+
// flows/wdl.ts and flowApi.ts -- gathered here so generic code calls the
|
|
3244
|
+
// entry, never a platform id.
|
|
3245
|
+
WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapter's own id
|
|
3246
|
+
id: "power-automate", // platform: its id, the row's config.platform
|
|
3247
|
+
label: "Power Automate flow",
|
|
3248
|
+
nouns: { workflow: "flow", action: "action" },
|
|
3249
|
+
signIn: "microsoft",
|
|
3250
|
+
uploads: { ".json": "application/json; charset=utf-8" },
|
|
3251
|
+
keepsDefinition: true,
|
|
3252
|
+
api: token => flowApi(token),
|
|
3253
|
+
stepsOf: flowStepsOf,
|
|
3254
|
+
actionsOf: actionsByName,
|
|
3255
|
+
readerFor,
|
|
3256
|
+
evaluate,
|
|
3257
|
+
asText,
|
|
3258
|
+
readsOf,
|
|
3259
|
+
// Copy request… writes the body back as a flow's HTTP action holds it: JSON
|
|
3260
|
+
// whose strings keep Power Automate's own `@{…}` expressions untouched, so a
|
|
3261
|
+
// rebuilt body round-trips into the flow (docs/power-automate.md).
|
|
3262
|
+
render: body => JSON.stringify(body, null, 2),
|
|
3018
3263
|
};
|
|
3019
3264
|
|
|
3265
|
+
// ---- Setup Wizards (docs/workflow-sources.md § Wizards) ---------------------
|
|
3266
|
+
//
|
|
3267
|
+
// A wizard is a registry entry whose steps each say where they happen and a
|
|
3268
|
+
// done-when predicate on lab state. Steps tick automatically: `done(lab)` reads
|
|
3269
|
+
// the cheap state the page already holds (a workflow Source exists, a pipeline
|
|
3270
|
+
// with a Recorded target exists, a run of it is in History), never a poll. One
|
|
3271
|
+
// wizard runs at a time, held in the browser-only `promptlab.wizard` store; a
|
|
3272
|
+
// header pill shows its progress and the Setup › Wizards section draws the same
|
|
3273
|
+
// walk-through. The registry lives here so a plugin registers a wizard through
|
|
3274
|
+
// the same host as every other kind (PluginHost.registerWizard, phase 7); the
|
|
3275
|
+
// page ships the built-in "Test a workflow" and owns its routes and selectors
|
|
3276
|
+
// (web/src/app/wizards.ts). The predicates run with the page's privileges and
|
|
3277
|
+
// only read lab state, so a plugin wizard is no new trust (docs/packs.md).
|
|
3278
|
+
|
|
3279
|
+
/** The slice of lab state a wizard step's `done` reads: what the page already
|
|
3280
|
+
holds, so a tick costs a predicate and not a request. A reader names a field
|
|
3281
|
+
it needs; a wizard that reads more declares it here. */
|
|
3282
|
+
|
|
3283
|
+
|
|
3284
|
+
|
|
3285
|
+
|
|
3286
|
+
|
|
3287
|
+
|
|
3288
|
+
|
|
3289
|
+
|
|
3290
|
+
|
|
3291
|
+
|
|
3292
|
+
|
|
3293
|
+
|
|
3294
|
+
|
|
3295
|
+
|
|
3296
|
+
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
|
|
3300
|
+
|
|
3301
|
+
|
|
3302
|
+
|
|
3303
|
+
|
|
3304
|
+
|
|
3305
|
+
|
|
3306
|
+
|
|
3307
|
+
|
|
3308
|
+
|
|
3309
|
+
|
|
3310
|
+
|
|
3311
|
+
|
|
3312
|
+
|
|
3313
|
+
|
|
3314
|
+
|
|
3315
|
+
|
|
3316
|
+
const WIZARDS = Object.create(null);
|
|
3317
|
+
|
|
3318
|
+
/** How far a wizard has got on [lab]: each step's tick, how many are done, and
|
|
3319
|
+
the first step not yet done -- the one the checklist marks current and Go
|
|
3320
|
+
to step heads for. `current` is -1 once every step is done. */
|
|
3321
|
+
function wizardProgress(entry , lab )
|
|
3322
|
+
|
|
3323
|
+
{
|
|
3324
|
+
const ticks = entry.steps.map((s) => s.done(lab));
|
|
3325
|
+
const done = ticks.filter(Boolean).length;
|
|
3326
|
+
return { ticks, done, total: entry.steps.length, current: ticks.findIndex((t) => !t) };
|
|
3327
|
+
}
|
|
3328
|
+
|
|
3020
3329
|
CONTENT_TYPES.source = {
|
|
3021
3330
|
label: "Source",
|
|
3022
3331
|
description: "Every file or record in a Source from the Library.",
|
|
@@ -3194,7 +3503,12 @@ function metricInput(res , kase , more )
|
|
|
3194
3503
|
const last = res.transcript?.at(-1);
|
|
3195
3504
|
const text = last?.got ?? res.raw ?? "";
|
|
3196
3505
|
return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
|
|
3197
|
-
|
|
3506
|
+
asked: last?.sent ?? null,
|
|
3507
|
+
// A case that has been given its own expected reply ("Take B as
|
|
3508
|
+
// expected") is compared with that; otherwise the Source's recorded
|
|
3509
|
+
// reply for the item.
|
|
3510
|
+
error: res.error ?? null, ms: res.ms ?? 0,
|
|
3511
|
+
recorded: typeof kase?.recorded === "string" ? kase.recorded : (more.production ?? null), kase };
|
|
3198
3512
|
}
|
|
3199
3513
|
|
|
3200
3514
|
/** A whole run's replies as one input: every reply's text together, and
|
|
@@ -3208,7 +3522,7 @@ function runInput(ress , plain )
|
|
|
3208
3522
|
terms.push(...(r.terms || []));
|
|
3209
3523
|
ms += r.ms ?? 0;
|
|
3210
3524
|
}
|
|
3211
|
-
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms,
|
|
3525
|
+
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, asked: null, kase: null };
|
|
3212
3526
|
}
|
|
3213
3527
|
|
|
3214
3528
|
/** One metric's reading of [input]: `of`, `not` and a failure all applied,
|
|
@@ -3469,16 +3783,21 @@ EVAL_TYPES.group = {
|
|
|
3469
3783
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3470
3784
|
.filter(Boolean).join(" ") })),
|
|
3471
3785
|
profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
|
|
3472
|
-
// A run carries the
|
|
3473
|
-
//
|
|
3474
|
-
//
|
|
3786
|
+
// A run carries the grader each group asks, so its profile goes with the
|
|
3787
|
+
// run: a link's is the Library group's own, or the lab's where it names
|
|
3788
|
+
// none -- the run's copy of the link carries it. A private group takes the
|
|
3789
|
+
// lab's where it names none and may ask one: a model-graded metric of its
|
|
3790
|
+
// own, or a case's.
|
|
3475
3791
|
resolve(t, ctx){
|
|
3476
|
-
|
|
3477
|
-
const
|
|
3792
|
+
const ref = (g ) => (isRef(g) ? { id: g.id, name: g.name } : null);
|
|
3793
|
+
const lab = ref(ctx.grader);
|
|
3478
3794
|
if (isObj(t.own)) {
|
|
3479
|
-
return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3795
|
+
return lab && !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3796
|
+
? { ...t, own: { ...t.own, grader: lab } } : t;
|
|
3480
3797
|
}
|
|
3481
|
-
|
|
3798
|
+
if (!isRef(t.group) || isRef(t.grader)) return t;
|
|
3799
|
+
const grader = ref(ctx.groups?.find(g => g.id === (t.group ).id)?.grader) ?? lab;
|
|
3800
|
+
return grader ? { ...t, grader } : t;
|
|
3482
3801
|
},
|
|
3483
3802
|
wholeRun: wholeRunGroup,
|
|
3484
3803
|
verdict: (t, ress, kind) => {
|
|
@@ -3629,10 +3948,38 @@ function readGroupRun(group , ress
|
|
|
3629
3948
|
return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
|
|
3630
3949
|
}
|
|
3631
3950
|
|
|
3951
|
+
/**
|
|
3952
|
+
* [body] with the cases named in [replies] expecting the reply given for
|
|
3953
|
+
* their item -- Results' "Take B as expected" (docs/workflow-sources.md §
|
|
3954
|
+
* Calls are graded cases). Each such case keeps everything else and gains
|
|
3955
|
+
* `recorded` (the reply the recorded-reply metrics compare against, read in
|
|
3956
|
+
* place of the Source's) and `todo: false`, so a later run grades the item
|
|
3957
|
+
* against that reply. A selected item the group has no case for, or one whose
|
|
3958
|
+
* reply is missing (Target B errored, or is the Recorded target), is left
|
|
3959
|
+
* untouched and named in `missing`; `taken` names the items whose expected was
|
|
3960
|
+
* written. Pure: the same body and replies give the same body, so Undo is the
|
|
3961
|
+
* body as it was.
|
|
3962
|
+
*/
|
|
3963
|
+
function takeAsExpected(body , replies )
|
|
3964
|
+
{
|
|
3965
|
+
const byItem = new Map ();
|
|
3966
|
+
body.cases.forEach((c, i) => { const item = caseItem(c); if (item) byItem.set(item, i); });
|
|
3967
|
+
const cases = body.cases.slice();
|
|
3968
|
+
const taken = [], missing = [];
|
|
3969
|
+
for (const item of Object.keys(replies)) {
|
|
3970
|
+
const reply = replies[item];
|
|
3971
|
+
const at = byItem.get(item);
|
|
3972
|
+
if (at == null || typeof reply !== "string") { missing.push(item); continue; }
|
|
3973
|
+
cases[at] = { ...cases[at], recorded: reply, todo: false };
|
|
3974
|
+
taken.push(item);
|
|
3975
|
+
}
|
|
3976
|
+
return { body: { ...body, cases }, taken, missing };
|
|
3977
|
+
}
|
|
3978
|
+
|
|
3632
3979
|
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3633
3980
|
function graderCtx(t , more ) {
|
|
3634
3981
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
3635
|
-
return ask ? { ask } : {};
|
|
3982
|
+
return { ...(ask ? { ask } : {}), ...(more.recordedTarget ? { recordedTarget: true } : {}) };
|
|
3636
3983
|
}
|
|
3637
3984
|
|
|
3638
3985
|
/** A metric's options in a few words -- "watering", "^\\{" -- or nothing
|
|
@@ -3647,6 +3994,9 @@ function metricSummary(m ) {
|
|
|
3647
3994
|
const v = m[o.key];
|
|
3648
3995
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3649
3996
|
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
3997
|
+
// The choices ticked, by their labels, where they are not a new one's.
|
|
3998
|
+
if (o.type === "multi") return !Array.isArray(v) || JSON.stringify(v) === JSON.stringify(fresh[o.key]) ? ""
|
|
3999
|
+
: o.choices.filter(c => v.includes(c.value)).map(c => c.label).join(" or ");
|
|
3650
4000
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3651
4001
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3652
4002
|
}).filter(Boolean);
|
|
@@ -3808,9 +4158,30 @@ STEP_TYPES.echo = {
|
|
|
3808
4158
|
},
|
|
3809
4159
|
};
|
|
3810
4160
|
|
|
3811
|
-
|
|
4161
|
+
// Recorded (#299): a target that replays each call's recorded reply, read as
|
|
4162
|
+
// the flow reads one. No profile, no prompt -- the record is the reply.
|
|
4163
|
+
STEP_TYPES.recorded = {
|
|
4164
|
+
label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
|
|
4165
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
4166
|
+
firstJobOnly: "a call's recorded reply is job 1's",
|
|
4167
|
+
in: "item", out: "text",
|
|
4168
|
+
apply: "runPipeline",
|
|
4169
|
+
// `prompt` only so the step satisfies the target-step shape; it is never
|
|
4170
|
+
// sent (the record is the reply), and the prompt rule exempts it below.
|
|
4171
|
+
fields: ["type", "prompt"],
|
|
4172
|
+
validate(){},
|
|
4173
|
+
};
|
|
4174
|
+
|
|
4175
|
+
/** What a local target step is answered as, by the connection type it stands
|
|
4176
|
+
in for (`replaces`): Echo's own connection, or Recorded's -- each with no
|
|
3812
4177
|
address, key or model. */
|
|
3813
4178
|
const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
|
|
4179
|
+
const RECORDED_CONNECTION = { name: "Recorded", url: "", model: "", type: "recorded" };
|
|
4180
|
+
const LOCAL_CONNECTIONS =
|
|
4181
|
+
{ echo: ECHO_CONNECTION, recorded: RECORDED_CONNECTION };
|
|
4182
|
+
/** The connection a local target step stands in for, by its entry. */
|
|
4183
|
+
const localConnOf = (entry ) =>
|
|
4184
|
+
LOCAL_CONNECTIONS[entry?.replaces ?? ""] ?? ECHO_CONNECTION;
|
|
3814
4185
|
|
|
3815
4186
|
// Responses: how a job's reply is read, in order -- possibly not at all.
|
|
3816
4187
|
|
|
@@ -3950,6 +4321,16 @@ function targetProfileOf(doc
|
|
|
3950
4321
|
return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
|
|
3951
4322
|
}
|
|
3952
4323
|
|
|
4324
|
+
/** Whether target [i] is a Recorded target: it replays each call's recorded
|
|
4325
|
+
reply rather than sending anything (its job-1 step answers locally with a
|
|
4326
|
+
connection that `replays`). A recorded-reply metric compares the recorded
|
|
4327
|
+
reply with itself there, so it says nothing (n/a). Asked through the
|
|
4328
|
+
registry, never a step id. */
|
|
4329
|
+
function recordedTargetOf(doc , i ) {
|
|
4330
|
+
const entry = targetEntryOf(doc, i, 0);
|
|
4331
|
+
return !!(entry?.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays);
|
|
4332
|
+
}
|
|
4333
|
+
|
|
3953
4334
|
/** [doc] with target [i]'s step in job [k] changed: [change] merged into
|
|
3954
4335
|
it, or the step [change] makes of it. */
|
|
3955
4336
|
function withTarget (doc , i , k ,
|
|
@@ -4641,11 +5022,16 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4641
5022
|
// what it is read against. Over Prompt only the prompt is the reply.
|
|
4642
5023
|
// What target [i] asks in job [k]: Echo's own connection for a step
|
|
4643
5024
|
// answered here, else its profile as the lookups hold it.
|
|
4644
|
-
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
5025
|
+
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k))
|
|
4645
5026
|
: c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
|
|
4646
5027
|
const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
|
|
4647
5028
|
for (let k = 0; k < n; k++) {
|
|
4648
5029
|
const i = targets.findIndex((_, i) => {
|
|
5030
|
+
const entry = targetEntryOf(doc, i, k);
|
|
5031
|
+
// A step with no prompt of its own, or a replaying target (Recorded,
|
|
5032
|
+
// which answers from the record), is never asked for one.
|
|
5033
|
+
if (!entry?.fields?.includes("prompt")) return false;
|
|
5034
|
+
if (entry.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays) return false;
|
|
4649
5035
|
const words = targetStepOf(doc, i, k)?.prompt;
|
|
4650
5036
|
return (!isStr(words) || !words.trim())
|
|
4651
5037
|
&& !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
|
|
@@ -4658,12 +5044,14 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4658
5044
|
const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
|
|
4659
5045
|
.find(n => !/\.(?:txt|md|csv)$/i.test(n));
|
|
4660
5046
|
targets.forEach((_, i) => {
|
|
4661
|
-
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ?
|
|
5047
|
+
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k)) : lookup(targetProfileOf(doc, i, k) )));
|
|
4662
5048
|
conns.forEach((p , k ) => {
|
|
4663
5049
|
if (!local(p)) return;
|
|
4664
|
-
const
|
|
4665
|
-
|
|
4666
|
-
|
|
5050
|
+
const entry = CONNECTION_TYPES[typeOf(p )] ;
|
|
5051
|
+
// A replaying target (Recorded) answers from the record, not the item
|
|
5052
|
+
// text, so the text rules do not apply -- only the job-1 one does.
|
|
5053
|
+
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${entry.label} answers only job 1${entry.replays ? "" : ", with the item's own text"}`);
|
|
5054
|
+
else if (nonText && !entry.replays) bad.push(`${targetLabel(doc, i)}: ${entry.label} answers each text item with its own text, and ${nonText} is not text`);
|
|
4667
5055
|
});
|
|
4668
5056
|
// A connection answers what its target's step asks: words, or a whole request.
|
|
4669
5057
|
conns.forEach((p , k ) => {
|
|
@@ -5102,7 +5490,8 @@ function exportBundle(doc , ctx
|
|
|
5102
5490
|
// What the lab would supply at submit, written down: no lab supplies it later.
|
|
5103
5491
|
const out = clone(doc) ;
|
|
5104
5492
|
out.evals = (Array.isArray(out.evals) ? out.evals : [])
|
|
5105
|
-
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null
|
|
5493
|
+
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null,
|
|
5494
|
+
groups: Object.entries(ctx.datasets ?? {}).map(([id, d]) => ({ id, name: d.name, grader: d.body.grader ?? null })) }) ?? t);
|
|
5106
5495
|
const table = {};
|
|
5107
5496
|
for (const id of profileIds(out)) {
|
|
5108
5497
|
const p = profiles.find(x => x.id === id);
|
|
@@ -5261,7 +5650,7 @@ function stagesFor(run , i )
|
|
|
5261
5650
|
if (targetEntryOf(run, i, k)?.local) {
|
|
5262
5651
|
const id = targetProfileOf(run, i, k)?.id;
|
|
5263
5652
|
const held = id != null ? run.profiles?.[id] : undefined;
|
|
5264
|
-
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...
|
|
5653
|
+
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...localConnOf(targetEntryOf(run, i, k)) };
|
|
5265
5654
|
}
|
|
5266
5655
|
const id = targetProfileOf(run, i, k) .id;
|
|
5267
5656
|
return { id, ...(run.profiles?.[id] || {}) };
|
|
@@ -5276,7 +5665,7 @@ function stagesFor(run , i )
|
|
|
5276
5665
|
function scenarioProfile(run , i ) {
|
|
5277
5666
|
const id = targetsOf(run)[i]?.profile?.id;
|
|
5278
5667
|
// A target that asks nothing ran as Echo.
|
|
5279
|
-
if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name:
|
|
5668
|
+
if (id == null && targetEntryOf(run, i, 0)?.local) { const lc = localConnOf(targetEntryOf(run, i, 0)); return { id: "", name: lc.name, settings: { ...lc } }; }
|
|
5280
5669
|
const conn = id != null ? run.profiles?.[id] : null;
|
|
5281
5670
|
return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
|
|
5282
5671
|
}
|
|
@@ -5337,9 +5726,9 @@ function itemScores(run , kase , res
|
|
|
5337
5726
|
return Object.keys(out).length ? out : null;
|
|
5338
5727
|
}
|
|
5339
5728
|
|
|
5340
|
-
/** What production
|
|
5341
|
-
call says, from the item's record, or nobody does. */
|
|
5342
|
-
function
|
|
5729
|
+
/** What production recorded as the reply to an item, for a run's metrics: its
|
|
5730
|
+
last job's call says, from the item's record, or nobody does. */
|
|
5731
|
+
function recordedReplyOf(run , record ) {
|
|
5343
5732
|
const last = run.jobs.at(-1), flow = flowStepOf(last);
|
|
5344
5733
|
return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
|
|
5345
5734
|
}
|
|
@@ -5635,20 +6024,20 @@ registerMetrics({ registerKinds });
|
|
|
5635
6024
|
export {
|
|
5636
6025
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
5637
6026
|
loopReplyError, preparedSize,
|
|
5638
|
-
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
6027
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, takeAsExpected, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
5639
6028
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
5640
6029
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
5641
6030
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
5642
6031
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
5643
6032
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
5644
6033
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
5645
|
-
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync,
|
|
5646
|
-
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
5647
|
-
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
6034
|
+
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
|
|
6035
|
+
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf, WIZARDS, wizardProgress,
|
|
6036
|
+
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, addressOf, connectionSettings,
|
|
5648
6037
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5649
6038
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5650
6039
|
contentOf, withContent, replyOf,
|
|
5651
|
-
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
6040
|
+
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
|
|
5652
6041
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5653
6042
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5654
6043
|
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|