evals-lab 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +34 -24
- package/lab/demo/pipelines/demo-2.json +34 -24
- package/lab/evals-core.mjs +701 -284
- package/lab/run-evals.js +42 -16
- package/lab/server.py +251 -75
- package/lab/web/dist/assets/gallery-DsetJSXv.js +3 -0
- package/lab/web/dist/assets/{main-DjQQums6.css → main-B-VtDGxC.css} +1 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +19 -0
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +51 -0
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-61NS6C4m.js +0 -18
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -94,49 +94,76 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
94
94
|
|
|
95
95
|
|
|
96
96
|
|
|
97
|
-
/**
|
|
98
|
-
|
|
99
|
-
|
|
97
|
+
/** The Power Automate step a job's records are calls of (job 1, Content):
|
|
98
|
+
the action's name its expressions read it by, the loop item() means,
|
|
99
|
+
and the HTTP API its replies speak -- how production's reply is read. */
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
/** The item's image goes beside every target's words in this job. */
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
/** The token mappings a job's words are resolved under. */
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
/** What the flow does to the reply before it reads it -- its Parse JSON's
|
|
119
|
+
content -- as an expression over body('<step>'). First of Responses. */
|
|
100
120
|
|
|
101
121
|
|
|
102
|
-
|
|
103
|
-
|
|
122
|
+
|
|
104
123
|
|
|
105
124
|
|
|
106
|
-
/**
|
|
125
|
+
/** Response Format Validation: the output kind a reply is read as, and that
|
|
126
|
+
kind's own settings. A job with none reads its reply as text. */
|
|
107
127
|
|
|
108
128
|
|
|
109
|
-
|
|
129
|
+
|
|
110
130
|
|
|
111
131
|
|
|
112
|
-
/**
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
132
|
+
/** One change applied to the parsed reply, after the Format Validation
|
|
133
|
+
whose kind accepts it. */
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
/** A job's steps: its Content stage, then its Responses (§16). */
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
/** A target step that prompts a model: its words, resolved under the job's
|
|
144
|
+
token mappings, and a profile of its own where it names one. */
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
/** A target step that sends a Power Automate step's request
|
|
150
|
+
(docs/power-automate.md § The HTTP Request step): the flow's own template
|
|
151
|
+
-- method, path, query and body, Workflow Definition Language expressions
|
|
152
|
+
and all -- with the target's words and fields placed in the body by its
|
|
153
|
+
HTTP API, evaluated against each item's record. The profile gives the
|
|
154
|
+
address and the key. */
|
|
155
|
+
|
|
119
156
|
|
|
120
|
-
|
|
121
|
-
|
|
122
157
|
|
|
123
158
|
|
|
124
159
|
|
|
125
160
|
|
|
126
161
|
|
|
127
162
|
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
163
|
|
|
135
164
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
/** A job is its steps, in fixed slots (pipeline-model §3): an Attach Content
|
|
139
|
-
on job 1 only, one Call, then Read Reply. */
|
|
165
|
+
/** A job is its steps, staged (pipeline-model §16): Content, then
|
|
166
|
+
Responses. What each target sends in it is the target's own. */
|
|
140
167
|
|
|
141
168
|
|
|
142
169
|
|
|
@@ -170,13 +197,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
170
197
|
|
|
171
198
|
|
|
172
199
|
|
|
173
|
-
/**
|
|
200
|
+
/** What a target's step says beside its type: what to ask in that job,
|
|
201
|
+
and whom, where it names a profile of its own. */
|
|
174
202
|
|
|
175
203
|
|
|
176
204
|
|
|
177
|
-
|
|
178
205
|
|
|
179
|
-
|
|
206
|
+
|
|
207
|
+
|
|
180
208
|
|
|
181
209
|
|
|
182
210
|
|
|
@@ -187,13 +215,21 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
187
215
|
|
|
188
216
|
|
|
189
217
|
|
|
190
|
-
|
|
191
|
-
|
|
218
|
+
/** What a target sends in one job: one of the target step types. */
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
/** A target (pipeline-model §16): one column through every job, compared
|
|
222
|
+
side by side in Results -- what version 9 called a scenario. Its step in
|
|
223
|
+
each job says what it sends there. */
|
|
224
|
+
|
|
225
|
+
|
|
192
226
|
|
|
193
227
|
|
|
194
228
|
|
|
195
|
-
|
|
196
|
-
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
|
|
197
233
|
|
|
198
234
|
|
|
199
235
|
/** What every test carries whatever its type: an id minted once and never
|
|
@@ -255,7 +291,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
255
291
|
|
|
256
292
|
|
|
257
293
|
|
|
258
|
-
|
|
294
|
+
|
|
259
295
|
|
|
260
296
|
|
|
261
297
|
|
|
@@ -323,7 +359,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
323
359
|
/** A pipeline read from YAML, references remapped to this lab. */
|
|
324
360
|
|
|
325
361
|
|
|
326
|
-
|
|
362
|
+
|
|
327
363
|
|
|
328
364
|
|
|
329
365
|
|
|
@@ -335,6 +371,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
335
371
|
|
|
336
372
|
|
|
337
373
|
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
|
|
338
377
|
|
|
339
378
|
|
|
340
379
|
/** An import's outcome: the remapped pipeline, or one sentence refusing it. */
|
|
@@ -417,12 +456,23 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
417
456
|
|
|
418
457
|
|
|
419
458
|
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
/** What an expression token reads: a record's scope, its time, the loop. */
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
|
|
420
470
|
|
|
421
471
|
|
|
422
472
|
/** A stage of a run as a transport asks it: stagesFor's answer. */
|
|
423
473
|
|
|
424
474
|
|
|
425
|
-
|
|
475
|
+
|
|
426
476
|
|
|
427
477
|
|
|
428
478
|
|
|
@@ -894,45 +944,54 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
894
944
|
|
|
895
945
|
|
|
896
946
|
|
|
897
|
-
/** A job's
|
|
898
|
-
|
|
899
|
-
|
|
947
|
+
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
948
|
+
step is in no job: it is the target's, in that job's place. */
|
|
949
|
+
|
|
950
|
+
const SLOTS = ["content", "target", "responses"];
|
|
900
951
|
|
|
901
952
|
|
|
902
953
|
|
|
903
954
|
|
|
904
955
|
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
956
|
+
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
|
|
908
960
|
|
|
909
|
-
|
|
910
961
|
|
|
911
962
|
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
|
|
966
|
+
|
|
967
|
+
|
|
912
968
|
|
|
913
969
|
|
|
914
970
|
|
|
915
971
|
|
|
916
|
-
|
|
972
|
+
|
|
973
|
+
|
|
917
974
|
|
|
975
|
+
|
|
976
|
+
|
|
977
|
+
|
|
978
|
+
|
|
979
|
+
|
|
918
980
|
|
|
919
981
|
|
|
920
982
|
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
983
|
|
|
984
|
+
|
|
925
985
|
|
|
926
|
-
|
|
986
|
+
|
|
927
987
|
|
|
988
|
+
|
|
928
989
|
|
|
929
|
-
|
|
930
|
-
|
|
990
|
+
|
|
991
|
+
|
|
931
992
|
|
|
932
|
-
|
|
993
|
+
|
|
933
994
|
|
|
934
|
-
|
|
935
|
-
|
|
936
995
|
|
|
937
996
|
|
|
938
997
|
|
|
@@ -1023,6 +1082,10 @@ function datasetRules(body ) {
|
|
|
1023
1082
|
|
|
1024
1083
|
|
|
1025
1084
|
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
|
|
1026
1089
|
|
|
1027
1090
|
|
|
1028
1091
|
|
|
@@ -1067,7 +1130,7 @@ function datasetRules(body ) {
|
|
|
1067
1130
|
|
|
1068
1131
|
|
|
1069
1132
|
/** What a Call hands its connection. */
|
|
1070
|
-
|
|
1133
|
+
|
|
1071
1134
|
|
|
1072
1135
|
/** What the core hands a registry module (kinds/list.ts) to register with. */
|
|
1073
1136
|
|
|
@@ -1120,7 +1183,9 @@ function mappingsFor(prompt ) {
|
|
|
1120
1183
|
|
|
1121
1184
|
|
|
1122
1185
|
|
|
1123
|
-
|
|
1186
|
+
|
|
1187
|
+
|
|
1188
|
+
|
|
1124
1189
|
|
|
1125
1190
|
|
|
1126
1191
|
|
|
@@ -1157,6 +1222,32 @@ TOKEN_TYPES.block = {
|
|
|
1157
1222
|
validate(m, at, bad){ if (typeof m.enabled !== "boolean") bad.push(`${at}: a block's enabled has to be true or false`); },
|
|
1158
1223
|
};
|
|
1159
1224
|
|
|
1225
|
+
// Expression: a Power Automate expression, evaluated against the item's
|
|
1226
|
+
// record -- `@triggerBody()?['note']`, or text with `@{…}` in it -- so a
|
|
1227
|
+
// prompt can say what the flow's own request says (docs/pipeline-model.md
|
|
1228
|
+
// §16 › Targets).
|
|
1229
|
+
TOKEN_TYPES.expression = {
|
|
1230
|
+
label: "Expression",
|
|
1231
|
+
fields: ["value"],
|
|
1232
|
+
defaults: () => ({ value: "" }),
|
|
1233
|
+
notation: name => `{${name}}`,
|
|
1234
|
+
names: name => [name],
|
|
1235
|
+
order: 1,
|
|
1236
|
+
apply(prompt, m, record){
|
|
1237
|
+
const token = `{${m.name}}`;
|
|
1238
|
+
if (!prompt.includes(token)) return prompt;
|
|
1239
|
+
if (!record) throw new WdlError(`{${m.name}} reads the item's record, and this item is not one`);
|
|
1240
|
+
try {
|
|
1241
|
+
return prompt.replaceAll(token, asText(evaluate(m.value ?? "", { scope: record.scope || {}, now: record.at ?? null, loop: record.loop ?? null })));
|
|
1242
|
+
} catch (e) {
|
|
1243
|
+
throw new WdlError(`{${m.name}}: ${(e ).message}`);
|
|
1244
|
+
}
|
|
1245
|
+
},
|
|
1246
|
+
validate(m, at, bad){
|
|
1247
|
+
if (!isStr(m.value)) bad.push(`${at}: an expression's value has to be text`);
|
|
1248
|
+
},
|
|
1249
|
+
};
|
|
1250
|
+
|
|
1160
1251
|
/** A mapping of [type] named [name], holding its type's defaults. */
|
|
1161
1252
|
function tokenMapping(name , type ) {
|
|
1162
1253
|
return { name, type, ...(TOKEN_TYPES[type]?.defaults() ?? {}) };
|
|
@@ -1186,12 +1277,12 @@ function tokenMappingsProblems(list , at , bad ) {
|
|
|
1186
1277
|
* The tokens are a parameter and not a global, because only one of the three
|
|
1187
1278
|
* callers has a localStorage to have loaded a set from.
|
|
1188
1279
|
*/
|
|
1189
|
-
function resolvePrompt(tpl , tokens ) {
|
|
1280
|
+
function resolvePrompt(tpl , tokens , record = null) {
|
|
1190
1281
|
const order = (m ) => TOKEN_TYPES[m.type]?.order ?? Infinity;
|
|
1191
1282
|
let out = tpl;
|
|
1192
1283
|
for (const m of [...tokens].sort((a, b) => order(a) - order(b))) {
|
|
1193
1284
|
const type = TOKEN_TYPES[m.type];
|
|
1194
|
-
if (type) out = type.apply(out, m);
|
|
1285
|
+
if (type) out = type.apply(out, m, record);
|
|
1195
1286
|
}
|
|
1196
1287
|
// Dropping a block leaves the spaces that surrounded it.
|
|
1197
1288
|
return out.replace(/[ \t]{2,}/g, " ").trim();
|
|
@@ -1376,8 +1467,10 @@ CONNECTION_TYPES["ollama-cloud"] = ollamaType("ollama-cloud", "Ollama (Cloud)",
|
|
|
1376
1467
|
// each scenario holds one. Only job 1 can use it: a later job's item is the
|
|
1377
1468
|
// job before's answer, not a file.
|
|
1378
1469
|
const sendsNothing = () => { throw new Error("Echo answers from the item and sends no request"); };
|
|
1470
|
+
// Echo is a target step now (STEP_TYPES.echo), with no profile; the type
|
|
1471
|
+
// stays so a stored run, or a profile made before, still reads.
|
|
1379
1472
|
CONNECTION_TYPES.echo = {
|
|
1380
|
-
id: "echo", label: "Echo", settings: [], keyless: true,
|
|
1473
|
+
id: "echo", label: "Echo", settings: [], keyless: true, picker: false,
|
|
1381
1474
|
description: "No model: answers with each text item itself, to grade replies recorded earlier.",
|
|
1382
1475
|
local: (item, sent) => item.text ?? sent,
|
|
1383
1476
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
@@ -1547,16 +1640,19 @@ CONNECTION_TYPES["llama.cpp"] = {
|
|
|
1547
1640
|
|
|
1548
1641
|
|
|
1549
1642
|
|
|
1550
|
-
|
|
1643
|
+
|
|
1551
1644
|
|
|
1552
|
-
|
|
1645
|
+
|
|
1553
1646
|
|
|
1554
|
-
|
|
1647
|
+
|
|
1555
1648
|
|
|
1556
1649
|
|
|
1557
1650
|
|
|
1558
1651
|
|
|
1559
1652
|
|
|
1653
|
+
|
|
1654
|
+
|
|
1655
|
+
|
|
1560
1656
|
|
|
1561
1657
|
|
|
1562
1658
|
|
|
@@ -1622,6 +1718,7 @@ HTTP_APIS.anthropic = {
|
|
|
1622
1718
|
const r = j ;
|
|
1623
1719
|
return { raw: (r?.content || []).filter(b => b?.type === "text").map(b => b.text).join(""), finishReason: r?.stop_reason ?? null };
|
|
1624
1720
|
},
|
|
1721
|
+
wrap: text => ({ type: "message", role: "assistant", content: [{ type: "text", text }], stop_reason: "end_turn" }),
|
|
1625
1722
|
withPrefill(j, prefill) {
|
|
1626
1723
|
const r = j ;
|
|
1627
1724
|
if (!isObj(r) || !Array.isArray(r.content)) return j;
|
|
@@ -1667,6 +1764,7 @@ const openAiApi = (label , matches ) =>
|
|
|
1667
1764
|
return { ...b, messages: msgs };
|
|
1668
1765
|
},
|
|
1669
1766
|
reply: chatReply,
|
|
1767
|
+
wrap: text => ({ object: "chat.completion", choices: [{ index: 0, message: { role: "assistant", content: text }, finish_reason: "stop" }] }),
|
|
1670
1768
|
});
|
|
1671
1769
|
HTTP_APIS.openai = openAiApi("OpenAI", u => u.hostname === "api.openai.com");
|
|
1672
1770
|
HTTP_APIS["azure-openai"] = openAiApi("Azure OpenAI", u => u.hostname.endsWith(".openai.azure.com"));
|
|
@@ -1684,6 +1782,9 @@ HTTP_APIS.raw = {
|
|
|
1684
1782
|
try { return JSON.parse(cell.prompt); } catch { throw new Error("the body is not JSON"); }
|
|
1685
1783
|
},
|
|
1686
1784
|
reply: j => ({ raw: isStr(j) ? j : JSON.stringify(j), finishReason: null }),
|
|
1785
|
+
// A body the flow reads is JSON where it parses as JSON, as the HTTP
|
|
1786
|
+
// action hands it on, and text otherwise.
|
|
1787
|
+
wrap: text => { try { return JSON.parse(text); } catch { return text; } },
|
|
1687
1788
|
};
|
|
1688
1789
|
|
|
1689
1790
|
/** The HTTP API a flow's request to [uri] speaks: the first entry that
|
|
@@ -2432,7 +2533,7 @@ function jobProblem(stages , call
|
|
|
2432
2533
|
// 1 the way `textPrompt` says, and any later stage that places {text}.
|
|
2433
2534
|
async function runPipeline(stages , dataUrl , call ,
|
|
2434
2535
|
opts = {}) {
|
|
2435
|
-
const { tokens = TOKEN_DEFAULTS, mode, text = null } = opts;
|
|
2536
|
+
const { tokens = TOKEN_DEFAULTS, mode, text = null, record = null } = opts;
|
|
2436
2537
|
const transcript = [];
|
|
2437
2538
|
const problem = jobProblem(stages, call, tokens, text);
|
|
2438
2539
|
if (problem) return { error: problem, terms: [], transcript, ms: 0 };
|
|
@@ -2466,8 +2567,15 @@ async function runPipeline(stages , dataUrl , cal
|
|
|
2466
2567
|
forwarded.set("reply", prev.reply);
|
|
2467
2568
|
for (const [name, v] of Object.entries(prev.rendered)) forwarded.set(name, v);
|
|
2468
2569
|
}
|
|
2469
|
-
// A verbatim stage's words are
|
|
2470
|
-
|
|
2570
|
+
// A verbatim stage's words are its target step's own template, sent as written.
|
|
2571
|
+
// An expression token that cannot be read is the stage's error, said
|
|
2572
|
+
// before anything is sent.
|
|
2573
|
+
let instruction ;
|
|
2574
|
+
try { instruction = st.verbatim ? st.text : resolvePrompt(st.text, tokenSet(tokens, i), record); }
|
|
2575
|
+
catch (e) {
|
|
2576
|
+
if (!(e instanceof WdlError)) throw e;
|
|
2577
|
+
return result({ error: named(e.message) });
|
|
2578
|
+
}
|
|
2471
2579
|
const fill = (to ) => instruction.replace(TOKEN, (m, name ) =>
|
|
2472
2580
|
forwarded.has(name) ? to(forwarded.get(name) ) : m);
|
|
2473
2581
|
const sent = st.verbatim ? instruction : i === 0 && text != null ? textPrompt(instruction, text) : fill(v => v);
|
|
@@ -2580,7 +2688,11 @@ function applyModifiers (list , kind ,
|
|
|
2580
2688
|
// 9: a test is Metrics. A Single Test and a Graded set are read as the Metrics
|
|
2581
2689
|
// they convert to (LEGACY_TESTS' toMetrics, proven equal by
|
|
2582
2690
|
// metrics-parity-check.js).
|
|
2583
|
-
|
|
2691
|
+
// 10: a job's steps are its stages (pipeline-model §16) -- Content (Attach
|
|
2692
|
+
// Content, the flow step, Attach Image, Token mappings) and Responses (Read
|
|
2693
|
+
// as, Response Format Validation, modifiers) -- and its call is gone: each
|
|
2694
|
+
// scenario is a target, whose own step in each job is what it sends there.
|
|
2695
|
+
const PIPELINE_VERSION = 10 ;
|
|
2584
2696
|
|
|
2585
2697
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2586
2698
|
// to know which registered it.
|
|
@@ -2741,8 +2853,7 @@ function encoderQuality(quality ) {
|
|
|
2741
2853
|
// always shown, so a sentence names what the reader sees.
|
|
2742
2854
|
const jobLabel = (doc , k ) =>
|
|
2743
2855
|
(doc.jobs?.[k]?.name || "").trim() || `Job ${k + 1}`;
|
|
2744
|
-
const scenarioLabel = (doc
|
|
2745
|
-
(doc.scenarios?.[i]?.name || "").trim() || `Scenario ${i + 1}`;
|
|
2856
|
+
const scenarioLabel = (doc , i ) => targetLabel(doc, i);
|
|
2746
2857
|
// The runner calls a job's link a stage; the model calls it a job.
|
|
2747
2858
|
const jobWords = (s ) => s && String(s).replace(/\bstage[ ](\d+)/g, "job $1");
|
|
2748
2859
|
|
|
@@ -2850,7 +2961,7 @@ CONTENT_TYPES.text = {
|
|
|
2850
2961
|
// being the recorded reply.
|
|
2851
2962
|
CONTENT_TYPES.prompt = {
|
|
2852
2963
|
label: "Prompt only",
|
|
2853
|
-
description: "No item: each
|
|
2964
|
+
description: "No item: each target's prompt is sent on its own.",
|
|
2854
2965
|
inline: true,
|
|
2855
2966
|
bare: true,
|
|
2856
2967
|
fields: ["type"],
|
|
@@ -3079,14 +3190,14 @@ TEST_TYPES.metrics = {
|
|
|
3079
3190
|
if (t.grader != null && !isRef(t.grader)) bad.push("the Metrics name their grader as { id, name }");
|
|
3080
3191
|
// The test's own grader, or the lab's.
|
|
3081
3192
|
const grader = isRef(t.grader) ? t.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3082
|
-
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a
|
|
3193
|
+
if (gradedIn(t.metrics) && !grader) bad.push("a model-graded metric needs a grader: pick one on the test, or make a Target profile the lab's grader");
|
|
3083
3194
|
// A grader is asked words, and needs a model to ask.
|
|
3084
3195
|
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3085
|
-
if (grader && ctx.profiles && !p) bad.push(`
|
|
3196
|
+
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
3086
3197
|
else if (p) {
|
|
3087
3198
|
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3088
3199
|
if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
|
|
3089
|
-
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's
|
|
3200
|
+
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
|
|
3090
3201
|
}
|
|
3091
3202
|
},
|
|
3092
3203
|
rules: t => (Array.isArray(t.metrics) ? t.metrics : []).map((m , x ) => ({
|
|
@@ -3244,15 +3355,19 @@ function metricSummary(m ) {
|
|
|
3244
3355
|
return said.join(", ");
|
|
3245
3356
|
}
|
|
3246
3357
|
|
|
3247
|
-
// ---- a job's steps (pipeline-model §
|
|
3358
|
+
// ---- a job's steps, and a target's (pipeline-model §16) ----------------------
|
|
3359
|
+
|
|
3360
|
+
// Content: what a job is given. Attach Content and the flow step are job
|
|
3361
|
+
// 1's; any job may send the item's image and map tokens.
|
|
3248
3362
|
|
|
3249
3363
|
// Attach Content: the items a run goes over, job 1's first step.
|
|
3250
3364
|
STEP_TYPES.attachContent = {
|
|
3251
|
-
label: "Attach Content", slot: "content",
|
|
3252
|
-
description: "The items the run goes through: each is sent to every
|
|
3365
|
+
label: "Attach Content", slot: "content", rank: 0,
|
|
3366
|
+
description: "The items the run goes through: each is sent to every target.",
|
|
3253
3367
|
in: null, out: "items",
|
|
3254
3368
|
apply: "itemContent", // run-evals.js, through its file-type registry
|
|
3255
3369
|
fields: ["type", "content"],
|
|
3370
|
+
firstJobOnly: "only job 1 attaches content -- a later job is handed the reply before it",
|
|
3256
3371
|
validate(step, ctx, bad){
|
|
3257
3372
|
const c = step?.content;
|
|
3258
3373
|
if (!c) return void bad.push("set the content first");
|
|
@@ -3263,20 +3378,65 @@ STEP_TYPES.attachContent = {
|
|
|
3263
3378
|
},
|
|
3264
3379
|
};
|
|
3265
3380
|
|
|
3266
|
-
//
|
|
3381
|
+
// The flow step: the Power Automate action a job's records are calls of.
|
|
3382
|
+
STEP_TYPES.flowStep = {
|
|
3383
|
+
label: "Flow step", slot: "content", rank: 1,
|
|
3384
|
+
description: "The Power Automate step each record is a call of.",
|
|
3385
|
+
in: "item", out: "item",
|
|
3386
|
+
apply: "runPipeline",
|
|
3387
|
+
fields: ["type", "step", "loop", "api"],
|
|
3388
|
+
firstJobOnly: "a flow step's records are job 1's items",
|
|
3389
|
+
// Production's reply is the record's result, read as the flow reads it.
|
|
3390
|
+
production(flow, record){
|
|
3391
|
+
const result = isObj(record) && isObj(record.result) ? record.result : null;
|
|
3392
|
+
if (!result || typeof result.status !== "number") return null;
|
|
3393
|
+
try {
|
|
3394
|
+
return httpReplyOf(flow, { prompt: "" }, result.body, result.status,
|
|
3395
|
+
{ scope: record.scope || {}, now: record.at ?? null }).raw;
|
|
3396
|
+
} catch { return null; }
|
|
3397
|
+
},
|
|
3398
|
+
validate(step, _ctx, bad, at){
|
|
3399
|
+
if (!isStr(step.step) || !step.step.trim()) bad.push(`${at}: a flow step names the action it is`);
|
|
3400
|
+
if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
|
|
3401
|
+
if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3402
|
+
},
|
|
3403
|
+
};
|
|
3404
|
+
|
|
3405
|
+
// Attach Image: the item's image beside every target's words.
|
|
3406
|
+
STEP_TYPES.attachImage = {
|
|
3407
|
+
label: "Attach Image", slot: "content", rank: 2,
|
|
3408
|
+
description: "Sends each item's image beside the words.",
|
|
3409
|
+
in: "item", out: "item",
|
|
3410
|
+
apply: "runPipeline",
|
|
3411
|
+
fields: ["type"],
|
|
3412
|
+
validate(){},
|
|
3413
|
+
};
|
|
3414
|
+
|
|
3415
|
+
// Token mappings: what the words' {tokens} are resolved to.
|
|
3416
|
+
STEP_TYPES.tokenMappings = {
|
|
3417
|
+
label: "Token mappings", slot: "content", rank: 3,
|
|
3418
|
+
description: "What each {token} in the words is filled in with.",
|
|
3419
|
+
in: "item", out: "item",
|
|
3420
|
+
apply: "runPipeline",
|
|
3421
|
+
fields: ["type", "tokenMappings"],
|
|
3422
|
+
validate(step, _ctx, bad, at){ tokenMappingsProblems(step.tokenMappings, at, bad); },
|
|
3423
|
+
};
|
|
3424
|
+
|
|
3425
|
+
// Targets: what each target sends. A target's own steps, one per job.
|
|
3426
|
+
|
|
3427
|
+
// Prompt: a target's words, asked of its model.
|
|
3267
3428
|
STEP_TYPES.prompt = {
|
|
3268
|
-
label: "Prompt", slot: "
|
|
3269
|
-
description: "Asks each
|
|
3429
|
+
label: "Prompt", slot: "target", asks: "prompt", modelFrom: "profile",
|
|
3430
|
+
description: "Asks each target's model its prompt about the item.",
|
|
3270
3431
|
in: "item", out: "text",
|
|
3271
3432
|
apply: "runPipeline",
|
|
3272
|
-
fields: ["type", "
|
|
3433
|
+
fields: ["type", "prompt", "profile", "from"],
|
|
3273
3434
|
validate(step, _ctx, bad, at){
|
|
3274
|
-
if (
|
|
3275
|
-
tokenMappingsProblems(step.tokenMappings, at, bad);
|
|
3435
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3276
3436
|
},
|
|
3277
3437
|
};
|
|
3278
3438
|
|
|
3279
|
-
// HTTP Request: a
|
|
3439
|
+
// HTTP Request: a target that sends a Power Automate step's own request,
|
|
3280
3440
|
// rebuilt for each item from its record (docs/power-automate.md).
|
|
3281
3441
|
const HTTP_METHODS = ["GET", "POST", "PUT", "PATCH", "DELETE"];
|
|
3282
3442
|
/** The first field under [v] named like a key, as a path, or null. */
|
|
@@ -3293,62 +3453,84 @@ function keyField(v , at = "body") {
|
|
|
3293
3453
|
return null;
|
|
3294
3454
|
}
|
|
3295
3455
|
STEP_TYPES.httpRequest = {
|
|
3296
|
-
label: "HTTP Request", slot: "
|
|
3297
|
-
description: "Sends a Power Automate step's own request, rebuilt from each record, with each
|
|
3298
|
-
|
|
3299
|
-
|
|
3300
|
-
|
|
3301
|
-
|
|
3302
|
-
|
|
3303
|
-
|
|
3304
|
-
|
|
3305
|
-
|
|
3306
|
-
|
|
3307
|
-
|
|
3308
|
-
|
|
3309
|
-
|
|
3310
|
-
|
|
3311
|
-
|
|
3312
|
-
|
|
3313
|
-
|
|
3456
|
+
label: "HTTP Request", slot: "target", asks: "request", verbatim: true, modelFrom: "step",
|
|
3457
|
+
description: "Sends a Power Automate step's own request, rebuilt from each record, with each target's words in it.",
|
|
3458
|
+
firstJobOnly: "an HTTP Request is sent from its item's record, so it is job 1's",
|
|
3459
|
+
// A new target starts from the flow's own words and fields.
|
|
3460
|
+
newStep: (step) => ({ type: "httpRequest", api: step.api, method: step.method, path: step.path,
|
|
3461
|
+
query: clone(step.query), body: clone(step.body),
|
|
3462
|
+
...(HTTP_APIS[step.api] ?? HTTP_APIS.raw ).cellOf(step.body) }),
|
|
3463
|
+
in: "item", out: "text",
|
|
3464
|
+
apply: "runPipeline",
|
|
3465
|
+
fields: ["type", "api", "method", "path", "query", "body", "prompt", "profile", "from", "system", "model", "settings", "prefill"],
|
|
3466
|
+
validate(step, _ctx, bad, at){
|
|
3467
|
+
const api = HTTP_APIS[step.api];
|
|
3468
|
+
if (!api) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3469
|
+
if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
|
|
3470
|
+
if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
|
|
3471
|
+
if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
|
|
3472
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3473
|
+
// A key is the Setup profile's, sent by its Key header; one in the
|
|
3474
|
+
// template would be kept in the pipeline and every run of it.
|
|
3475
|
+
const hit = keyField(step.body) ?? keyField(step.query, "query");
|
|
3476
|
+
if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Target profile sends it`);
|
|
3314
3477
|
if (!api) return;
|
|
3478
|
+
// Its fields, where its HTTP API has a place for them.
|
|
3315
3479
|
for (const f of ["system", "model", "prefill"] ) {
|
|
3316
|
-
if (
|
|
3317
|
-
if (!isStr(
|
|
3480
|
+
if (step[f] == null) continue;
|
|
3481
|
+
if (!isStr(step[f])) bad.push(`${at}: ${f} has to be text`);
|
|
3318
3482
|
else if (!api.fields.includes(f)) bad.push(`${at}: ${api.label} has no place for a ${f}`);
|
|
3319
3483
|
}
|
|
3320
|
-
if (
|
|
3321
|
-
if (!isObj(
|
|
3322
|
-
else for (const k of Object.keys(
|
|
3484
|
+
if (step.settings != null) {
|
|
3485
|
+
if (!isObj(step.settings) || !Object.values(step.settings).every(isStr)) bad.push(`${at}: settings are names to text`);
|
|
3486
|
+
else for (const k of Object.keys(step.settings)) {
|
|
3323
3487
|
if (!api.settings.includes(k)) bad.push(`${at}: ${api.label} has no setting ${k}`);
|
|
3324
3488
|
}
|
|
3325
3489
|
}
|
|
3326
|
-
if (isStr(
|
|
3327
|
-
try { api.place(
|
|
3490
|
+
if (isStr(step.prompt)) {
|
|
3491
|
+
try { api.place(step.body, step ); }
|
|
3328
3492
|
catch (e) { bad.push(`${at}: ${(e ).message}`); }
|
|
3329
3493
|
}
|
|
3330
3494
|
},
|
|
3495
|
+
};
|
|
3496
|
+
|
|
3497
|
+
// Echo: a target that sends nothing and needs no profile -- each text item
|
|
3498
|
+
// is its own reply, to grade replies recorded earlier (#97). Job 1's, over
|
|
3499
|
+
// text items only. Its words, where it has any, are what the reply is read
|
|
3500
|
+
// against; over Prompt only they are the reply.
|
|
3501
|
+
STEP_TYPES.echo = {
|
|
3502
|
+
label: "Echo", slot: "target", asks: "nothing", local: true, replaces: "echo",
|
|
3503
|
+
description: "Sends nothing: each text item is its own reply, to grade replies recorded earlier.",
|
|
3504
|
+
firstJobOnly: "Echo answers only job 1, with the item's own text",
|
|
3331
3505
|
in: "item", out: "text",
|
|
3332
3506
|
apply: "runPipeline",
|
|
3333
|
-
fields: ["type", "
|
|
3507
|
+
fields: ["type", "prompt", "from"],
|
|
3334
3508
|
validate(step, _ctx, bad, at){
|
|
3335
|
-
if (!isStr(step.
|
|
3336
|
-
if (!HTTP_APIS[step.api]) bad.push(`${at}: "${step.api}" is not an HTTP API this lab has`);
|
|
3337
|
-
if (!HTTP_METHODS.includes(step.method)) bad.push(`${at}: the method is one of ${HTTP_METHODS.join(", ")}`);
|
|
3338
|
-
if (!isStr(step.path) || !/^[/@]/.test(step.path)) bad.push(`${at}: the path starts with /`);
|
|
3339
|
-
if (!isObj(step.query) || !Object.values(step.query).every(isStr)) bad.push(`${at}: the query is names to text`);
|
|
3340
|
-
if (step.readAs != null && !isStr(step.readAs)) bad.push(`${at}: readAs is an expression or null`);
|
|
3341
|
-
if (step.loop != null && !isStr(step.loop)) bad.push(`${at}: loop names a loop or is null`);
|
|
3342
|
-
// A key is the Setup profile's, sent by its Key header; one in the
|
|
3343
|
-
// template would be kept in the pipeline and every run of it.
|
|
3344
|
-
const hit = keyField(step.body) ?? keyField(step.query, "query");
|
|
3345
|
-
if (hit) bad.push(`${at} carries a key at ${hit}, and a key never goes in a pipeline -- the Setup profile sends it`);
|
|
3509
|
+
if (!isStr(step.prompt)) bad.push(`${at}: prompt has to be text`);
|
|
3346
3510
|
},
|
|
3347
3511
|
};
|
|
3348
3512
|
|
|
3349
|
-
|
|
3513
|
+
/** What a local target step is answered as: Echo's own connection, with no
|
|
3514
|
+
address, key or model. */
|
|
3515
|
+
const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
|
|
3516
|
+
|
|
3517
|
+
// Responses: how a job's reply is read, in order -- possibly not at all.
|
|
3518
|
+
|
|
3519
|
+
// Read as: what the flow does to the reply before it reads it.
|
|
3520
|
+
STEP_TYPES.readAs = {
|
|
3521
|
+
label: "Read as", slot: "responses", rank: 0,
|
|
3522
|
+
description: "What the flow makes of the reply before it reads it.",
|
|
3523
|
+
in: "text", out: "text",
|
|
3524
|
+
apply: "runPipeline",
|
|
3525
|
+
fields: ["type", "readAs"],
|
|
3526
|
+
validate(step, _ctx, bad, at){
|
|
3527
|
+
if (!isStr(step.readAs) || !step.readAs.trim()) bad.push(`${at}: readAs is an expression`);
|
|
3528
|
+
},
|
|
3529
|
+
};
|
|
3530
|
+
|
|
3531
|
+
// Response Format Validation: the kind a reply is read as.
|
|
3350
3532
|
STEP_TYPES.readReply = {
|
|
3351
|
-
label: "Read Reply", slot: "
|
|
3533
|
+
label: "Read Reply", slot: "responses", rank: 1,
|
|
3352
3534
|
description: "How the reply is read before tests and later jobs see it.",
|
|
3353
3535
|
in: "text", out: step => step.out?.kind,
|
|
3354
3536
|
// Read inside the stage: runPipeline parses each reply as it comes back.
|
|
@@ -3360,30 +3542,42 @@ STEP_TYPES.readReply = {
|
|
|
3360
3542
|
return void bad.push(`${at} answers as "${out?.kind}", which is not an output kind this lab has`);
|
|
3361
3543
|
}
|
|
3362
3544
|
const kindEntry = OUTPUT_KINDS[out.kind] ;
|
|
3363
|
-
onlyFields(out, `${at}'s output`, ["kind",
|
|
3545
|
+
onlyFields(out, `${at}'s output`, ["kind", ...(kindEntry.settings || []).map(o => o.key)], bad);
|
|
3364
3546
|
kindEntry.validateSettings?.(out, `${at}'s output`, bad);
|
|
3365
|
-
|
|
3366
|
-
|
|
3367
|
-
|
|
3368
|
-
|
|
3369
|
-
|
|
3370
|
-
|
|
3371
|
-
|
|
3372
|
-
|
|
3547
|
+
},
|
|
3548
|
+
};
|
|
3549
|
+
|
|
3550
|
+
// A modifier: one change to the parsed reply, after the kind that takes it.
|
|
3551
|
+
STEP_TYPES.modifier = {
|
|
3552
|
+
label: "Modifier", slot: "responses", rank: 2, many: true,
|
|
3553
|
+
description: "Changes the reply after it is read.",
|
|
3554
|
+
in: "value", out: "value",
|
|
3555
|
+
apply: "runPipeline",
|
|
3556
|
+
fields: ["type", "modifier"],
|
|
3557
|
+
validate(step, _ctx, bad, at, kind){
|
|
3558
|
+
const m = step.modifier;
|
|
3559
|
+
const entry = isObj(m) && MODIFIERS[m.type];
|
|
3560
|
+
if (!entry) return void bad.push(`${at}: "${m?.type}" is not a modifier this lab has`);
|
|
3561
|
+
if (entry.accepts && !entry.accepts.includes(kind)) {
|
|
3562
|
+
bad.push(`${at}: ${entry.label || m.type} does not apply to ${OUTPUT_KINDS[kind]?.noun || kind}`);
|
|
3373
3563
|
}
|
|
3564
|
+
entry.validate?.(m , `${at}: ${entry.label || m.type}`, bad);
|
|
3374
3565
|
},
|
|
3375
3566
|
};
|
|
3376
3567
|
|
|
3377
|
-
// ---- reading a job's steps
|
|
3378
|
-
// Every reader asks these, never a step's position
|
|
3379
|
-
//
|
|
3568
|
+
// ---- reading a job's steps, and a target's -----------------------------------
|
|
3569
|
+
// Every reader asks these, never a step's position or the document's shape:
|
|
3570
|
+
// which steps a job holds is its own, and a version that moves a field
|
|
3571
|
+
// changes these alone (pipeline-model §16).
|
|
3572
|
+
|
|
3573
|
+
|
|
3380
3574
|
|
|
3381
3575
|
const slotOf = (st ) =>
|
|
3382
3576
|
(isObj(st) ? STEP_TYPES[st.type ]?.slot : undefined);
|
|
3383
3577
|
|
|
3384
|
-
/**
|
|
3385
|
-
const
|
|
3386
|
-
(job?.steps || []).find(st =>
|
|
3578
|
+
/** [job]'s step of [type], when it holds one. */
|
|
3579
|
+
const stepOf = (job , type ) =>
|
|
3580
|
+
(job?.steps || []).find(st => isObj(st) && st.type === type) ;
|
|
3387
3581
|
|
|
3388
3582
|
/** The registry entry that fills [slot] by default: what a new step there is. */
|
|
3389
3583
|
function slotEntry(slot ) {
|
|
@@ -3391,9 +3585,26 @@ function slotEntry(slot )
|
|
|
3391
3585
|
return hit ? { type: hit[0], entry: hit[1] } : undefined;
|
|
3392
3586
|
}
|
|
3393
3587
|
|
|
3588
|
+
/** Where a step sits: its stage, then its place in it. */
|
|
3589
|
+
const placeOf = (st ) => {
|
|
3590
|
+
const e = isObj(st) ? STEP_TYPES[st.type ] : undefined;
|
|
3591
|
+
return (e?.slot ? SLOTS.indexOf(e.slot) : SLOTS.length) * 100 + (e?.rank ?? 0);
|
|
3592
|
+
};
|
|
3593
|
+
|
|
3594
|
+
/** [steps] in stage order, each stage in its own order; steps that tie
|
|
3595
|
+
(modifiers) keep theirs. */
|
|
3596
|
+
const ordered = (steps ) =>
|
|
3597
|
+
steps.map((st, x) => ({ st, x })).sort((a, b) => placeOf(a.st) - placeOf(b.st) || a.x - b.x).map(o => o.st);
|
|
3598
|
+
|
|
3599
|
+
/** [job] with its step of [type] set to [step], or taken away for null. */
|
|
3600
|
+
function withStep(job , type , step ) {
|
|
3601
|
+
const steps = job.steps.filter(st => st.type !== type);
|
|
3602
|
+
return { ...job, steps: ordered(step ? [...steps, step] : steps) };
|
|
3603
|
+
}
|
|
3604
|
+
|
|
3394
3605
|
/** A pipeline's content: job 1's content step, or none. */
|
|
3395
3606
|
function contentOf(doc ) {
|
|
3396
|
-
return
|
|
3607
|
+
return stepOf (doc?.jobs?.[0], "attachContent")?.content ?? null;
|
|
3397
3608
|
}
|
|
3398
3609
|
|
|
3399
3610
|
/** [doc] with its content set -- job 1's content step added, replaced, or
|
|
@@ -3401,26 +3612,94 @@ function contentOf(doc )
|
|
|
3401
3612
|
function withContent (doc , content ) {
|
|
3402
3613
|
const [first, ...rest] = doc.jobs;
|
|
3403
3614
|
if (!first) return doc;
|
|
3404
|
-
|
|
3405
|
-
|
|
3406
|
-
|
|
3615
|
+
return { ...doc, jobs: [withStep(first, "attachContent", content ? { type: "attachContent", content } : null), ...rest] };
|
|
3616
|
+
}
|
|
3617
|
+
|
|
3618
|
+
/** A document's targets, in column order. */
|
|
3619
|
+
function targetsOf(doc ) {
|
|
3620
|
+
return Array.isArray(doc?.targets) ? doc .targets : [];
|
|
3621
|
+
}
|
|
3622
|
+
|
|
3623
|
+
/** Target [i]'s name, or the number it has always shown. */
|
|
3624
|
+
function targetLabel(doc , i ) {
|
|
3625
|
+
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
|
|
3626
|
+
}
|
|
3627
|
+
|
|
3628
|
+
/** What target [i] sends in job [k]: its step there. */
|
|
3629
|
+
function targetStepOf(doc , i , k ) {
|
|
3630
|
+
const st = targetsOf(doc)[i]?.steps?.[k];
|
|
3631
|
+
return isObj(st) ? st : undefined;
|
|
3632
|
+
}
|
|
3633
|
+
|
|
3634
|
+
/** The registry entry of target [i]'s step in job [k]: what it asks, and of whom. */
|
|
3635
|
+
const targetEntryOf = (doc , i , k ) =>
|
|
3636
|
+
STEP_TYPES[targetStepOf(doc, i, k)?.type ?? ""];
|
|
3637
|
+
|
|
3638
|
+
/** What job [k]'s targets send, as the job's editor shows it: the first
|
|
3639
|
+
target's step type there, or Prompt's while there is none. */
|
|
3640
|
+
function jobTargetType(doc , k ) {
|
|
3641
|
+
return targetStepOf(doc, 0, k)?.type ?? "prompt";
|
|
3642
|
+
}
|
|
3643
|
+
|
|
3644
|
+
/** The Setup profile target [i] asks in job [k]: its step's own, or the target's. */
|
|
3645
|
+
function targetProfileOf(doc , i , k ) {
|
|
3646
|
+
return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
|
|
3647
|
+
}
|
|
3648
|
+
|
|
3649
|
+
/** [doc] with target [i]'s step in job [k] changed: [change] merged into
|
|
3650
|
+
it, or the step [change] makes of it. */
|
|
3651
|
+
function withTarget (doc , i , k ,
|
|
3652
|
+
change ) {
|
|
3653
|
+
return { ...doc, targets: doc.targets.map((t, x) => (x !== i ? t : {
|
|
3654
|
+
...t, steps: t.steps.map((st, j) => (j !== k ? st : typeof change === "function" ? change(st) : { ...st, ...change } )),
|
|
3655
|
+
})) };
|
|
3407
3656
|
}
|
|
3408
3657
|
|
|
3409
|
-
/**
|
|
3410
|
-
const
|
|
3411
|
-
stepIn (job, "call") ;
|
|
3658
|
+
/** The token mappings [job]'s words are resolved under. */
|
|
3659
|
+
const tokensOf = (job ) => stepOf (job, "tokenMappings")?.tokenMappings ?? [];
|
|
3412
3660
|
|
|
3413
|
-
/**
|
|
3414
|
-
const
|
|
3415
|
-
|
|
3661
|
+
/** [job] resolving its words under [list]: none takes the step away. */
|
|
3662
|
+
const withTokens = (job , list ) =>
|
|
3663
|
+
withStep(job, "tokenMappings", list.length ? { type: "tokenMappings", tokenMappings: list } : null);
|
|
3664
|
+
|
|
3665
|
+
/** Whether [job] sends the item's image beside its words. */
|
|
3666
|
+
const sendsImage = (job ) => !!stepOf(job, "attachImage");
|
|
3667
|
+
|
|
3668
|
+
/** [job] sending the item's image, or not. */
|
|
3669
|
+
const withImage = (job , on ) => withStep(job, "attachImage", on ? { type: "attachImage" } : null);
|
|
3670
|
+
|
|
3671
|
+
/** The Power Automate step [job]'s records are calls of: its name, which the
|
|
3672
|
+
flow's expressions read it by, the loop item() means, and the HTTP API
|
|
3673
|
+
its replies speak. None for a job that is not one. */
|
|
3674
|
+
function flowStepOf(job ) {
|
|
3675
|
+
const f = stepOf (job, "flowStep");
|
|
3676
|
+
return f ? { step: f.step, loop: f.loop ?? null, api: f.api } : null;
|
|
3677
|
+
}
|
|
3678
|
+
|
|
3679
|
+
/** What the flow does to [job]'s reply before it reads it, as an expression
|
|
3680
|
+
over the step's body; null reads the reply's own text. */
|
|
3681
|
+
const readAsOf = (job ) => stepOf (job, "readAs")?.readAs ?? null;
|
|
3682
|
+
|
|
3683
|
+
/** What a transport builds target [i]'s request in job [k] from: the
|
|
3684
|
+
target's step, with the flow step and reading the job holds. */
|
|
3685
|
+
function requestStepOf(doc , i , k ) {
|
|
3686
|
+
const job = doc.jobs[k];
|
|
3687
|
+
return { ...flowStepOf(job), ...targetStepOf(doc, i, k), readAs: readAsOf(job) } ;
|
|
3688
|
+
}
|
|
3689
|
+
|
|
3690
|
+
/** What httpRequestOf builds a request from: a target's template, and the
|
|
3691
|
+
loop its flow step sits in. */
|
|
3692
|
+
|
|
3693
|
+
/** What httpReplyOf reads a reply with: the flow step, its API, its reading. */
|
|
3694
|
+
|
|
3416
3695
|
|
|
3417
3696
|
/** The request an HTTP Request [step] sends for [cell] over an item whose
|
|
3418
3697
|
record read [ctx]'s scope: the cell's words and fields placed in the
|
|
3419
3698
|
flow's body by its HTTP API, then every expression evaluated. */
|
|
3420
|
-
function httpRequestOf(step
|
|
3699
|
+
function httpRequestOf(step , cell , ctx )
|
|
3421
3700
|
{
|
|
3422
3701
|
const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
|
|
3423
|
-
const at = { ...ctx, loop: step.loop };
|
|
3702
|
+
const at = { ...ctx, loop: step.loop ?? null };
|
|
3424
3703
|
const path = String(evaluate(step.path, at));
|
|
3425
3704
|
// A path an expression wrote whole -- an address -- keeps its path and query.
|
|
3426
3705
|
const [p, q] = /^https?:\/\//i.test(path) ? (() => { const u = new URL(path); return [u.pathname, u.search]; })() : [path, ""];
|
|
@@ -3431,35 +3710,50 @@ function httpRequestOf(step , cell , ctx
|
|
|
3431
3710
|
/** What the flow reads of a reply [j] to [step]: its readAs evaluated with the
|
|
3432
3711
|
reply as the step's body -- the prefill put back in front, as the flow has
|
|
3433
3712
|
to -- or the HTTP API's own text. */
|
|
3434
|
-
function httpReplyOf(step
|
|
3713
|
+
function httpReplyOf(step , cell , j , status ,
|
|
3435
3714
|
ctx ) {
|
|
3436
3715
|
const api = HTTP_APIS[step.api] ?? HTTP_APIS.raw ;
|
|
3437
3716
|
const whole = cell.prefill && api.withPrefill ? api.withPrefill(j, cell.prefill) : j;
|
|
3438
3717
|
const { raw: said, finishReason } = api.reply(whole);
|
|
3439
3718
|
if (!step.readAs) return { raw: said, said, finishReason };
|
|
3440
3719
|
const scope = { ...ctx.scope, actions: { ...(ctx.scope.actions || {}), [step.step]: { body: whole, statusCode: status } } };
|
|
3441
|
-
return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop })), said, finishReason };
|
|
3720
|
+
return { raw: asText(evaluate(step.readAs, { ...ctx, scope, loop: step.loop ?? null })), said, finishReason };
|
|
3442
3721
|
}
|
|
3443
3722
|
|
|
3444
|
-
/** A
|
|
3445
|
-
|
|
3446
|
-
|
|
3447
|
-
|
|
3448
|
-
|
|
3449
|
-
const
|
|
3723
|
+
/** A reply [text] from a model asked in words, read as the flow reads its
|
|
3724
|
+
step's reply: put in the body the flow's API would have sent it in
|
|
3725
|
+
(`HttpApiEntry.wrap`), then read through the job's Read as -- so a
|
|
3726
|
+
model's words and production's are read alike. */
|
|
3727
|
+
function readFlowReply(flow , text , record ) {
|
|
3728
|
+
const api = HTTP_APIS[flow.api] ?? HTTP_APIS.raw ;
|
|
3729
|
+
const read = httpReplyOf({ ...flow, api: HTTP_APIS[flow.api] ? flow.api : "raw" }, { prompt: "" }, api.wrap(text), 200,
|
|
3730
|
+
{ scope: record.scope || {}, now: record.at ?? null });
|
|
3731
|
+
return { raw: read.raw, said: text };
|
|
3732
|
+
}
|
|
3450
3733
|
|
|
3451
|
-
/**
|
|
3452
|
-
|
|
3453
|
-
|
|
3734
|
+
/** A job's Response Format Validation, when it has one. */
|
|
3735
|
+
const replyOf = (job ) => stepOf (job, "readReply");
|
|
3736
|
+
|
|
3737
|
+
/** How a job's reply is read: its kind, settings and modifiers. A job with
|
|
3738
|
+
no Format Validation reads its reply as text -- the text kind, by name,
|
|
3739
|
+
so a stored document never changes meaning when a plugin registers a
|
|
3740
|
+
kind of its own. */
|
|
3741
|
+
function outOf(job ) {
|
|
3742
|
+
const read = replyOf(job)?.out ?? { kind: "text", ...(OUTPUT_KINDS.text?.settingDefaults?.() || {}) };
|
|
3743
|
+
const modifiers = (job?.steps || []).filter((st) => isObj(st) && st.type === "modifier").map(st => st.modifier);
|
|
3744
|
+
return { ...read, modifiers } ;
|
|
3454
3745
|
}
|
|
3455
3746
|
|
|
3456
|
-
/** [job] reading its reply as [out]. */
|
|
3457
|
-
|
|
3747
|
+
/** [job] reading its reply as [out]: its Format Validation and modifiers. */
|
|
3748
|
+
function withOut(job , out ) {
|
|
3749
|
+
const { modifiers, ...read } = out;
|
|
3750
|
+
const steps = job.steps.filter(st => st.type !== "readReply" && st.type !== "modifier");
|
|
3751
|
+
return { ...job, steps: ordered([...steps, { type: "readReply", out: read },
|
|
3752
|
+
...(modifiers || []).map(modifier => ({ type: "modifier", modifier }) )]) };
|
|
3753
|
+
}
|
|
3458
3754
|
|
|
3459
|
-
/** [job] with its call changed by [patch]. */
|
|
3460
|
-
const withCall = (job , patch ) => withSlot(job, "call", st => ({ ...st, ...patch } ));
|
|
3461
3755
|
/** A new id: random, so one minted in one browser never collides with one
|
|
3462
|
-
minted in another. A pipeline, its jobs and its
|
|
3756
|
+
minted in another. A pipeline, its jobs and its targets each carry one
|
|
3463
3757
|
(#93), minted once and never shown. */
|
|
3464
3758
|
function newId() {
|
|
3465
3759
|
const bytes = new Uint8Array(6);
|
|
@@ -3470,53 +3764,47 @@ function newId() {
|
|
|
3470
3764
|
/** A new job: the kind a stage answers in, that kind's own modifiers, and
|
|
3471
3765
|
a copy of the token set. */
|
|
3472
3766
|
function jobDefaults(kind = defaultOutputKind, tokens = TOKEN_DEFAULTS) {
|
|
3473
|
-
|
|
3474
|
-
|
|
3475
|
-
steps: [
|
|
3476
|
-
{ type: "prompt", withImage: false, tokenMappings: clone(tokens) },
|
|
3477
|
-
{ type: "readReply",
|
|
3478
|
-
out: { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] } },
|
|
3479
|
-
],
|
|
3480
|
-
};
|
|
3767
|
+
const job = withTokens({ type: "job", id: newId(), name: "", steps: [] }, clone(tokens));
|
|
3768
|
+
return withOut(job, { kind, ...(OUTPUT_KINDS[kind]?.settingDefaults?.() || {}), modifiers: OUTPUT_KINDS[kind]?.modifiers?.() || [] });
|
|
3481
3769
|
}
|
|
3482
3770
|
STEP_TYPES.job = {
|
|
3483
3771
|
in: "item", out: job => outOf(job )?.kind,
|
|
3484
3772
|
apply: "runPipeline",
|
|
3485
3773
|
fields: ["type", "id", "name", "steps"],
|
|
3486
3774
|
defaults: jobDefaults,
|
|
3487
|
-
// A job's steps
|
|
3488
|
-
//
|
|
3775
|
+
// A job's steps run in stage order: Content, then Responses, each in its
|
|
3776
|
+
// own order. A target's step is the target's, never a job's. Each step is
|
|
3777
|
+
// its entry's to check.
|
|
3489
3778
|
validate(c, ctx, bad, k, doc){
|
|
3490
3779
|
const at = jobLabel(doc, k);
|
|
3491
3780
|
if (!isObj(c) || c.type !== "job") return void bad.push(`${at} has to be a job step`);
|
|
3492
3781
|
onlyFields(c, at, STEP_TYPES.job .fields , bad);
|
|
3493
|
-
// An id is a job's own, like a
|
|
3782
|
+
// An id is a job's own, like a target's: stable across renames, so a
|
|
3494
3783
|
// stored run keeps pointing at the job that actually ran (#93).
|
|
3495
3784
|
if (!isStr(c.id) || !c.id.trim()) bad.push(`${at} has no id`);
|
|
3496
3785
|
else if ((doc.jobs ).findIndex((o) => isObj(o) && o.id === c.id) !== k) bad.push(`${at} has the id of another job`);
|
|
3497
3786
|
if (c.name != null && !isStr(c.name)) bad.push(`${at}: name has to be text`);
|
|
3498
3787
|
if (!Array.isArray(c.steps)) return void bad.push(`${at}: steps has to be a list`);
|
|
3499
3788
|
const steps = c.steps;
|
|
3500
|
-
// A job's steps are the entries with a
|
|
3789
|
+
// A job's steps are the entries with a stage; one without (a job, the
|
|
3501
3790
|
// tests) or of no type the lab has is not one.
|
|
3502
3791
|
const unknown = steps.find(st => !slotOf(st));
|
|
3503
3792
|
if (unknown) return void bad.push(`${at}: "${isObj(unknown) ? unknown.type : unknown}" is not a step this lab has`);
|
|
3504
|
-
|
|
3505
|
-
|
|
3506
|
-
if (slots.join() !== want.join()) {
|
|
3507
|
-
// One sentence for the one mistake, the likeliest first.
|
|
3508
|
-
const label = (slot ) => slotEntry(slot)?.entry.label ?? slot;
|
|
3509
|
-
bad.push(k > 0 && slots.includes("content")
|
|
3510
|
-
? `${at}: only job 1 attaches content -- a later job is handed the reply before it`
|
|
3511
|
-
: slots.filter(x => x === "call").length !== 1 ? `${at} has to have one call step`
|
|
3512
|
-
: `${at}: steps are ${label("content")} (job 1), one call, then ${label("reply")}`);
|
|
3513
|
-
return;
|
|
3793
|
+
if (steps.some(st => slotOf(st) === "target")) {
|
|
3794
|
+
return void bad.push(`${at}: what a target sends is the target's own step, not the job's`);
|
|
3514
3795
|
}
|
|
3796
|
+
const twice = steps.find((st, x) => !STEP_TYPES[st.type] .many && steps.findIndex(o => o.type === st.type) !== x);
|
|
3797
|
+
if (twice) return void bad.push(`${at} has two ${STEP_TYPES[twice.type] .label ?? twice.type} steps`);
|
|
3798
|
+
if (steps.some((st, x) => x > 0 && placeOf(st) < placeOf(steps[x - 1]))) {
|
|
3799
|
+
return void bad.push(`${at}: steps are its Content, then its Responses, each in order`);
|
|
3800
|
+
}
|
|
3801
|
+
const kind = outOf(c ).kind;
|
|
3515
3802
|
for (const st of steps) {
|
|
3516
3803
|
const entry = STEP_TYPES[st.type] ;
|
|
3804
|
+
const label = `${at}'s ${entry.label ?? st.type}`;
|
|
3517
3805
|
if (k > 0 && entry.firstJobOnly) { bad.push(`${at}: ${entry.firstJobOnly}`); continue; }
|
|
3518
|
-
onlyFields(st,
|
|
3519
|
-
entry.validate(st, ctx, bad,
|
|
3806
|
+
onlyFields(st, label, entry.fields ?? [], bad);
|
|
3807
|
+
entry.validate(st, ctx, bad, label, kind);
|
|
3520
3808
|
}
|
|
3521
3809
|
},
|
|
3522
3810
|
};
|
|
@@ -3583,7 +3871,7 @@ STEP_TYPES.tests = {
|
|
|
3583
3871
|
|
|
3584
3872
|
// ---- the document -------------------------------------------------------------
|
|
3585
3873
|
|
|
3586
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "
|
|
3874
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "tests"];
|
|
3587
3875
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3588
3876
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3589
3877
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3603,6 +3891,10 @@ function versionProblem(doc ) {
|
|
|
3603
3891
|
rules, as a run with no graded test parsed. */
|
|
3604
3892
|
|
|
3605
3893
|
|
|
3894
|
+
|
|
3895
|
+
|
|
3896
|
+
|
|
3897
|
+
|
|
3606
3898
|
|
|
3607
3899
|
|
|
3608
3900
|
/**
|
|
@@ -3637,25 +3929,30 @@ function upgradeIds(doc ) {
|
|
|
3637
3929
|
return doc;
|
|
3638
3930
|
}
|
|
3639
3931
|
|
|
3932
|
+
/** A document's columns and their steps, wherever its version keeps them:
|
|
3933
|
+
version 10's targets, or an earlier one's scenarios and their cells. */
|
|
3934
|
+
function columnsOf(doc ) {
|
|
3935
|
+
const list = Array.isArray(doc?.targets) ? doc.targets : Array.isArray(doc?.scenarios) ? doc.scenarios : [];
|
|
3936
|
+
return list.filter(isObj).map((column ) => ({
|
|
3937
|
+
column, steps: (Array.isArray(column.steps) ? column.steps : Array.isArray(column.stages) ? column.stages : []).filter(isObj),
|
|
3938
|
+
}));
|
|
3939
|
+
}
|
|
3940
|
+
|
|
3640
3941
|
/** Every profile reference cut to { id, name }. The Runs tab once kept the
|
|
3641
3942
|
whole Setup profile a scenario picked -- key and all -- so a document saved
|
|
3642
3943
|
then is cleaned as it is read, and validation refuses one that is not. */
|
|
3643
3944
|
function profileRefs(doc ) {
|
|
3644
3945
|
const cut = (r ) => (isObj(r) && isStr(r.id) ? { id: r.id, name: isStr(r.name) ? r.name : "" } : r);
|
|
3645
|
-
for (const
|
|
3646
|
-
if (
|
|
3647
|
-
if (
|
|
3648
|
-
for (const cell of Array.isArray(sc.stages) ? sc.stages : []) {
|
|
3649
|
-
if (isObj(cell) && cell.profile != null) cell.profile = cut(cell.profile);
|
|
3650
|
-
}
|
|
3946
|
+
for (const { column, steps } of columnsOf(doc)) {
|
|
3947
|
+
if ("profile" in column) column.profile = cut(column.profile);
|
|
3948
|
+
for (const st of steps) if (st.profile != null) st.profile = cut(st.profile);
|
|
3651
3949
|
}
|
|
3652
3950
|
}
|
|
3653
3951
|
|
|
3654
3952
|
/** Whether any profile reference in a pipeline holds more than { id, name }. */
|
|
3655
3953
|
function fatProfileRef(doc ) {
|
|
3656
3954
|
const fat = (r ) => isObj(r) && Object.keys(r).some(k => k !== "id" && k !== "name");
|
|
3657
|
-
return isObj(doc) &&
|
|
3658
|
-
isObj(sc) && (fat(sc.profile) || (Array.isArray(sc.stages) && sc.stages.some((c ) => isObj(c) && fat(c.profile)))));
|
|
3955
|
+
return isObj(doc) && columnsOf(doc).some(({ column, steps }) => fat(column.profile) || steps.some(st => fat(st.profile)));
|
|
3659
3956
|
}
|
|
3660
3957
|
|
|
3661
3958
|
/** Version 5 to 6: the one test, or none, as a list. Its id is `t1`, so
|
|
@@ -3723,22 +4020,107 @@ function metricsOf(next ) {
|
|
|
3723
4020
|
return next;
|
|
3724
4021
|
}
|
|
3725
4022
|
|
|
4023
|
+
/** Version 9 to 10: a job's stages, and each scenario a target
|
|
4024
|
+
(pipeline-model §16). A job's call becomes what it held for the whole
|
|
4025
|
+
job -- a Prompt's image flag and token mappings, an HTTP Request's flow
|
|
4026
|
+
step and reading -- and each scenario's cell in it, with the call's
|
|
4027
|
+
template where it has one, that target's own step there. Read Reply's
|
|
4028
|
+
modifiers are steps of their own after it. What each held is moved as it
|
|
4029
|
+
was, so the run sends the same requests and reads them the same way
|
|
4030
|
+
(fixtures/stages-v9.json, held to it by pipeline-check.js). */
|
|
4031
|
+
function targetsFromScenarios(next ) {
|
|
4032
|
+
next.version = PIPELINE_VERSION;
|
|
4033
|
+
const jobs = Array.isArray(next.jobs) ? next.jobs : [];
|
|
4034
|
+
const calls = jobs.map(job => (isObj(job) && Array.isArray(job.steps)
|
|
4035
|
+
? job.steps.find((st ) => isObj(st) && (st.type === "prompt" || st.type === "httpRequest")) : undefined));
|
|
4036
|
+
jobs.forEach((job, k) => {
|
|
4037
|
+
if (!isObj(job) || !Array.isArray(job.steps)) return;
|
|
4038
|
+
job.steps = job.steps.flatMap((st ) => {
|
|
4039
|
+
if (!isObj(st)) return [st];
|
|
4040
|
+
if (st === calls[k] && st.type === "prompt") {
|
|
4041
|
+
return [...(st.withImage ? [{ type: "attachImage" }] : []),
|
|
4042
|
+
...(Array.isArray(st.tokenMappings) && st.tokenMappings.length ? [{ type: "tokenMappings", tokenMappings: st.tokenMappings }] : [])];
|
|
4043
|
+
}
|
|
4044
|
+
if (st === calls[k]) {
|
|
4045
|
+
return [{ type: "flowStep", step: st.step, loop: st.loop ?? null, api: st.api },
|
|
4046
|
+
...(st.readAs != null ? [{ type: "readAs", readAs: st.readAs }] : [])];
|
|
4047
|
+
}
|
|
4048
|
+
if (st.type === "readReply" && isObj(st.out)) {
|
|
4049
|
+
const { modifiers, ...out } = st.out;
|
|
4050
|
+
return [{ type: "readReply", out },
|
|
4051
|
+
...(Array.isArray(modifiers) ? modifiers : []).map((modifier ) => ({ type: "modifier", modifier }))];
|
|
4052
|
+
}
|
|
4053
|
+
return [st];
|
|
4054
|
+
});
|
|
4055
|
+
});
|
|
4056
|
+
const targets = (Array.isArray(next.scenarios) ? next.scenarios : []).map((sc ) => {
|
|
4057
|
+
if (!isObj(sc)) return sc;
|
|
4058
|
+
const { stages, ...rest } = sc;
|
|
4059
|
+
return { ...rest, steps: (Array.isArray(stages) ? stages : []).map((cell , k ) => {
|
|
4060
|
+
const call = calls[k];
|
|
4061
|
+
if (!isObj(cell) || !isObj(call)) return cell;
|
|
4062
|
+
const { step: _step, loop: _loop, readAs: _readAs, withImage: _image, tokenMappings: _tokens, ...shape } = call;
|
|
4063
|
+
return { ...clone(shape), ...cell };
|
|
4064
|
+
}) };
|
|
4065
|
+
});
|
|
4066
|
+
// In the scenarios' place, so the document reads in the same order.
|
|
4067
|
+
return Object.fromEntries(Object.entries(next).map(([key, v]) => (key === "scenarios" ? ["targets", targets] : [key, v])));
|
|
4068
|
+
}
|
|
4069
|
+
|
|
3726
4070
|
function upgradePipeline (doc , ctx = {}) {
|
|
3727
4071
|
// A current document is read as it is, but for a profile reference the Runs
|
|
3728
|
-
// tab saved whole (see profileRefs), which is cut back
|
|
4072
|
+
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
4073
|
+
// profile a target step stands in for (localSteps).
|
|
3729
4074
|
if (isObj(doc) && doc.version === PIPELINE_VERSION) {
|
|
3730
|
-
|
|
3731
|
-
|
|
3732
|
-
|
|
3733
|
-
|
|
4075
|
+
let out = doc ;
|
|
4076
|
+
if (fatProfileRef(out)) {
|
|
4077
|
+
out = clone(out) ;
|
|
4078
|
+
profileRefs(out);
|
|
4079
|
+
}
|
|
4080
|
+
return localSteps(out, ctx) ;
|
|
4081
|
+
}
|
|
4082
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9].includes(doc.version )) return doc;
|
|
4083
|
+
// Every version before 10 reads as version 9 first, then as 10.
|
|
4084
|
+
return localSteps(targetsFromScenarios(nineOf(clone(doc) , ctx)), ctx) ;
|
|
4085
|
+
}
|
|
4086
|
+
|
|
4087
|
+
/** [doc] with each step asked of a profile whose type a target step stands
|
|
4088
|
+
in for (`replaces`: Echo for an Echo profile) read as that step -- its
|
|
4089
|
+
words kept, and no profile, since it asks none -- once the profiles'
|
|
4090
|
+
types are known: [ctx]'s, or a run document's own. A target whose every
|
|
4091
|
+
step then asks none, or names its own, keeps no profile either. The same
|
|
4092
|
+
document, untouched, when there is nothing to read so. */
|
|
4093
|
+
function localSteps(doc , ctx ) {
|
|
4094
|
+
const typeOfProfile = ctx.profileType
|
|
4095
|
+
?? (isObj(doc.profiles) ? (id ) => (isObj(doc.profiles[id]) ? doc.profiles[id].type : undefined) : null);
|
|
4096
|
+
if (!typeOfProfile) return doc;
|
|
4097
|
+
const stands = new Map(Object.entries(STEP_TYPES).filter(([, e]) => e.slot === "target" && e.replaces).map(([type, e]) => [e.replaces , type]));
|
|
4098
|
+
const as = (ref ) => (isObj(ref) && isStr(ref.id) ? stands.get(typeOfProfile(ref.id) ?? "") : undefined);
|
|
4099
|
+
const turns = (t , st ) => isObj(st) && STEP_TYPES[st.type]?.asks === "prompt" && as(st.profile ?? t.profile);
|
|
4100
|
+
const targets = Array.isArray(doc.targets) ? doc.targets : [];
|
|
4101
|
+
if (!targets.some(t => isObj(t) && Array.isArray(t.steps) && t.steps.some((st ) => turns(t, st)))) return doc;
|
|
4102
|
+
const next = clone(doc) ;
|
|
4103
|
+
for (const t of next.targets ) {
|
|
4104
|
+
if (!isObj(t) || !Array.isArray(t.steps)) continue;
|
|
4105
|
+
t.steps = t.steps.map((st ) => {
|
|
4106
|
+
const type = turns(t, st);
|
|
4107
|
+
return type ? { type, prompt: st.prompt, ...(st.from ? { from: st.from } : {}) } : st;
|
|
4108
|
+
});
|
|
4109
|
+
// A run keeps the profile it ran as, so its report names it as it did.
|
|
4110
|
+
if (!isObj(doc.profiles) && as(t.profile)
|
|
4111
|
+
&& t.steps.every((st ) => isObj(st) && (STEP_TYPES[st.type]?.local || st.profile != null))) t.profile = null;
|
|
3734
4112
|
}
|
|
3735
|
-
|
|
3736
|
-
|
|
3737
|
-
|
|
3738
|
-
|
|
3739
|
-
|
|
3740
|
-
if (next.version ===
|
|
3741
|
-
if (next.version ===
|
|
4113
|
+
return next;
|
|
4114
|
+
}
|
|
4115
|
+
|
|
4116
|
+
/** [next] as version 9 held it, from any version before 10. */
|
|
4117
|
+
function nineOf(next , ctx ) {
|
|
4118
|
+
if (next.version === 9) { profileRefs(next); return next; }
|
|
4119
|
+
if (next.version === 8) return metricsOf(next);
|
|
4120
|
+
if (next.version === 7) return metricsOf(stepsOf(testsList(next)));
|
|
4121
|
+
if (next.version === 6) return metricsOf(stepsOf(jobsOf(next)));
|
|
4122
|
+
if (next.version === 5) return metricsOf(stepsOf(jobsOf(testsList(next))));
|
|
4123
|
+
if (next.version === 3 || next.version === 4) return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
|
|
3742
4124
|
if (next.version === 1) {
|
|
3743
4125
|
next.version = 2;
|
|
3744
4126
|
if (Array.isArray(next.chains)) {
|
|
@@ -3764,7 +4146,7 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
3764
4146
|
if (isObj(cell)) cell.prompt = renamed(cell.prompt);
|
|
3765
4147
|
}
|
|
3766
4148
|
}
|
|
3767
|
-
return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))))
|
|
4149
|
+
return metricsOf(stepsOf(jobsOf(testsList(upgradeIds(next)))));
|
|
3768
4150
|
}
|
|
3769
4151
|
|
|
3770
4152
|
/** Version 1's two maps as a list of mappings. */
|
|
@@ -3781,9 +4163,9 @@ function tokenMappingsFromV1(set ) {
|
|
|
3781
4163
|
the item's image, as a job over a Source would; one over text content
|
|
3782
4164
|
has none to send, and the page turns it off with the content. */
|
|
3783
4165
|
function blankPipeline(opts = {}) {
|
|
3784
|
-
const job =
|
|
4166
|
+
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
3785
4167
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
3786
|
-
jobs: [job],
|
|
4168
|
+
jobs: [job], targets: [], tests: [] };
|
|
3787
4169
|
}
|
|
3788
4170
|
|
|
3789
4171
|
/**
|
|
@@ -3808,37 +4190,50 @@ function validatePipeline(input , ctx = {}) {
|
|
|
3808
4190
|
if (!Array.isArray(doc.jobs) || !doc.jobs.length) {
|
|
3809
4191
|
return [...bad, "jobs has to be a list of at least one job"];
|
|
3810
4192
|
}
|
|
3811
|
-
if (!
|
|
4193
|
+
if (!targetsOf(doc).length) bad.push("add a target first");
|
|
3812
4194
|
// Content is job 1's Attach Content step; a job 1 without one has none yet.
|
|
3813
|
-
if (!
|
|
4195
|
+
if (!contentOf(doc)) bad.push("set the content first");
|
|
3814
4196
|
if (run) profilesProblems(doc.profiles, bad);
|
|
3815
4197
|
doc.jobs.forEach((ch , k ) => STEP_TYPES.job .validate(ch, c, bad, k, doc));
|
|
3816
4198
|
if (bad.length) return bad;
|
|
3817
4199
|
const content = contentOf(doc) ;
|
|
3818
4200
|
|
|
3819
|
-
const
|
|
4201
|
+
const targets = targetsOf(doc);
|
|
3820
4202
|
const n = doc.jobs.length;
|
|
3821
|
-
|
|
3822
|
-
const at =
|
|
3823
|
-
if (!isObj(
|
|
3824
|
-
onlyFields(
|
|
3825
|
-
if (!isStr(
|
|
3826
|
-
else if (
|
|
3827
|
-
if (
|
|
3828
|
-
if (!isRef(
|
|
3829
|
-
else onlyFields(
|
|
3830
|
-
if (!Array.isArray(
|
|
3831
|
-
bad.push(`${at} has ${Array.isArray(
|
|
4203
|
+
targets.forEach((t, i) => {
|
|
4204
|
+
const at = targetLabel(doc, i);
|
|
4205
|
+
if (!isObj(t)) return void bad.push(`${at} has to be an object`);
|
|
4206
|
+
onlyFields(t, at, ["id", "name", "profile", "steps"], bad);
|
|
4207
|
+
if (!isStr(t.id) || !t.id.trim()) bad.push(`${at} has no id`);
|
|
4208
|
+
else if (targets.findIndex((o) => isObj(o) && o.id === t.id) !== i) bad.push(`${at} has the id of another target`);
|
|
4209
|
+
if (t.name != null && !isStr(t.name)) bad.push(`${at}: name has to be text`);
|
|
4210
|
+
if (t.profile != null && !isRef(t.profile)) bad.push(`${at} has to name its Target profile as { id, name }`);
|
|
4211
|
+
else if (t.profile != null) onlyFields(t.profile, `${at}'s profile`, ["id", "name"], bad);
|
|
4212
|
+
if (!Array.isArray(t.steps) || t.steps.length !== n) {
|
|
4213
|
+
bad.push(`${at} has ${Array.isArray(t.steps) ? t.steps.length : "no"} prompts for ${n} job${n === 1 ? "" : "s"}`);
|
|
3832
4214
|
return;
|
|
3833
4215
|
}
|
|
3834
|
-
|
|
3835
|
-
|
|
3836
|
-
|
|
3837
|
-
|
|
3838
|
-
|
|
3839
|
-
|
|
3840
|
-
|
|
3841
|
-
if (
|
|
4216
|
+
// A target with no profile of its own is one whose steps need none, or
|
|
4217
|
+
// each name theirs.
|
|
4218
|
+
if (t.profile == null && t.steps.some((st ) => !(isObj(st) && STEP_TYPES[st.type]?.local) && !(isObj(st) && st.profile != null))) {
|
|
4219
|
+
bad.push(`${at} has to name its Target profile as { id, name }`);
|
|
4220
|
+
}
|
|
4221
|
+
t.steps.forEach((st , k ) => {
|
|
4222
|
+
const where = `${at}, ${jobLabel(doc, k)}`;
|
|
4223
|
+
if (!isObj(st)) return void bad.push(`${where} has to be an object`);
|
|
4224
|
+
const entry = STEP_TYPES[st.type];
|
|
4225
|
+
if (entry?.slot !== "target") return void bad.push(`${where}: "${st.type}" is not something a target can send`);
|
|
4226
|
+
if (entry.local && st.profile != null) return void bad.push(`${where}: ${entry.label} asks no profile`);
|
|
4227
|
+
if (k > 0 && entry.firstJobOnly) return void bad.push(`${where}: ${entry.firstJobOnly}`);
|
|
4228
|
+
onlyFields(st, where, entry.fields ?? [], bad);
|
|
4229
|
+
entry.validate(st, c, bad, where);
|
|
4230
|
+
if (st.profile != null && !isRef(st.profile)) bad.push(`${where} has to name its Target profile as { id, name }`);
|
|
4231
|
+
else if (st.profile != null) onlyFields(st.profile, `${where}'s profile`, ["id", "name"], bad);
|
|
4232
|
+
if (st.from != null && !isRef(st.from)) bad.push(`${where} has to name the prompt it was picked from as { id, name }`);
|
|
4233
|
+
// A request is a flow step's, rebuilt from its records, and its
|
|
4234
|
+
// template has no place for an image.
|
|
4235
|
+
if (entry.asks === "request" && !flowStepOf(doc.jobs[k])) bad.push(`${where}: ${entry.label} needs the job's flow step`);
|
|
4236
|
+
if (entry.asks === "request" && sendsImage(doc.jobs[k])) bad.push(`${where}: ${entry.label} has no place for an image`);
|
|
3842
4237
|
});
|
|
3843
4238
|
});
|
|
3844
4239
|
if (bad.length) return bad;
|
|
@@ -3849,55 +4244,63 @@ function validatePipeline(input , ctx = {}) {
|
|
|
3849
4244
|
const p = c.profiles ? c.profiles(ref.id) : {};
|
|
3850
4245
|
if (!p && !missing.has(ref.id)) {
|
|
3851
4246
|
missing.add(ref.id);
|
|
3852
|
-
bad.push(`
|
|
4247
|
+
bad.push(`Target profile ${ref.name || ref.id} not found`);
|
|
3853
4248
|
}
|
|
3854
4249
|
return p;
|
|
3855
4250
|
};
|
|
3856
4251
|
// A type that answers from the item (Echo) has no model, answers only
|
|
3857
4252
|
// job 1, and answers only a text item.
|
|
3858
4253
|
const local = (p ) => !!p && typeof p === "object" && !!CONNECTION_TYPES[typeOf(p )]?.local;
|
|
3859
|
-
// Every
|
|
4254
|
+
// Every target needs a prompt in every job. A blank one is an error
|
|
3860
4255
|
// rather than a stage dropped, which would hand the job before it
|
|
3861
4256
|
// straight to the job after it, under numbers nobody sees. The one it
|
|
3862
4257
|
// may leave blank is job 1 answered from the item's own text (Echo over
|
|
3863
4258
|
// a Source or Text): the reply is the item, and a prompt would only be
|
|
3864
4259
|
// what it is read against. Over Prompt only the prompt is the reply.
|
|
4260
|
+
// What target [i] asks in job [k]: Echo's own connection for a step
|
|
4261
|
+
// answered here, else its profile as the lookups hold it.
|
|
4262
|
+
const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION
|
|
4263
|
+
: c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
|
|
3865
4264
|
const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
|
|
3866
4265
|
for (let k = 0; k < n; k++) {
|
|
3867
|
-
const i =
|
|
3868
|
-
|
|
3869
|
-
|
|
3870
|
-
|
|
4266
|
+
const i = targets.findIndex((_, i) => {
|
|
4267
|
+
const words = targetStepOf(doc, i, k)?.prompt;
|
|
4268
|
+
return (!isStr(words) || !words.trim())
|
|
4269
|
+
&& !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
|
|
4270
|
+
});
|
|
4271
|
+
if (i >= 0) bad.push(n > 1 ? `${jobLabel(doc, k)}, ${targetLabel(doc, i)} has no prompt`
|
|
4272
|
+
: `${targetLabel(doc, i)} has no prompt`);
|
|
3871
4273
|
}
|
|
3872
4274
|
const contentFiles = content?.type === "source"
|
|
3873
4275
|
? (c.sources?.(content.ref?.id)?.files ?? content.files ?? null) : null;
|
|
3874
4276
|
const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
|
|
3875
4277
|
.find(n => !/\.(?:txt|md|csv)$/i.test(n));
|
|
3876
|
-
|
|
3877
|
-
const conns =
|
|
4278
|
+
targets.forEach((_, i) => {
|
|
4279
|
+
const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION : lookup(targetProfileOf(doc, i, k) )));
|
|
3878
4280
|
conns.forEach((p , k ) => {
|
|
3879
4281
|
if (!local(p)) return;
|
|
3880
4282
|
const label = CONNECTION_TYPES[typeOf(p )] .label;
|
|
3881
|
-
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${
|
|
3882
|
-
else if (nonText) bad.push(`${
|
|
4283
|
+
if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
|
|
4284
|
+
else if (nonText) bad.push(`${targetLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
|
|
3883
4285
|
});
|
|
3884
|
-
// A connection answers what its
|
|
4286
|
+
// A connection answers what its target's step asks: words, or a whole request.
|
|
3885
4287
|
conns.forEach((p , k ) => {
|
|
3886
|
-
const
|
|
3887
|
-
// Only a profile that was looked up says what it answers
|
|
3888
|
-
|
|
4288
|
+
const step = targetEntryOf(doc, i, k);
|
|
4289
|
+
// Only a profile that was looked up says what it answers; a step
|
|
4290
|
+
// answered here asks none.
|
|
4291
|
+
if (!c.profiles || !p || !isObj(p) || !step || step.local) return;
|
|
3889
4292
|
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3890
|
-
if (type && !(type.answers ?? ["prompt"]).includes(
|
|
3891
|
-
bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${
|
|
4293
|
+
if (type && !(type.answers ?? ["prompt"]).includes(step.asks ?? "prompt")) {
|
|
4294
|
+
bad.push(`${n > 1 ? `${jobLabel(doc, k)}, ` : ""}${targetLabel(doc, i)}: ${step.label ?? "this step"} needs a profile that answers it, and ${type.label} does not`);
|
|
3892
4295
|
}
|
|
3893
4296
|
});
|
|
3894
|
-
// The model is the profile's unless the
|
|
4297
|
+
// The model is the profile's unless the target's step carries it.
|
|
3895
4298
|
const bare = conns.findIndex((p , k ) => p && c.profiles && !local(p)
|
|
3896
|
-
&&
|
|
4299
|
+
&& targetEntryOf(doc, i, k)?.modelFrom !== "step" && !String(p.model || "").trim());
|
|
3897
4300
|
if (bare >= 0) {
|
|
3898
4301
|
bad.push(n > 1
|
|
3899
|
-
? `no model on the profile ${jobLabel(doc, bare)} of ${
|
|
3900
|
-
: `no model on ${
|
|
4302
|
+
? `no model on the profile ${jobLabel(doc, bare)} of ${targetLabel(doc, i)} uses — manage profiles on the Setup tab`
|
|
4303
|
+
: `no model on ${targetLabel(doc, i)}'s Target profile — manage profiles on the Setup tab`);
|
|
3901
4304
|
}
|
|
3902
4305
|
});
|
|
3903
4306
|
STEP_TYPES.tests .validate(doc.tests, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
@@ -3906,11 +4309,11 @@ function validatePipeline(input , ctx = {}) {
|
|
|
3906
4309
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
3907
4310
|
// says which job, instead of failing every item the same way.
|
|
3908
4311
|
const text = CONTENT_TYPES[content?.type]?.text?.(content, c) ?? null;
|
|
3909
|
-
|
|
3910
|
-
const stages = doc.jobs.map((ch , k ) => ({ text:
|
|
3911
|
-
verbatim: !!
|
|
3912
|
-
const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) =>
|
|
3913
|
-
if (problem) bad.push(`${
|
|
4312
|
+
targets.forEach((_, i) => {
|
|
4313
|
+
const stages = doc.jobs.map((ch , k ) => ({ text: targetStepOf(doc, i, k) .prompt, kind: outOf(ch).kind,
|
|
4314
|
+
verbatim: !!targetEntryOf(doc, i, k)?.verbatim }));
|
|
4315
|
+
const problem = jobProblem(stages, stages.map(() => () => {}), doc.jobs.map((ch ) => tokensOf(ch)), text);
|
|
4316
|
+
if (problem) bad.push(`${targetLabel(doc, i)}: ${jobWords(problem)}`);
|
|
3914
4317
|
});
|
|
3915
4318
|
return bad;
|
|
3916
4319
|
}
|
|
@@ -3990,10 +4393,10 @@ function connectionProblems(conn , at , from
|
|
|
3990
4393
|
}
|
|
3991
4394
|
|
|
3992
4395
|
/** The profile ids a pipeline references, scenarios first, in order. */
|
|
3993
|
-
function profileIds(doc
|
|
4396
|
+
function profileIds(doc ) {
|
|
3994
4397
|
const ids = [];
|
|
3995
|
-
for (const
|
|
3996
|
-
for (const ref of [
|
|
4398
|
+
for (const t of targetsOf(doc)) {
|
|
4399
|
+
for (const ref of [t.profile, ...(Array.isArray(t.steps) ? t.steps : []).map(st => st?.profile)] ) {
|
|
3997
4400
|
if (ref?.id && !ids.includes(ref.id)) ids.push(ref.id);
|
|
3998
4401
|
}
|
|
3999
4402
|
}
|
|
@@ -4076,7 +4479,7 @@ function importPipeline(input , ctx = {}) {
|
|
|
4076
4479
|
const missing = [], seen = new Set ();
|
|
4077
4480
|
const lists = { profile: profiles, source: sources, dataset: datasets };
|
|
4078
4481
|
const words = {
|
|
4079
|
-
profile: (ref ) => `
|
|
4482
|
+
profile: (ref ) => `Target profile ${ref.name || ref.id} not found`,
|
|
4080
4483
|
source: (ref ) => `Source ${ref.name || ref.id} not found`,
|
|
4081
4484
|
dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
|
|
4082
4485
|
};
|
|
@@ -4089,15 +4492,20 @@ function importPipeline(input , ctx = {}) {
|
|
|
4089
4492
|
};
|
|
4090
4493
|
const content = contentOf(next) ;
|
|
4091
4494
|
if (content?.type === "source") content.ref = remap(content.ref, "source");
|
|
4092
|
-
for (const
|
|
4093
|
-
|
|
4094
|
-
for (const
|
|
4095
|
-
if (
|
|
4495
|
+
for (const t of targetsOf(next)) {
|
|
4496
|
+
if (t.profile) t.profile = remap(t.profile, "profile");
|
|
4497
|
+
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4498
|
+
if (st?.profile) st.profile = remap(st.profile, "profile");
|
|
4096
4499
|
}
|
|
4097
4500
|
}
|
|
4098
4501
|
for (const t of Array.isArray(next.tests) ? next.tests : []) {
|
|
4099
4502
|
if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
|
|
4100
4503
|
}
|
|
4504
|
+
// Its references are this lab's now, so a step asking this lab's Echo
|
|
4505
|
+
// profile is read as an Echo step (upgradePipeline did it for ids that
|
|
4506
|
+
// already matched).
|
|
4507
|
+
const local = localSteps(next, ctx);
|
|
4508
|
+
if (local !== next) Object.assign(next, local);
|
|
4101
4509
|
// A hand-written file needs no id of its own: mint the ones it lacks, and
|
|
4102
4510
|
// keep the ones it carries, so export → import → export is the same
|
|
4103
4511
|
// document and a re-import can recognise the same pipeline (#93).
|
|
@@ -4119,8 +4527,8 @@ function mintIds(doc ) {
|
|
|
4119
4527
|
for (const ch of Array.isArray(doc.jobs) ? doc.jobs : []) {
|
|
4120
4528
|
if (isObj(ch) && (!isStr(ch.id) || !ch.id)) ch.id = newId();
|
|
4121
4529
|
}
|
|
4122
|
-
for (const
|
|
4123
|
-
if (isObj(
|
|
4530
|
+
for (const t of Array.isArray(doc.targets) ? doc.targets : []) {
|
|
4531
|
+
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4124
4532
|
}
|
|
4125
4533
|
for (const t of Array.isArray(doc.tests) ? doc.tests : []) {
|
|
4126
4534
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
@@ -4146,26 +4554,34 @@ function yamlToPipeline(text , ctx ) {
|
|
|
4146
4554
|
*/
|
|
4147
4555
|
function stagesFor(run , i )
|
|
4148
4556
|
|
|
4149
|
-
|
|
4150
|
-
const sc = run.scenarios[i] ;
|
|
4557
|
+
{
|
|
4151
4558
|
const stages = run.jobs.map((ch, k) => {
|
|
4152
4559
|
const { kind, modifiers, ...settings } = outOf(ch);
|
|
4153
|
-
return { text:
|
|
4154
|
-
verbatim: !!
|
|
4560
|
+
return { text: targetStepOf(run, i, k) .prompt, kind, withImage: sendsImage(ch),
|
|
4561
|
+
verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: modifiers || [], settings };
|
|
4155
4562
|
});
|
|
4156
|
-
const connections = run.jobs.map((
|
|
4157
|
-
|
|
4563
|
+
const connections = run.jobs.map((_, k) => {
|
|
4564
|
+
// A step answered here asks nothing: the profile a run made before it
|
|
4565
|
+
// kept is what it ran as, else Echo itself.
|
|
4566
|
+
if (targetEntryOf(run, i, k)?.local) {
|
|
4567
|
+
const id = targetProfileOf(run, i, k)?.id;
|
|
4568
|
+
const held = id != null ? run.profiles?.[id] : undefined;
|
|
4569
|
+
return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...ECHO_CONNECTION };
|
|
4570
|
+
}
|
|
4571
|
+
const id = targetProfileOf(run, i, k) .id;
|
|
4158
4572
|
return { id, ...(run.profiles?.[id] || {}) };
|
|
4159
4573
|
});
|
|
4160
|
-
//
|
|
4161
|
-
//
|
|
4162
|
-
return { stages, tokens: run.jobs.map(
|
|
4163
|
-
calls: run.jobs.map(
|
|
4574
|
+
// What a transport that builds its own request (an HTTP Request) builds
|
|
4575
|
+
// it from in each job, and this target's step there.
|
|
4576
|
+
return { stages, tokens: run.jobs.map(tokensOf), connections,
|
|
4577
|
+
calls: run.jobs.map((_, k) => requestStepOf(run, i, k)), cells: run.jobs.map((_, k) => targetStepOf(run, i, k) ) };
|
|
4164
4578
|
}
|
|
4165
4579
|
|
|
4166
4580
|
/** The Setup profile scenario [i] of a run ran under, as History shows it: keyless. */
|
|
4167
4581
|
function scenarioProfile(run , i ) {
|
|
4168
|
-
const id = run
|
|
4582
|
+
const id = targetsOf(run)[i]?.profile?.id;
|
|
4583
|
+
// A target that asks nothing ran as Echo.
|
|
4584
|
+
if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name: ECHO_CONNECTION.name, settings: { ...ECHO_CONNECTION } };
|
|
4169
4585
|
const conn = id != null ? run.profiles?.[id] : null;
|
|
4170
4586
|
return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
|
|
4171
4587
|
}
|
|
@@ -4229,9 +4645,8 @@ function itemScores(run , kase , res
|
|
|
4229
4645
|
/** What production replied to an item, for a run's metrics: its last job's
|
|
4230
4646
|
call says, from the item's record, or nobody does. */
|
|
4231
4647
|
function productionOf(run , record ) {
|
|
4232
|
-
const last = run.jobs.at(-1);
|
|
4233
|
-
|
|
4234
|
-
return record ? STEP_TYPES[(call )?.type ?? ""]?.production?.(call, record) ?? null : null;
|
|
4648
|
+
const last = run.jobs.at(-1), flow = flowStepOf(last);
|
|
4649
|
+
return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
|
|
4235
4650
|
}
|
|
4236
4651
|
|
|
4237
4652
|
/**
|
|
@@ -4507,7 +4922,7 @@ export {
|
|
|
4507
4922
|
loopReplyError, preparedSize,
|
|
4508
4923
|
termIn, forbiddenIn, scoreCase, gradedSetFrom, caseFile, canonicalCases, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4509
4924
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
4510
|
-
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf,
|
|
4925
|
+
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
4511
4926
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
|
4512
4927
|
tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
|
|
4513
4928
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
@@ -4517,7 +4932,9 @@ export {
|
|
|
4517
4932
|
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
4518
4933
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
4519
4934
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
4520
|
-
contentOf, withContent,
|
|
4935
|
+
contentOf, withContent, replyOf,
|
|
4936
|
+
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
4937
|
+
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
4521
4938
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newTest, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
4522
4939
|
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioTests, scenarioPasses, isSkipped, failedScore,
|
|
4523
4940
|
testLabel, testsOf, testsDataset, isWholeRun, TEST_FIELDS,
|