evals-lab 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/README.md +1 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +1 -1
- package/lab/demo/pipelines/demo-2.json +1 -1
- package/lab/evals-core.mjs +428 -49
- package/lab/kinds/list.mjs +258 -14
- package/lab/metrics/builtin.mjs +76 -20
- package/lab/run-evals.js +42 -7
- package/lab/server.py +1200 -31
- package/lab/web/dist/assets/gallery-BpR9b4EM.js +3 -0
- package/lab/web/dist/assets/main-BMRmyJvo.css +1 -0
- package/lab/web/dist/assets/main-BfiFtaYW.js +19 -0
- package/lab/web/dist/assets/tokens-C67OCDd-.css +1 -0
- package/lab/web/dist/assets/tokens-DzZqM5IZ.js +59 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +15 -2
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +0 -3
- package/lab/web/dist/assets/main-BQL5j5oF.js +0 -20
- package/lab/web/dist/assets/main-Cza2gwQd.css +0 -1
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +0 -61
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +0 -1
package/lab/evals-core.mjs
CHANGED
|
@@ -33,12 +33,14 @@
|
|
|
33
33
|
// and that is a measurement.
|
|
34
34
|
|
|
35
35
|
import yaml from "./js-yaml.mjs";
|
|
36
|
-
|
|
36
|
+
|
|
37
37
|
import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
|
|
38
38
|
import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
|
|
39
39
|
import { flowApi, } from "./flows/flowApi.mjs";
|
|
40
40
|
import registerMetrics from "./metrics/builtin.mjs";
|
|
41
|
-
import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher
|
|
41
|
+
import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
42
|
+
FILTER_TYPES, FILTER_SEPARATORS, FILTER_MODIFIER, applyFilter, applyFilterSet, tryFilter, splitSample, filterProblems, filterFromItemRule,
|
|
43
|
+
} from "./kinds/list.mjs";
|
|
42
44
|
|
|
43
45
|
// ---- The types -------------------------------------------------------------
|
|
44
46
|
// docs/pipeline-model.md as types, and the shapes of the registries every
|
|
@@ -138,9 +140,19 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
138
140
|
|
|
139
141
|
|
|
140
142
|
|
|
143
|
+
/** The filter sets a job cleans and gates its reply by (#339): each a link to
|
|
144
|
+
a Library filter set (followed latest or pinned) or a private one held
|
|
145
|
+
inline. A Responses step beside the modifiers, at the same place, so its
|
|
146
|
+
sets' filters apply in document order with them -- what the Drop items and
|
|
147
|
+
Reject the answer modifiers did inline, lifted into linked sets. */
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
|
|
141
153
|
/** A job's steps: its Content stage, then its Responses (§16). */
|
|
142
154
|
|
|
143
|
-
|
|
155
|
+
|
|
144
156
|
|
|
145
157
|
/** A target step that prompts a model: its words, resolved under the job's
|
|
146
158
|
token mappings, and a profile of its own where it names one. */
|
|
@@ -316,6 +328,45 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
316
328
|
/** A pipeline's evals, in the order they read a run. */
|
|
317
329
|
|
|
318
330
|
|
|
331
|
+
/**
|
|
332
|
+
* A filter set (issue #335): a versioned Library body holding the filters
|
|
333
|
+
* that shape a list reply, mirroring DatasetBody / the eval group. The lab
|
|
334
|
+
* will keep each as a row and a run grade against the body kept with it, as a
|
|
335
|
+
* dataset does (#336); for now it is the format a filter set reads and writes
|
|
336
|
+
* at. `filters` are tried in order -- each drop-item narrows the items the
|
|
337
|
+
* next reads, and the first reject-reply that fires gates the answer.
|
|
338
|
+
*/
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
/** A Library filter set by reference; a run's copy says the version it used:
|
|
345
|
+
the body's fingerprint, and its number. Mirrors GroupRef. */
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
/** A private filter set: a filter-set body held in the pipeline rather than
|
|
349
|
+
the Library -- what Runs makes when a link is not to a Library set (#339),
|
|
350
|
+
as a private eval group (OwnGroup) is. A filter set has no cases, so this
|
|
351
|
+
is the body itself. */
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
/**
|
|
355
|
+
* A pipeline's link to a filter set in Responses (issue #335, mirrors
|
|
356
|
+
* GroupEval / OwnGroup): `set` names a Library set, followed at its newest
|
|
357
|
+
* version (`pin: null`) or pinned at one (`pin: n`); a private set is `own`,
|
|
358
|
+
* with `set: null`. From version 14 a job's Responses stage carries these on
|
|
359
|
+
* a `filters` step (#339); the upgrade from 13 migrates the inline Drop items
|
|
360
|
+
* / Reject rules into a private one, and the worker cleans and gates a run's
|
|
361
|
+
* replies against the kept bodies.
|
|
362
|
+
*/
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
|
|
319
370
|
|
|
320
371
|
|
|
321
372
|
|
|
@@ -657,6 +708,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
657
708
|
|
|
658
709
|
|
|
659
710
|
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
|
|
660
714
|
|
|
661
715
|
|
|
662
716
|
|
|
@@ -667,9 +721,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
667
721
|
|
|
668
722
|
/** What a metric needs besides the reply: a model to grade with. */
|
|
669
723
|
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
|
|
673
728
|
|
|
674
729
|
|
|
675
730
|
|
|
@@ -723,6 +778,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
723
778
|
|
|
724
779
|
|
|
725
780
|
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
|
|
726
788
|
|
|
727
789
|
|
|
728
790
|
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
@@ -797,6 +859,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
797
859
|
|
|
798
860
|
|
|
799
861
|
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
|
|
800
865
|
|
|
801
866
|
|
|
802
867
|
|
|
@@ -860,6 +925,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
860
925
|
|
|
861
926
|
|
|
862
927
|
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
|
|
863
933
|
|
|
864
934
|
|
|
865
935
|
|
|
@@ -881,9 +951,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
881
951
|
|
|
882
952
|
|
|
883
953
|
|
|
954
|
+
|
|
884
955
|
|
|
885
|
-
|
|
886
|
-
|
|
956
|
+
|
|
887
957
|
|
|
888
958
|
|
|
889
959
|
/** Lookups, and whether it is a run document being validated. */
|
|
@@ -1041,7 +1111,7 @@ const platformOf = (src
|
|
|
1041
1111
|
|
|
1042
1112
|
|
|
1043
1113
|
|
|
1044
|
-
|
|
1114
|
+
|
|
1045
1115
|
|
|
1046
1116
|
|
|
1047
1117
|
|
|
@@ -1219,6 +1289,57 @@ function upgradeDatasetBody (body ) {
|
|
|
1219
1289
|
return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
|
|
1220
1290
|
}
|
|
1221
1291
|
|
|
1292
|
+
/** The filter-set body's version (issue #335). Version 1 is the first: the
|
|
1293
|
+
`{ version, filters }` body #334 introduces. A later format adds a step to
|
|
1294
|
+
`upgradeFilterSetBody`, as `upgradeDatasetBody` grew, with server.py's
|
|
1295
|
+
`upgrade_filter_set_body` its twin. */
|
|
1296
|
+
const FILTER_SET_VERSION = 1 ;
|
|
1297
|
+
|
|
1298
|
+
/** [body] as this version of a filter set (1), from any earlier one. There is
|
|
1299
|
+
no earlier version yet, so a version-1 body reads as it is and anything
|
|
1300
|
+
else comes back untouched for `filterSetBodyProblems` to refuse. Pure: the
|
|
1301
|
+
same body gives the same answer, and server.py's upgrade_filter_set_body is
|
|
1302
|
+
its twin. */
|
|
1303
|
+
function upgradeFilterSetBody (body ) {
|
|
1304
|
+
if (!isObj(body)) return body;
|
|
1305
|
+
if (body.version === FILTER_SET_VERSION) return body;
|
|
1306
|
+
return body;
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
/** A filter set with no filters, at this version: what New filter set starts
|
|
1310
|
+
from, and what a lab with none holds. */
|
|
1311
|
+
function blankFilterSetBody() {
|
|
1312
|
+
return { version: FILTER_SET_VERSION, filters: [] };
|
|
1313
|
+
}
|
|
1314
|
+
|
|
1315
|
+
/** Why [body] is not a filter set the lab reads, one sentence each: an object
|
|
1316
|
+
at this version whose filters are a valid filter list (filterProblems). For
|
|
1317
|
+
the store (#336) and the pipeline link (#339). */
|
|
1318
|
+
function filterSetBodyProblems(body , at = "the filter set") {
|
|
1319
|
+
if (!isObj(body)) return [`${at} has to be a JSON object`];
|
|
1320
|
+
if (body.version !== FILTER_SET_VERSION) {
|
|
1321
|
+
return [`${at} is version ${JSON.stringify(body.version)}, and the lab reads version ${FILTER_SET_VERSION}`];
|
|
1322
|
+
}
|
|
1323
|
+
return filterProblems(body.filters, at);
|
|
1324
|
+
}
|
|
1325
|
+
|
|
1326
|
+
/** Why [link] is not a filter-set link a run's Responses step may carry, one
|
|
1327
|
+
sentence each (#339): an id, exactly one of a Library `set` reference or a
|
|
1328
|
+
private `own` body (whose shape is the core's to judge), and a `pin` that
|
|
1329
|
+
is null or a whole number. A Library set is the server's to resolve, so its
|
|
1330
|
+
body is not checked here. */
|
|
1331
|
+
function filterSetLinkProblems(link , at = "a filter set") {
|
|
1332
|
+
if (!isObj(link)) return [`${at}: a filter-set link has to be a JSON object`];
|
|
1333
|
+
const bad = [];
|
|
1334
|
+
if (!isStr(link.id) || !link.id.trim()) bad.push(`${at}: a filter-set link has no id`);
|
|
1335
|
+
const hasSet = isObj(link.set) && isStr(link.set.id);
|
|
1336
|
+
const hasOwn = link.own != null;
|
|
1337
|
+
if (hasSet === hasOwn) bad.push(`${at}: a filter-set link names a Library set or holds a private one, not both`);
|
|
1338
|
+
if (hasOwn) bad.push(...filterSetBodyProblems(link.own, `${at}: the private filter set`));
|
|
1339
|
+
if (link.pin != null && !Number.isInteger(link.pin)) bad.push(`${at}: a pin is a whole number or none`);
|
|
1340
|
+
return bad;
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1222
1343
|
/** A version-6 body as a version-8 eval group: scored All, the lab's
|
|
1223
1344
|
grader, and no metrics of its own for every item or the whole run --
|
|
1224
1345
|
what a Metrics eval naming the dataset with none of its own graded, which
|
|
@@ -1323,6 +1444,10 @@ function datasetRules(body ) {
|
|
|
1323
1444
|
|
|
1324
1445
|
|
|
1325
1446
|
|
|
1447
|
+
|
|
1448
|
+
|
|
1449
|
+
|
|
1450
|
+
|
|
1326
1451
|
|
|
1327
1452
|
|
|
1328
1453
|
/** An HTTP request as a connection type builds it, for the relay to send. */
|
|
@@ -1346,6 +1471,10 @@ function datasetRules(body ) {
|
|
|
1346
1471
|
|
|
1347
1472
|
|
|
1348
1473
|
|
|
1474
|
+
|
|
1475
|
+
|
|
1476
|
+
|
|
1477
|
+
|
|
1349
1478
|
|
|
1350
1479
|
|
|
1351
1480
|
|
|
@@ -1354,8 +1483,9 @@ function datasetRules(body ) {
|
|
|
1354
1483
|
|
|
1355
1484
|
|
|
1356
1485
|
|
|
1357
|
-
|
|
1358
|
-
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
|
|
1359
1489
|
|
|
1360
1490
|
|
|
1361
1491
|
|
|
@@ -1708,6 +1838,8 @@ function ollamaType(id , label , cloud ) {
|
|
|
1708
1838
|
body: { model: conn.model, messages: [message], options, stream: false },
|
|
1709
1839
|
};
|
|
1710
1840
|
},
|
|
1841
|
+
// Ollama's own: the schema is the format.
|
|
1842
|
+
structured: (body, schema) => ({ ...body, format: schema }),
|
|
1711
1843
|
parseReply(j) {
|
|
1712
1844
|
const r = j ;
|
|
1713
1845
|
return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
|
|
@@ -1754,7 +1886,7 @@ function recordedRaw(record ) {
|
|
|
1754
1886
|
// the flow reads a reply; `local` is the no-Read-as fallback.
|
|
1755
1887
|
CONNECTION_TYPES.recorded = {
|
|
1756
1888
|
id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
|
|
1757
|
-
description: "
|
|
1889
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
1758
1890
|
local: (_item, _sent, record) => recordedRaw(record),
|
|
1759
1891
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1760
1892
|
};
|
|
@@ -1769,19 +1901,25 @@ function localAnswer(conn , text , sent , record
|
|
|
1769
1901
|
return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
|
|
1770
1902
|
}
|
|
1771
1903
|
|
|
1904
|
+
/** A chat completions body asking for a reply held to [schema]: OpenAI's
|
|
1905
|
+
response_format, which llama-server and most compatible servers take too. */
|
|
1906
|
+
const chatSchema = (body , schema ) =>
|
|
1907
|
+
({ ...body, response_format: { type: "json_schema", json_schema: { name: "reply", strict: true, schema } } });
|
|
1908
|
+
|
|
1772
1909
|
CONNECTION_TYPES["openai-compatible"] = {
|
|
1773
1910
|
id: "openai-compatible", label: "OpenAI-compatible",
|
|
1774
1911
|
description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
|
|
1775
1912
|
settings: [
|
|
1776
|
-
|
|
1913
|
+
// A reasoning model refuses a temperature.
|
|
1914
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1915
|
+
appliesTo: (conn) => !/^(gpt-5|o\d)/i.test(String(conn.model ?? "")) },
|
|
1777
1916
|
{ key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1778
1917
|
],
|
|
1779
|
-
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1780
|
-
//
|
|
1781
|
-
//
|
|
1918
|
+
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1919
|
+
// max_completion_tokens stays with this type, as the hosted providers
|
|
1920
|
+
// reject the fields a llama.cpp server takes.
|
|
1782
1921
|
request(conn, prompt, dataUrl, base, key) {
|
|
1783
1922
|
const hosted = isHostedUrl(base);
|
|
1784
|
-
const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
|
|
1785
1923
|
const temp = asNumber(conn.temperature);
|
|
1786
1924
|
const predict = asNumber(conn.nPredict);
|
|
1787
1925
|
return {
|
|
@@ -1789,7 +1927,7 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1789
1927
|
headers: bearer(key),
|
|
1790
1928
|
body: {
|
|
1791
1929
|
model: conn.model,
|
|
1792
|
-
...(
|
|
1930
|
+
...(temp == null ? {} : { temperature: temp }),
|
|
1793
1931
|
...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
|
|
1794
1932
|
messages: [{ role: "user", content: [
|
|
1795
1933
|
{ type: "text", text: prompt },
|
|
@@ -1798,15 +1936,23 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1798
1936
|
},
|
|
1799
1937
|
};
|
|
1800
1938
|
},
|
|
1939
|
+
structured: chatSchema,
|
|
1801
1940
|
parseReply: chatReply,
|
|
1802
1941
|
listModels, parseModels: idsFrom,
|
|
1803
1942
|
};
|
|
1804
1943
|
|
|
1944
|
+
/** The Claude models that refuse a sampling setting: Opus from 4.7, and
|
|
1945
|
+
every Opus, Sonnet, Fable and Mythos from 5. Older ones (Opus 4.6, Sonnet
|
|
1946
|
+
4.6, Haiku 4.5) take one. */
|
|
1947
|
+
const CLAUDE_FIXED_SAMPLING = /claude-(opus-4-[7-9]|(opus|sonnet|fable|mythos)-[5-9])/i;
|
|
1948
|
+
|
|
1805
1949
|
CONNECTION_TYPES.anthropic = {
|
|
1806
1950
|
id: "anthropic", label: "Anthropic",
|
|
1807
1951
|
description: "Claude, through Anthropic's Messages API.",
|
|
1952
|
+
url: "https://api.anthropic.com",
|
|
1808
1953
|
settings: [
|
|
1809
|
-
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: ""
|
|
1954
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1955
|
+
appliesTo: (conn) => !CLAUDE_FIXED_SAMPLING.test(String(conn.model ?? "")) },
|
|
1810
1956
|
{ key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1811
1957
|
],
|
|
1812
1958
|
// The Messages API: the x-api-key and anthropic-version headers, base64
|
|
@@ -1831,6 +1977,9 @@ CONNECTION_TYPES.anthropic = {
|
|
|
1831
1977
|
},
|
|
1832
1978
|
};
|
|
1833
1979
|
},
|
|
1980
|
+
// The Messages API's structured outputs. A model that cannot hold one
|
|
1981
|
+
// refuses the request, and the grader asks again without it.
|
|
1982
|
+
structured: (body, schema) => ({ ...body, output_config: { ...(body.output_config ), format: { type: "json_schema", schema } } }),
|
|
1834
1983
|
parseReply(j) {
|
|
1835
1984
|
const r = j ;
|
|
1836
1985
|
const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
|
|
@@ -1903,6 +2052,8 @@ CONNECTION_TYPES["llama.cpp"] = {
|
|
|
1903
2052
|
},
|
|
1904
2053
|
};
|
|
1905
2054
|
},
|
|
2055
|
+
// llama-server takes OpenAI's response_format, a schema and all.
|
|
2056
|
+
structured: chatSchema,
|
|
1906
2057
|
parseReply: chatReply,
|
|
1907
2058
|
listModels, parseModels: idsFrom,
|
|
1908
2059
|
// A reply that arrived is still a failure when this profile asked for the
|
|
@@ -2098,12 +2249,14 @@ function extraHeaders(raw ) {
|
|
|
2098
2249
|
}
|
|
2099
2250
|
return out;
|
|
2100
2251
|
}
|
|
2101
|
-
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2252
|
+
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2253
|
+
path. A step's path is the whole path production called, so an address
|
|
2254
|
+
entered with one (`…/v1/messages`) would send it twice. */
|
|
2102
2255
|
function originBase(raw ) {
|
|
2103
|
-
let s = String(raw || "").trim()
|
|
2256
|
+
let s = String(raw || "").trim();
|
|
2104
2257
|
if (!s) return "";
|
|
2105
2258
|
if (!/^https?:\/\//.test(s)) s = "https://" + s;
|
|
2106
|
-
return s;
|
|
2259
|
+
try { return new URL(s).origin; } catch { return s.replace(/\/+$/, ""); }
|
|
2107
2260
|
}
|
|
2108
2261
|
const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
|
|
2109
2262
|
CONNECTION_TYPES.http = {
|
|
@@ -2230,9 +2383,14 @@ function splitOllama(p ) {
|
|
|
2230
2383
|
* connection carries its type's settings under `options`, so they are merged
|
|
2231
2384
|
* up before the type reads them.
|
|
2232
2385
|
*/
|
|
2233
|
-
function connectionRequest(conn , prompt , dataUrl , base , key
|
|
2386
|
+
function connectionRequest(conn , prompt , dataUrl , base , key ,
|
|
2387
|
+
schema ) {
|
|
2234
2388
|
const flat = { ...conn, ...(conn?.options || {}) };
|
|
2235
|
-
|
|
2389
|
+
const type = typeEntry(flat);
|
|
2390
|
+
// A setting the model refuses is not sent, whatever the profile holds.
|
|
2391
|
+
for (const s of type.settings) if (s.appliesTo && !s.appliesTo(flat)) delete flat[s.key];
|
|
2392
|
+
const req = type.request(flat, prompt, dataUrl, base, key);
|
|
2393
|
+
return schema && type.structured ? { ...req, body: type.structured(req.body , schema) } : req;
|
|
2236
2394
|
}
|
|
2237
2395
|
|
|
2238
2396
|
/** Where [conn]'s requests hang off: its type's own reading of its address,
|
|
@@ -2885,7 +3043,7 @@ function applyModifiers (list , kind ,
|
|
|
2885
3043
|
// 12: a Contains metric's Ignore case holds item by item too.
|
|
2886
3044
|
// 13: an eval is a link to an eval group, or a group of the pipeline's own,
|
|
2887
3045
|
// and the document has an overall pass rule (pipeline-model §17).
|
|
2888
|
-
const PIPELINE_VERSION =
|
|
3046
|
+
const PIPELINE_VERSION = 14 ;
|
|
2889
3047
|
|
|
2890
3048
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2891
3049
|
// to know which registered it.
|
|
@@ -2934,6 +3092,15 @@ function registerKinds(k ) {
|
|
|
2934
3092
|
|
|
2935
3093
|
|
|
2936
3094
|
|
|
3095
|
+
|
|
3096
|
+
|
|
3097
|
+
|
|
3098
|
+
|
|
3099
|
+
|
|
3100
|
+
|
|
3101
|
+
|
|
3102
|
+
|
|
3103
|
+
|
|
2937
3104
|
|
|
2938
3105
|
|
|
2939
3106
|
function pluginHost(pluginId ) {
|
|
@@ -2961,6 +3128,14 @@ function pluginHost(pluginId ) {
|
|
|
2961
3128
|
taken(CONNECTION_TYPES, "connection type", [t.id]);
|
|
2962
3129
|
CONNECTION_TYPES[t.id] = t;
|
|
2963
3130
|
},
|
|
3131
|
+
registerWorkflowPlatform(t) {
|
|
3132
|
+
taken(WORKFLOW_PLATFORMS, "workflow platform", [t.id]);
|
|
3133
|
+
WORKFLOW_PLATFORMS[t.id] = t;
|
|
3134
|
+
},
|
|
3135
|
+
registerWizard(w) {
|
|
3136
|
+
taken(WIZARDS, "wizard", [w.id]);
|
|
3137
|
+
WIZARDS[w.id] = w;
|
|
3138
|
+
},
|
|
2964
3139
|
};
|
|
2965
3140
|
}
|
|
2966
3141
|
|
|
@@ -3058,6 +3233,13 @@ function withSlugs (list
|
|
|
3058
3233
|
/** Whether [s] is a slug as a profile may hold one. */
|
|
3059
3234
|
const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
|
|
3060
3235
|
|
|
3236
|
+
/** Where [p] connects: its address, or blank, its type's own -- an Anthropic
|
|
3237
|
+
profile left blank reaches Anthropic. Blank for a type with none means the
|
|
3238
|
+
lab's own Ollama, which the runner and the relay fill in. */
|
|
3239
|
+
function addressOf(p ) {
|
|
3240
|
+
return String(p.url ?? "").trim() || CONNECTION_TYPES[typeOf(p )]?.url || "";
|
|
3241
|
+
}
|
|
3242
|
+
|
|
3061
3243
|
/** A stored profile as a run carries it: its request settings, no key. */
|
|
3062
3244
|
function connectionOf(p ) {
|
|
3063
3245
|
const type = typeOf(p);
|
|
@@ -3066,7 +3248,7 @@ function connectionOf(p ) {
|
|
|
3066
3248
|
// temperature is a common field, carried at the top like px/format/quality;
|
|
3067
3249
|
// the options bag holds the type's own settings only.
|
|
3068
3250
|
for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
|
|
3069
|
-
const out = { name: p.name || "unnamed", url: p
|
|
3251
|
+
const out = { name: p.name || "unnamed", url: addressOf(p), model: p.model || "", type };
|
|
3070
3252
|
if (isSlug(p.slug)) out.slug = p.slug;
|
|
3071
3253
|
for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
|
|
3072
3254
|
if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
|
|
@@ -3187,6 +3369,70 @@ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapt
|
|
|
3187
3369
|
render: body => JSON.stringify(body, null, 2),
|
|
3188
3370
|
};
|
|
3189
3371
|
|
|
3372
|
+
// ---- Setup Wizards (docs/workflow-sources.md § Wizards) ---------------------
|
|
3373
|
+
//
|
|
3374
|
+
// A wizard is a registry entry whose steps each say where they happen and a
|
|
3375
|
+
// done-when predicate on lab state. Steps tick automatically: `done(lab)` reads
|
|
3376
|
+
// the cheap state the page already holds (a workflow Source exists, a pipeline
|
|
3377
|
+
// with a Recorded target exists, a run of it is in History), never a poll. One
|
|
3378
|
+
// wizard runs at a time, held in the browser-only `promptlab.wizard` store; a
|
|
3379
|
+
// header pill shows its progress and the Setup › Wizards section draws the same
|
|
3380
|
+
// walk-through. The registry lives here so a plugin registers a wizard through
|
|
3381
|
+
// the same host as every other kind (PluginHost.registerWizard, phase 7); the
|
|
3382
|
+
// page ships the built-in "Test a workflow" and owns its routes and selectors
|
|
3383
|
+
// (web/src/app/wizards.ts). The predicates run with the page's privileges and
|
|
3384
|
+
// only read lab state, so a plugin wizard is no new trust (docs/packs.md).
|
|
3385
|
+
|
|
3386
|
+
/** The slice of lab state a wizard step's `done` reads: what the page already
|
|
3387
|
+
holds, so a tick costs a predicate and not a request. A reader names a field
|
|
3388
|
+
it needs; a wizard that reads more declares it here. */
|
|
3389
|
+
|
|
3390
|
+
|
|
3391
|
+
|
|
3392
|
+
|
|
3393
|
+
|
|
3394
|
+
|
|
3395
|
+
|
|
3396
|
+
|
|
3397
|
+
|
|
3398
|
+
|
|
3399
|
+
|
|
3400
|
+
|
|
3401
|
+
|
|
3402
|
+
|
|
3403
|
+
|
|
3404
|
+
|
|
3405
|
+
|
|
3406
|
+
|
|
3407
|
+
|
|
3408
|
+
|
|
3409
|
+
|
|
3410
|
+
|
|
3411
|
+
|
|
3412
|
+
|
|
3413
|
+
|
|
3414
|
+
|
|
3415
|
+
|
|
3416
|
+
|
|
3417
|
+
|
|
3418
|
+
|
|
3419
|
+
|
|
3420
|
+
|
|
3421
|
+
|
|
3422
|
+
|
|
3423
|
+
const WIZARDS = Object.create(null);
|
|
3424
|
+
|
|
3425
|
+
/** How far a wizard has got on [lab]: each step's tick, how many are done, and
|
|
3426
|
+
the first step not yet done -- the one the checklist marks current and Go
|
|
3427
|
+
to step heads for. `current` is -1 once every step is done. */
|
|
3428
|
+
function wizardProgress(entry , lab )
|
|
3429
|
+
|
|
3430
|
+
{
|
|
3431
|
+
const ticks = entry.steps.map((s) => s.done(lab));
|
|
3432
|
+
const done = ticks.filter(Boolean).length;
|
|
3433
|
+
return { ticks, done, total: entry.steps.length, current: ticks.findIndex((t) => !t) };
|
|
3434
|
+
}
|
|
3435
|
+
|
|
3190
3436
|
CONTENT_TYPES.source = {
|
|
3191
3437
|
label: "Source",
|
|
3192
3438
|
description: "Every file or record in a Source from the Library.",
|
|
@@ -3364,6 +3610,7 @@ function metricInput(res , kase , more )
|
|
|
3364
3610
|
const last = res.transcript?.at(-1);
|
|
3365
3611
|
const text = last?.got ?? res.raw ?? "";
|
|
3366
3612
|
return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
|
|
3613
|
+
asked: last?.sent ?? null,
|
|
3367
3614
|
// A case that has been given its own expected reply ("Take B as
|
|
3368
3615
|
// expected") is compared with that; otherwise the Source's recorded
|
|
3369
3616
|
// reply for the item.
|
|
@@ -3382,7 +3629,7 @@ function runInput(ress , plain )
|
|
|
3382
3629
|
terms.push(...(r.terms || []));
|
|
3383
3630
|
ms += r.ms ?? 0;
|
|
3384
3631
|
}
|
|
3385
|
-
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
|
|
3632
|
+
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, asked: null, kase: null };
|
|
3386
3633
|
}
|
|
3387
3634
|
|
|
3388
3635
|
/** One metric's reading of [input]: `of`, `not` and a failure all applied,
|
|
@@ -3643,16 +3890,21 @@ EVAL_TYPES.group = {
|
|
|
3643
3890
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3644
3891
|
.filter(Boolean).join(" ") })),
|
|
3645
3892
|
profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
|
|
3646
|
-
// A run carries the
|
|
3647
|
-
//
|
|
3648
|
-
//
|
|
3893
|
+
// A run carries the grader each group asks, so its profile goes with the
|
|
3894
|
+
// run: a link's is the Library group's own, or the lab's where it names
|
|
3895
|
+
// none -- the run's copy of the link carries it. A private group takes the
|
|
3896
|
+
// lab's where it names none and may ask one: a model-graded metric of its
|
|
3897
|
+
// own, or a case's.
|
|
3649
3898
|
resolve(t, ctx){
|
|
3650
|
-
|
|
3651
|
-
const
|
|
3899
|
+
const ref = (g ) => (isRef(g) ? { id: g.id, name: g.name } : null);
|
|
3900
|
+
const lab = ref(ctx.grader);
|
|
3652
3901
|
if (isObj(t.own)) {
|
|
3653
|
-
return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3902
|
+
return lab && !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3903
|
+
? { ...t, own: { ...t.own, grader: lab } } : t;
|
|
3654
3904
|
}
|
|
3655
|
-
|
|
3905
|
+
if (!isRef(t.group) || isRef(t.grader)) return t;
|
|
3906
|
+
const grader = ref(ctx.groups?.find(g => g.id === (t.group ).id)?.grader) ?? lab;
|
|
3907
|
+
return grader ? { ...t, grader } : t;
|
|
3656
3908
|
},
|
|
3657
3909
|
wholeRun: wholeRunGroup,
|
|
3658
3910
|
verdict: (t, ress, kind) => {
|
|
@@ -3849,6 +4101,9 @@ function metricSummary(m ) {
|
|
|
3849
4101
|
const v = m[o.key];
|
|
3850
4102
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3851
4103
|
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
4104
|
+
// The choices ticked, by their labels, where they are not a new one's.
|
|
4105
|
+
if (o.type === "multi") return !Array.isArray(v) || JSON.stringify(v) === JSON.stringify(fresh[o.key]) ? ""
|
|
4106
|
+
: o.choices.filter(c => v.includes(c.value)).map(c => c.label).join(" or ");
|
|
3852
4107
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3853
4108
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3854
4109
|
}).filter(Boolean);
|
|
@@ -4014,7 +4269,7 @@ STEP_TYPES.echo = {
|
|
|
4014
4269
|
// the flow reads one. No profile, no prompt -- the record is the reply.
|
|
4015
4270
|
STEP_TYPES.recorded = {
|
|
4016
4271
|
label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
|
|
4017
|
-
description: "
|
|
4272
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
4018
4273
|
firstJobOnly: "a call's recorded reply is job 1's",
|
|
4019
4274
|
in: "item", out: "text",
|
|
4020
4275
|
apply: "runPipeline",
|
|
@@ -4086,6 +4341,23 @@ STEP_TYPES.modifier = {
|
|
|
4086
4341
|
},
|
|
4087
4342
|
};
|
|
4088
4343
|
|
|
4344
|
+
// A filters step: the filter sets a job cleans and gates its reply by, each a
|
|
4345
|
+
// link or a private set (#339). It sits with the modifiers (rank 2), so its
|
|
4346
|
+
// sets apply in document order among them. A private set's body is the core's
|
|
4347
|
+
// to judge (filterSetBodyProblems); a link to a Library set is the server's to
|
|
4348
|
+
// resolve, so only its shape is checked here.
|
|
4349
|
+
STEP_TYPES.filters = {
|
|
4350
|
+
label: "Filter sets", slot: "responses", rank: 2, many: true,
|
|
4351
|
+
description: "Cleans and gates the reply with linked or private filter sets.",
|
|
4352
|
+
in: "value", out: "value",
|
|
4353
|
+
apply: "runPipeline",
|
|
4354
|
+
fields: ["type", "filterSets"],
|
|
4355
|
+
validate(step, _ctx, bad, at){
|
|
4356
|
+
if (!Array.isArray(step.filterSets)) return void bad.push(`${at}: filterSets has to be a list`);
|
|
4357
|
+
for (const link of step.filterSets ) bad.push(...filterSetLinkProblems(link, at));
|
|
4358
|
+
},
|
|
4359
|
+
};
|
|
4360
|
+
|
|
4089
4361
|
// ---- reading a job's steps, and a target's -----------------------------------
|
|
4090
4362
|
// Every reader asks these, never a step's position or the document's shape:
|
|
4091
4363
|
// which steps a job holds is its own, and a version that moves a field
|
|
@@ -4281,12 +4553,58 @@ function outOf(job ) {
|
|
|
4281
4553
|
return { ...read, modifiers } ;
|
|
4282
4554
|
}
|
|
4283
4555
|
|
|
4284
|
-
/** [job] reading its reply as [out]: its Format Validation and modifiers.
|
|
4556
|
+
/** [job] reading its reply as [out]: its Format Validation and modifiers.
|
|
4557
|
+
The modifiers replace the job's modifier steps in place, so a modifier step
|
|
4558
|
+
that interleaves with a filters step (#339) keeps its place among them --
|
|
4559
|
+
an edit to one modifier never reorders another step past it. Extra
|
|
4560
|
+
modifiers (one added) come after the last, and a step with no modifier left
|
|
4561
|
+
(one removed) drops out. */
|
|
4285
4562
|
function withOut(job , out ) {
|
|
4286
4563
|
const { modifiers, ...read } = out;
|
|
4287
|
-
const
|
|
4288
|
-
|
|
4289
|
-
|
|
4564
|
+
const mods = modifiers || [];
|
|
4565
|
+
let mi = 0;
|
|
4566
|
+
const steps = [];
|
|
4567
|
+
for (const st of job.steps) {
|
|
4568
|
+
if (st.type === "readReply") continue; // re-added below, in its place
|
|
4569
|
+
if (st.type === "modifier") { if (mi < mods.length) steps.push({ type: "modifier", modifier: mods[mi++] }); }
|
|
4570
|
+
else steps.push(st); // readAs, filters, content -- kept where they are
|
|
4571
|
+
}
|
|
4572
|
+
for (; mi < mods.length; mi++) steps.push({ type: "modifier", modifier: mods[mi] });
|
|
4573
|
+
return { ...job, steps: ordered([...steps, { type: "readReply", out: read }]) };
|
|
4574
|
+
}
|
|
4575
|
+
|
|
4576
|
+
/** The filter-set links a job's Responses stage carries, in document order
|
|
4577
|
+
(#339): read from its one filters step, or none. */
|
|
4578
|
+
function filterSetsOf(job ) {
|
|
4579
|
+
return (stepOf (job, "filters")?.filterSets ?? []).slice();
|
|
4580
|
+
}
|
|
4581
|
+
|
|
4582
|
+
/** [job] with its filter-set links set to [links]: its filters step added,
|
|
4583
|
+
replaced, or taken away when none are left. */
|
|
4584
|
+
function withFilterSets(job , links ) {
|
|
4585
|
+
return withStep(job, "filters", links.length ? { type: "filters", filterSets: [...links] } : null);
|
|
4586
|
+
}
|
|
4587
|
+
|
|
4588
|
+
/** A job's Responses as an ordered modifier list for a run (#339): each
|
|
4589
|
+
modifier step as it is, and each filter-set link as a synthetic modifier
|
|
4590
|
+
that applies the set's filters where the link sits -- so a linked or
|
|
4591
|
+
private set cleans and gates a list reply exactly as the Drop items /
|
|
4592
|
+
Reject modifiers it replaced did, in the same order. [resolve] gives a
|
|
4593
|
+
Library link's body (the worker's kept bodies); a private (`own`) link
|
|
4594
|
+
carries its own, and a Library link with no body resolves to nothing. */
|
|
4595
|
+
function stageModifiers(job , resolve ) {
|
|
4596
|
+
const out = [];
|
|
4597
|
+
for (const st of job?.steps || []) {
|
|
4598
|
+
if (!isObj(st)) continue;
|
|
4599
|
+
if (st.type === "modifier") out.push(st.modifier);
|
|
4600
|
+
else if (st.type === "filters") {
|
|
4601
|
+
for (const link of (st.filterSets ) || []) {
|
|
4602
|
+
const body = link.own ?? (link.set && resolve ? resolve(link.set) : null);
|
|
4603
|
+
out.push({ type: FILTER_MODIFIER, filters: body?.filters || [] });
|
|
4604
|
+
}
|
|
4605
|
+
}
|
|
4606
|
+
}
|
|
4607
|
+
return out;
|
|
4290
4608
|
}
|
|
4291
4609
|
|
|
4292
4610
|
/** A new id: random, so one minted in one browser never collides with one
|
|
@@ -4528,6 +4846,13 @@ function newLink(group , name = "") {
|
|
|
4528
4846
|
return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
|
|
4529
4847
|
}
|
|
4530
4848
|
|
|
4849
|
+
/** A new pipeline link to the Library filter set [set], followed at its newest
|
|
4850
|
+
version. Mirrors newLink; for Runs to use when it links a set in Responses
|
|
4851
|
+
(#339). */
|
|
4852
|
+
function newFilterLink(set ) {
|
|
4853
|
+
return { id: newId(), set: { id: set.id, name: set.name }, pin: null };
|
|
4854
|
+
}
|
|
4855
|
+
|
|
4531
4856
|
function jobsOf(next ) {
|
|
4532
4857
|
next.version = PIPELINE_VERSION;
|
|
4533
4858
|
if (Array.isArray(next.chains)) {
|
|
@@ -4674,16 +4999,62 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
4674
4999
|
}
|
|
4675
5000
|
return localSteps(out, ctx) ;
|
|
4676
5001
|
}
|
|
4677
|
-
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
|
|
5002
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13].includes(doc.version )) return doc;
|
|
5003
|
+
// Version 13 reads as 14 by migrating a job's inline Drop items / Reject
|
|
5004
|
+
// rules into a private filter set (#339); nothing else about it changes.
|
|
5005
|
+
if (doc.version === 13) {
|
|
5006
|
+
return localSteps(filtersFromRules(clone(doc) ), ctx) ;
|
|
5007
|
+
}
|
|
4678
5008
|
if (doc.version === 11 || doc.version === 12) {
|
|
4679
|
-
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
|
|
5009
|
+
return localSteps(filtersFromRules(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))))), ctx) ;
|
|
4680
5010
|
}
|
|
4681
5011
|
// Every version before 10 reads as version 9 first, then as 10, then as 11,
|
|
4682
|
-
// 12 and
|
|
5012
|
+
// 12, 13 and 14.
|
|
4683
5013
|
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4684
5014
|
// earlier ones are cut on their way through nineOf.
|
|
4685
5015
|
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4686
|
-
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
|
|
5016
|
+
return localSteps(filtersFromRules(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten))))), ctx) ;
|
|
5017
|
+
}
|
|
5018
|
+
|
|
5019
|
+
/** [doc] with each job's inline Drop items / Reject modifier steps migrated
|
|
5020
|
+
into one private filter set on a `filters` step, version 13 to 14 (#339).
|
|
5021
|
+
Every drop/reject modifier's rules become filters in document order -- a
|
|
5022
|
+
`drop` modifier's rules drop-item filters, a `reject` modifier's
|
|
5023
|
+
reject-reply filters (filterFromItemRule, which carries the matcher
|
|
5024
|
+
unchanged) -- in one `own` set, and the step takes the place of the first
|
|
5025
|
+
of those modifiers, so the reply is cleaned and gated as it was. A job with
|
|
5026
|
+
no such modifier, or only empty ones, is left as it is; this is a no-op on
|
|
5027
|
+
every path that reaches version 14, so it is safe to run on all of them. */
|
|
5028
|
+
function filtersFromRules(next ) {
|
|
5029
|
+
next.version = PIPELINE_VERSION;
|
|
5030
|
+
if (!Array.isArray(next.jobs)) return next;
|
|
5031
|
+
const TYPE = { drop: "drop-item", reject: "reject-reply" };
|
|
5032
|
+
const isRule = (st ) =>
|
|
5033
|
+
isObj(st) && st.type === "modifier" && isObj(st.modifier) && (st.modifier.type ) in TYPE;
|
|
5034
|
+
next.jobs = next.jobs.map((job ) => {
|
|
5035
|
+
if (!isObj(job) || !Array.isArray(job.steps) || !job.steps.some(isRule)) return job;
|
|
5036
|
+
const filters = [];
|
|
5037
|
+
for (const st of job.steps ) {
|
|
5038
|
+
if (!isRule(st)) continue;
|
|
5039
|
+
const m = (st ).modifier ;
|
|
5040
|
+
for (const r of Array.isArray(m.rules) ? m.rules : []) {
|
|
5041
|
+
filters.push(filterFromItemRule(r, TYPE[m.type ] ));
|
|
5042
|
+
}
|
|
5043
|
+
}
|
|
5044
|
+
// Empty Drop items / Reject modifiers held nothing, so they simply drop.
|
|
5045
|
+
// The link's id is the job's, so re-upgrading a document mints the same
|
|
5046
|
+
// one (the migration is idempotent) and a run keyed by it reads the same.
|
|
5047
|
+
const link = filters.length
|
|
5048
|
+
? { id: `filters-${isStr(job.id) ? job.id : "job"}`, set: null, pin: null, own: { version: FILTER_SET_VERSION, filters } } : null;
|
|
5049
|
+
let placed = false;
|
|
5050
|
+
const steps = [];
|
|
5051
|
+
for (const st of job.steps ) {
|
|
5052
|
+
if (!isRule(st)) { steps.push(st); continue; }
|
|
5053
|
+
if (link && !placed) { steps.push({ type: "filters", filterSets: [link] }); placed = true; }
|
|
5054
|
+
}
|
|
5055
|
+
return { ...job, steps };
|
|
5056
|
+
});
|
|
5057
|
+
return next;
|
|
4687
5058
|
}
|
|
4688
5059
|
|
|
4689
5060
|
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
@@ -5342,7 +5713,8 @@ function exportBundle(doc , ctx
|
|
|
5342
5713
|
// What the lab would supply at submit, written down: no lab supplies it later.
|
|
5343
5714
|
const out = clone(doc) ;
|
|
5344
5715
|
out.evals = (Array.isArray(out.evals) ? out.evals : [])
|
|
5345
|
-
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null
|
|
5716
|
+
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null,
|
|
5717
|
+
groups: Object.entries(ctx.datasets ?? {}).map(([id, d]) => ({ id, name: d.name, grader: d.body.grader ?? null })) }) ?? t);
|
|
5346
5718
|
const table = {};
|
|
5347
5719
|
for (const id of profileIds(out)) {
|
|
5348
5720
|
const p = profiles.find(x => x.id === id);
|
|
@@ -5487,13 +5859,16 @@ function readBundle(files , ctx = {})
|
|
|
5487
5859
|
* the job it sits in, and the connection it asks -- its own profile, or
|
|
5488
5860
|
* its scenario's.
|
|
5489
5861
|
*/
|
|
5490
|
-
function stagesFor(run , i )
|
|
5862
|
+
function stagesFor(run , i , resolveFilters )
|
|
5491
5863
|
|
|
5492
5864
|
{
|
|
5493
5865
|
const stages = run.jobs.map((ch, k) => {
|
|
5494
|
-
const { kind, modifiers, ...settings } = outOf(ch);
|
|
5866
|
+
const { kind, modifiers: _mods, ...settings } = outOf(ch);
|
|
5867
|
+
// The modifiers a run applies, with each linked or private filter set a
|
|
5868
|
+
// synthetic modifier where its link sits (#339); without a resolver, a
|
|
5869
|
+
// Library link reads nothing, a private one its own body.
|
|
5495
5870
|
return { text: targetStepOf(run, i, k) .prompt, kind, withImage: sendsImage(ch),
|
|
5496
|
-
verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers:
|
|
5871
|
+
verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: stageModifiers(ch, resolveFilters), settings };
|
|
5497
5872
|
});
|
|
5498
5873
|
const connections = run.jobs.map((_, k) => {
|
|
5499
5874
|
// A step answered here asks nothing: the profile a run made before it
|
|
@@ -5883,16 +6258,20 @@ export {
|
|
|
5883
6258
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
5884
6259
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
5885
6260
|
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
|
|
5886
|
-
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
|
|
5887
|
-
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
6261
|
+
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf, WIZARDS, wizardProgress,
|
|
6262
|
+
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, addressOf, connectionSettings,
|
|
5888
6263
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5889
6264
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5890
6265
|
contentOf, withContent, replyOf,
|
|
5891
6266
|
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
|
|
5892
6267
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
6268
|
+
filterSetsOf, withFilterSets, stageModifiers,
|
|
5893
6269
|
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5894
6270
|
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|
|
5895
6271
|
evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
|
|
5896
6272
|
pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
|
|
5897
6273
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
6274
|
+
FILTER_SET_VERSION, FILTER_TYPES, FILTER_SEPARATORS, applyFilter, applyFilterSet, tryFilter, splitSample,
|
|
6275
|
+
filterProblems, filterFromItemRule, upgradeFilterSetBody, blankFilterSetBody, filterSetBodyProblems, filterSetLinkProblems,
|
|
6276
|
+
newFilterLink, FILTER_MODIFIER,
|
|
5898
6277
|
};
|