evals-lab 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,12 +33,14 @@
33
33
  // and that is a measurement.
34
34
 
35
35
  import yaml from "./js-yaml.mjs";
36
-
36
+
37
37
  import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
38
38
  import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
39
39
  import { flowApi, } from "./flows/flowApi.mjs";
40
40
  import registerMetrics from "./metrics/builtin.mjs";
41
- import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
41
+ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
42
+ FILTER_TYPES, FILTER_SEPARATORS, FILTER_MODIFIER, applyFilter, applyFilterSet, tryFilter, splitSample, filterProblems, filterFromItemRule,
43
+ } from "./kinds/list.mjs";
42
44
 
43
45
  // ---- The types -------------------------------------------------------------
44
46
  // docs/pipeline-model.md as types, and the shapes of the registries every
@@ -138,9 +140,19 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
138
140
 
139
141
 
140
142
 
143
+ /** The filter sets a job cleans and gates its reply by (#339): each a link to
144
+ a Library filter set (followed latest or pinned) or a private one held
145
+ inline. A Responses step beside the modifiers, at the same place, so its
146
+ sets' filters apply in document order with them -- what the Drop items and
147
+ Reject the answer modifiers did inline, lifted into linked sets. */
148
+
149
+
150
+
151
+
152
+
141
153
  /** A job's steps: its Content stage, then its Responses (§16). */
142
154
 
143
-
155
+
144
156
 
145
157
  /** A target step that prompts a model: its words, resolved under the job's
146
158
  token mappings, and a profile of its own where it names one. */
@@ -316,6 +328,45 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
316
328
  /** A pipeline's evals, in the order they read a run. */
317
329
 
318
330
 
331
+ /**
332
+ * A filter set (issue #335): a versioned Library body holding the filters
333
+ * that shape a list reply, mirroring DatasetBody / the eval group. The lab
334
+ * will keep each as a row and a run grade against the body kept with it, as a
335
+ * dataset does (#336); for now it is the format a filter set reads and writes
336
+ * at. `filters` are tried in order -- each drop-item narrows the items the
337
+ * next reads, and the first reject-reply that fires gates the answer.
338
+ */
339
+
340
+
341
+
342
+
343
+
344
+ /** A Library filter set by reference; a run's copy says the version it used:
345
+ the body's fingerprint, and its number. Mirrors GroupRef. */
346
+
347
+
348
+ /** A private filter set: a filter-set body held in the pipeline rather than
349
+ the Library -- what Runs makes when a link is not to a Library set (#339),
350
+ as a private eval group (OwnGroup) is. A filter set has no cases, so this
351
+ is the body itself. */
352
+
353
+
354
+ /**
355
+ * A pipeline's link to a filter set in Responses (issue #335, mirrors
356
+ * GroupEval / OwnGroup): `set` names a Library set, followed at its newest
357
+ * version (`pin: null`) or pinned at one (`pin: n`); a private set is `own`,
358
+ * with `set: null`. From version 14 a job's Responses stage carries these on
359
+ * a `filters` step (#339); the upgrade from 13 migrates the inline Drop items
360
+ * / Reject rules into a private one, and the worker cleans and gates a run's
361
+ * replies against the kept bodies.
362
+ */
363
+
364
+
365
+
366
+
367
+
368
+
369
+
319
370
 
320
371
 
321
372
 
@@ -657,6 +708,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
657
708
 
658
709
 
659
710
 
711
+
712
+
713
+
660
714
 
661
715
 
662
716
 
@@ -667,9 +721,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
667
721
 
668
722
  /** What a metric needs besides the reply: a model to grade with. */
669
723
 
670
-
671
-
672
-
724
+
725
+
726
+
727
+
673
728
 
674
729
 
675
730
 
@@ -723,6 +778,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
723
778
 
724
779
 
725
780
 
781
+
782
+
783
+
784
+
785
+
786
+
787
+
726
788
 
727
789
 
728
790
  /** What an earlier eval that failed and does not continue leaves a later one. */
@@ -797,6 +859,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
797
859
 
798
860
 
799
861
 
862
+
863
+
864
+
800
865
 
801
866
 
802
867
 
@@ -860,6 +925,11 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
860
925
 
861
926
 
862
927
 
928
+
929
+
930
+
931
+
932
+
863
933
 
864
934
 
865
935
 
@@ -881,9 +951,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
881
951
 
882
952
 
883
953
 
954
+
884
955
 
885
-
886
-
956
+
887
957
 
888
958
 
889
959
  /** Lookups, and whether it is a run document being validated. */
@@ -1041,7 +1111,7 @@ const platformOf = (src
1041
1111
 
1042
1112
 
1043
1113
 
1044
-
1114
+
1045
1115
 
1046
1116
 
1047
1117
 
@@ -1219,6 +1289,57 @@ function upgradeDatasetBody (body ) {
1219
1289
  return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
1220
1290
  }
1221
1291
 
1292
+ /** The filter-set body's version (issue #335). Version 1 is the first: the
1293
+ `{ version, filters }` body #334 introduces. A later format adds a step to
1294
+ `upgradeFilterSetBody`, as `upgradeDatasetBody` grew, with server.py's
1295
+ `upgrade_filter_set_body` its twin. */
1296
+ const FILTER_SET_VERSION = 1 ;
1297
+
1298
+ /** [body] as this version of a filter set (1), from any earlier one. There is
1299
+ no earlier version yet, so a version-1 body reads as it is and anything
1300
+ else comes back untouched for `filterSetBodyProblems` to refuse. Pure: the
1301
+ same body gives the same answer, and server.py's upgrade_filter_set_body is
1302
+ its twin. */
1303
+ function upgradeFilterSetBody (body ) {
1304
+ if (!isObj(body)) return body;
1305
+ if (body.version === FILTER_SET_VERSION) return body;
1306
+ return body;
1307
+ }
1308
+
1309
+ /** A filter set with no filters, at this version: what New filter set starts
1310
+ from, and what a lab with none holds. */
1311
+ function blankFilterSetBody() {
1312
+ return { version: FILTER_SET_VERSION, filters: [] };
1313
+ }
1314
+
1315
+ /** Why [body] is not a filter set the lab reads, one sentence each: an object
1316
+ at this version whose filters are a valid filter list (filterProblems). For
1317
+ the store (#336) and the pipeline link (#339). */
1318
+ function filterSetBodyProblems(body , at = "the filter set") {
1319
+ if (!isObj(body)) return [`${at} has to be a JSON object`];
1320
+ if (body.version !== FILTER_SET_VERSION) {
1321
+ return [`${at} is version ${JSON.stringify(body.version)}, and the lab reads version ${FILTER_SET_VERSION}`];
1322
+ }
1323
+ return filterProblems(body.filters, at);
1324
+ }
1325
+
1326
+ /** Why [link] is not a filter-set link a run's Responses step may carry, one
1327
+ sentence each (#339): an id, exactly one of a Library `set` reference or a
1328
+ private `own` body (whose shape is the core's to judge), and a `pin` that
1329
+ is null or a whole number. A Library set is the server's to resolve, so its
1330
+ body is not checked here. */
1331
+ function filterSetLinkProblems(link , at = "a filter set") {
1332
+ if (!isObj(link)) return [`${at}: a filter-set link has to be a JSON object`];
1333
+ const bad = [];
1334
+ if (!isStr(link.id) || !link.id.trim()) bad.push(`${at}: a filter-set link has no id`);
1335
+ const hasSet = isObj(link.set) && isStr(link.set.id);
1336
+ const hasOwn = link.own != null;
1337
+ if (hasSet === hasOwn) bad.push(`${at}: a filter-set link names a Library set or holds a private one, not both`);
1338
+ if (hasOwn) bad.push(...filterSetBodyProblems(link.own, `${at}: the private filter set`));
1339
+ if (link.pin != null && !Number.isInteger(link.pin)) bad.push(`${at}: a pin is a whole number or none`);
1340
+ return bad;
1341
+ }
1342
+
1222
1343
  /** A version-6 body as a version-8 eval group: scored All, the lab's
1223
1344
  grader, and no metrics of its own for every item or the whole run --
1224
1345
  what a Metrics eval naming the dataset with none of its own graded, which
@@ -1323,6 +1444,10 @@ function datasetRules(body ) {
1323
1444
 
1324
1445
 
1325
1446
 
1447
+
1448
+
1449
+
1450
+
1326
1451
 
1327
1452
 
1328
1453
  /** An HTTP request as a connection type builds it, for the relay to send. */
@@ -1346,6 +1471,10 @@ function datasetRules(body ) {
1346
1471
 
1347
1472
 
1348
1473
 
1474
+
1475
+
1476
+
1477
+
1349
1478
 
1350
1479
 
1351
1480
 
@@ -1354,8 +1483,9 @@ function datasetRules(body ) {
1354
1483
 
1355
1484
 
1356
1485
 
1357
-
1358
-
1486
+
1487
+
1488
+
1359
1489
 
1360
1490
 
1361
1491
 
@@ -1708,6 +1838,8 @@ function ollamaType(id , label , cloud ) {
1708
1838
  body: { model: conn.model, messages: [message], options, stream: false },
1709
1839
  };
1710
1840
  },
1841
+ // Ollama's own: the schema is the format.
1842
+ structured: (body, schema) => ({ ...body, format: schema }),
1711
1843
  parseReply(j) {
1712
1844
  const r = j ;
1713
1845
  return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
@@ -1754,7 +1886,7 @@ function recordedRaw(record ) {
1754
1886
  // the flow reads a reply; `local` is the no-Read-as fallback.
1755
1887
  CONNECTION_TYPES.recorded = {
1756
1888
  id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
1757
- description: "Replays the reply production recorded for each call. Sends nothing.",
1889
+ description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
1758
1890
  local: (_item, _sent, record) => recordedRaw(record),
1759
1891
  request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1760
1892
  };
@@ -1769,19 +1901,25 @@ function localAnswer(conn , text , sent , record
1769
1901
  return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
1770
1902
  }
1771
1903
 
1904
+ /** A chat completions body asking for a reply held to [schema]: OpenAI's
1905
+ response_format, which llama-server and most compatible servers take too. */
1906
+ const chatSchema = (body , schema ) =>
1907
+ ({ ...body, response_format: { type: "json_schema", json_schema: { name: "reply", strict: true, schema } } });
1908
+
1772
1909
  CONNECTION_TYPES["openai-compatible"] = {
1773
1910
  id: "openai-compatible", label: "OpenAI-compatible",
1774
1911
  description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
1775
1912
  settings: [
1776
- { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1913
+ // A reasoning model refuses a temperature.
1914
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
1915
+ appliesTo: (conn) => !/^(gpt-5|o\d)/i.test(String(conn.model ?? "")) },
1777
1916
  { key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
1778
1917
  ],
1779
- // /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio. The
1780
- // reasoning-model temperature rule and max_completion_tokens stay with this
1781
- // type, as the hosted providers reject the fields a llama.cpp server takes.
1918
+ // /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
1919
+ // max_completion_tokens stays with this type, as the hosted providers
1920
+ // reject the fields a llama.cpp server takes.
1782
1921
  request(conn, prompt, dataUrl, base, key) {
1783
1922
  const hosted = isHostedUrl(base);
1784
- const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
1785
1923
  const temp = asNumber(conn.temperature);
1786
1924
  const predict = asNumber(conn.nPredict);
1787
1925
  return {
@@ -1789,7 +1927,7 @@ CONNECTION_TYPES["openai-compatible"] = {
1789
1927
  headers: bearer(key),
1790
1928
  body: {
1791
1929
  model: conn.model,
1792
- ...(isReasoning || temp == null ? {} : { temperature: temp }),
1930
+ ...(temp == null ? {} : { temperature: temp }),
1793
1931
  ...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
1794
1932
  messages: [{ role: "user", content: [
1795
1933
  { type: "text", text: prompt },
@@ -1798,15 +1936,23 @@ CONNECTION_TYPES["openai-compatible"] = {
1798
1936
  },
1799
1937
  };
1800
1938
  },
1939
+ structured: chatSchema,
1801
1940
  parseReply: chatReply,
1802
1941
  listModels, parseModels: idsFrom,
1803
1942
  };
1804
1943
 
1944
+ /** The Claude models that refuse a sampling setting: Opus from 4.7, and
1945
+ every Opus, Sonnet, Fable and Mythos from 5. Older ones (Opus 4.6, Sonnet
1946
+ 4.6, Haiku 4.5) take one. */
1947
+ const CLAUDE_FIXED_SAMPLING = /claude-(opus-4-[7-9]|(opus|sonnet|fable|mythos)-[5-9])/i;
1948
+
1805
1949
  CONNECTION_TYPES.anthropic = {
1806
1950
  id: "anthropic", label: "Anthropic",
1807
1951
  description: "Claude, through Anthropic's Messages API.",
1952
+ url: "https://api.anthropic.com",
1808
1953
  settings: [
1809
- { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1954
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
1955
+ appliesTo: (conn) => !CLAUDE_FIXED_SAMPLING.test(String(conn.model ?? "")) },
1810
1956
  { key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
1811
1957
  ],
1812
1958
  // The Messages API: the x-api-key and anthropic-version headers, base64
@@ -1831,6 +1977,9 @@ CONNECTION_TYPES.anthropic = {
1831
1977
  },
1832
1978
  };
1833
1979
  },
1980
+ // The Messages API's structured outputs. A model that cannot hold one
1981
+ // refuses the request, and the grader asks again without it.
1982
+ structured: (body, schema) => ({ ...body, output_config: { ...(body.output_config ), format: { type: "json_schema", schema } } }),
1834
1983
  parseReply(j) {
1835
1984
  const r = j ;
1836
1985
  const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
@@ -1903,6 +2052,8 @@ CONNECTION_TYPES["llama.cpp"] = {
1903
2052
  },
1904
2053
  };
1905
2054
  },
2055
+ // llama-server takes OpenAI's response_format, a schema and all.
2056
+ structured: chatSchema,
1906
2057
  parseReply: chatReply,
1907
2058
  listModels, parseModels: idsFrom,
1908
2059
  // A reply that arrived is still a failure when this profile asked for the
@@ -2098,12 +2249,14 @@ function extraHeaders(raw ) {
2098
2249
  }
2099
2250
  return out;
2100
2251
  }
2101
- /** An address as a whole-request type hangs paths off it: its origin, no /v1. */
2252
+ /** An address as a whole-request type hangs paths off it: its origin, no
2253
+ path. A step's path is the whole path production called, so an address
2254
+ entered with one (`…/v1/messages`) would send it twice. */
2102
2255
  function originBase(raw ) {
2103
- let s = String(raw || "").trim().replace(/\/+$/, "");
2256
+ let s = String(raw || "").trim();
2104
2257
  if (!s) return "";
2105
2258
  if (!/^https?:\/\//.test(s)) s = "https://" + s;
2106
- return s;
2259
+ try { return new URL(s).origin; } catch { return s.replace(/\/+$/, ""); }
2107
2260
  }
2108
2261
  const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
2109
2262
  CONNECTION_TYPES.http = {
@@ -2230,9 +2383,14 @@ function splitOllama(p ) {
2230
2383
  * connection carries its type's settings under `options`, so they are merged
2231
2384
  * up before the type reads them.
2232
2385
  */
2233
- function connectionRequest(conn , prompt , dataUrl , base , key ) {
2386
+ function connectionRequest(conn , prompt , dataUrl , base , key ,
2387
+ schema ) {
2234
2388
  const flat = { ...conn, ...(conn?.options || {}) };
2235
- return typeEntry(flat).request(flat, prompt, dataUrl, base, key);
2389
+ const type = typeEntry(flat);
2390
+ // A setting the model refuses is not sent, whatever the profile holds.
2391
+ for (const s of type.settings) if (s.appliesTo && !s.appliesTo(flat)) delete flat[s.key];
2392
+ const req = type.request(flat, prompt, dataUrl, base, key);
2393
+ return schema && type.structured ? { ...req, body: type.structured(req.body , schema) } : req;
2236
2394
  }
2237
2395
 
2238
2396
  /** Where [conn]'s requests hang off: its type's own reading of its address,
@@ -2885,7 +3043,7 @@ function applyModifiers (list , kind ,
2885
3043
  // 12: a Contains metric's Ignore case holds item by item too.
2886
3044
  // 13: an eval is a link to an eval group, or a group of the pipeline's own,
2887
3045
  // and the document has an overall pass rule (pipeline-model §17).
2888
- const PIPELINE_VERSION = 13 ;
3046
+ const PIPELINE_VERSION = 14 ;
2889
3047
 
2890
3048
  // Plain objects, so an entry is added by assignment and a reader never needs
2891
3049
  // to know which registered it.
@@ -2934,6 +3092,15 @@ function registerKinds(k ) {
2934
3092
 
2935
3093
 
2936
3094
 
3095
+
3096
+
3097
+
3098
+  
3099
+
3100
+
3101
+
3102
+
3103
+
2937
3104
 
2938
3105
 
2939
3106
  function pluginHost(pluginId ) {
@@ -2961,6 +3128,14 @@ function pluginHost(pluginId ) {
2961
3128
  taken(CONNECTION_TYPES, "connection type", [t.id]);
2962
3129
  CONNECTION_TYPES[t.id] = t;
2963
3130
  },
3131
+ registerWorkflowPlatform(t) {
3132
+ taken(WORKFLOW_PLATFORMS, "workflow platform", [t.id]);
3133
+ WORKFLOW_PLATFORMS[t.id] = t;
3134
+ },
3135
+ registerWizard(w) {
3136
+ taken(WIZARDS, "wizard", [w.id]);
3137
+ WIZARDS[w.id] = w;
3138
+ },
2964
3139
  };
2965
3140
  }
2966
3141
 
@@ -3058,6 +3233,13 @@ function withSlugs (list
3058
3233
  /** Whether [s] is a slug as a profile may hold one. */
3059
3234
  const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
3060
3235
 
3236
+ /** Where [p] connects: its address, or blank, its type's own -- an Anthropic
3237
+ profile left blank reaches Anthropic. Blank for a type with none means the
3238
+ lab's own Ollama, which the runner and the relay fill in. */
3239
+ function addressOf(p ) {
3240
+ return String(p.url ?? "").trim() || CONNECTION_TYPES[typeOf(p )]?.url || "";
3241
+ }
3242
+
3061
3243
  /** A stored profile as a run carries it: its request settings, no key. */
3062
3244
  function connectionOf(p ) {
3063
3245
  const type = typeOf(p);
@@ -3066,7 +3248,7 @@ function connectionOf(p ) {
3066
3248
  // temperature is a common field, carried at the top like px/format/quality;
3067
3249
  // the options bag holds the type's own settings only.
3068
3250
  for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
3069
- const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
3251
+ const out = { name: p.name || "unnamed", url: addressOf(p), model: p.model || "", type };
3070
3252
  if (isSlug(p.slug)) out.slug = p.slug;
3071
3253
  for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
3072
3254
  if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
@@ -3187,6 +3369,70 @@ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapt
3187
3369
  render: body => JSON.stringify(body, null, 2),
3188
3370
  };
3189
3371
 
3372
+ // ---- Setup Wizards (docs/workflow-sources.md § Wizards) ---------------------
3373
+ //
3374
+ // A wizard is a registry entry whose steps each say where they happen and a
3375
+ // done-when predicate on lab state. Steps tick automatically: `done(lab)` reads
3376
+ // the cheap state the page already holds (a workflow Source exists, a pipeline
3377
+ // with a Recorded target exists, a run of it is in History), never a poll. One
3378
+ // wizard runs at a time, held in the browser-only `promptlab.wizard` store; a
3379
+ // header pill shows its progress and the Setup › Wizards section draws the same
3380
+ // walk-through. The registry lives here so a plugin registers a wizard through
3381
+ // the same host as every other kind (PluginHost.registerWizard, phase 7); the
3382
+ // page ships the built-in "Test a workflow" and owns its routes and selectors
3383
+ // (web/src/app/wizards.ts). The predicates run with the page's privileges and
3384
+ // only read lab state, so a plugin wizard is no new trust (docs/packs.md).
3385
+
3386
+ /** The slice of lab state a wizard step's `done` reads: what the page already
3387
+ holds, so a tick costs a predicate and not a request. A reader names a field
3388
+ it needs; a wizard that reads more declares it here. */
3389
+
3390
+
3391
+
3392
+
3393
+
3394
+
3395
+
3396
+
3397
+
3398
+
3399
+
3400
+
3401
+
3402
+
3403
+
3404
+
3405
+
3406
+
3407
+
3408
+
3409
+
3410
+
3411
+
3412
+
3413
+
3414
+
3415
+
3416
+
3417
+
3418
+
3419
+
3420
+
3421
+
3422
+
3423
+ const WIZARDS = Object.create(null);
3424
+
3425
+ /** How far a wizard has got on [lab]: each step's tick, how many are done, and
3426
+ the first step not yet done -- the one the checklist marks current and Go
3427
+ to step heads for. `current` is -1 once every step is done. */
3428
+ function wizardProgress(entry , lab )
3429
+
3430
+ {
3431
+ const ticks = entry.steps.map((s) => s.done(lab));
3432
+ const done = ticks.filter(Boolean).length;
3433
+ return { ticks, done, total: entry.steps.length, current: ticks.findIndex((t) => !t) };
3434
+ }
3435
+
3190
3436
  CONTENT_TYPES.source = {
3191
3437
  label: "Source",
3192
3438
  description: "Every file or record in a Source from the Library.",
@@ -3364,6 +3610,7 @@ function metricInput(res , kase , more )
3364
3610
  const last = res.transcript?.at(-1);
3365
3611
  const text = last?.got ?? res.raw ?? "";
3366
3612
  return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
3613
+ asked: last?.sent ?? null,
3367
3614
  // A case that has been given its own expected reply ("Take B as
3368
3615
  // expected") is compared with that; otherwise the Source's recorded
3369
3616
  // reply for the item.
@@ -3382,7 +3629,7 @@ function runInput(ress , plain )
3382
3629
  terms.push(...(r.terms || []));
3383
3630
  ms += r.ms ?? 0;
3384
3631
  }
3385
- return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
3632
+ return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, asked: null, kase: null };
3386
3633
  }
3387
3634
 
3388
3635
  /** One metric's reading of [input]: `of`, `not` and a failure all applied,
@@ -3643,16 +3890,21 @@ EVAL_TYPES.group = {
3643
3890
  want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3644
3891
  .filter(Boolean).join(" ") })),
3645
3892
  profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
3646
- // A run carries the lab's grader where the group names none and may ask
3647
- // one: a model-graded metric of its own, or a case's. A link's group is
3648
- // the Library's, so the run's copy of the link carries it.
3893
+ // A run carries the grader each group asks, so its profile goes with the
3894
+ // run: a link's is the Library group's own, or the lab's where it names
3895
+ // none -- the run's copy of the link carries it. A private group takes the
3896
+ // lab's where it names none and may ask one: a model-graded metric of its
3897
+ // own, or a case's.
3649
3898
  resolve(t, ctx){
3650
- if (!isRef(ctx.grader)) return t;
3651
- const grader = { id: ctx.grader.id, name: ctx.grader.name };
3899
+ const ref = (g ) => (isRef(g) ? { id: g.id, name: g.name } : null);
3900
+ const lab = ref(ctx.grader);
3652
3901
  if (isObj(t.own)) {
3653
- return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
3902
+ return lab && !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
3903
+ ? { ...t, own: { ...t.own, grader: lab } } : t;
3654
3904
  }
3655
- return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
3905
+ if (!isRef(t.group) || isRef(t.grader)) return t;
3906
+ const grader = ref(ctx.groups?.find(g => g.id === (t.group ).id)?.grader) ?? lab;
3907
+ return grader ? { ...t, grader } : t;
3656
3908
  },
3657
3909
  wholeRun: wholeRunGroup,
3658
3910
  verdict: (t, ress, kind) => {
@@ -3849,6 +4101,9 @@ function metricSummary(m ) {
3849
4101
  const v = m[o.key];
3850
4102
  if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3851
4103
  if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
4104
+ // The choices ticked, by their labels, where they are not a new one's.
4105
+ if (o.type === "multi") return !Array.isArray(v) || JSON.stringify(v) === JSON.stringify(fresh[o.key]) ? ""
4106
+ : o.choices.filter(c => v.includes(c.value)).map(c => c.label).join(" or ");
3852
4107
  // Text that runs to lines (a schema) reads as its first words, run together.
3853
4108
  return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3854
4109
  }).filter(Boolean);
@@ -4014,7 +4269,7 @@ STEP_TYPES.echo = {
4014
4269
  // the flow reads one. No profile, no prompt -- the record is the reply.
4015
4270
  STEP_TYPES.recorded = {
4016
4271
  label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
4017
- description: "Replays the reply production recorded for each call. Sends nothing.",
4272
+ description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
4018
4273
  firstJobOnly: "a call's recorded reply is job 1's",
4019
4274
  in: "item", out: "text",
4020
4275
  apply: "runPipeline",
@@ -4086,6 +4341,23 @@ STEP_TYPES.modifier = {
4086
4341
  },
4087
4342
  };
4088
4343
 
4344
+ // A filters step: the filter sets a job cleans and gates its reply by, each a
4345
+ // link or a private set (#339). It sits with the modifiers (rank 2), so its
4346
+ // sets apply in document order among them. A private set's body is the core's
4347
+ // to judge (filterSetBodyProblems); a link to a Library set is the server's to
4348
+ // resolve, so only its shape is checked here.
4349
+ STEP_TYPES.filters = {
4350
+ label: "Filter sets", slot: "responses", rank: 2, many: true,
4351
+ description: "Cleans and gates the reply with linked or private filter sets.",
4352
+ in: "value", out: "value",
4353
+ apply: "runPipeline",
4354
+ fields: ["type", "filterSets"],
4355
+ validate(step, _ctx, bad, at){
4356
+ if (!Array.isArray(step.filterSets)) return void bad.push(`${at}: filterSets has to be a list`);
4357
+ for (const link of step.filterSets ) bad.push(...filterSetLinkProblems(link, at));
4358
+ },
4359
+ };
4360
+
4089
4361
  // ---- reading a job's steps, and a target's -----------------------------------
4090
4362
  // Every reader asks these, never a step's position or the document's shape:
4091
4363
  // which steps a job holds is its own, and a version that moves a field
@@ -4281,12 +4553,58 @@ function outOf(job ) {
4281
4553
  return { ...read, modifiers } ;
4282
4554
  }
4283
4555
 
4284
- /** [job] reading its reply as [out]: its Format Validation and modifiers. */
4556
+ /** [job] reading its reply as [out]: its Format Validation and modifiers.
4557
+ The modifiers replace the job's modifier steps in place, so a modifier step
4558
+ that interleaves with a filters step (#339) keeps its place among them --
4559
+ an edit to one modifier never reorders another step past it. Extra
4560
+ modifiers (one added) come after the last, and a step with no modifier left
4561
+ (one removed) drops out. */
4285
4562
  function withOut(job , out ) {
4286
4563
  const { modifiers, ...read } = out;
4287
- const steps = job.steps.filter(st => st.type !== "readReply" && st.type !== "modifier");
4288
- return { ...job, steps: ordered([...steps, { type: "readReply", out: read },
4289
- ...(modifiers || []).map(modifier => ({ type: "modifier", modifier }) )]) };
4564
+ const mods = modifiers || [];
4565
+ let mi = 0;
4566
+ const steps = [];
4567
+ for (const st of job.steps) {
4568
+ if (st.type === "readReply") continue; // re-added below, in its place
4569
+ if (st.type === "modifier") { if (mi < mods.length) steps.push({ type: "modifier", modifier: mods[mi++] }); }
4570
+ else steps.push(st); // readAs, filters, content -- kept where they are
4571
+ }
4572
+ for (; mi < mods.length; mi++) steps.push({ type: "modifier", modifier: mods[mi] });
4573
+ return { ...job, steps: ordered([...steps, { type: "readReply", out: read }]) };
4574
+ }
4575
+
4576
+ /** The filter-set links a job's Responses stage carries, in document order
4577
+ (#339): read from its one filters step, or none. */
4578
+ function filterSetsOf(job ) {
4579
+ return (stepOf (job, "filters")?.filterSets ?? []).slice();
4580
+ }
4581
+
4582
+ /** [job] with its filter-set links set to [links]: its filters step added,
4583
+ replaced, or taken away when none are left. */
4584
+ function withFilterSets(job , links ) {
4585
+ return withStep(job, "filters", links.length ? { type: "filters", filterSets: [...links] } : null);
4586
+ }
4587
+
4588
+ /** A job's Responses as an ordered modifier list for a run (#339): each
4589
+ modifier step as it is, and each filter-set link as a synthetic modifier
4590
+ that applies the set's filters where the link sits -- so a linked or
4591
+ private set cleans and gates a list reply exactly as the Drop items /
4592
+ Reject modifiers it replaced did, in the same order. [resolve] gives a
4593
+ Library link's body (the worker's kept bodies); a private (`own`) link
4594
+ carries its own, and a Library link with no body resolves to nothing. */
4595
+ function stageModifiers(job , resolve ) {
4596
+ const out = [];
4597
+ for (const st of job?.steps || []) {
4598
+ if (!isObj(st)) continue;
4599
+ if (st.type === "modifier") out.push(st.modifier);
4600
+ else if (st.type === "filters") {
4601
+ for (const link of (st.filterSets ) || []) {
4602
+ const body = link.own ?? (link.set && resolve ? resolve(link.set) : null);
4603
+ out.push({ type: FILTER_MODIFIER, filters: body?.filters || [] });
4604
+ }
4605
+ }
4606
+ }
4607
+ return out;
4290
4608
  }
4291
4609
 
4292
4610
  /** A new id: random, so one minted in one browser never collides with one
@@ -4528,6 +4846,13 @@ function newLink(group , name = "") {
4528
4846
  return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
4529
4847
  }
4530
4848
 
4849
+ /** A new pipeline link to the Library filter set [set], followed at its newest
4850
+ version. Mirrors newLink; for Runs to use when it links a set in Responses
4851
+ (#339). */
4852
+ function newFilterLink(set ) {
4853
+ return { id: newId(), set: { id: set.id, name: set.name }, pin: null };
4854
+ }
4855
+
4531
4856
  function jobsOf(next ) {
4532
4857
  next.version = PIPELINE_VERSION;
4533
4858
  if (Array.isArray(next.chains)) {
@@ -4674,16 +4999,62 @@ function upgradePipeline (doc , ctx = {}) {
4674
4999
  }
4675
5000
  return localSteps(out, ctx) ;
4676
5001
  }
4677
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
5002
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13].includes(doc.version )) return doc;
5003
+ // Version 13 reads as 14 by migrating a job's inline Drop items / Reject
5004
+ // rules into a private filter set (#339); nothing else about it changes.
5005
+ if (doc.version === 13) {
5006
+ return localSteps(filtersFromRules(clone(doc) ), ctx) ;
5007
+ }
4678
5008
  if (doc.version === 11 || doc.version === 12) {
4679
- return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
5009
+ return localSteps(filtersFromRules(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))))), ctx) ;
4680
5010
  }
4681
5011
  // Every version before 10 reads as version 9 first, then as 10, then as 11,
4682
- // 12 and 13.
5012
+ // 12, 13 and 14.
4683
5013
  // A version-10 document is cut as a current one was (profileRefs); the
4684
5014
  // earlier ones are cut on their way through nineOf.
4685
5015
  const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4686
- return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
5016
+ return localSteps(filtersFromRules(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten))))), ctx) ;
5017
+ }
5018
+
5019
+ /** [doc] with each job's inline Drop items / Reject modifier steps migrated
5020
+ into one private filter set on a `filters` step, version 13 to 14 (#339).
5021
+ Every drop/reject modifier's rules become filters in document order -- a
5022
+ `drop` modifier's rules drop-item filters, a `reject` modifier's
5023
+ reject-reply filters (filterFromItemRule, which carries the matcher
5024
+ unchanged) -- in one `own` set, and the step takes the place of the first
5025
+ of those modifiers, so the reply is cleaned and gated as it was. A job with
5026
+ no such modifier, or only empty ones, is left as it is; this is a no-op on
5027
+ every path that reaches version 14, so it is safe to run on all of them. */
5028
+ function filtersFromRules(next ) {
5029
+ next.version = PIPELINE_VERSION;
5030
+ if (!Array.isArray(next.jobs)) return next;
5031
+ const TYPE = { drop: "drop-item", reject: "reject-reply" };
5032
+ const isRule = (st ) =>
5033
+ isObj(st) && st.type === "modifier" && isObj(st.modifier) && (st.modifier.type ) in TYPE;
5034
+ next.jobs = next.jobs.map((job ) => {
5035
+ if (!isObj(job) || !Array.isArray(job.steps) || !job.steps.some(isRule)) return job;
5036
+ const filters = [];
5037
+ for (const st of job.steps ) {
5038
+ if (!isRule(st)) continue;
5039
+ const m = (st ).modifier ;
5040
+ for (const r of Array.isArray(m.rules) ? m.rules : []) {
5041
+ filters.push(filterFromItemRule(r, TYPE[m.type ] ));
5042
+ }
5043
+ }
5044
+ // Empty Drop items / Reject modifiers held nothing, so they simply drop.
5045
+ // The link's id is the job's, so re-upgrading a document mints the same
5046
+ // one (the migration is idempotent) and a run keyed by it reads the same.
5047
+ const link = filters.length
5048
+ ? { id: `filters-${isStr(job.id) ? job.id : "job"}`, set: null, pin: null, own: { version: FILTER_SET_VERSION, filters } } : null;
5049
+ let placed = false;
5050
+ const steps = [];
5051
+ for (const st of job.steps ) {
5052
+ if (!isRule(st)) { steps.push(st); continue; }
5053
+ if (link && !placed) { steps.push({ type: "filters", filterSets: [link] }); placed = true; }
5054
+ }
5055
+ return { ...job, steps };
5056
+ });
5057
+ return next;
4687
5058
  }
4688
5059
 
4689
5060
  /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
@@ -5342,7 +5713,8 @@ function exportBundle(doc , ctx
5342
5713
  // What the lab would supply at submit, written down: no lab supplies it later.
5343
5714
  const out = clone(doc) ;
5344
5715
  out.evals = (Array.isArray(out.evals) ? out.evals : [])
5345
- .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
5716
+ .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null,
5717
+ groups: Object.entries(ctx.datasets ?? {}).map(([id, d]) => ({ id, name: d.name, grader: d.body.grader ?? null })) }) ?? t);
5346
5718
  const table = {};
5347
5719
  for (const id of profileIds(out)) {
5348
5720
  const p = profiles.find(x => x.id === id);
@@ -5487,13 +5859,16 @@ function readBundle(files , ctx = {})
5487
5859
  * the job it sits in, and the connection it asks -- its own profile, or
5488
5860
  * its scenario's.
5489
5861
  */
5490
- function stagesFor(run , i )
5862
+ function stagesFor(run , i , resolveFilters )
5491
5863
 
5492
5864
  {
5493
5865
  const stages = run.jobs.map((ch, k) => {
5494
- const { kind, modifiers, ...settings } = outOf(ch);
5866
+ const { kind, modifiers: _mods, ...settings } = outOf(ch);
5867
+ // The modifiers a run applies, with each linked or private filter set a
5868
+ // synthetic modifier where its link sits (#339); without a resolver, a
5869
+ // Library link reads nothing, a private one its own body.
5495
5870
  return { text: targetStepOf(run, i, k) .prompt, kind, withImage: sendsImage(ch),
5496
- verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: modifiers || [], settings };
5871
+ verbatim: !!targetEntryOf(run, i, k)?.verbatim, modifiers: stageModifiers(ch, resolveFilters), settings };
5497
5872
  });
5498
5873
  const connections = run.jobs.map((_, k) => {
5499
5874
  // A step answered here asks nothing: the profile a run made before it
@@ -5883,16 +6258,20 @@ export {
5883
6258
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
5884
6259
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
5885
6260
  PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
5886
- SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
5887
- registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
6261
+ SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf, WIZARDS, wizardProgress,
6262
+ registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, addressOf, connectionSettings,
5888
6263
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5889
6264
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5890
6265
  contentOf, withContent, replyOf,
5891
6266
  targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
5892
6267
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
6268
+ filterSetsOf, withFilterSets, stageModifiers,
5893
6269
  blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5894
6270
  scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
5895
6271
  evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
5896
6272
  pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
5897
6273
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
6274
+ FILTER_SET_VERSION, FILTER_TYPES, FILTER_SEPARATORS, applyFilter, applyFilterSet, tryFilter, splitSample,
6275
+ filterProblems, filterFromItemRule, upgradeFilterSetBody, blankFilterSetBody, filterSetBodyProblems, filterSetLinkProblems,
6276
+ newFilterLink, FILTER_MODIFIER,
5898
6277
  };