evals-lab 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -5,6 +5,74 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
5
5
  upgrade can rewrite what the lab keeps in your data directory, and an older
6
6
  version cannot always read it back.
7
7
 
8
+ ## 0.8.0
9
+
10
+ ### Upgrade notes
11
+
12
+ - No stored document changes format: pipelines stay version 13, and eval
13
+ groups and exports stay version 8, so 0.7.0 still reads what 0.8.0 saves.
14
+ - On first start, 0.8.0 adds a table recording which workspaces may use each
15
+ Target profile and account. Nothing is narrowed: every connection stays
16
+ shared with every workspace until you change it.
17
+
18
+ ### Added
19
+
20
+ - Several workspaces. Make, rename and archive them in Setup › Workspaces,
21
+ and switch between them from the header's workspace menu. Each one has its
22
+ own address (`/w/<slug>`) and its own datasets, prompts, Sources, pipelines
23
+ and runs. The theme choice (Dark, Light, Match system) has moved into that
24
+ menu.
25
+ - Target profiles and accounts can be shared with every workspace, with some,
26
+ or with each new one. A run in a workspace a profile isn't shared with says
27
+ so and offers to share it.
28
+ - `evals-lab run --lab … --workspace <slug>` runs a lab's pipeline in a named
29
+ workspace. Without it, the default workspace is used, as before.
30
+ - Setup › Wizards, and a header pill showing a wizard's progress. The first
31
+ wizard, Test a Power Automate workflow, ticks off each step as you do it.
32
+ Plugins can add workflow platforms and wizards.
33
+ - Judged against recorded has an **Expected** setting: tick which of Better,
34
+ Same and Worse pass (Better and Same by default, as before).
35
+ - A Recorded target's menu has **Preview replies…**, showing what each call
36
+ recorded.
37
+ - Test… asks for a grader when a check needs one. You can create the test
38
+ without one and add it later.
39
+
40
+ ### Changed
41
+
42
+ - Model-graded checks no longer need a prompt written. The grader is told
43
+ what the target was asked; Judged against recorded's Task is now optional
44
+ Additional guidance.
45
+ - A grader's answer is held to a fixed JSON shape where its API supports
46
+ that (Anthropic, OpenAI-compatible, llama.cpp, Ollama). An answer that
47
+ still can't be read is asked for once more, then reported as a grader
48
+ error rather than a verdict.
49
+ - Test… on a step whose reply is plain words judges the reply against the
50
+ recorded one, instead of requiring the exact same words.
51
+ - Rubric's and Factual's threshold reads **Expected: score at least**.
52
+ - New profile: Anthropic and Ollama (Cloud) no longer ask for an address. A
53
+ proxy or gateway address goes under Advanced. Temperature is hidden, and
54
+ never sent, for models that reject it (Claude Opus 4.7 and later, the
55
+ 5-series Claude models, OpenAI reasoning models). Image settings are now
56
+ the folded Vision Capabilities group.
57
+ - The Test… dialog is "Create new pipeline test", with a Create button.
58
+ - Library › Sources › a flow: Actions come before Calls, and each call's
59
+ request reads In sync or Out of sync.
60
+ - Run results: Take B as expected is in each call's row menu, and the
61
+ Results table no longer has checkboxes.
62
+ - The wizard pill's button is Open wizard, which shows the wizard's steps.
63
+
64
+ ### Fixed
65
+
66
+ - An HTTP endpoint profile whose address included a path (such as
67
+ `https://api.anthropic.com/v1/messages`) sent the path twice, so every
68
+ call came back Not found.
69
+ - An Anthropic profile with a blank address was sent to the lab's own Ollama.
70
+ - A run said "this run has none" for a grader that its Library eval group
71
+ named.
72
+ - The wizard pill didn't tick a step until the page was reloaded.
73
+ - A Microsoft 365 sign-in now shows which tenants it reaches, so a lab that
74
+ works across tenants is not tied to one.
75
+
8
76
  ## 0.7.0
9
77
 
10
78
  ### Upgrade notes
package/bin/run.js CHANGED
@@ -51,6 +51,8 @@ const USAGE = `Usage: evals-lab run <bundle-dir | pipeline.yaml> [options]
51
51
  and their keys, and the run in its History. LAB_PASSWORD
52
52
  is sent when it is set.
53
53
  --pipeline <name> the lab's pipeline to run, by its name or its id.
54
+ --workspace <slug> the lab's workspace to run in, by its /w/<slug> address.
55
+ Default: the lab's default workspace.
54
56
  --wait wait for the lab's run to finish, and exit with its
55
57
  verdict. Without it the run's id is printed once it is
56
58
  queued. --json, --junit, --summary, --min-pass and
@@ -70,9 +72,9 @@ class Refused extends Error {
70
72
  // What the command line asks for, or a Refused saying what is wrong with it.
71
73
  function runOptions(argv) {
72
74
  const o = { bundle: null, items: null, json: null, junit: null, summary: null, minPass: null, progress: false,
73
- lab: null, pipeline: null, wait: false };
75
+ lab: null, pipeline: null, workspace: null, wait: false };
74
76
  const value = { "--items": "items", "--json": "json", "--junit": "junit", "--summary": "summary", "--min-pass": "minPass",
75
- "--lab": "lab", "--pipeline": "pipeline" };
77
+ "--lab": "lab", "--pipeline": "pipeline", "--workspace": "workspace" };
76
78
  for (let i = 0; i < argv.length; i++) {
77
79
  const a = argv[i];
78
80
  if (a === "--help" || a === "-h") o.help = true;
@@ -96,6 +98,8 @@ function runOptions(argv) {
96
98
  if (waits.length && !o.wait) {
97
99
  throw new Refused(`--${waits[0].replace("minPass", "min-pass")} reads the finished run: add --wait`, true);
98
100
  }
101
+ } else if (o.workspace != null) {
102
+ throw new Refused("--workspace is for a lab's run: name the lab with --lab", true);
99
103
  } else if (o.wait) {
100
104
  throw new Refused("--wait is for a lab's run: a bundle's always waits", true);
101
105
  } else if (o.bundle == null) {
@@ -276,15 +280,18 @@ const POLL_MISSES = 20;
276
280
 
277
281
  // The lab's API at [base], as [env]'s LAB_PASSWORD opens it: HTTP Basic, the
278
282
  // password alone, as server.py takes it. A refusal says what the lab said.
279
- function labApi(base, env) {
283
+ // [workspace], when given, is the slug every call carries as X-Workspace, so
284
+ // the run reads and queues in that workspace rather than the lab's default.
285
+ function labApi(base, env, workspace) {
280
286
  const auth = env.LAB_PASSWORD
281
287
  ? { Authorization: `Basic ${Buffer.from(`:${env.LAB_PASSWORD}`).toString("base64")}` } : {};
288
+ const scope = workspace ? { "X-Workspace": workspace } : {};
282
289
  return async (method, route, body, { missing = false } = {}) => {
283
290
  let res;
284
291
  try {
285
292
  res = await fetch(base + route, {
286
293
  method, redirect: "manual",
287
- headers: { ...auth, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
294
+ headers: { ...auth, ...scope, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
288
295
  ...(body !== undefined ? { body: JSON.stringify(body) } : {}),
289
296
  });
290
297
  } catch (e) {
@@ -319,6 +326,21 @@ function pipelineNamed(saved, name, base) {
319
326
  return found[0];
320
327
  }
321
328
 
329
+ // The lab's workspace --workspace [slug] names, by its /w/<slug> address -- or
330
+ // a Refused naming the slugs it has, as pipelineNamed does for pipelines. The
331
+ // server falls back to its default for an unknown slug, so the CLI checks the
332
+ // slug itself; an archived workspace cannot be run against.
333
+ function workspaceNamed(workspaces, slug, base) {
334
+ const active = workspaces.filter(w => w && !w.archived);
335
+ const found = active.find(w => w.slug === slug);
336
+ if (found) return found;
337
+ if (workspaces.some(w => w && w.slug === slug)) {
338
+ throw new Refused(`${base}'s workspace at /w/${slug} is archived`);
339
+ }
340
+ throw new Refused(`${base} has no workspace at /w/${slug}`
341
+ + (active.length ? `: it has ${active.map(w => JSON.stringify(w.slug)).join(", ")}` : ""));
342
+ }
343
+
322
344
  // The row's verdicts with --min-pass relaxing each eval read item by item,
323
345
  // as the worker's --min-pass does: it only ever turns a fail into a pass.
324
346
  const relaxed = (verdicts, minPass) => minPass == null ? verdicts : verdicts.map(evals =>
@@ -356,7 +378,16 @@ function rowSuites(core, run, verdicts, items) {
356
378
  /** [o]'s pipeline, queued on its lab; with --wait, followed to its verdict. */
357
379
  async function runOnLab(core, o, { env, stdout, say }) {
358
380
  const base = o.lab.replace(/\/+$/, "");
359
- const call = labApi(base, env);
381
+ // Resolve --workspace to the slug every call carries; the lab's pipelines,
382
+ // Sources and eval groups are per-workspace, so the whole run reads and
383
+ // queues inside it. Without it, the lab's default workspace is used and the
384
+ // X-Workspace header is omitted, so existing CI is unchanged.
385
+ let workspace = null;
386
+ if (o.workspace != null) {
387
+ const { workspaces } = await labApi(base, env)("GET", "/api/workspaces");
388
+ workspace = workspaceNamed(Array.isArray(workspaces) ? workspaces : [], o.workspace, base).slug;
389
+ }
390
+ const call = labApi(base, env, workspace);
360
391
 
361
392
  // What the page reads before it submits: the lab's pipelines and Target
362
393
  // profiles, the Source the pipeline reads and the eval groups it may link.
package/lab/VERSION CHANGED
@@ -1 +1 @@
1
- 0.7.0 (2026.10.06-456)
1
+ 0.8.0 (2026.10.07-484)
@@ -657,6 +657,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
657
657
 
658
658
 
659
659
 
660
+
661
+
662
+
660
663
 
661
664
 
662
665
 
@@ -667,9 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
667
670
 
668
671
  /** What a metric needs besides the reply: a model to grade with. */
669
672
 
670
-
671
-
672
-
673
+
674
+
675
+
676
+
673
677
 
674
678
 
675
679
 
@@ -723,6 +727,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
723
727
 
724
728
 
725
729
 
730
+
731
+
732
+
733
+
734
+
735
+
736
+
726
737
 
727
738
 
728
739
  /** What an earlier eval that failed and does not continue leaves a later one. */
@@ -797,6 +808,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
797
808
 
798
809
 
799
810
 
811
+
812
+
813
+
800
814
 
801
815
 
802
816
 
@@ -881,9 +895,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
881
895
 
882
896
 
883
897
 
898
+
884
899
 
885
-
886
-
900
+
887
901
 
888
902
 
889
903
  /** Lookups, and whether it is a run document being validated. */
@@ -1041,7 +1055,7 @@ const platformOf = (src
1041
1055
 
1042
1056
 
1043
1057
 
1044
-
1058
+
1045
1059
 
1046
1060
 
1047
1061
 
@@ -1323,6 +1337,10 @@ function datasetRules(body ) {
1323
1337
 
1324
1338
 
1325
1339
 
1340
+
1341
+
1342
+
1343
+
1326
1344
 
1327
1345
 
1328
1346
  /** An HTTP request as a connection type builds it, for the relay to send. */
@@ -1346,6 +1364,10 @@ function datasetRules(body ) {
1346
1364
 
1347
1365
 
1348
1366
 
1367
+
1368
+
1369
+
1370
+
1349
1371
 
1350
1372
 
1351
1373
 
@@ -1354,8 +1376,9 @@ function datasetRules(body ) {
1354
1376
 
1355
1377
 
1356
1378
 
1357
-
1358
-
1379
+
1380
+
1381
+
1359
1382
 
1360
1383
 
1361
1384
 
@@ -1708,6 +1731,8 @@ function ollamaType(id , label , cloud ) {
1708
1731
  body: { model: conn.model, messages: [message], options, stream: false },
1709
1732
  };
1710
1733
  },
1734
+ // Ollama's own: the schema is the format.
1735
+ structured: (body, schema) => ({ ...body, format: schema }),
1711
1736
  parseReply(j) {
1712
1737
  const r = j ;
1713
1738
  return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
@@ -1754,7 +1779,7 @@ function recordedRaw(record ) {
1754
1779
  // the flow reads a reply; `local` is the no-Read-as fallback.
1755
1780
  CONNECTION_TYPES.recorded = {
1756
1781
  id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
1757
- description: "Replays the reply production recorded for each call. Sends nothing.",
1782
+ description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
1758
1783
  local: (_item, _sent, record) => recordedRaw(record),
1759
1784
  request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1760
1785
  };
@@ -1769,19 +1794,25 @@ function localAnswer(conn , text , sent , record
1769
1794
  return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
1770
1795
  }
1771
1796
 
1797
+ /** A chat completions body asking for a reply held to [schema]: OpenAI's
1798
+ response_format, which llama-server and most compatible servers take too. */
1799
+ const chatSchema = (body , schema ) =>
1800
+ ({ ...body, response_format: { type: "json_schema", json_schema: { name: "reply", strict: true, schema } } });
1801
+
1772
1802
  CONNECTION_TYPES["openai-compatible"] = {
1773
1803
  id: "openai-compatible", label: "OpenAI-compatible",
1774
1804
  description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
1775
1805
  settings: [
1776
- { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1806
+ // A reasoning model refuses a temperature.
1807
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
1808
+ appliesTo: (conn) => !/^(gpt-5|o\d)/i.test(String(conn.model ?? "")) },
1777
1809
  { key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
1778
1810
  ],
1779
- // /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio. The
1780
- // reasoning-model temperature rule and max_completion_tokens stay with this
1781
- // type, as the hosted providers reject the fields a llama.cpp server takes.
1811
+ // /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
1812
+ // max_completion_tokens stays with this type, as the hosted providers
1813
+ // reject the fields a llama.cpp server takes.
1782
1814
  request(conn, prompt, dataUrl, base, key) {
1783
1815
  const hosted = isHostedUrl(base);
1784
- const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
1785
1816
  const temp = asNumber(conn.temperature);
1786
1817
  const predict = asNumber(conn.nPredict);
1787
1818
  return {
@@ -1789,7 +1820,7 @@ CONNECTION_TYPES["openai-compatible"] = {
1789
1820
  headers: bearer(key),
1790
1821
  body: {
1791
1822
  model: conn.model,
1792
- ...(isReasoning || temp == null ? {} : { temperature: temp }),
1823
+ ...(temp == null ? {} : { temperature: temp }),
1793
1824
  ...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
1794
1825
  messages: [{ role: "user", content: [
1795
1826
  { type: "text", text: prompt },
@@ -1798,15 +1829,23 @@ CONNECTION_TYPES["openai-compatible"] = {
1798
1829
  },
1799
1830
  };
1800
1831
  },
1832
+ structured: chatSchema,
1801
1833
  parseReply: chatReply,
1802
1834
  listModels, parseModels: idsFrom,
1803
1835
  };
1804
1836
 
1837
+ /** The Claude models that refuse a sampling setting: Opus from 4.7, and
1838
+ every Opus, Sonnet, Fable and Mythos from 5. Older ones (Opus 4.6, Sonnet
1839
+ 4.6, Haiku 4.5) take one. */
1840
+ const CLAUDE_FIXED_SAMPLING = /claude-(opus-4-[7-9]|(opus|sonnet|fable|mythos)-[5-9])/i;
1841
+
1805
1842
  CONNECTION_TYPES.anthropic = {
1806
1843
  id: "anthropic", label: "Anthropic",
1807
1844
  description: "Claude, through Anthropic's Messages API.",
1845
+ url: "https://api.anthropic.com",
1808
1846
  settings: [
1809
- { key: "temperature", label: "Temperature", control: "text", default: "", hint: "" },
1847
+ { key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
1848
+ appliesTo: (conn) => !CLAUDE_FIXED_SAMPLING.test(String(conn.model ?? "")) },
1810
1849
  { key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
1811
1850
  ],
1812
1851
  // The Messages API: the x-api-key and anthropic-version headers, base64
@@ -1831,6 +1870,9 @@ CONNECTION_TYPES.anthropic = {
1831
1870
  },
1832
1871
  };
1833
1872
  },
1873
+ // The Messages API's structured outputs. A model that cannot hold one
1874
+ // refuses the request, and the grader asks again without it.
1875
+ structured: (body, schema) => ({ ...body, output_config: { ...(body.output_config ), format: { type: "json_schema", schema } } }),
1834
1876
  parseReply(j) {
1835
1877
  const r = j ;
1836
1878
  const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
@@ -1903,6 +1945,8 @@ CONNECTION_TYPES["llama.cpp"] = {
1903
1945
  },
1904
1946
  };
1905
1947
  },
1948
+ // llama-server takes OpenAI's response_format, a schema and all.
1949
+ structured: chatSchema,
1906
1950
  parseReply: chatReply,
1907
1951
  listModels, parseModels: idsFrom,
1908
1952
  // A reply that arrived is still a failure when this profile asked for the
@@ -2098,12 +2142,14 @@ function extraHeaders(raw ) {
2098
2142
  }
2099
2143
  return out;
2100
2144
  }
2101
- /** An address as a whole-request type hangs paths off it: its origin, no /v1. */
2145
+ /** An address as a whole-request type hangs paths off it: its origin, no
2146
+ path. A step's path is the whole path production called, so an address
2147
+ entered with one (`…/v1/messages`) would send it twice. */
2102
2148
  function originBase(raw ) {
2103
- let s = String(raw || "").trim().replace(/\/+$/, "");
2149
+ let s = String(raw || "").trim();
2104
2150
  if (!s) return "";
2105
2151
  if (!/^https?:\/\//.test(s)) s = "https://" + s;
2106
- return s;
2152
+ try { return new URL(s).origin; } catch { return s.replace(/\/+$/, ""); }
2107
2153
  }
2108
2154
  const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
2109
2155
  CONNECTION_TYPES.http = {
@@ -2230,9 +2276,14 @@ function splitOllama(p ) {
2230
2276
  * connection carries its type's settings under `options`, so they are merged
2231
2277
  * up before the type reads them.
2232
2278
  */
2233
- function connectionRequest(conn , prompt , dataUrl , base , key ) {
2279
+ function connectionRequest(conn , prompt , dataUrl , base , key ,
2280
+ schema ) {
2234
2281
  const flat = { ...conn, ...(conn?.options || {}) };
2235
- return typeEntry(flat).request(flat, prompt, dataUrl, base, key);
2282
+ const type = typeEntry(flat);
2283
+ // A setting the model refuses is not sent, whatever the profile holds.
2284
+ for (const s of type.settings) if (s.appliesTo && !s.appliesTo(flat)) delete flat[s.key];
2285
+ const req = type.request(flat, prompt, dataUrl, base, key);
2286
+ return schema && type.structured ? { ...req, body: type.structured(req.body , schema) } : req;
2236
2287
  }
2237
2288
 
2238
2289
  /** Where [conn]'s requests hang off: its type's own reading of its address,
@@ -2934,6 +2985,15 @@ function registerKinds(k ) {
2934
2985
 
2935
2986
 
2936
2987
 
2988
+
2989
+
2990
+
2991
+  
2992
+
2993
+
2994
+
2995
+
2996
+
2937
2997
 
2938
2998
 
2939
2999
  function pluginHost(pluginId ) {
@@ -2961,6 +3021,14 @@ function pluginHost(pluginId ) {
2961
3021
  taken(CONNECTION_TYPES, "connection type", [t.id]);
2962
3022
  CONNECTION_TYPES[t.id] = t;
2963
3023
  },
3024
+ registerWorkflowPlatform(t) {
3025
+ taken(WORKFLOW_PLATFORMS, "workflow platform", [t.id]);
3026
+ WORKFLOW_PLATFORMS[t.id] = t;
3027
+ },
3028
+ registerWizard(w) {
3029
+ taken(WIZARDS, "wizard", [w.id]);
3030
+ WIZARDS[w.id] = w;
3031
+ },
2964
3032
  };
2965
3033
  }
2966
3034
 
@@ -3058,6 +3126,13 @@ function withSlugs (list
3058
3126
  /** Whether [s] is a slug as a profile may hold one. */
3059
3127
  const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
3060
3128
 
3129
+ /** Where [p] connects: its address, or blank, its type's own -- an Anthropic
3130
+ profile left blank reaches Anthropic. Blank for a type with none means the
3131
+ lab's own Ollama, which the runner and the relay fill in. */
3132
+ function addressOf(p ) {
3133
+ return String(p.url ?? "").trim() || CONNECTION_TYPES[typeOf(p )]?.url || "";
3134
+ }
3135
+
3061
3136
  /** A stored profile as a run carries it: its request settings, no key. */
3062
3137
  function connectionOf(p ) {
3063
3138
  const type = typeOf(p);
@@ -3066,7 +3141,7 @@ function connectionOf(p ) {
3066
3141
  // temperature is a common field, carried at the top like px/format/quality;
3067
3142
  // the options bag holds the type's own settings only.
3068
3143
  for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
3069
- const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
3144
+ const out = { name: p.name || "unnamed", url: addressOf(p), model: p.model || "", type };
3070
3145
  if (isSlug(p.slug)) out.slug = p.slug;
3071
3146
  for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
3072
3147
  if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
@@ -3187,6 +3262,70 @@ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapt
3187
3262
  render: body => JSON.stringify(body, null, 2),
3188
3263
  };
3189
3264
 
3265
+ // ---- Setup Wizards (docs/workflow-sources.md § Wizards) ---------------------
3266
+ //
3267
+ // A wizard is a registry entry whose steps each say where they happen and a
3268
+ // done-when predicate on lab state. Steps tick automatically: `done(lab)` reads
3269
+ // the cheap state the page already holds (a workflow Source exists, a pipeline
3270
+ // with a Recorded target exists, a run of it is in History), never a poll. One
3271
+ // wizard runs at a time, held in the browser-only `promptlab.wizard` store; a
3272
+ // header pill shows its progress and the Setup › Wizards section draws the same
3273
+ // walk-through. The registry lives here so a plugin registers a wizard through
3274
+ // the same host as every other kind (PluginHost.registerWizard, phase 7); the
3275
+ // page ships the built-in "Test a workflow" and owns its routes and selectors
3276
+ // (web/src/app/wizards.ts). The predicates run with the page's privileges and
3277
+ // only read lab state, so a plugin wizard is no new trust (docs/packs.md).
3278
+
3279
+ /** The slice of lab state a wizard step's `done` reads: what the page already
3280
+ holds, so a tick costs a predicate and not a request. A reader names a field
3281
+ it needs; a wizard that reads more declares it here. */
3282
+
3283
+
3284
+
3285
+
3286
+
3287
+
3288
+
3289
+
3290
+
3291
+
3292
+
3293
+
3294
+
3295
+
3296
+
3297
+
3298
+
3299
+
3300
+
3301
+
3302
+
3303
+
3304
+
3305
+
3306
+
3307
+
3308
+
3309
+
3310
+
3311
+
3312
+
3313
+
3314
+
3315
+
3316
+ const WIZARDS = Object.create(null);
3317
+
3318
+ /** How far a wizard has got on [lab]: each step's tick, how many are done, and
3319
+ the first step not yet done -- the one the checklist marks current and Go
3320
+ to step heads for. `current` is -1 once every step is done. */
3321
+ function wizardProgress(entry , lab )
3322
+
3323
+ {
3324
+ const ticks = entry.steps.map((s) => s.done(lab));
3325
+ const done = ticks.filter(Boolean).length;
3326
+ return { ticks, done, total: entry.steps.length, current: ticks.findIndex((t) => !t) };
3327
+ }
3328
+
3190
3329
  CONTENT_TYPES.source = {
3191
3330
  label: "Source",
3192
3331
  description: "Every file or record in a Source from the Library.",
@@ -3364,6 +3503,7 @@ function metricInput(res , kase , more )
3364
3503
  const last = res.transcript?.at(-1);
3365
3504
  const text = last?.got ?? res.raw ?? "";
3366
3505
  return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
3506
+ asked: last?.sent ?? null,
3367
3507
  // A case that has been given its own expected reply ("Take B as
3368
3508
  // expected") is compared with that; otherwise the Source's recorded
3369
3509
  // reply for the item.
@@ -3382,7 +3522,7 @@ function runInput(ress , plain )
3382
3522
  terms.push(...(r.terms || []));
3383
3523
  ms += r.ms ?? 0;
3384
3524
  }
3385
- return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
3525
+ return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, asked: null, kase: null };
3386
3526
  }
3387
3527
 
3388
3528
  /** One metric's reading of [input]: `of`, `not` and a failure all applied,
@@ -3643,16 +3783,21 @@ EVAL_TYPES.group = {
3643
3783
  want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3644
3784
  .filter(Boolean).join(" ") })),
3645
3785
  profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
3646
- // A run carries the lab's grader where the group names none and may ask
3647
- // one: a model-graded metric of its own, or a case's. A link's group is
3648
- // the Library's, so the run's copy of the link carries it.
3786
+ // A run carries the grader each group asks, so its profile goes with the
3787
+ // run: a link's is the Library group's own, or the lab's where it names
3788
+ // none -- the run's copy of the link carries it. A private group takes the
3789
+ // lab's where it names none and may ask one: a model-graded metric of its
3790
+ // own, or a case's.
3649
3791
  resolve(t, ctx){
3650
- if (!isRef(ctx.grader)) return t;
3651
- const grader = { id: ctx.grader.id, name: ctx.grader.name };
3792
+ const ref = (g ) => (isRef(g) ? { id: g.id, name: g.name } : null);
3793
+ const lab = ref(ctx.grader);
3652
3794
  if (isObj(t.own)) {
3653
- return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
3795
+ return lab && !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
3796
+ ? { ...t, own: { ...t.own, grader: lab } } : t;
3654
3797
  }
3655
- return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
3798
+ if (!isRef(t.group) || isRef(t.grader)) return t;
3799
+ const grader = ref(ctx.groups?.find(g => g.id === (t.group ).id)?.grader) ?? lab;
3800
+ return grader ? { ...t, grader } : t;
3656
3801
  },
3657
3802
  wholeRun: wholeRunGroup,
3658
3803
  verdict: (t, ress, kind) => {
@@ -3849,6 +3994,9 @@ function metricSummary(m ) {
3849
3994
  const v = m[o.key];
3850
3995
  if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
3851
3996
  if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
3997
+ // The choices ticked, by their labels, where they are not a new one's.
3998
+ if (o.type === "multi") return !Array.isArray(v) || JSON.stringify(v) === JSON.stringify(fresh[o.key]) ? ""
3999
+ : o.choices.filter(c => v.includes(c.value)).map(c => c.label).join(" or ");
3852
4000
  // Text that runs to lines (a schema) reads as its first words, run together.
3853
4001
  return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
3854
4002
  }).filter(Boolean);
@@ -4014,7 +4162,7 @@ STEP_TYPES.echo = {
4014
4162
  // the flow reads one. No profile, no prompt -- the record is the reply.
4015
4163
  STEP_TYPES.recorded = {
4016
4164
  label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
4017
- description: "Replays the reply production recorded for each call. Sends nothing.",
4165
+ description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
4018
4166
  firstJobOnly: "a call's recorded reply is job 1's",
4019
4167
  in: "item", out: "text",
4020
4168
  apply: "runPipeline",
@@ -5342,7 +5490,8 @@ function exportBundle(doc , ctx
5342
5490
  // What the lab would supply at submit, written down: no lab supplies it later.
5343
5491
  const out = clone(doc) ;
5344
5492
  out.evals = (Array.isArray(out.evals) ? out.evals : [])
5345
- .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
5493
+ .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null,
5494
+ groups: Object.entries(ctx.datasets ?? {}).map(([id, d]) => ({ id, name: d.name, grader: d.body.grader ?? null })) }) ?? t);
5346
5495
  const table = {};
5347
5496
  for (const id of profileIds(out)) {
5348
5497
  const p = profiles.find(x => x.id === id);
@@ -5883,8 +6032,8 @@ export {
5883
6032
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
5884
6033
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
5885
6034
  PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
5886
- SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
5887
- registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
6035
+ SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf, WIZARDS, wizardProgress,
6036
+ registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, addressOf, connectionSettings,
5888
6037
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5889
6038
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5890
6039
  contentOf, withContent, replyOf,