evals-lab 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +183 -34
- package/lab/metrics/builtin.mjs +76 -20
- package/lab/run-evals.js +8 -4
- package/lab/server.py +588 -12
- package/lab/web/dist/assets/gallery-SnUhXRBn.js +3 -0
- package/lab/web/dist/assets/main-Ca7o-nM0.css +1 -0
- package/lab/web/dist/assets/main-Dru4_P5G.js +20 -0
- package/lab/web/dist/assets/tokens-0az9gfTq.js +58 -0
- package/lab/web/dist/assets/tokens-CqWJKhOx.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +0 -3
- package/lab/web/dist/assets/main-BQL5j5oF.js +0 -20
- package/lab/web/dist/assets/main-Cza2gwQd.css +0 -1
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +0 -61
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,74 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.8.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- No stored document changes format: pipelines stay version 13, and eval
|
|
13
|
+
groups and exports stay version 8, so 0.7.0 still reads what 0.8.0 saves.
|
|
14
|
+
- On first start, 0.8.0 adds a table recording which workspaces may use each
|
|
15
|
+
Target profile and account. Nothing is narrowed: every connection stays
|
|
16
|
+
shared with every workspace until you change it.
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
|
|
20
|
+
- Several workspaces. Make, rename and archive them in Setup › Workspaces,
|
|
21
|
+
and switch between them from the header's workspace menu. Each one has its
|
|
22
|
+
own address (`/w/<slug>`) and its own datasets, prompts, Sources, pipelines
|
|
23
|
+
and runs. The theme choice (Dark, Light, Match system) has moved into that
|
|
24
|
+
menu.
|
|
25
|
+
- Target profiles and accounts can be shared with every workspace, with some,
|
|
26
|
+
or with each new one. A run in a workspace a profile isn't shared with says
|
|
27
|
+
so and offers to share it.
|
|
28
|
+
- `evals-lab run --lab … --workspace <slug>` runs a lab's pipeline in a named
|
|
29
|
+
workspace. Without it, the default workspace is used, as before.
|
|
30
|
+
- Setup › Wizards, and a header pill showing a wizard's progress. The first
|
|
31
|
+
wizard, Test a Power Automate workflow, ticks off each step as you do it.
|
|
32
|
+
Plugins can add workflow platforms and wizards.
|
|
33
|
+
- Judged against recorded has an **Expected** setting: tick which of Better,
|
|
34
|
+
Same and Worse pass (Better and Same by default, as before).
|
|
35
|
+
- A Recorded target's menu has **Preview replies…**, showing what each call
|
|
36
|
+
recorded.
|
|
37
|
+
- Test… asks for a grader when a check needs one. You can create the test
|
|
38
|
+
without one and add it later.
|
|
39
|
+
|
|
40
|
+
### Changed
|
|
41
|
+
|
|
42
|
+
- Model-graded checks no longer need a prompt written. The grader is told
|
|
43
|
+
what the target was asked; Judged against recorded's Task is now optional
|
|
44
|
+
Additional guidance.
|
|
45
|
+
- A grader's answer is held to a fixed JSON shape where its API supports
|
|
46
|
+
that (Anthropic, OpenAI-compatible, llama.cpp, Ollama). An answer that
|
|
47
|
+
still can't be read is asked for once more, then reported as a grader
|
|
48
|
+
error rather than a verdict.
|
|
49
|
+
- Test… on a step whose reply is plain words judges the reply against the
|
|
50
|
+
recorded one, instead of requiring the exact same words.
|
|
51
|
+
- Rubric's and Factual's threshold reads **Expected: score at least**.
|
|
52
|
+
- New profile: Anthropic and Ollama (Cloud) no longer ask for an address. A
|
|
53
|
+
proxy or gateway address goes under Advanced. Temperature is hidden, and
|
|
54
|
+
never sent, for models that reject it (Claude Opus 4.7 and later, the
|
|
55
|
+
5-series Claude models, OpenAI reasoning models). Image settings are now
|
|
56
|
+
the folded Vision Capabilities group.
|
|
57
|
+
- The Test… dialog is "Create new pipeline test", with a Create button.
|
|
58
|
+
- Library › Sources › a flow: Actions come before Calls, and each call's
|
|
59
|
+
request reads In sync or Out of sync.
|
|
60
|
+
- Run results: Take B as expected is in each call's row menu, and the
|
|
61
|
+
Results table no longer has checkboxes.
|
|
62
|
+
- The wizard pill's button is Open wizard, which shows the wizard's steps.
|
|
63
|
+
|
|
64
|
+
### Fixed
|
|
65
|
+
|
|
66
|
+
- An HTTP endpoint profile whose address included a path (such as
|
|
67
|
+
`https://api.anthropic.com/v1/messages`) sent the path twice, so every
|
|
68
|
+
call came back Not found.
|
|
69
|
+
- An Anthropic profile with a blank address was sent to the lab's own Ollama.
|
|
70
|
+
- A run said "this run has none" for a grader that its Library eval group
|
|
71
|
+
named.
|
|
72
|
+
- The wizard pill didn't tick a step until the page was reloaded.
|
|
73
|
+
- A Microsoft 365 sign-in now shows which tenants it reaches, so a lab that
|
|
74
|
+
works across tenants is not tied to one.
|
|
75
|
+
|
|
8
76
|
## 0.7.0
|
|
9
77
|
|
|
10
78
|
### Upgrade notes
|
package/bin/run.js
CHANGED
|
@@ -51,6 +51,8 @@ const USAGE = `Usage: evals-lab run <bundle-dir | pipeline.yaml> [options]
|
|
|
51
51
|
and their keys, and the run in its History. LAB_PASSWORD
|
|
52
52
|
is sent when it is set.
|
|
53
53
|
--pipeline <name> the lab's pipeline to run, by its name or its id.
|
|
54
|
+
--workspace <slug> the lab's workspace to run in, by its /w/<slug> address.
|
|
55
|
+
Default: the lab's default workspace.
|
|
54
56
|
--wait wait for the lab's run to finish, and exit with its
|
|
55
57
|
verdict. Without it the run's id is printed once it is
|
|
56
58
|
queued. --json, --junit, --summary, --min-pass and
|
|
@@ -70,9 +72,9 @@ class Refused extends Error {
|
|
|
70
72
|
// What the command line asks for, or a Refused saying what is wrong with it.
|
|
71
73
|
function runOptions(argv) {
|
|
72
74
|
const o = { bundle: null, items: null, json: null, junit: null, summary: null, minPass: null, progress: false,
|
|
73
|
-
lab: null, pipeline: null, wait: false };
|
|
75
|
+
lab: null, pipeline: null, workspace: null, wait: false };
|
|
74
76
|
const value = { "--items": "items", "--json": "json", "--junit": "junit", "--summary": "summary", "--min-pass": "minPass",
|
|
75
|
-
"--lab": "lab", "--pipeline": "pipeline" };
|
|
77
|
+
"--lab": "lab", "--pipeline": "pipeline", "--workspace": "workspace" };
|
|
76
78
|
for (let i = 0; i < argv.length; i++) {
|
|
77
79
|
const a = argv[i];
|
|
78
80
|
if (a === "--help" || a === "-h") o.help = true;
|
|
@@ -96,6 +98,8 @@ function runOptions(argv) {
|
|
|
96
98
|
if (waits.length && !o.wait) {
|
|
97
99
|
throw new Refused(`--${waits[0].replace("minPass", "min-pass")} reads the finished run: add --wait`, true);
|
|
98
100
|
}
|
|
101
|
+
} else if (o.workspace != null) {
|
|
102
|
+
throw new Refused("--workspace is for a lab's run: name the lab with --lab", true);
|
|
99
103
|
} else if (o.wait) {
|
|
100
104
|
throw new Refused("--wait is for a lab's run: a bundle's always waits", true);
|
|
101
105
|
} else if (o.bundle == null) {
|
|
@@ -276,15 +280,18 @@ const POLL_MISSES = 20;
|
|
|
276
280
|
|
|
277
281
|
// The lab's API at [base], as [env]'s LAB_PASSWORD opens it: HTTP Basic, the
|
|
278
282
|
// password alone, as server.py takes it. A refusal says what the lab said.
|
|
279
|
-
|
|
283
|
+
// [workspace], when given, is the slug every call carries as X-Workspace, so
|
|
284
|
+
// the run reads and queues in that workspace rather than the lab's default.
|
|
285
|
+
function labApi(base, env, workspace) {
|
|
280
286
|
const auth = env.LAB_PASSWORD
|
|
281
287
|
? { Authorization: `Basic ${Buffer.from(`:${env.LAB_PASSWORD}`).toString("base64")}` } : {};
|
|
288
|
+
const scope = workspace ? { "X-Workspace": workspace } : {};
|
|
282
289
|
return async (method, route, body, { missing = false } = {}) => {
|
|
283
290
|
let res;
|
|
284
291
|
try {
|
|
285
292
|
res = await fetch(base + route, {
|
|
286
293
|
method, redirect: "manual",
|
|
287
|
-
headers: { ...auth, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
294
|
+
headers: { ...auth, ...scope, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
288
295
|
...(body !== undefined ? { body: JSON.stringify(body) } : {}),
|
|
289
296
|
});
|
|
290
297
|
} catch (e) {
|
|
@@ -319,6 +326,21 @@ function pipelineNamed(saved, name, base) {
|
|
|
319
326
|
return found[0];
|
|
320
327
|
}
|
|
321
328
|
|
|
329
|
+
// The lab's workspace --workspace [slug] names, by its /w/<slug> address -- or
|
|
330
|
+
// a Refused naming the slugs it has, as pipelineNamed does for pipelines. The
|
|
331
|
+
// server falls back to its default for an unknown slug, so the CLI checks the
|
|
332
|
+
// slug itself; an archived workspace cannot be run against.
|
|
333
|
+
function workspaceNamed(workspaces, slug, base) {
|
|
334
|
+
const active = workspaces.filter(w => w && !w.archived);
|
|
335
|
+
const found = active.find(w => w.slug === slug);
|
|
336
|
+
if (found) return found;
|
|
337
|
+
if (workspaces.some(w => w && w.slug === slug)) {
|
|
338
|
+
throw new Refused(`${base}'s workspace at /w/${slug} is archived`);
|
|
339
|
+
}
|
|
340
|
+
throw new Refused(`${base} has no workspace at /w/${slug}`
|
|
341
|
+
+ (active.length ? `: it has ${active.map(w => JSON.stringify(w.slug)).join(", ")}` : ""));
|
|
342
|
+
}
|
|
343
|
+
|
|
322
344
|
// The row's verdicts with --min-pass relaxing each eval read item by item,
|
|
323
345
|
// as the worker's --min-pass does: it only ever turns a fail into a pass.
|
|
324
346
|
const relaxed = (verdicts, minPass) => minPass == null ? verdicts : verdicts.map(evals =>
|
|
@@ -356,7 +378,16 @@ function rowSuites(core, run, verdicts, items) {
|
|
|
356
378
|
/** [o]'s pipeline, queued on its lab; with --wait, followed to its verdict. */
|
|
357
379
|
async function runOnLab(core, o, { env, stdout, say }) {
|
|
358
380
|
const base = o.lab.replace(/\/+$/, "");
|
|
359
|
-
|
|
381
|
+
// Resolve --workspace to the slug every call carries; the lab's pipelines,
|
|
382
|
+
// Sources and eval groups are per-workspace, so the whole run reads and
|
|
383
|
+
// queues inside it. Without it, the lab's default workspace is used and the
|
|
384
|
+
// X-Workspace header is omitted, so existing CI is unchanged.
|
|
385
|
+
let workspace = null;
|
|
386
|
+
if (o.workspace != null) {
|
|
387
|
+
const { workspaces } = await labApi(base, env)("GET", "/api/workspaces");
|
|
388
|
+
workspace = workspaceNamed(Array.isArray(workspaces) ? workspaces : [], o.workspace, base).slug;
|
|
389
|
+
}
|
|
390
|
+
const call = labApi(base, env, workspace);
|
|
360
391
|
|
|
361
392
|
// What the page reads before it submits: the lab's pipelines and Target
|
|
362
393
|
// profiles, the Source the pipeline reads and the eval groups it may link.
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.8.0 (2026.10.07-484)
|
package/lab/evals-core.mjs
CHANGED
|
@@ -657,6 +657,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
657
657
|
|
|
658
658
|
|
|
659
659
|
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
|
|
660
663
|
|
|
661
664
|
|
|
662
665
|
|
|
@@ -667,9 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
667
670
|
|
|
668
671
|
/** What a metric needs besides the reply: a model to grade with. */
|
|
669
672
|
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
|
|
673
677
|
|
|
674
678
|
|
|
675
679
|
|
|
@@ -723,6 +727,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
723
727
|
|
|
724
728
|
|
|
725
729
|
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
|
|
726
737
|
|
|
727
738
|
|
|
728
739
|
/** What an earlier eval that failed and does not continue leaves a later one. */
|
|
@@ -797,6 +808,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
797
808
|
|
|
798
809
|
|
|
799
810
|
|
|
811
|
+
|
|
812
|
+
|
|
813
|
+
|
|
800
814
|
|
|
801
815
|
|
|
802
816
|
|
|
@@ -881,9 +895,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
881
895
|
|
|
882
896
|
|
|
883
897
|
|
|
898
|
+
|
|
884
899
|
|
|
885
|
-
|
|
886
|
-
|
|
900
|
+
|
|
887
901
|
|
|
888
902
|
|
|
889
903
|
/** Lookups, and whether it is a run document being validated. */
|
|
@@ -1041,7 +1055,7 @@ const platformOf = (src
|
|
|
1041
1055
|
|
|
1042
1056
|
|
|
1043
1057
|
|
|
1044
|
-
|
|
1058
|
+
|
|
1045
1059
|
|
|
1046
1060
|
|
|
1047
1061
|
|
|
@@ -1323,6 +1337,10 @@ function datasetRules(body ) {
|
|
|
1323
1337
|
|
|
1324
1338
|
|
|
1325
1339
|
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
|
|
1343
|
+
|
|
1326
1344
|
|
|
1327
1345
|
|
|
1328
1346
|
/** An HTTP request as a connection type builds it, for the relay to send. */
|
|
@@ -1346,6 +1364,10 @@ function datasetRules(body ) {
|
|
|
1346
1364
|
|
|
1347
1365
|
|
|
1348
1366
|
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
|
|
1349
1371
|
|
|
1350
1372
|
|
|
1351
1373
|
|
|
@@ -1354,8 +1376,9 @@ function datasetRules(body ) {
|
|
|
1354
1376
|
|
|
1355
1377
|
|
|
1356
1378
|
|
|
1357
|
-
|
|
1358
|
-
|
|
1379
|
+
|
|
1380
|
+
|
|
1381
|
+
|
|
1359
1382
|
|
|
1360
1383
|
|
|
1361
1384
|
|
|
@@ -1708,6 +1731,8 @@ function ollamaType(id , label , cloud ) {
|
|
|
1708
1731
|
body: { model: conn.model, messages: [message], options, stream: false },
|
|
1709
1732
|
};
|
|
1710
1733
|
},
|
|
1734
|
+
// Ollama's own: the schema is the format.
|
|
1735
|
+
structured: (body, schema) => ({ ...body, format: schema }),
|
|
1711
1736
|
parseReply(j) {
|
|
1712
1737
|
const r = j ;
|
|
1713
1738
|
return { raw: r?.message?.content ?? "", finishReason: r?.done_reason ?? null };
|
|
@@ -1754,7 +1779,7 @@ function recordedRaw(record ) {
|
|
|
1754
1779
|
// the flow reads a reply; `local` is the no-Read-as fallback.
|
|
1755
1780
|
CONNECTION_TYPES.recorded = {
|
|
1756
1781
|
id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
|
|
1757
|
-
description: "
|
|
1782
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
1758
1783
|
local: (_item, _sent, record) => recordedRaw(record),
|
|
1759
1784
|
request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
|
|
1760
1785
|
};
|
|
@@ -1769,19 +1794,25 @@ function localAnswer(conn , text , sent , record
|
|
|
1769
1794
|
return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
|
|
1770
1795
|
}
|
|
1771
1796
|
|
|
1797
|
+
/** A chat completions body asking for a reply held to [schema]: OpenAI's
|
|
1798
|
+
response_format, which llama-server and most compatible servers take too. */
|
|
1799
|
+
const chatSchema = (body , schema ) =>
|
|
1800
|
+
({ ...body, response_format: { type: "json_schema", json_schema: { name: "reply", strict: true, schema } } });
|
|
1801
|
+
|
|
1772
1802
|
CONNECTION_TYPES["openai-compatible"] = {
|
|
1773
1803
|
id: "openai-compatible", label: "OpenAI-compatible",
|
|
1774
1804
|
description: "OpenAI, OpenRouter, vLLM, LM Studio: any chat completions endpoint.",
|
|
1775
1805
|
settings: [
|
|
1776
|
-
|
|
1806
|
+
// A reasoning model refuses a temperature.
|
|
1807
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1808
|
+
appliesTo: (conn) => !/^(gpt-5|o\d)/i.test(String(conn.model ?? "")) },
|
|
1777
1809
|
{ key: "nPredict", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1778
1810
|
],
|
|
1779
|
-
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1780
|
-
//
|
|
1781
|
-
//
|
|
1811
|
+
// /v1/chat/completions, covering OpenAI, OpenRouter, vLLM, LM Studio.
|
|
1812
|
+
// max_completion_tokens stays with this type, as the hosted providers
|
|
1813
|
+
// reject the fields a llama.cpp server takes.
|
|
1782
1814
|
request(conn, prompt, dataUrl, base, key) {
|
|
1783
1815
|
const hosted = isHostedUrl(base);
|
|
1784
|
-
const isReasoning = /^(gpt-5|o\d)/i.test(String(conn.model ?? ""));
|
|
1785
1816
|
const temp = asNumber(conn.temperature);
|
|
1786
1817
|
const predict = asNumber(conn.nPredict);
|
|
1787
1818
|
return {
|
|
@@ -1789,7 +1820,7 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1789
1820
|
headers: bearer(key),
|
|
1790
1821
|
body: {
|
|
1791
1822
|
model: conn.model,
|
|
1792
|
-
...(
|
|
1823
|
+
...(temp == null ? {} : { temperature: temp }),
|
|
1793
1824
|
...(predict == null ? {} : hosted ? { max_completion_tokens: predict } : { max_tokens: predict }),
|
|
1794
1825
|
messages: [{ role: "user", content: [
|
|
1795
1826
|
{ type: "text", text: prompt },
|
|
@@ -1798,15 +1829,23 @@ CONNECTION_TYPES["openai-compatible"] = {
|
|
|
1798
1829
|
},
|
|
1799
1830
|
};
|
|
1800
1831
|
},
|
|
1832
|
+
structured: chatSchema,
|
|
1801
1833
|
parseReply: chatReply,
|
|
1802
1834
|
listModels, parseModels: idsFrom,
|
|
1803
1835
|
};
|
|
1804
1836
|
|
|
1837
|
+
/** The Claude models that refuse a sampling setting: Opus from 4.7, and
|
|
1838
|
+
every Opus, Sonnet, Fable and Mythos from 5. Older ones (Opus 4.6, Sonnet
|
|
1839
|
+
4.6, Haiku 4.5) take one. */
|
|
1840
|
+
const CLAUDE_FIXED_SAMPLING = /claude-(opus-4-[7-9]|(opus|sonnet|fable|mythos)-[5-9])/i;
|
|
1841
|
+
|
|
1805
1842
|
CONNECTION_TYPES.anthropic = {
|
|
1806
1843
|
id: "anthropic", label: "Anthropic",
|
|
1807
1844
|
description: "Claude, through Anthropic's Messages API.",
|
|
1845
|
+
url: "https://api.anthropic.com",
|
|
1808
1846
|
settings: [
|
|
1809
|
-
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: ""
|
|
1847
|
+
{ key: "temperature", label: "Temperature", control: "text", default: "", hint: "",
|
|
1848
|
+
appliesTo: (conn) => !CLAUDE_FIXED_SAMPLING.test(String(conn.model ?? "")) },
|
|
1810
1849
|
{ key: "maxTokens", label: "Reply tokens", control: "text", default: "", hint: "" },
|
|
1811
1850
|
],
|
|
1812
1851
|
// The Messages API: the x-api-key and anthropic-version headers, base64
|
|
@@ -1831,6 +1870,9 @@ CONNECTION_TYPES.anthropic = {
|
|
|
1831
1870
|
},
|
|
1832
1871
|
};
|
|
1833
1872
|
},
|
|
1873
|
+
// The Messages API's structured outputs. A model that cannot hold one
|
|
1874
|
+
// refuses the request, and the grader asks again without it.
|
|
1875
|
+
structured: (body, schema) => ({ ...body, output_config: { ...(body.output_config ), format: { type: "json_schema", schema } } }),
|
|
1834
1876
|
parseReply(j) {
|
|
1835
1877
|
const r = j ;
|
|
1836
1878
|
const raw = (r?.content || []).filter(b => b.type === "text").map(b => b.text).join("");
|
|
@@ -1903,6 +1945,8 @@ CONNECTION_TYPES["llama.cpp"] = {
|
|
|
1903
1945
|
},
|
|
1904
1946
|
};
|
|
1905
1947
|
},
|
|
1948
|
+
// llama-server takes OpenAI's response_format, a schema and all.
|
|
1949
|
+
structured: chatSchema,
|
|
1906
1950
|
parseReply: chatReply,
|
|
1907
1951
|
listModels, parseModels: idsFrom,
|
|
1908
1952
|
// A reply that arrived is still a failure when this profile asked for the
|
|
@@ -2098,12 +2142,14 @@ function extraHeaders(raw ) {
|
|
|
2098
2142
|
}
|
|
2099
2143
|
return out;
|
|
2100
2144
|
}
|
|
2101
|
-
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2145
|
+
/** An address as a whole-request type hangs paths off it: its origin, no
|
|
2146
|
+
path. A step's path is the whole path production called, so an address
|
|
2147
|
+
entered with one (`…/v1/messages`) would send it twice. */
|
|
2102
2148
|
function originBase(raw ) {
|
|
2103
|
-
let s = String(raw || "").trim()
|
|
2149
|
+
let s = String(raw || "").trim();
|
|
2104
2150
|
if (!s) return "";
|
|
2105
2151
|
if (!/^https?:\/\//.test(s)) s = "https://" + s;
|
|
2106
|
-
return s;
|
|
2152
|
+
try { return new URL(s).origin; } catch { return s.replace(/\/+$/, ""); }
|
|
2107
2153
|
}
|
|
2108
2154
|
const wholeRequestsOnly = () => { throw new Error("an HTTP endpoint is sent an HTTP Request step's request, not a prompt"); };
|
|
2109
2155
|
CONNECTION_TYPES.http = {
|
|
@@ -2230,9 +2276,14 @@ function splitOllama(p ) {
|
|
|
2230
2276
|
* connection carries its type's settings under `options`, so they are merged
|
|
2231
2277
|
* up before the type reads them.
|
|
2232
2278
|
*/
|
|
2233
|
-
function connectionRequest(conn , prompt , dataUrl , base , key
|
|
2279
|
+
function connectionRequest(conn , prompt , dataUrl , base , key ,
|
|
2280
|
+
schema ) {
|
|
2234
2281
|
const flat = { ...conn, ...(conn?.options || {}) };
|
|
2235
|
-
|
|
2282
|
+
const type = typeEntry(flat);
|
|
2283
|
+
// A setting the model refuses is not sent, whatever the profile holds.
|
|
2284
|
+
for (const s of type.settings) if (s.appliesTo && !s.appliesTo(flat)) delete flat[s.key];
|
|
2285
|
+
const req = type.request(flat, prompt, dataUrl, base, key);
|
|
2286
|
+
return schema && type.structured ? { ...req, body: type.structured(req.body , schema) } : req;
|
|
2236
2287
|
}
|
|
2237
2288
|
|
|
2238
2289
|
/** Where [conn]'s requests hang off: its type's own reading of its address,
|
|
@@ -2934,6 +2985,15 @@ function registerKinds(k ) {
|
|
|
2934
2985
|
|
|
2935
2986
|
|
|
2936
2987
|
|
|
2988
|
+
|
|
2989
|
+
|
|
2990
|
+
|
|
2991
|
+
|
|
2992
|
+
|
|
2993
|
+
|
|
2994
|
+
|
|
2995
|
+
|
|
2996
|
+
|
|
2937
2997
|
|
|
2938
2998
|
|
|
2939
2999
|
function pluginHost(pluginId ) {
|
|
@@ -2961,6 +3021,14 @@ function pluginHost(pluginId ) {
|
|
|
2961
3021
|
taken(CONNECTION_TYPES, "connection type", [t.id]);
|
|
2962
3022
|
CONNECTION_TYPES[t.id] = t;
|
|
2963
3023
|
},
|
|
3024
|
+
registerWorkflowPlatform(t) {
|
|
3025
|
+
taken(WORKFLOW_PLATFORMS, "workflow platform", [t.id]);
|
|
3026
|
+
WORKFLOW_PLATFORMS[t.id] = t;
|
|
3027
|
+
},
|
|
3028
|
+
registerWizard(w) {
|
|
3029
|
+
taken(WIZARDS, "wizard", [w.id]);
|
|
3030
|
+
WIZARDS[w.id] = w;
|
|
3031
|
+
},
|
|
2964
3032
|
};
|
|
2965
3033
|
}
|
|
2966
3034
|
|
|
@@ -3058,6 +3126,13 @@ function withSlugs (list
|
|
|
3058
3126
|
/** Whether [s] is a slug as a profile may hold one. */
|
|
3059
3127
|
const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
|
|
3060
3128
|
|
|
3129
|
+
/** Where [p] connects: its address, or blank, its type's own -- an Anthropic
|
|
3130
|
+
profile left blank reaches Anthropic. Blank for a type with none means the
|
|
3131
|
+
lab's own Ollama, which the runner and the relay fill in. */
|
|
3132
|
+
function addressOf(p ) {
|
|
3133
|
+
return String(p.url ?? "").trim() || CONNECTION_TYPES[typeOf(p )]?.url || "";
|
|
3134
|
+
}
|
|
3135
|
+
|
|
3061
3136
|
/** A stored profile as a run carries it: its request settings, no key. */
|
|
3062
3137
|
function connectionOf(p ) {
|
|
3063
3138
|
const type = typeOf(p);
|
|
@@ -3066,7 +3141,7 @@ function connectionOf(p ) {
|
|
|
3066
3141
|
// temperature is a common field, carried at the top like px/format/quality;
|
|
3067
3142
|
// the options bag holds the type's own settings only.
|
|
3068
3143
|
for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
|
|
3069
|
-
const out = { name: p.name || "unnamed", url: p
|
|
3144
|
+
const out = { name: p.name || "unnamed", url: addressOf(p), model: p.model || "", type };
|
|
3070
3145
|
if (isSlug(p.slug)) out.slug = p.slug;
|
|
3071
3146
|
for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
|
|
3072
3147
|
if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
|
|
@@ -3187,6 +3262,70 @@ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapt
|
|
|
3187
3262
|
render: body => JSON.stringify(body, null, 2),
|
|
3188
3263
|
};
|
|
3189
3264
|
|
|
3265
|
+
// ---- Setup Wizards (docs/workflow-sources.md § Wizards) ---------------------
|
|
3266
|
+
//
|
|
3267
|
+
// A wizard is a registry entry whose steps each say where they happen and a
|
|
3268
|
+
// done-when predicate on lab state. Steps tick automatically: `done(lab)` reads
|
|
3269
|
+
// the cheap state the page already holds (a workflow Source exists, a pipeline
|
|
3270
|
+
// with a Recorded target exists, a run of it is in History), never a poll. One
|
|
3271
|
+
// wizard runs at a time, held in the browser-only `promptlab.wizard` store; a
|
|
3272
|
+
// header pill shows its progress and the Setup › Wizards section draws the same
|
|
3273
|
+
// walk-through. The registry lives here so a plugin registers a wizard through
|
|
3274
|
+
// the same host as every other kind (PluginHost.registerWizard, phase 7); the
|
|
3275
|
+
// page ships the built-in "Test a workflow" and owns its routes and selectors
|
|
3276
|
+
// (web/src/app/wizards.ts). The predicates run with the page's privileges and
|
|
3277
|
+
// only read lab state, so a plugin wizard is no new trust (docs/packs.md).
|
|
3278
|
+
|
|
3279
|
+
/** The slice of lab state a wizard step's `done` reads: what the page already
|
|
3280
|
+
holds, so a tick costs a predicate and not a request. A reader names a field
|
|
3281
|
+
it needs; a wizard that reads more declares it here. */
|
|
3282
|
+
|
|
3283
|
+
|
|
3284
|
+
|
|
3285
|
+
|
|
3286
|
+
|
|
3287
|
+
|
|
3288
|
+
|
|
3289
|
+
|
|
3290
|
+
|
|
3291
|
+
|
|
3292
|
+
|
|
3293
|
+
|
|
3294
|
+
|
|
3295
|
+
|
|
3296
|
+
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
|
|
3300
|
+
|
|
3301
|
+
|
|
3302
|
+
|
|
3303
|
+
|
|
3304
|
+
|
|
3305
|
+
|
|
3306
|
+
|
|
3307
|
+
|
|
3308
|
+
|
|
3309
|
+
|
|
3310
|
+
|
|
3311
|
+
|
|
3312
|
+
|
|
3313
|
+
|
|
3314
|
+
|
|
3315
|
+
|
|
3316
|
+
const WIZARDS = Object.create(null);
|
|
3317
|
+
|
|
3318
|
+
/** How far a wizard has got on [lab]: each step's tick, how many are done, and
|
|
3319
|
+
the first step not yet done -- the one the checklist marks current and Go
|
|
3320
|
+
to step heads for. `current` is -1 once every step is done. */
|
|
3321
|
+
function wizardProgress(entry , lab )
|
|
3322
|
+
|
|
3323
|
+
{
|
|
3324
|
+
const ticks = entry.steps.map((s) => s.done(lab));
|
|
3325
|
+
const done = ticks.filter(Boolean).length;
|
|
3326
|
+
return { ticks, done, total: entry.steps.length, current: ticks.findIndex((t) => !t) };
|
|
3327
|
+
}
|
|
3328
|
+
|
|
3190
3329
|
CONTENT_TYPES.source = {
|
|
3191
3330
|
label: "Source",
|
|
3192
3331
|
description: "Every file or record in a Source from the Library.",
|
|
@@ -3364,6 +3503,7 @@ function metricInput(res , kase , more )
|
|
|
3364
3503
|
const last = res.transcript?.at(-1);
|
|
3365
3504
|
const text = last?.got ?? res.raw ?? "";
|
|
3366
3505
|
return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
|
|
3506
|
+
asked: last?.sent ?? null,
|
|
3367
3507
|
// A case that has been given its own expected reply ("Take B as
|
|
3368
3508
|
// expected") is compared with that; otherwise the Source's recorded
|
|
3369
3509
|
// reply for the item.
|
|
@@ -3382,7 +3522,7 @@ function runInput(ress , plain )
|
|
|
3382
3522
|
terms.push(...(r.terms || []));
|
|
3383
3523
|
ms += r.ms ?? 0;
|
|
3384
3524
|
}
|
|
3385
|
-
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
|
|
3525
|
+
return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, asked: null, kase: null };
|
|
3386
3526
|
}
|
|
3387
3527
|
|
|
3388
3528
|
/** One metric's reading of [input]: `of`, `not` and a failure all applied,
|
|
@@ -3643,16 +3783,21 @@ EVAL_TYPES.group = {
|
|
|
3643
3783
|
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3644
3784
|
.filter(Boolean).join(" ") })),
|
|
3645
3785
|
profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
|
|
3646
|
-
// A run carries the
|
|
3647
|
-
//
|
|
3648
|
-
//
|
|
3786
|
+
// A run carries the grader each group asks, so its profile goes with the
|
|
3787
|
+
// run: a link's is the Library group's own, or the lab's where it names
|
|
3788
|
+
// none -- the run's copy of the link carries it. A private group takes the
|
|
3789
|
+
// lab's where it names none and may ask one: a model-graded metric of its
|
|
3790
|
+
// own, or a case's.
|
|
3649
3791
|
resolve(t, ctx){
|
|
3650
|
-
|
|
3651
|
-
const
|
|
3792
|
+
const ref = (g ) => (isRef(g) ? { id: g.id, name: g.name } : null);
|
|
3793
|
+
const lab = ref(ctx.grader);
|
|
3652
3794
|
if (isObj(t.own)) {
|
|
3653
|
-
return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3795
|
+
return lab && !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom))
|
|
3796
|
+
? { ...t, own: { ...t.own, grader: lab } } : t;
|
|
3654
3797
|
}
|
|
3655
|
-
|
|
3798
|
+
if (!isRef(t.group) || isRef(t.grader)) return t;
|
|
3799
|
+
const grader = ref(ctx.groups?.find(g => g.id === (t.group ).id)?.grader) ?? lab;
|
|
3800
|
+
return grader ? { ...t, grader } : t;
|
|
3656
3801
|
},
|
|
3657
3802
|
wholeRun: wholeRunGroup,
|
|
3658
3803
|
verdict: (t, ress, kind) => {
|
|
@@ -3849,6 +3994,9 @@ function metricSummary(m ) {
|
|
|
3849
3994
|
const v = m[o.key];
|
|
3850
3995
|
if (o.type === "select") return o.choices.find(c => c.value === v)?.label ?? "";
|
|
3851
3996
|
if (o.type === "checkbox") return !!v === !!fresh[o.key] ? "" : v ? o.label.toLowerCase() : `${o.label.toLowerCase()} off`;
|
|
3997
|
+
// The choices ticked, by their labels, where they are not a new one's.
|
|
3998
|
+
if (o.type === "multi") return !Array.isArray(v) || JSON.stringify(v) === JSON.stringify(fresh[o.key]) ? ""
|
|
3999
|
+
: o.choices.filter(c => v.includes(c.value)).map(c => c.label).join(" or ");
|
|
3852
4000
|
// Text that runs to lines (a schema) reads as its first words, run together.
|
|
3853
4001
|
return v == null || isObj(v) ? "" : String(v).replace(/\s+/g, " ").trim().slice(0, 40);
|
|
3854
4002
|
}).filter(Boolean);
|
|
@@ -4014,7 +4162,7 @@ STEP_TYPES.echo = {
|
|
|
4014
4162
|
// the flow reads one. No profile, no prompt -- the record is the reply.
|
|
4015
4163
|
STEP_TYPES.recorded = {
|
|
4016
4164
|
label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
|
|
4017
|
-
description: "
|
|
4165
|
+
description: "Answers each call with the reply it got when it was recorded. Sends nothing.",
|
|
4018
4166
|
firstJobOnly: "a call's recorded reply is job 1's",
|
|
4019
4167
|
in: "item", out: "text",
|
|
4020
4168
|
apply: "runPipeline",
|
|
@@ -5342,7 +5490,8 @@ function exportBundle(doc , ctx
|
|
|
5342
5490
|
// What the lab would supply at submit, written down: no lab supplies it later.
|
|
5343
5491
|
const out = clone(doc) ;
|
|
5344
5492
|
out.evals = (Array.isArray(out.evals) ? out.evals : [])
|
|
5345
|
-
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null
|
|
5493
|
+
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null,
|
|
5494
|
+
groups: Object.entries(ctx.datasets ?? {}).map(([id, d]) => ({ id, name: d.name, grader: d.body.grader ?? null })) }) ?? t);
|
|
5346
5495
|
const table = {};
|
|
5347
5496
|
for (const id of profileIds(out)) {
|
|
5348
5497
|
const p = profiles.find(x => x.id === id);
|
|
@@ -5883,8 +6032,8 @@ export {
|
|
|
5883
6032
|
scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
|
|
5884
6033
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
5885
6034
|
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
|
|
5886
|
-
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
|
|
5887
|
-
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
6035
|
+
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf, WIZARDS, wizardProgress,
|
|
6036
|
+
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, addressOf, connectionSettings,
|
|
5888
6037
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5889
6038
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5890
6039
|
contentOf, withContent, replyOf,
|