evals-lab 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/README.md +1 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +1 -1
- package/lab/demo/pipelines/demo-2.json +1 -1
- package/lab/evals-core.mjs +428 -49
- package/lab/kinds/list.mjs +258 -14
- package/lab/metrics/builtin.mjs +76 -20
- package/lab/run-evals.js +42 -7
- package/lab/server.py +1200 -31
- package/lab/web/dist/assets/gallery-BpR9b4EM.js +3 -0
- package/lab/web/dist/assets/main-BMRmyJvo.css +1 -0
- package/lab/web/dist/assets/main-BfiFtaYW.js +19 -0
- package/lab/web/dist/assets/tokens-C67OCDd-.css +1 -0
- package/lab/web/dist/assets/tokens-DzZqM5IZ.js +59 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +15 -2
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +0 -3
- package/lab/web/dist/assets/main-BQL5j5oF.js +0 -20
- package/lab/web/dist/assets/main-Cza2gwQd.css +0 -1
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +0 -61
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,116 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.9.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- **Copy your data directory before upgrading** (see "User data" in the
|
|
13
|
+
README). A pipeline 0.9.0 saves is pipeline version 14, which 0.8.x
|
|
14
|
+
cannot read. To go back, reinstall 0.8.0 and restore the copy.
|
|
15
|
+
- A job's Drop items and Reject the answer rules become one private filter
|
|
16
|
+
set in that job, which keeps and rejects exactly what they did. Pipelines
|
|
17
|
+
read as version 14, and are saved as it at their next edit.
|
|
18
|
+
- On first start, 0.9.0 adds the tables filter sets are kept in, and each
|
|
19
|
+
run keeps the filter sets it ran with.
|
|
20
|
+
|
|
21
|
+
### Added
|
|
22
|
+
|
|
23
|
+
- Library › Filter sets: named, versioned lists of filters that clean a list
|
|
24
|
+
reply. A Drop item filter drops the items its conditions match; a Reject
|
|
25
|
+
reply filter rejects the whole reply. Filters run in their list's order.
|
|
26
|
+
- A job's Responses stage links filter sets, following each one's latest
|
|
27
|
+
version or pinned to one, or holds a private set of its own.
|
|
28
|
+
- Each Target profile shows its connection status: Not tested, Connected,
|
|
29
|
+
Key needed or Unreachable, checked quietly when Setup opens. The dot opens
|
|
30
|
+
the Test connection result.
|
|
31
|
+
- Target profiles list each profile's Model and what uses it, with quick
|
|
32
|
+
filters: All, Connected, Needs attention and Grader. A profile's edit
|
|
33
|
+
sheet links the pipelines and eval groups that use it.
|
|
34
|
+
- Selected profiles can be tested, or have their workspaces changed, all
|
|
35
|
+
at once.
|
|
36
|
+
- An HTTP Request target's prompt can be edited. Insert field… adds one of
|
|
37
|
+
the item's fields, and the preview highlights the words changed from the
|
|
38
|
+
recorded prompt.
|
|
39
|
+
- The Recorded target shows the flow's original prompt.
|
|
40
|
+
|
|
41
|
+
### Changed
|
|
42
|
+
|
|
43
|
+
- Target profiles no longer show Type, Address and Key columns, or the
|
|
44
|
+
Type, Key and Sort selects.
|
|
45
|
+
- The Edited target's request template is in Flow's request, folded until
|
|
46
|
+
opened.
|
|
47
|
+
- Drop items and Reject the answer are no longer offered under a job's
|
|
48
|
+
Responses: link a filter set instead.
|
|
49
|
+
|
|
50
|
+
## 0.8.0
|
|
51
|
+
|
|
52
|
+
### Upgrade notes
|
|
53
|
+
|
|
54
|
+
- No stored document changes format: pipelines stay version 13, and eval
|
|
55
|
+
groups and exports stay version 8, so 0.7.0 still reads what 0.8.0 saves.
|
|
56
|
+
- On first start, 0.8.0 adds a table recording which workspaces may use each
|
|
57
|
+
Target profile and account. Nothing is narrowed: every connection stays
|
|
58
|
+
shared with every workspace until you change it.
|
|
59
|
+
|
|
60
|
+
### Added
|
|
61
|
+
|
|
62
|
+
- Several workspaces. Make, rename and archive them in Setup › Workspaces,
|
|
63
|
+
and switch between them from the header's workspace menu. Each one has its
|
|
64
|
+
own address (`/w/<slug>`) and its own datasets, prompts, Sources, pipelines
|
|
65
|
+
and runs. The theme choice (Dark, Light, Match system) has moved into that
|
|
66
|
+
menu.
|
|
67
|
+
- Target profiles and accounts can be shared with every workspace, with some,
|
|
68
|
+
or with each new one. A run in a workspace a profile isn't shared with says
|
|
69
|
+
so and offers to share it.
|
|
70
|
+
- `evals-lab run --lab … --workspace <slug>` runs a lab's pipeline in a named
|
|
71
|
+
workspace. Without it, the default workspace is used, as before.
|
|
72
|
+
- Setup › Wizards, and a header pill showing a wizard's progress. The first
|
|
73
|
+
wizard, Test a Power Automate workflow, ticks off each step as you do it.
|
|
74
|
+
Plugins can add workflow platforms and wizards.
|
|
75
|
+
- Judged against recorded has an **Expected** setting: tick which of Better,
|
|
76
|
+
Same and Worse pass (Better and Same by default, as before).
|
|
77
|
+
- A Recorded target's menu has **Preview replies…**, showing what each call
|
|
78
|
+
recorded.
|
|
79
|
+
- Test… asks for a grader when a check needs one. You can create the test
|
|
80
|
+
without one and add it later.
|
|
81
|
+
|
|
82
|
+
### Changed
|
|
83
|
+
|
|
84
|
+
- Model-graded checks no longer need a prompt written. The grader is told
|
|
85
|
+
what the target was asked; Judged against recorded's Task is now optional
|
|
86
|
+
Additional guidance.
|
|
87
|
+
- A grader's answer is held to a fixed JSON shape where its API supports
|
|
88
|
+
that (Anthropic, OpenAI-compatible, llama.cpp, Ollama). An answer that
|
|
89
|
+
still can't be read is asked for once more, then reported as a grader
|
|
90
|
+
error rather than a verdict.
|
|
91
|
+
- Test… on a step whose reply is plain words judges the reply against the
|
|
92
|
+
recorded one, instead of requiring the exact same words.
|
|
93
|
+
- Rubric's and Factual's threshold reads **Expected: score at least**.
|
|
94
|
+
- New profile: Anthropic and Ollama (Cloud) no longer ask for an address. A
|
|
95
|
+
proxy or gateway address goes under Advanced. Temperature is hidden, and
|
|
96
|
+
never sent, for models that reject it (Claude Opus 4.7 and later, the
|
|
97
|
+
5-series Claude models, OpenAI reasoning models). Image settings are now
|
|
98
|
+
the folded Vision Capabilities group.
|
|
99
|
+
- The Test… dialog is "Create new pipeline test", with a Create button.
|
|
100
|
+
- Library › Sources › a flow: Actions come before Calls, and each call's
|
|
101
|
+
request reads In sync or Out of sync.
|
|
102
|
+
- Run results: Take B as expected is in each call's row menu, and the
|
|
103
|
+
Results table no longer has checkboxes.
|
|
104
|
+
- The wizard pill's button is Open wizard, which shows the wizard's steps.
|
|
105
|
+
|
|
106
|
+
### Fixed
|
|
107
|
+
|
|
108
|
+
- An HTTP endpoint profile whose address included a path (such as
|
|
109
|
+
`https://api.anthropic.com/v1/messages`) sent the path twice, so every
|
|
110
|
+
call came back Not found.
|
|
111
|
+
- An Anthropic profile with a blank address was sent to the lab's own Ollama.
|
|
112
|
+
- A run said "this run has none" for a grader that its Library eval group
|
|
113
|
+
named.
|
|
114
|
+
- The wizard pill didn't tick a step until the page was reloaded.
|
|
115
|
+
- A Microsoft 365 sign-in now shows which tenants it reaches, so a lab that
|
|
116
|
+
works across tenants is not tied to one.
|
|
117
|
+
|
|
8
118
|
## 0.7.0
|
|
9
119
|
|
|
10
120
|
### Upgrade notes
|
package/README.md
CHANGED
|
@@ -11,6 +11,7 @@ Runs are kept in history which can be exported.
|
|
|
11
11
|
Currently in early release. Runs locally and doesn't send your data anywhere else.
|
|
12
12
|
|
|
13
13
|
Example use-cases:
|
|
14
|
+
- Keywording images using a model's vision capability, and running evals on the tags it returns
|
|
14
15
|
- Providing images to an LLM and comparing responses for accuracy
|
|
15
16
|
- Checking which model can meet evals in a Power Automate workflow
|
|
16
17
|
- Determining which prompt gets the highest score in metrics
|
package/bin/run.js
CHANGED
|
@@ -51,6 +51,8 @@ const USAGE = `Usage: evals-lab run <bundle-dir | pipeline.yaml> [options]
|
|
|
51
51
|
and their keys, and the run in its History. LAB_PASSWORD
|
|
52
52
|
is sent when it is set.
|
|
53
53
|
--pipeline <name> the lab's pipeline to run, by its name or its id.
|
|
54
|
+
--workspace <slug> the lab's workspace to run in, by its /w/<slug> address.
|
|
55
|
+
Default: the lab's default workspace.
|
|
54
56
|
--wait wait for the lab's run to finish, and exit with its
|
|
55
57
|
verdict. Without it the run's id is printed once it is
|
|
56
58
|
queued. --json, --junit, --summary, --min-pass and
|
|
@@ -70,9 +72,9 @@ class Refused extends Error {
|
|
|
70
72
|
// What the command line asks for, or a Refused saying what is wrong with it.
|
|
71
73
|
function runOptions(argv) {
|
|
72
74
|
const o = { bundle: null, items: null, json: null, junit: null, summary: null, minPass: null, progress: false,
|
|
73
|
-
lab: null, pipeline: null, wait: false };
|
|
75
|
+
lab: null, pipeline: null, workspace: null, wait: false };
|
|
74
76
|
const value = { "--items": "items", "--json": "json", "--junit": "junit", "--summary": "summary", "--min-pass": "minPass",
|
|
75
|
-
"--lab": "lab", "--pipeline": "pipeline" };
|
|
77
|
+
"--lab": "lab", "--pipeline": "pipeline", "--workspace": "workspace" };
|
|
76
78
|
for (let i = 0; i < argv.length; i++) {
|
|
77
79
|
const a = argv[i];
|
|
78
80
|
if (a === "--help" || a === "-h") o.help = true;
|
|
@@ -96,6 +98,8 @@ function runOptions(argv) {
|
|
|
96
98
|
if (waits.length && !o.wait) {
|
|
97
99
|
throw new Refused(`--${waits[0].replace("minPass", "min-pass")} reads the finished run: add --wait`, true);
|
|
98
100
|
}
|
|
101
|
+
} else if (o.workspace != null) {
|
|
102
|
+
throw new Refused("--workspace is for a lab's run: name the lab with --lab", true);
|
|
99
103
|
} else if (o.wait) {
|
|
100
104
|
throw new Refused("--wait is for a lab's run: a bundle's always waits", true);
|
|
101
105
|
} else if (o.bundle == null) {
|
|
@@ -276,15 +280,18 @@ const POLL_MISSES = 20;
|
|
|
276
280
|
|
|
277
281
|
// The lab's API at [base], as [env]'s LAB_PASSWORD opens it: HTTP Basic, the
|
|
278
282
|
// password alone, as server.py takes it. A refusal says what the lab said.
|
|
279
|
-
|
|
283
|
+
// [workspace], when given, is the slug every call carries as X-Workspace, so
|
|
284
|
+
// the run reads and queues in that workspace rather than the lab's default.
|
|
285
|
+
function labApi(base, env, workspace) {
|
|
280
286
|
const auth = env.LAB_PASSWORD
|
|
281
287
|
? { Authorization: `Basic ${Buffer.from(`:${env.LAB_PASSWORD}`).toString("base64")}` } : {};
|
|
288
|
+
const scope = workspace ? { "X-Workspace": workspace } : {};
|
|
282
289
|
return async (method, route, body, { missing = false } = {}) => {
|
|
283
290
|
let res;
|
|
284
291
|
try {
|
|
285
292
|
res = await fetch(base + route, {
|
|
286
293
|
method, redirect: "manual",
|
|
287
|
-
headers: { ...auth, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
294
|
+
headers: { ...auth, ...scope, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
288
295
|
...(body !== undefined ? { body: JSON.stringify(body) } : {}),
|
|
289
296
|
});
|
|
290
297
|
} catch (e) {
|
|
@@ -319,6 +326,21 @@ function pipelineNamed(saved, name, base) {
|
|
|
319
326
|
return found[0];
|
|
320
327
|
}
|
|
321
328
|
|
|
329
|
+
// The lab's workspace --workspace [slug] names, by its /w/<slug> address -- or
|
|
330
|
+
// a Refused naming the slugs it has, as pipelineNamed does for pipelines. The
|
|
331
|
+
// server falls back to its default for an unknown slug, so the CLI checks the
|
|
332
|
+
// slug itself; an archived workspace cannot be run against.
|
|
333
|
+
function workspaceNamed(workspaces, slug, base) {
|
|
334
|
+
const active = workspaces.filter(w => w && !w.archived);
|
|
335
|
+
const found = active.find(w => w.slug === slug);
|
|
336
|
+
if (found) return found;
|
|
337
|
+
if (workspaces.some(w => w && w.slug === slug)) {
|
|
338
|
+
throw new Refused(`${base}'s workspace at /w/${slug} is archived`);
|
|
339
|
+
}
|
|
340
|
+
throw new Refused(`${base} has no workspace at /w/${slug}`
|
|
341
|
+
+ (active.length ? `: it has ${active.map(w => JSON.stringify(w.slug)).join(", ")}` : ""));
|
|
342
|
+
}
|
|
343
|
+
|
|
322
344
|
// The row's verdicts with --min-pass relaxing each eval read item by item,
|
|
323
345
|
// as the worker's --min-pass does: it only ever turns a fail into a pass.
|
|
324
346
|
const relaxed = (verdicts, minPass) => minPass == null ? verdicts : verdicts.map(evals =>
|
|
@@ -356,7 +378,16 @@ function rowSuites(core, run, verdicts, items) {
|
|
|
356
378
|
/** [o]'s pipeline, queued on its lab; with --wait, followed to its verdict. */
|
|
357
379
|
async function runOnLab(core, o, { env, stdout, say }) {
|
|
358
380
|
const base = o.lab.replace(/\/+$/, "");
|
|
359
|
-
|
|
381
|
+
// Resolve --workspace to the slug every call carries; the lab's pipelines,
|
|
382
|
+
// Sources and eval groups are per-workspace, so the whole run reads and
|
|
383
|
+
// queues inside it. Without it, the lab's default workspace is used and the
|
|
384
|
+
// X-Workspace header is omitted, so existing CI is unchanged.
|
|
385
|
+
let workspace = null;
|
|
386
|
+
if (o.workspace != null) {
|
|
387
|
+
const { workspaces } = await labApi(base, env)("GET", "/api/workspaces");
|
|
388
|
+
workspace = workspaceNamed(Array.isArray(workspaces) ? workspaces : [], o.workspace, base).slug;
|
|
389
|
+
}
|
|
390
|
+
const call = labApi(base, env, workspace);
|
|
360
391
|
|
|
361
392
|
// What the page reads before it submits: the lab's pipelines and Target
|
|
362
393
|
// profiles, the Source the pipeline reads and the eval groups it may link.
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.9.0 (2026.10.07-516)
|