evals-lab 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +108 -0
- package/bin/run.js +36 -5
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +464 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/metrics/builtin.mjs +113 -52
- package/lab/run-evals.js +32 -8
- package/lab/server.py +1174 -179
- package/lab/web/dist/assets/gallery-SnUhXRBn.js +3 -0
- package/lab/web/dist/assets/main-Ca7o-nM0.css +1 -0
- package/lab/web/dist/assets/main-Dru4_P5G.js +20 -0
- package/lab/web/dist/assets/tokens-0az9gfTq.js +58 -0
- package/lab/web/dist/assets/tokens-CqWJKhOx.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +0 -3
- package/lab/web/dist/assets/main-DDoeU6hq.css +0 -1
- package/lab/web/dist/assets/main-nz6Q4jVm.js +0 -21
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +0 -59
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,114 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.8.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- No stored document changes format: pipelines stay version 13, and eval
|
|
13
|
+
groups and exports stay version 8, so 0.7.0 still reads what 0.8.0 saves.
|
|
14
|
+
- On first start, 0.8.0 adds a table recording which workspaces may use each
|
|
15
|
+
Target profile and account. Nothing is narrowed: every connection stays
|
|
16
|
+
shared with every workspace until you change it.
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
|
|
20
|
+
- Several workspaces. Make, rename and archive them in Setup › Workspaces,
|
|
21
|
+
and switch between them from the header's workspace menu. Each one has its
|
|
22
|
+
own address (`/w/<slug>`) and its own datasets, prompts, Sources, pipelines
|
|
23
|
+
and runs. The theme choice (Dark, Light, Match system) has moved into that
|
|
24
|
+
menu.
|
|
25
|
+
- Target profiles and accounts can be shared with every workspace, with some,
|
|
26
|
+
or with each new one. A run in a workspace a profile isn't shared with says
|
|
27
|
+
so and offers to share it.
|
|
28
|
+
- `evals-lab run --lab … --workspace <slug>` runs a lab's pipeline in a named
|
|
29
|
+
workspace. Without it, the default workspace is used, as before.
|
|
30
|
+
- Setup › Wizards, and a header pill showing a wizard's progress. The first
|
|
31
|
+
wizard, Test a Power Automate workflow, ticks off each step as you do it.
|
|
32
|
+
Plugins can add workflow platforms and wizards.
|
|
33
|
+
- Judged against recorded has an **Expected** setting: tick which of Better,
|
|
34
|
+
Same and Worse pass (Better and Same by default, as before).
|
|
35
|
+
- A Recorded target's menu has **Preview replies…**, showing what each call
|
|
36
|
+
recorded.
|
|
37
|
+
- Test… asks for a grader when a check needs one. You can create the test
|
|
38
|
+
without one and add it later.
|
|
39
|
+
|
|
40
|
+
### Changed
|
|
41
|
+
|
|
42
|
+
- Model-graded checks no longer need a prompt written. The grader is told
|
|
43
|
+
what the target was asked; Judged against recorded's Task is now optional
|
|
44
|
+
Additional guidance.
|
|
45
|
+
- A grader's answer is held to a fixed JSON shape where its API supports
|
|
46
|
+
that (Anthropic, OpenAI-compatible, llama.cpp, Ollama). An answer that
|
|
47
|
+
still can't be read is asked for once more, then reported as a grader
|
|
48
|
+
error rather than a verdict.
|
|
49
|
+
- Test… on a step whose reply is plain words judges the reply against the
|
|
50
|
+
recorded one, instead of requiring the exact same words.
|
|
51
|
+
- Rubric's and Factual's threshold reads **Expected: score at least**.
|
|
52
|
+
- New profile: Anthropic and Ollama (Cloud) no longer ask for an address. A
|
|
53
|
+
proxy or gateway address goes under Advanced. Temperature is hidden, and
|
|
54
|
+
never sent, for models that reject it (Claude Opus 4.7 and later, the
|
|
55
|
+
5-series Claude models, OpenAI reasoning models). Image settings are now
|
|
56
|
+
the folded Vision Capabilities group.
|
|
57
|
+
- The Test… dialog is "Create new pipeline test", with a Create button.
|
|
58
|
+
- Library › Sources › a flow: Actions come before Calls, and each call's
|
|
59
|
+
request reads In sync or Out of sync.
|
|
60
|
+
- Run results: Take B as expected is in each call's row menu, and the
|
|
61
|
+
Results table no longer has checkboxes.
|
|
62
|
+
- The wizard pill's button is Open wizard, which shows the wizard's steps.
|
|
63
|
+
|
|
64
|
+
### Fixed
|
|
65
|
+
|
|
66
|
+
- An HTTP endpoint profile whose address included a path (such as
|
|
67
|
+
`https://api.anthropic.com/v1/messages`) sent the path twice, so every
|
|
68
|
+
call came back Not found.
|
|
69
|
+
- An Anthropic profile with a blank address was sent to the lab's own Ollama.
|
|
70
|
+
- A run said "this run has none" for a grader that its Library eval group
|
|
71
|
+
named.
|
|
72
|
+
- The wizard pill didn't tick a step until the page was reloaded.
|
|
73
|
+
- A Microsoft 365 sign-in now shows which tenants it reaches, so a lab that
|
|
74
|
+
works across tenants is not tied to one.
|
|
75
|
+
|
|
76
|
+
## 0.7.0
|
|
77
|
+
|
|
78
|
+
### Upgrade notes
|
|
79
|
+
|
|
80
|
+
- **Copy your data directory before upgrading** (see "User data" in the
|
|
81
|
+
README). An eval group 0.7.0 saves is dataset version 8, which 0.6.x
|
|
82
|
+
cannot read. To go back, reinstall 0.6.0 and restore the copy.
|
|
83
|
+
- On its first start, 0.7.0 moves everything the lab keeps into one
|
|
84
|
+
workspace, Default. Nothing on the page changes.
|
|
85
|
+
- Stored eval groups are not rewritten on upgrade: a version-7 group reads
|
|
86
|
+
as version 8, and is saved as version 8 at its next edit.
|
|
87
|
+
- Export JSON writes version 8, which 0.6.x refuses to import. Import reads
|
|
88
|
+
versions 1 to 8.
|
|
89
|
+
- The metrics that compare a reply with production's are now the Recorded
|
|
90
|
+
reply metrics. Groups that use them grade as before.
|
|
91
|
+
|
|
92
|
+
### Added
|
|
93
|
+
|
|
94
|
+
- Test… on a Power Automate Source's action makes an eval group from the
|
|
95
|
+
Source's calls, graded against the replies production recorded, and a
|
|
96
|
+
pipeline to run it, then opens it in Runs.
|
|
97
|
+
- The Recorded target replays each call's recorded reply and sends nothing,
|
|
98
|
+
so a run compares an edited request with what production did.
|
|
99
|
+
- Adding a Power Automate Source imports the calls of its last five runs in
|
|
100
|
+
the background, while the tab stays open.
|
|
101
|
+
- Results of a run with a Recorded target: a table of each check with a
|
|
102
|
+
column per Target (n/a where a check cannot apply), a call that opens both
|
|
103
|
+
replies side by side, and Take B as expected, which makes the selected
|
|
104
|
+
calls expect Target B's reply.
|
|
105
|
+
- An HTTP Request target shows its words with the item's fields as chips.
|
|
106
|
+
Select prompt… and Save to library… use the Prompt library, and Copy
|
|
107
|
+
request… writes the request back for Power Automate, or as JSON.
|
|
108
|
+
|
|
109
|
+
### Changed
|
|
110
|
+
|
|
111
|
+
- A Power Automate Source's records are its Calls, with Add call.
|
|
112
|
+
- Targets sit side by side on a wide screen. A job's Content and Responses
|
|
113
|
+
fold to one line until opened.
|
|
114
|
+
- A row with one action shows it as a button; two or more are in its ⋯ menu.
|
|
115
|
+
|
|
8
116
|
## 0.6.0
|
|
9
117
|
|
|
10
118
|
### Added
|
package/bin/run.js
CHANGED
|
@@ -51,6 +51,8 @@ const USAGE = `Usage: evals-lab run <bundle-dir | pipeline.yaml> [options]
|
|
|
51
51
|
and their keys, and the run in its History. LAB_PASSWORD
|
|
52
52
|
is sent when it is set.
|
|
53
53
|
--pipeline <name> the lab's pipeline to run, by its name or its id.
|
|
54
|
+
--workspace <slug> the lab's workspace to run in, by its /w/<slug> address.
|
|
55
|
+
Default: the lab's default workspace.
|
|
54
56
|
--wait wait for the lab's run to finish, and exit with its
|
|
55
57
|
verdict. Without it the run's id is printed once it is
|
|
56
58
|
queued. --json, --junit, --summary, --min-pass and
|
|
@@ -70,9 +72,9 @@ class Refused extends Error {
|
|
|
70
72
|
// What the command line asks for, or a Refused saying what is wrong with it.
|
|
71
73
|
function runOptions(argv) {
|
|
72
74
|
const o = { bundle: null, items: null, json: null, junit: null, summary: null, minPass: null, progress: false,
|
|
73
|
-
lab: null, pipeline: null, wait: false };
|
|
75
|
+
lab: null, pipeline: null, workspace: null, wait: false };
|
|
74
76
|
const value = { "--items": "items", "--json": "json", "--junit": "junit", "--summary": "summary", "--min-pass": "minPass",
|
|
75
|
-
"--lab": "lab", "--pipeline": "pipeline" };
|
|
77
|
+
"--lab": "lab", "--pipeline": "pipeline", "--workspace": "workspace" };
|
|
76
78
|
for (let i = 0; i < argv.length; i++) {
|
|
77
79
|
const a = argv[i];
|
|
78
80
|
if (a === "--help" || a === "-h") o.help = true;
|
|
@@ -96,6 +98,8 @@ function runOptions(argv) {
|
|
|
96
98
|
if (waits.length && !o.wait) {
|
|
97
99
|
throw new Refused(`--${waits[0].replace("minPass", "min-pass")} reads the finished run: add --wait`, true);
|
|
98
100
|
}
|
|
101
|
+
} else if (o.workspace != null) {
|
|
102
|
+
throw new Refused("--workspace is for a lab's run: name the lab with --lab", true);
|
|
99
103
|
} else if (o.wait) {
|
|
100
104
|
throw new Refused("--wait is for a lab's run: a bundle's always waits", true);
|
|
101
105
|
} else if (o.bundle == null) {
|
|
@@ -276,15 +280,18 @@ const POLL_MISSES = 20;
|
|
|
276
280
|
|
|
277
281
|
// The lab's API at [base], as [env]'s LAB_PASSWORD opens it: HTTP Basic, the
|
|
278
282
|
// password alone, as server.py takes it. A refusal says what the lab said.
|
|
279
|
-
|
|
283
|
+
// [workspace], when given, is the slug every call carries as X-Workspace, so
|
|
284
|
+
// the run reads and queues in that workspace rather than the lab's default.
|
|
285
|
+
function labApi(base, env, workspace) {
|
|
280
286
|
const auth = env.LAB_PASSWORD
|
|
281
287
|
? { Authorization: `Basic ${Buffer.from(`:${env.LAB_PASSWORD}`).toString("base64")}` } : {};
|
|
288
|
+
const scope = workspace ? { "X-Workspace": workspace } : {};
|
|
282
289
|
return async (method, route, body, { missing = false } = {}) => {
|
|
283
290
|
let res;
|
|
284
291
|
try {
|
|
285
292
|
res = await fetch(base + route, {
|
|
286
293
|
method, redirect: "manual",
|
|
287
|
-
headers: { ...auth, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
294
|
+
headers: { ...auth, ...scope, ...(body !== undefined ? { "Content-Type": "application/json" } : {}) },
|
|
288
295
|
...(body !== undefined ? { body: JSON.stringify(body) } : {}),
|
|
289
296
|
});
|
|
290
297
|
} catch (e) {
|
|
@@ -319,6 +326,21 @@ function pipelineNamed(saved, name, base) {
|
|
|
319
326
|
return found[0];
|
|
320
327
|
}
|
|
321
328
|
|
|
329
|
+
// The lab's workspace --workspace [slug] names, by its /w/<slug> address -- or
|
|
330
|
+
// a Refused naming the slugs it has, as pipelineNamed does for pipelines. The
|
|
331
|
+
// server falls back to its default for an unknown slug, so the CLI checks the
|
|
332
|
+
// slug itself; an archived workspace cannot be run against.
|
|
333
|
+
function workspaceNamed(workspaces, slug, base) {
|
|
334
|
+
const active = workspaces.filter(w => w && !w.archived);
|
|
335
|
+
const found = active.find(w => w.slug === slug);
|
|
336
|
+
if (found) return found;
|
|
337
|
+
if (workspaces.some(w => w && w.slug === slug)) {
|
|
338
|
+
throw new Refused(`${base}'s workspace at /w/${slug} is archived`);
|
|
339
|
+
}
|
|
340
|
+
throw new Refused(`${base} has no workspace at /w/${slug}`
|
|
341
|
+
+ (active.length ? `: it has ${active.map(w => JSON.stringify(w.slug)).join(", ")}` : ""));
|
|
342
|
+
}
|
|
343
|
+
|
|
322
344
|
// The row's verdicts with --min-pass relaxing each eval read item by item,
|
|
323
345
|
// as the worker's --min-pass does: it only ever turns a fail into a pass.
|
|
324
346
|
const relaxed = (verdicts, minPass) => minPass == null ? verdicts : verdicts.map(evals =>
|
|
@@ -356,7 +378,16 @@ function rowSuites(core, run, verdicts, items) {
|
|
|
356
378
|
/** [o]'s pipeline, queued on its lab; with --wait, followed to its verdict. */
|
|
357
379
|
async function runOnLab(core, o, { env, stdout, say }) {
|
|
358
380
|
const base = o.lab.replace(/\/+$/, "");
|
|
359
|
-
|
|
381
|
+
// Resolve --workspace to the slug every call carries; the lab's pipelines,
|
|
382
|
+
// Sources and eval groups are per-workspace, so the whole run reads and
|
|
383
|
+
// queues inside it. Without it, the lab's default workspace is used and the
|
|
384
|
+
// X-Workspace header is omitted, so existing CI is unchanged.
|
|
385
|
+
let workspace = null;
|
|
386
|
+
if (o.workspace != null) {
|
|
387
|
+
const { workspaces } = await labApi(base, env)("GET", "/api/workspaces");
|
|
388
|
+
workspace = workspaceNamed(Array.isArray(workspaces) ? workspaces : [], o.workspace, base).slug;
|
|
389
|
+
}
|
|
390
|
+
const call = labApi(base, env, workspace);
|
|
360
391
|
|
|
361
392
|
// What the page reads before it submits: the lab's pipelines and Target
|
|
362
393
|
// profiles, the Source the pipeline reads and the eval groups it may link.
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.8.0 (2026.10.07-484)
|