evals-lab 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +77 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +353 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/kinds/list.mjs +2 -1
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +220 -83
- package/lab/server.py +617 -176
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,83 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.7.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- **Copy your data directory before upgrading** (see "User data" in the
|
|
13
|
+
README). An eval group 0.7.0 saves is dataset version 8, which 0.6.x
|
|
14
|
+
cannot read. To go back, reinstall 0.6.0 and restore the copy.
|
|
15
|
+
- On its first start, 0.7.0 moves everything the lab keeps into one
|
|
16
|
+
workspace, Default. Nothing on the page changes.
|
|
17
|
+
- Stored eval groups are not rewritten on upgrade: a version-7 group reads
|
|
18
|
+
as version 8, and is saved as version 8 at its next edit.
|
|
19
|
+
- Export JSON writes version 8, which 0.6.x refuses to import. Import reads
|
|
20
|
+
versions 1 to 8.
|
|
21
|
+
- The metrics that compare a reply with production's are now the Recorded
|
|
22
|
+
reply metrics. Groups that use them grade as before.
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- Test… on a Power Automate Source's action makes an eval group from the
|
|
27
|
+
Source's calls, graded against the replies production recorded, and a
|
|
28
|
+
pipeline to run it, then opens it in Runs.
|
|
29
|
+
- The Recorded target replays each call's recorded reply and sends nothing,
|
|
30
|
+
so a run compares an edited request with what production did.
|
|
31
|
+
- Adding a Power Automate Source imports the calls of its last five runs in
|
|
32
|
+
the background, while the tab stays open.
|
|
33
|
+
- Results of a run with a Recorded target: a table of each check with a
|
|
34
|
+
column per Target (n/a where a check cannot apply), a call that opens both
|
|
35
|
+
replies side by side, and Take B as expected, which makes the selected
|
|
36
|
+
calls expect Target B's reply.
|
|
37
|
+
- An HTTP Request target shows its words with the item's fields as chips.
|
|
38
|
+
Select prompt… and Save to library… use the Prompt library, and Copy
|
|
39
|
+
request… writes the request back for Power Automate, or as JSON.
|
|
40
|
+
|
|
41
|
+
### Changed
|
|
42
|
+
|
|
43
|
+
- A Power Automate Source's records are its Calls, with Add call.
|
|
44
|
+
- Targets sit side by side on a wide screen. A job's Content and Responses
|
|
45
|
+
fold to one line until opened.
|
|
46
|
+
- A row with one action shows it as a button; two or more are in its ⋯ menu.
|
|
47
|
+
|
|
48
|
+
## 0.6.0
|
|
49
|
+
|
|
50
|
+
### Added
|
|
51
|
+
|
|
52
|
+
- Library › Evals (was Library › Datasets) lists every eval group: its
|
|
53
|
+
Source, cases, scoring, and the pipelines that link it. A group's page
|
|
54
|
+
holds its Grades, Every item, Whole run and cases.
|
|
55
|
+
- Runs › Evals links eval groups to a pipeline. Link eval group picks from
|
|
56
|
+
the Library and marks the groups made for this content. A link follows a
|
|
57
|
+
group's latest version, or is pinned to one. A pipeline's own group is
|
|
58
|
+
edited in place, and Move to Library shares it.
|
|
59
|
+
- A pipeline can link several eval groups. Its pass rule says whether every
|
|
60
|
+
one must pass, or at least a number. Results, History and `evals-lab run`
|
|
61
|
+
all grade by it.
|
|
62
|
+
- An Undo button in the header takes back the last 20 actions, newest
|
|
63
|
+
first, until the page reloads.
|
|
64
|
+
- Choose pipeline lists the 5 used most recently, then the rest, searchable
|
|
65
|
+
and sortable.
|
|
66
|
+
- A Profile, Source or Grader field opens a picker, most recent first.
|
|
67
|
+
|
|
68
|
+
### Changed
|
|
69
|
+
|
|
70
|
+
- Results shows a row per eval group, with each Target's share and verdict,
|
|
71
|
+
then the overall verdict. History's Outcome is the overall verdict.
|
|
72
|
+
- A dialog is a draft: nothing reaches the lab until Save & Close. Cancel,
|
|
73
|
+
× and Esc discard it.
|
|
74
|
+
- An unnamed target is Target A, B, C.
|
|
75
|
+
- The run bar sits under the pipeline's header. It says "12 of 42" while a
|
|
76
|
+
run goes and "42 in 2m" when it is done.
|
|
77
|
+
- Verdicts show as a check or a cross, in Results and History.
|
|
78
|
+
- A metric's "How it counts" is Options, and Grades is Reads: Parsed reply
|
|
79
|
+
or Raw reply.
|
|
80
|
+
|
|
81
|
+
### Fixed
|
|
82
|
+
|
|
83
|
+
- Add profile, then ×, added a profile.
|
|
84
|
+
|
|
8
85
|
## 0.5.0
|
|
9
86
|
|
|
10
87
|
### Upgrade notes
|
package/bin/run.js
CHANGED
|
@@ -487,10 +487,21 @@ async function run(argv, { lab, env = process.env, stdout = process.stdout, stde
|
|
|
487
487
|
const args = [path.join(lab, "run-evals.js"), "--run", path.join(tmp, "run.json"),
|
|
488
488
|
"--json", path.join(tmp, "report.json")];
|
|
489
489
|
if (items) args.push("--source", items);
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
490
|
+
// The eval group bodies the run grades against: its one, as --dataset, or
|
|
491
|
+
// a body per group it links, as --groups (#233), keyed by the id its
|
|
492
|
+
// reference carries (a bundle's refs record no version).
|
|
493
|
+
const ids = [...new Set(core.evalsOf(read.run).map(core.casesRef).filter(Boolean).map(r => r.id))];
|
|
494
|
+
if (ids.length === 1 && read.datasets[ids[0]]) {
|
|
495
|
+
fs.writeFileSync(path.join(tmp, "dataset.json"), JSON.stringify(read.datasets[ids[0]]));
|
|
493
496
|
args.push("--dataset", path.join(tmp, "dataset.json"));
|
|
497
|
+
} else if (ids.length > 1) {
|
|
498
|
+
const groups = {};
|
|
499
|
+
for (const id of ids) {
|
|
500
|
+
if (!read.datasets[id]) throw new Refused(`${o.bundle}: the bundle holds no eval group ${id}`);
|
|
501
|
+
groups[id] = read.datasets[id];
|
|
502
|
+
}
|
|
503
|
+
fs.writeFileSync(path.join(tmp, "groups.json"), JSON.stringify(groups));
|
|
504
|
+
args.push("--groups", path.join(tmp, "groups.json"));
|
|
494
505
|
}
|
|
495
506
|
if (plugins.length) args.push("--plugins", path.join(dir, "plugins"));
|
|
496
507
|
if (o.minPass != null) args.push("--min-pass", o.minPass);
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.7.0 (2026.10.06-456)
|