evals-lab 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +90 -19
- package/lab/run-evals.js +2 -2
- package/lab/server.py +165 -52
- package/lab/web/dist/assets/{gallery-DRBlZ8mP.js → gallery-B-7oyY37.js} +1 -1
- package/lab/web/dist/assets/{main-wfqC6HcM.css → main-C_b7QoTv.css} +1 -1
- package/lab/web/dist/assets/main-DyDG-V9N.js +21 -0
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +55 -0
- package/lab/web/dist/gallery.html +2 -2
- package/lab/web/dist/index.html +3 -3
- package/package.json +1 -1
- package/lab/web/dist/assets/main-DMpQmW8l.js +0 -21
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +0 -55
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,35 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
|
|
|
5
5
|
upgrade can rewrite what the lab keeps in your data directory, and an older
|
|
6
6
|
version cannot always read it back.
|
|
7
7
|
|
|
8
|
+
## 0.4.0
|
|
9
|
+
|
|
10
|
+
### Upgrade notes
|
|
11
|
+
|
|
12
|
+
- **Copy your data directory before upgrading** (see "Where your data is
|
|
13
|
+
kept" in the README). A dataset 0.4.0 saves is dataset version 7, which
|
|
14
|
+
0.3.x cannot read. To go back, reinstall 0.3.0 and restore the copy.
|
|
15
|
+
- Stored datasets are not rewritten on upgrade: 0.4.0 reads a version-6
|
|
16
|
+
dataset as version 7, and writes version 7 the next time it is saved.
|
|
17
|
+
- Export JSON writes dataset version 7, which 0.3.x refuses to import.
|
|
18
|
+
Import reads versions 1 to 7.
|
|
19
|
+
|
|
20
|
+
### Added
|
|
21
|
+
|
|
22
|
+
- Re-run, in a History row's menu and in Results' run menu: a new run of the
|
|
23
|
+
same pipeline over the same file revisions and the same dataset version.
|
|
24
|
+
It is refused while the run is still going, or when one of the files it
|
|
25
|
+
read is gone. The new run says which run it re-ran, and links back to it
|
|
26
|
+
in Results. Undo cancels it while it is still queued.
|
|
27
|
+
- A dataset carries four new sections, `scoring`, `grader`, `every` and
|
|
28
|
+
`run`: how it scores, which model grades it, and metrics for every item
|
|
29
|
+
and for the whole run. Edit raw… shows them. Runs do not read them yet.
|
|
30
|
+
A dataset from 0.3.x has them as scored All, by the lab's grader, with no
|
|
31
|
+
metrics of its own.
|
|
32
|
+
|
|
33
|
+
### Changed
|
|
34
|
+
|
|
35
|
+
- Resume is in a History row's menu, beside Re-run.
|
|
36
|
+
|
|
8
37
|
## 0.3.0
|
|
9
38
|
|
|
10
39
|
### Upgrade notes
|
package/lab/VERSION
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
0.
|
|
1
|
+
0.4.0 (2026.10.02-361)
|
package/lab/evals-core.mjs
CHANGED
|
@@ -983,22 +983,37 @@ const SLOTS = ["content", "target", "responses"];
|
|
|
983
983
|
|
|
984
984
|
|
|
985
985
|
|
|
986
|
-
|
|
986
|
+
|
|
987
987
|
|
|
988
|
-
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
|
|
997
|
+
|
|
989
998
|
|
|
990
999
|
|
|
991
1000
|
|
|
992
|
-
/**
|
|
993
|
-
|
|
994
|
-
|
|
1001
|
+
/** An eval group's scoring (docs/pipeline-model.md §17). */
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
/** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
|
|
1008
|
+
named its Source; 4 and earlier, neither. */
|
|
1009
|
+
const DATASET_BODY_VERSION = 7 ;
|
|
995
1010
|
|
|
996
1011
|
/** The metrics whose Ignore case version 6 made mean what it says for a
|
|
997
1012
|
reply read as a list, as well as one read as text. */
|
|
998
1013
|
const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
999
1014
|
|
|
1000
1015
|
/**
|
|
1001
|
-
* [body] as this version of a dataset (
|
|
1016
|
+
* [body] as this version of a dataset (7), from any earlier one. Version 1
|
|
1002
1017
|
* held `imageCases`, each with `minTags`/`maxTags`, and the parser's
|
|
1003
1018
|
* `replays` and `conformance` (now fixtures/replays.json beside the checks).
|
|
1004
1019
|
* Version 2 held `rules`, which clean a job's answer and so belong to the
|
|
@@ -1010,7 +1025,9 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
|
|
|
1010
1025
|
* `note`, `traits` go, and the body names no Source yet. Version 5 matched
|
|
1011
1026
|
* a list's items ignoring case whatever a Contains metric's Ignore case
|
|
1012
1027
|
* said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
|
|
1013
|
-
* the body says its version.
|
|
1028
|
+
* the body says its version. Version 6 is a group's cases alone: version 7
|
|
1029
|
+
* adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
|
|
1030
|
+
* stored row is read that way rather than rewritten. Every reader of a
|
|
1014
1031
|
* body calls this: the runner, the page, a run's kept copy. A reader that
|
|
1015
1032
|
* needs a version-2 body's rules -- to upgrade a pipeline graded against
|
|
1016
1033
|
* it -- takes them first (`datasetRules`). Pure: the same body gives the
|
|
@@ -1021,14 +1038,26 @@ function upgradeDatasetBody (body ) {
|
|
|
1021
1038
|
if (!isObj(body)) return body;
|
|
1022
1039
|
if (body.version === DATASET_BODY_VERSION) return body;
|
|
1023
1040
|
const b = body ;
|
|
1041
|
+
// A body saying any other version is one this lab does not read.
|
|
1042
|
+
if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
|
|
1024
1043
|
// A body that names its Source, even as null, is version 5.
|
|
1025
1044
|
if ("source" in b) {
|
|
1026
|
-
return {
|
|
1045
|
+
return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
|
|
1027
1046
|
}
|
|
1028
1047
|
const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
|
|
1029
1048
|
// Not a body of any version -- a copy kept while a dataset was an overlay.
|
|
1030
1049
|
if (!raw) return body;
|
|
1031
|
-
return {
|
|
1050
|
+
return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
/** A version-6 body as a version-7 eval group: scored All, the lab's
|
|
1054
|
+
grader, and no metrics of its own for every item or the whole run --
|
|
1055
|
+
what a Metrics eval naming the dataset with none of its own graded, which
|
|
1056
|
+
is what every Graded set converted to. */
|
|
1057
|
+
function groupOfV6(b ) {
|
|
1058
|
+
const { version: _v, ...rest } = b;
|
|
1059
|
+
return { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null, every: [], run: [],
|
|
1060
|
+
...rest } ;
|
|
1032
1061
|
}
|
|
1033
1062
|
|
|
1034
1063
|
/** A version-5 case as a version-6 one: each Contains metric says Ignore
|
|
@@ -3187,15 +3216,7 @@ EVAL_TYPES.metrics = {
|
|
|
3187
3216
|
? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
|
|
3188
3217
|
wholeRun: t => t.over === "run",
|
|
3189
3218
|
// Over the whole run: every metric over the replies together, or each alone.
|
|
3190
|
-
verdict(t, ress, kind)
|
|
3191
|
-
const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
|
|
3192
|
-
const metrics = readRun(t.metrics || [], run);
|
|
3193
|
-
const s = scoreOf(metrics, t.mode, t.threshold);
|
|
3194
|
-
const off = metrics.filter(r => !r.pass);
|
|
3195
|
-
return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
|
|
3196
|
-
ran: run.replies.length,
|
|
3197
|
-
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3198
|
-
},
|
|
3219
|
+
verdict: (t, ress, kind) => wholeRunVerdict(t.metrics || [], ress, kind, t.mode, t.threshold),
|
|
3199
3220
|
// Rule m{x} is the eval's own metric x, read in that place on every item.
|
|
3200
3221
|
// A score stored before Metrics -- a Graded set's, read as its conversion --
|
|
3201
3222
|
// has no readings of its own: its one metric's reading is the score's.
|
|
@@ -3296,6 +3317,38 @@ LEGACY_TESTS.single .toMetrics = t => {
|
|
|
3296
3317
|
return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
|
|
3297
3318
|
};
|
|
3298
3319
|
|
|
3320
|
+
/** [list] read over a whole run's replies so far, scored the way [mode]
|
|
3321
|
+
says: a Metrics eval's verdict over the run, and an eval group's Whole run. */
|
|
3322
|
+
function wholeRunVerdict(list , ress , kind ,
|
|
3323
|
+
mode = "all", threshold = null) {
|
|
3324
|
+
const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
|
|
3325
|
+
const metrics = readRun(list, run);
|
|
3326
|
+
const s = scoreOf(metrics, mode, threshold);
|
|
3327
|
+
const off = metrics.filter(r => !r.pass);
|
|
3328
|
+
return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
|
|
3329
|
+
ran: run.replies.length,
|
|
3330
|
+
checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
|
|
3331
|
+
}
|
|
3332
|
+
|
|
3333
|
+
/**
|
|
3334
|
+
* An eval group's verdict on one item (docs/pipeline-model.md §17): its
|
|
3335
|
+
* Every item metrics and [kase]'s own -- the item's case in the group, or
|
|
3336
|
+
* null where it has none -- read of the reply under the group's scoring.
|
|
3337
|
+
* The reading a Metrics eval naming the group as its dataset gives, which
|
|
3338
|
+
* groups-check.js holds it to. Null where no metric had anything to read.
|
|
3339
|
+
*/
|
|
3340
|
+
function readGroup(group , kase , res , more = {}) {
|
|
3341
|
+
return readMetrics([...group.every, ...(kase?.metrics ?? [])], metricInput(res, kase, more), graderCtx(group, more),
|
|
3342
|
+
group.scoring.mode, group.scoring.threshold);
|
|
3343
|
+
}
|
|
3344
|
+
|
|
3345
|
+
/** An eval group's Whole run verdict, from the replies so far: its `run`
|
|
3346
|
+
metrics under its scoring. Null for a group with no Whole run metrics,
|
|
3347
|
+
which has nothing to say of a run. */
|
|
3348
|
+
function readGroupRun(group , ress , kind = null) {
|
|
3349
|
+
return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
|
|
3350
|
+
}
|
|
3351
|
+
|
|
3299
3352
|
/** The grader a Metrics eval names, reached through what the runner hands it. */
|
|
3300
3353
|
function graderCtx(t , more ) {
|
|
3301
3354
|
const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
|
|
@@ -4768,6 +4821,24 @@ function validateEvals(ev , files
|
|
|
4768
4821
|
}
|
|
4769
4822
|
if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
|
|
4770
4823
|
if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
|
|
4824
|
+
// An eval group's own sections (version 7): what a Metrics eval held.
|
|
4825
|
+
if (ev.scoring != null) {
|
|
4826
|
+
const sc = ev.scoring;
|
|
4827
|
+
if (!isObj(sc) || !SCORING_MODES.includes(sc.mode)) bad.push(`a group is scored ${SCORING_MODES.join(" or ")}`);
|
|
4828
|
+
else if (sc.mode === "weighted" && typeof sc.threshold !== "number") bad.push("a group scored in points needs Pass at: the points an item has to reach");
|
|
4829
|
+
else if (sc.threshold != null && typeof sc.threshold !== "number") bad.push("a group's Pass at has to be a number");
|
|
4830
|
+
}
|
|
4831
|
+
if (ev.grader != null && !isRef(ev.grader)) bad.push("a group names its grader as { id, name }, or null");
|
|
4832
|
+
if (ev.every != null) metricsProblems(ev.every, "Every item", bad);
|
|
4833
|
+
if (ev.run != null) {
|
|
4834
|
+
metricsProblems(ev.run, "Whole run", bad);
|
|
4835
|
+
// Read once over every reply: nothing that reads one item, or asks a grader.
|
|
4836
|
+
for (const m of Array.isArray(ev.run) ? ev.run : []) {
|
|
4837
|
+
const entry = isObj(m) ? METRICS[m.type] : undefined;
|
|
4838
|
+
if (entry?.graded) bad.push(`Whole run: ${entry.label} is model-graded, so it cannot read a whole run`);
|
|
4839
|
+
else if (entry?.perItem) bad.push(`Whole run: ${entry.label} reads one item at a time, so it cannot read a whole run`);
|
|
4840
|
+
}
|
|
4841
|
+
}
|
|
4771
4842
|
if (bad.length) return bad;
|
|
4772
4843
|
|
|
4773
4844
|
const graded = ev.cases || [];
|
|
@@ -4915,7 +4986,7 @@ registerMetrics({ registerKinds });
|
|
|
4915
4986
|
export {
|
|
4916
4987
|
TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
|
|
4917
4988
|
loopReplyError, preparedSize,
|
|
4918
|
-
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4989
|
+
termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
|
|
4919
4990
|
SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
|
|
4920
4991
|
CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
|
|
4921
4992
|
EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
|
package/lab/run-evals.js
CHANGED
|
@@ -296,8 +296,8 @@ function readDataset(file) {
|
|
|
296
296
|
broken(`${file}: ${e.message}`);
|
|
297
297
|
}
|
|
298
298
|
if (isObject(doc) && doc.format === "evals-lab/dataset") {
|
|
299
|
-
if (![1, 2, 3, 4, 5].includes(doc.version)) {
|
|
300
|
-
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to
|
|
299
|
+
if (![1, 2, 3, 4, 5, 6, 7].includes(doc.version)) {
|
|
300
|
+
broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
|
|
301
301
|
}
|
|
302
302
|
doc = isObject(doc.dataset) ? doc.dataset.body : null;
|
|
303
303
|
}
|
package/lab/server.py
CHANGED
|
@@ -1356,23 +1356,25 @@ class Sources:
|
|
|
1356
1356
|
# A lab from before this held a one-time import's `meta` row saying it ran;
|
|
1357
1357
|
# it is left where it is, and nothing reads it.
|
|
1358
1358
|
|
|
1359
|
-
DATASET_FIELDS = ("version", "source", "cases")
|
|
1360
|
-
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version
|
|
1361
|
-
#
|
|
1362
|
-
|
|
1359
|
+
DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
|
|
1360
|
+
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 7 is an
|
|
1361
|
+
# eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
|
|
1362
|
+
# 5 was told by its `source` alone, and earlier ones by neither.
|
|
1363
|
+
DATASET_BODY_VERSION = 7
|
|
1363
1364
|
DATASET_NAME_MAX = 80
|
|
1364
|
-
# The file forms Export writes and Import reads. Export writes version
|
|
1365
|
-
# Import reads it and versions 1 to
|
|
1365
|
+
# The file forms Export writes and Import reads. Export writes version 7;
|
|
1366
|
+
# Import reads it and versions 1 to 6, upgraded, and refuses anything else,
|
|
1366
1367
|
# as a pipeline of another version is refused. Versions 1 to 3 carried a
|
|
1367
1368
|
# prompt, which an import gives to the Prompt library.
|
|
1368
1369
|
EXPORT_ONE = "evals-lab/dataset"
|
|
1369
1370
|
EXPORT_ALL = "evals-lab/datasets"
|
|
1370
|
-
EXPORT_VERSION =
|
|
1371
|
-
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
|
|
1371
|
+
EXPORT_VERSION = 7
|
|
1372
|
+
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
|
|
1373
|
+
SCORING_MODES = ("all", "weighted")
|
|
1372
1374
|
|
|
1373
1375
|
|
|
1374
1376
|
def blank_dataset() -> dict:
|
|
1375
|
-
return {"
|
|
1377
|
+
return group_of_v6({"source": None, "cases": []})
|
|
1376
1378
|
|
|
1377
1379
|
|
|
1378
1380
|
# vocab: the names older versions gave a case's fields
|
|
@@ -1400,7 +1402,7 @@ def _term_in(items, term) -> bool:
|
|
|
1400
1402
|
def case_metrics(c: dict) -> list:
|
|
1401
1403
|
"""A version-4 case's expectations as the metrics that say the same:
|
|
1402
1404
|
evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
|
|
1403
|
-
through fixtures/dataset-
|
|
1405
|
+
through fixtures/dataset-v7.json."""
|
|
1404
1406
|
def strs(v):
|
|
1405
1407
|
return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
|
|
1406
1408
|
if c.get("discarded") is True:
|
|
@@ -1466,8 +1468,18 @@ def case_of_v5(c):
|
|
|
1466
1468
|
else m for m in c["metrics"]]}
|
|
1467
1469
|
|
|
1468
1470
|
|
|
1471
|
+
def group_of_v6(body: dict) -> dict:
|
|
1472
|
+
"""A version-6 body as a version-7 eval group: evals-core.ts's
|
|
1473
|
+
groupOfV6. Scored All, the lab's grader, and no metrics of its own for
|
|
1474
|
+
every item or the whole run -- what a Metrics eval naming the dataset with
|
|
1475
|
+
none of its own graded."""
|
|
1476
|
+
rest = {k: v for k, v in body.items() if k != "version"}
|
|
1477
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "scoring": {"mode": "all", "threshold": None},
|
|
1478
|
+
"grader": None, "every": [], "run": [], **rest}
|
|
1479
|
+
|
|
1480
|
+
|
|
1469
1481
|
def upgrade_body(body):
|
|
1470
|
-
"""An earlier body as today's (version
|
|
1482
|
+
"""An earlier body as today's (version 7): evals-core.ts's
|
|
1471
1483
|
upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
|
|
1472
1484
|
its `replays` and `conformance` go (fixtures/replays.json holds the
|
|
1473
1485
|
parser's tests). Version 2's `rules` go -- they clean a job's answer, so
|
|
@@ -1477,19 +1489,25 @@ def upgrade_body(body):
|
|
|
1477
1489
|
`note`, `traits` go, and the body names no Source yet. A caller that
|
|
1478
1490
|
needs the rules or the prompt takes them first (`body_rules`,
|
|
1479
1491
|
`body_prompt`). A body naming its Source is version 5, whose Contains
|
|
1480
|
-
metrics each come to say Ignore case (`case_of_v5`).
|
|
1481
|
-
|
|
1492
|
+
metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
|
|
1493
|
+
group's scoring, grader, Every item and Whole run (`group_of_v6`). A body
|
|
1494
|
+
saying it is version 7 comes back as it was; so does anything that is
|
|
1495
|
+
not a body."""
|
|
1482
1496
|
if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
|
|
1483
1497
|
return body
|
|
1498
|
+
# A body saying any other version is one this lab does not read, and is
|
|
1499
|
+
# left for dataset_problem to refuse.
|
|
1500
|
+
if "version" in body:
|
|
1501
|
+
return group_of_v6(body) if body["version"] == 6 else body
|
|
1484
1502
|
if "source" in body:
|
|
1485
|
-
up =
|
|
1503
|
+
up = dict(body)
|
|
1486
1504
|
if isinstance(body.get("cases"), list):
|
|
1487
1505
|
up["cases"] = [case_of_v5(c) for c in body["cases"]]
|
|
1488
|
-
return up
|
|
1506
|
+
return group_of_v6(up)
|
|
1489
1507
|
cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
|
|
1490
1508
|
if not isinstance(cases, list):
|
|
1491
1509
|
return body
|
|
1492
|
-
return {"
|
|
1510
|
+
return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
|
|
1493
1511
|
|
|
1494
1512
|
|
|
1495
1513
|
def body_prompt(body):
|
|
@@ -1518,10 +1536,25 @@ def dataset_problem(body) -> str:
|
|
|
1518
1536
|
return f"a dataset's body is version {DATASET_BODY_VERSION}"
|
|
1519
1537
|
if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
|
|
1520
1538
|
return "cases has to be a list of cases"
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1539
|
+
def is_ref(v):
|
|
1540
|
+
return isinstance(v, dict) and isinstance(v.get("id"), str) and isinstance(v.get("name"), str)
|
|
1541
|
+
if body["source"] is not None and not is_ref(body["source"]):
|
|
1524
1542
|
return "a dataset names its Source as { id, name }, or null"
|
|
1543
|
+
# A group's own sections: their metrics are evals-core.ts's validateEvals'
|
|
1544
|
+
# to judge, as a case's are; the shape is the server's.
|
|
1545
|
+
sc = body["scoring"]
|
|
1546
|
+
if not isinstance(sc, dict) or sc.get("mode") not in SCORING_MODES:
|
|
1547
|
+
return "a group is scored all or weighted"
|
|
1548
|
+
number = lambda v: type(v) in (int, float)
|
|
1549
|
+
if sc["mode"] == "weighted" and not number(sc.get("threshold")):
|
|
1550
|
+
return "a group scored in points needs Pass at: the points an item has to reach"
|
|
1551
|
+
if sc.get("threshold") is not None and not number(sc.get("threshold")):
|
|
1552
|
+
return "a group's Pass at has to be a number"
|
|
1553
|
+
if body["grader"] is not None and not is_ref(body["grader"]):
|
|
1554
|
+
return "a group names its grader as { id, name }, or null"
|
|
1555
|
+
for k, label in (("every", "Every item"), ("run", "Whole run")):
|
|
1556
|
+
if not isinstance(body[k], list) or not all(isinstance(m, dict) for m in body[k]):
|
|
1557
|
+
return f"{label} has to be a list of metrics"
|
|
1525
1558
|
return ""
|
|
1526
1559
|
|
|
1527
1560
|
|
|
@@ -1926,6 +1959,10 @@ class Datasets:
|
|
|
1926
1959
|
given = False
|
|
1927
1960
|
for did, name, version, raw in rows:
|
|
1928
1961
|
body = json.loads(raw)
|
|
1962
|
+
# A version-6 body is read as version 7 (`_doc`) and saved as
|
|
1963
|
+
# one at its next edit, never rewritten here (§17).
|
|
1964
|
+
if isinstance(body, dict) and body.get("version") == 6:
|
|
1965
|
+
continue
|
|
1929
1966
|
up = upgrade_body(body)
|
|
1930
1967
|
if up is not body:
|
|
1931
1968
|
self._archive(db, did, body)
|
|
@@ -1956,8 +1993,9 @@ class Datasets:
|
|
|
1956
1993
|
|
|
1957
1994
|
@staticmethod
|
|
1958
1995
|
def _doc(r, body=True):
|
|
1959
|
-
"""A row as the API answers it: a DatasetSummary, and its body with it
|
|
1960
|
-
|
|
1996
|
+
"""A row as the API answers it: a DatasetSummary, and its body with it,
|
|
1997
|
+
read as today's version."""
|
|
1998
|
+
parsed = upgrade_body(json.loads(r["body"]))
|
|
1961
1999
|
out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
|
|
1962
2000
|
"version": r["version"], "updated": r["updated_at"]}
|
|
1963
2001
|
if body:
|
|
@@ -2101,7 +2139,7 @@ class Datasets:
|
|
|
2101
2139
|
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
|
|
2102
2140
|
"ORDER BY name COLLATE NOCASE, id").fetchall()
|
|
2103
2141
|
return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
|
|
2104
|
-
"datasets": [{"name": r["name"], "body": json.loads(r["body"])} for r in rows]}
|
|
2142
|
+
"datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
|
|
2105
2143
|
|
|
2106
2144
|
def import_file(self, doc):
|
|
2107
2145
|
"""Either export's file, as new datasets: import always creates, ids
|
|
@@ -3323,11 +3361,22 @@ class Queue:
|
|
|
3323
3361
|
"dataset TEXT)")
|
|
3324
3362
|
# A store from before runs kept their dataset gains the column;
|
|
3325
3363
|
# its rows have none, and are pinned when first they need one.
|
|
3326
|
-
|
|
3364
|
+
cols = [c[1] for c in db.execute("PRAGMA table_info(queue)")]
|
|
3365
|
+
if "dataset" not in cols:
|
|
3327
3366
|
db.execute("ALTER TABLE queue ADD COLUMN dataset TEXT")
|
|
3367
|
+
# The run a re-run was queued from (#245); every earlier row is
|
|
3368
|
+
# one of its own.
|
|
3369
|
+
if "rerun_of" not in cols:
|
|
3370
|
+
db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
|
|
3328
3371
|
|
|
3329
3372
|
# ---- rows -----------------------------------------------------------
|
|
3330
3373
|
|
|
3374
|
+
# A row read with the submit time of the run it re-runs, if any: the
|
|
3375
|
+
# page names a run by that time (History's Run ID), so a "Re-run of"
|
|
3376
|
+
# note reads without fetching the original.
|
|
3377
|
+
SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
|
|
3378
|
+
"LEFT JOIN queue o ON o.id = q.rerun_of")
|
|
3379
|
+
|
|
3331
3380
|
@staticmethod
|
|
3332
3381
|
def _row(r):
|
|
3333
3382
|
if r is None:
|
|
@@ -3338,6 +3387,8 @@ class Queue:
|
|
|
3338
3387
|
"snapshot": json.loads(r[6]), "results": json.loads(r[7]),
|
|
3339
3388
|
"progress": json.loads(r[8]), "totals": json.loads(r[9]),
|
|
3340
3389
|
"error": r[10],
|
|
3390
|
+
"rerunOf": r[12] if len(r) > 12 else None,
|
|
3391
|
+
"rerunOfAt": r[13] if len(r) > 13 else None,
|
|
3341
3392
|
}
|
|
3342
3393
|
|
|
3343
3394
|
# A row from before run documents has no version, and nothing here can
|
|
@@ -3351,12 +3402,12 @@ class Queue:
|
|
|
3351
3402
|
return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
|
|
3352
3403
|
|
|
3353
3404
|
def _all(self, db):
|
|
3354
|
-
return [row for row in (self._row(r) for r in db.execute(
|
|
3405
|
+
return [row for row in (self._row(r) for r in db.execute(self.SELECT))
|
|
3355
3406
|
if self._readable(row)]
|
|
3356
3407
|
|
|
3357
3408
|
def get(self, rid):
|
|
3358
3409
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3359
|
-
row = self._row(db.execute(
|
|
3410
|
+
row = self._row(db.execute(self.SELECT + " WHERE q.id = ?",
|
|
3360
3411
|
(rid,)).fetchone())
|
|
3361
3412
|
return row if self._readable(row) else None
|
|
3362
3413
|
|
|
@@ -3379,13 +3430,13 @@ class Queue:
|
|
|
3379
3430
|
|
|
3380
3431
|
# ---- submit ---------------------------------------------------------
|
|
3381
3432
|
|
|
3382
|
-
def submit(self, run: dict, dataset=None):
|
|
3433
|
+
def submit(self, run: dict, dataset=None, rerun_of=None):
|
|
3383
3434
|
"""
|
|
3384
3435
|
A new queued run. `run` is the run document (docs/pipeline-model.md
|
|
3385
3436
|
§5): the pipeline, the profiles it resolved to without their keys, its
|
|
3386
3437
|
content's file list in order and with its repeats, and the dataset's
|
|
3387
|
-
version; `dataset` is that version's body, for a graded run
|
|
3388
|
-
the row. Its items are that list, or the one inline text, each through
|
|
3438
|
+
version; `dataset` is that version's body, for a graded run;
|
|
3439
|
+
`rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
|
|
3389
3440
|
every scenario -- so the total is the list's length, repeats and all,
|
|
3390
3441
|
the same count the runner and the page make.
|
|
3391
3442
|
"""
|
|
@@ -3394,11 +3445,12 @@ class Queue:
|
|
|
3394
3445
|
total = len(run_items(run))
|
|
3395
3446
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3396
3447
|
db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
|
|
3397
|
-
"snapshot, results, progress, totals, dataset
|
|
3448
|
+
"snapshot, results, progress, totals, dataset, rerun_of) "
|
|
3449
|
+
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
|
|
3398
3450
|
(rid, "queued", now, json.dumps(run), "[]",
|
|
3399
3451
|
json.dumps({"current": None, "n": 0, "total": total}),
|
|
3400
3452
|
json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
|
|
3401
|
-
None if dataset is None else json.dumps(dataset)))
|
|
3453
|
+
None if dataset is None else json.dumps(dataset), rerun_of))
|
|
3402
3454
|
# Its prompts' uses, in the same transaction: a run is in the
|
|
3403
3455
|
# library the moment it is queued, or not queued at all.
|
|
3404
3456
|
if self.prompts is not None:
|
|
@@ -3406,7 +3458,7 @@ class Queue:
|
|
|
3406
3458
|
# Read back under the lock the worker dequeues under, so the
|
|
3407
3459
|
# answer is the run as it was queued: an idle worker can take it
|
|
3408
3460
|
# the moment the lock is let go.
|
|
3409
|
-
return self._row(db.execute(
|
|
3461
|
+
return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
|
|
3410
3462
|
|
|
3411
3463
|
def dataset(self, rid, raw=False):
|
|
3412
3464
|
"""The dataset body a readable run was submitted against, or None --
|
|
@@ -3525,6 +3577,46 @@ class Queue:
|
|
|
3525
3577
|
shutil.rmtree(d, ignore_errors=True)
|
|
3526
3578
|
return n
|
|
3527
3579
|
|
|
3580
|
+
def rerun(self, rid):
|
|
3581
|
+
"""
|
|
3582
|
+
A new run from a finished one's document (#245): a new id and submit
|
|
3583
|
+
time, the same document -- the file revisions and the plugins it
|
|
3584
|
+
pinned -- and the dataset body it was graded by, so a re-run measures
|
|
3585
|
+
the model again rather than changed data. The original is left as it
|
|
3586
|
+
was; the new row names it in `rerunOf`. Refused, naming what is
|
|
3587
|
+
missing, when a pinned file or dataset version is no longer kept. A
|
|
3588
|
+
Target's key is read from its profile now, at dequeue, as for any
|
|
3589
|
+
run: none is ever stored in one.
|
|
3590
|
+
"""
|
|
3591
|
+
run = self.get(rid)
|
|
3592
|
+
if run is None:
|
|
3593
|
+
return None, (404, "no such run")
|
|
3594
|
+
if run["status"] in ("queued", "running"):
|
|
3595
|
+
return None, (409, "a run still in progress cannot be re-run")
|
|
3596
|
+
snap = run["snapshot"]
|
|
3597
|
+
_, err = self._pinned_files(snap)
|
|
3598
|
+
if err:
|
|
3599
|
+
return None, (409, err)
|
|
3600
|
+
ref = evals_dataset(snap)
|
|
3601
|
+
body = None
|
|
3602
|
+
if ref is not None:
|
|
3603
|
+
body = self.dataset(rid, raw=True)
|
|
3604
|
+
if body is None:
|
|
3605
|
+
# A run that never started kept no body: the dataset's, if it
|
|
3606
|
+
# still reads as the version the run was submitted against.
|
|
3607
|
+
now = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
|
|
3608
|
+
if now is None or not ref.get("version") or now[1] != ref.get("version"):
|
|
3609
|
+
return None, (409, f"the version of the dataset {ref.get('name') or ref.get('id')!r} "
|
|
3610
|
+
"this run was submitted against is no longer kept")
|
|
3611
|
+
body = now[0]
|
|
3612
|
+
_, err = self._plugin_args(run)
|
|
3613
|
+
if err:
|
|
3614
|
+
return None, (409, err)
|
|
3615
|
+
_, err = worker_destinations(snap)
|
|
3616
|
+
if err:
|
|
3617
|
+
return None, (403, err)
|
|
3618
|
+
return self.submit(snap, body, rerun_of=rid), None
|
|
3619
|
+
|
|
3528
3620
|
def rerun_item(self, rid, index):
|
|
3529
3621
|
"""
|
|
3530
3622
|
One item against the run's snapshot, by its index, updating the row in
|
|
@@ -3671,34 +3763,48 @@ class Queue:
|
|
|
3671
3763
|
has gone fails the run with the reason stated. Returns the run dir,
|
|
3672
3764
|
or (None, error)."""
|
|
3673
3765
|
snap = run["snapshot"]
|
|
3674
|
-
|
|
3766
|
+
files, err = self._pinned_files(snap)
|
|
3767
|
+
if err:
|
|
3768
|
+
return None, err
|
|
3675
3769
|
rundir = self.dir / run["id"]
|
|
3676
3770
|
files_dir = rundir / "files"
|
|
3677
3771
|
shutil.rmtree(rundir, ignore_errors=True)
|
|
3678
3772
|
files_dir.mkdir(parents=True)
|
|
3679
|
-
|
|
3680
|
-
|
|
3681
|
-
label = ref.get("name") or ref.get("id")
|
|
3682
|
-
if SOURCES.get(ref.get("id")) is None:
|
|
3683
|
-
return None, f"the source {label!r} is gone"
|
|
3684
|
-
revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
|
|
3685
|
-
changed = []
|
|
3686
|
-
for name in dict.fromkeys(content.get("files") or []):
|
|
3687
|
-
path = SOURCES.file_path(ref["id"], name)
|
|
3688
|
-
if path is None or not path.is_file():
|
|
3689
|
-
return None, f"{name!r} is gone from the source {label!r}"
|
|
3690
|
-
if revs is not None:
|
|
3691
|
-
path = SOURCES.rev_path(ref["id"], name, revs.get(name))
|
|
3692
|
-
if path is None:
|
|
3693
|
-
changed.append(repr(name))
|
|
3694
|
-
continue
|
|
3695
|
-
shutil.copy2(path, files_dir / name)
|
|
3696
|
-
if changed:
|
|
3697
|
-
return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
|
|
3698
|
-
f"changed in the source {label!r} since this run was queued")
|
|
3773
|
+
for name, path in files:
|
|
3774
|
+
shutil.copy2(path, files_dir / name)
|
|
3699
3775
|
(rundir / "run.json").write_text(json.dumps(snap))
|
|
3700
3776
|
return rundir, None
|
|
3701
3777
|
|
|
3778
|
+
@staticmethod
|
|
3779
|
+
def _pinned_files(snap):
|
|
3780
|
+
"""Each file a run document reads, by name, and the path holding the
|
|
3781
|
+
bytes it pinned at submit -- or (None, why), naming the Source that
|
|
3782
|
+
has gone, or the files gone from it or changed since. A document
|
|
3783
|
+
with no Source reads no files."""
|
|
3784
|
+
content = content_of(snap) or {}
|
|
3785
|
+
if content.get("type") != "source":
|
|
3786
|
+
return [], None
|
|
3787
|
+
ref = content.get("ref") or {}
|
|
3788
|
+
label = ref.get("name") or ref.get("id")
|
|
3789
|
+
if SOURCES.get(ref.get("id")) is None:
|
|
3790
|
+
return None, f"the source {label!r} is gone"
|
|
3791
|
+
revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
|
|
3792
|
+
files, changed = [], []
|
|
3793
|
+
for name in dict.fromkeys(content.get("files") or []):
|
|
3794
|
+
path = SOURCES.file_path(ref["id"], name)
|
|
3795
|
+
if path is None or not path.is_file():
|
|
3796
|
+
return None, f"{name!r} is gone from the source {label!r}"
|
|
3797
|
+
if revs is not None:
|
|
3798
|
+
path = SOURCES.rev_path(ref["id"], name, revs.get(name))
|
|
3799
|
+
if path is None:
|
|
3800
|
+
changed.append(repr(name))
|
|
3801
|
+
continue
|
|
3802
|
+
files.append((name, path))
|
|
3803
|
+
if changed:
|
|
3804
|
+
return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
|
|
3805
|
+
f"changed in the source {label!r} since this run was queued")
|
|
3806
|
+
return files, None
|
|
3807
|
+
|
|
3702
3808
|
def _execute(self, run):
|
|
3703
3809
|
"""One run, FIFO. Never more than one of these at a time: the worker
|
|
3704
3810
|
is a single thread, so the queue is single-flight by construction."""
|
|
@@ -5199,6 +5305,9 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5199
5305
|
return self._send(404, b"not found", "text/plain")
|
|
5200
5306
|
|
|
5201
5307
|
def _queue_action(self, path):
|
|
5308
|
+
# Drained, so a keep-alive connection is not left holding the
|
|
5309
|
+
# page's `{}` in front of its next request.
|
|
5310
|
+
self._payload()
|
|
5202
5311
|
parts = path.split("/")
|
|
5203
5312
|
rid = parts[3]
|
|
5204
5313
|
action = parts[4] if len(parts) > 4 else ""
|
|
@@ -5206,6 +5315,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5206
5315
|
run, err = QUEUE.cancel(rid)
|
|
5207
5316
|
elif action == "resume":
|
|
5208
5317
|
run, err = QUEUE.resume(rid)
|
|
5318
|
+
elif action == "rerun" and len(parts) == 5:
|
|
5319
|
+
run, err = QUEUE.rerun(rid)
|
|
5320
|
+
if not err:
|
|
5321
|
+
return self._json(201, {"run": run})
|
|
5209
5322
|
elif action == "items" and len(parts) == 6:
|
|
5210
5323
|
run, err = QUEUE.rerun_item(rid, urllib.parse.unquote(parts[5]))
|
|
5211
5324
|
elif action == "rescore" and len(parts) == 6:
|