evals-lab 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -5,6 +5,35 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
5
5
  upgrade can rewrite what the lab keeps in your data directory, and an older
6
6
  version cannot always read it back.
7
7
 
8
+ ## 0.4.0
9
+
10
+ ### Upgrade notes
11
+
12
+ - **Copy your data directory before upgrading** (see "Where your data is
13
+ kept" in the README). A dataset 0.4.0 saves is dataset version 7, which
14
+ 0.3.x cannot read. To go back, reinstall 0.3.0 and restore the copy.
15
+ - Stored datasets are not rewritten on upgrade: 0.4.0 reads a version-6
16
+ dataset as version 7, and writes version 7 the next time it is saved.
17
+ - Export JSON writes dataset version 7, which 0.3.x refuses to import.
18
+ Import reads versions 1 to 7.
19
+
20
+ ### Added
21
+
22
+ - Re-run, in a History row's menu and in Results' run menu: a new run of the
23
+ same pipeline over the same file revisions and the same dataset version.
24
+ It is refused while the run is still going, or when one of the files it
25
+ read is gone. The new run says which run it re-ran, and links back to it
26
+ in Results. Undo cancels it while it is still queued.
27
+ - A dataset carries four new sections, `scoring`, `grader`, `every` and
28
+ `run`: how it scores, which model grades it, and metrics for every item
29
+ and for the whole run. Edit raw… shows them. Runs do not read them yet.
30
+ A dataset from 0.3.x has them as scored All, by the lab's grader, with no
31
+ metrics of its own.
32
+
33
+ ### Changed
34
+
35
+ - Resume is in a History row's menu, beside Re-run.
36
+
8
37
  ## 0.3.0
9
38
 
10
39
  ### Upgrade notes
package/lab/VERSION CHANGED
@@ -1 +1 @@
1
- 0.3.0 (2026.10.02-355)
1
+ 0.4.0 (2026.10.02-361)
@@ -983,22 +983,37 @@ const SLOTS = ["content", "target", "responses"];
983
983
 
984
984
 
985
985
 
986
-
986
+
987
987
 
988
-
988
+
989
+
990
+
991
+
992
+
993
+
994
+
995
+
996
+
997
+
989
998
 
990
999
 
991
1000
 
992
- /** The dataset body's version: 6 marks itself; 5 named its Source; 4 and
993
- earlier, neither. */
994
- const DATASET_BODY_VERSION = 6 ;
1001
+ /** An eval group's scoring (docs/pipeline-model.md §17). */
1002
+
1003
+
1004
+
1005
+
1006
+
1007
+ /** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
1008
+ named its Source; 4 and earlier, neither. */
1009
+ const DATASET_BODY_VERSION = 7 ;
995
1010
 
996
1011
  /** The metrics whose Ignore case version 6 made mean what it says for a
997
1012
  reply read as a list, as well as one read as text. */
998
1013
  const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
999
1014
 
1000
1015
  /**
1001
- * [body] as this version of a dataset (6), from any earlier one. Version 1
1016
+ * [body] as this version of a dataset (7), from any earlier one. Version 1
1002
1017
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
1003
1018
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
1004
1019
  * Version 2 held `rules`, which clean a job's answer and so belong to the
@@ -1010,7 +1025,9 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1010
1025
  * `note`, `traits` go, and the body names no Source yet. Version 5 matched
1011
1026
  * a list's items ignoring case whatever a Contains metric's Ignore case
1012
1027
  * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1013
- * the body says its version. Every reader of a
1028
+ * the body says its version. Version 6 is a group's cases alone: version 7
1029
+ * adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
1030
+ * stored row is read that way rather than rewritten. Every reader of a
1014
1031
  * body calls this: the runner, the page, a run's kept copy. A reader that
1015
1032
  * needs a version-2 body's rules -- to upgrade a pipeline graded against
1016
1033
  * it -- takes them first (`datasetRules`). Pure: the same body gives the
@@ -1021,14 +1038,26 @@ function upgradeDatasetBody (body ) {
1021
1038
  if (!isObj(body)) return body;
1022
1039
  if (body.version === DATASET_BODY_VERSION) return body;
1023
1040
  const b = body ;
1041
+ // A body saying any other version is one this lab does not read.
1042
+ if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
1024
1043
  // A body that names its Source, even as null, is version 5.
1025
1044
  if ("source" in b) {
1026
- return { version: DATASET_BODY_VERSION, ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) };
1045
+ return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
1027
1046
  }
1028
1047
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
1029
1048
  // Not a body of any version -- a copy kept while a dataset was an overlay.
1030
1049
  if (!raw) return body;
1031
- return { version: DATASET_BODY_VERSION, source: null, cases: raw.map(caseOfV4).map(caseOfV5) };
1050
+ return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
1051
+ }
1052
+
1053
+ /** A version-6 body as a version-7 eval group: scored All, the lab's
1054
+ grader, and no metrics of its own for every item or the whole run --
1055
+ what a Metrics eval naming the dataset with none of its own graded, which
1056
+ is what every Graded set converted to. */
1057
+ function groupOfV6(b ) {
1058
+ const { version: _v, ...rest } = b;
1059
+ return { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null, every: [], run: [],
1060
+ ...rest } ;
1032
1061
  }
1033
1062
 
1034
1063
  /** A version-5 case as a version-6 one: each Contains metric says Ignore
@@ -3187,15 +3216,7 @@ EVAL_TYPES.metrics = {
3187
3216
  ? { ...t, grader: { id: ctx.grader.id, name: ctx.grader.name } } : t),
3188
3217
  wholeRun: t => t.over === "run",
3189
3218
  // Over the whole run: every metric over the replies together, or each alone.
3190
- verdict(t, ress, kind){
3191
- const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
3192
- const metrics = readRun(t.metrics || [], run);
3193
- const s = scoreOf(metrics, t.mode, t.threshold);
3194
- const off = metrics.filter(r => !r.pass);
3195
- return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
3196
- ran: run.replies.length,
3197
- checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3198
- },
3219
+ verdict: (t, ress, kind) => wholeRunVerdict(t.metrics || [], ress, kind, t.mode, t.threshold),
3199
3220
  // Rule m{x} is the eval's own metric x, read in that place on every item.
3200
3221
  // A score stored before Metrics -- a Graded set's, read as its conversion --
3201
3222
  // has no readings of its own: its one metric's reading is the score's.
@@ -3296,6 +3317,38 @@ LEGACY_TESTS.single .toMetrics = t => {
3296
3317
  return { ...common(t), type: "metrics", mode: "all", threshold: null, grader: null, dataset: null, over: "run", metrics };
3297
3318
  };
3298
3319
 
3320
+ /** [list] read over a whole run's replies so far, scored the way [mode]
3321
+ says: a Metrics eval's verdict over the run, and an eval group's Whole run. */
3322
+ function wholeRunVerdict(list , ress , kind ,
3323
+ mode = "all", threshold = null) {
3324
+ const run = runInput(ress, !(kind != null && OUTPUT_KINDS[kind]?.terms));
3325
+ const metrics = readRun(list, run);
3326
+ const s = scoreOf(metrics, mode, threshold);
3327
+ const off = metrics.filter(r => !r.pass);
3328
+ return { pass: s ? s.pass : true, detail: off.length ? off.map(r => `${r.label}: ${r.reason}`).join(" · ") : "all matched",
3329
+ ran: run.replies.length,
3330
+ checks: metrics.map((r, x) => ({ key: `m${x}`, label: r.label, want: "", pass: r.pass, ...(r.pass ? {} : { note: r.reason }) })) };
3331
+ }
3332
+
3333
+ /**
3334
+ * An eval group's verdict on one item (docs/pipeline-model.md §17): its
3335
+ * Every item metrics and [kase]'s own -- the item's case in the group, or
3336
+ * null where it has none -- read of the reply under the group's scoring.
3337
+ * The reading a Metrics eval naming the group as its dataset gives, which
3338
+ * groups-check.js holds it to. Null where no metric had anything to read.
3339
+ */
3340
+ function readGroup(group , kase , res , more = {}) {
3341
+ return readMetrics([...group.every, ...(kase?.metrics ?? [])], metricInput(res, kase, more), graderCtx(group, more),
3342
+ group.scoring.mode, group.scoring.threshold);
3343
+ }
3344
+
3345
+ /** An eval group's Whole run verdict, from the replies so far: its `run`
3346
+ metrics under its scoring. Null for a group with no Whole run metrics,
3347
+ which has nothing to say of a run. */
3348
+ function readGroupRun(group , ress , kind = null) {
3349
+ return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
3350
+ }
3351
+
3299
3352
  /** The grader a Metrics eval names, reached through what the runner hands it. */
3300
3353
  function graderCtx(t , more ) {
3301
3354
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
@@ -4768,6 +4821,24 @@ function validateEvals(ev , files
4768
4821
  }
4769
4822
  if (ev.cases != null && !Array.isArray(ev.cases)) bad.push("cases has to be a list");
4770
4823
  if (ev.source != null && !isRef(ev.source)) bad.push("a dataset names its Source as { id, name }, or null");
4824
+ // An eval group's own sections (version 7): what a Metrics eval held.
4825
+ if (ev.scoring != null) {
4826
+ const sc = ev.scoring;
4827
+ if (!isObj(sc) || !SCORING_MODES.includes(sc.mode)) bad.push(`a group is scored ${SCORING_MODES.join(" or ")}`);
4828
+ else if (sc.mode === "weighted" && typeof sc.threshold !== "number") bad.push("a group scored in points needs Pass at: the points an item has to reach");
4829
+ else if (sc.threshold != null && typeof sc.threshold !== "number") bad.push("a group's Pass at has to be a number");
4830
+ }
4831
+ if (ev.grader != null && !isRef(ev.grader)) bad.push("a group names its grader as { id, name }, or null");
4832
+ if (ev.every != null) metricsProblems(ev.every, "Every item", bad);
4833
+ if (ev.run != null) {
4834
+ metricsProblems(ev.run, "Whole run", bad);
4835
+ // Read once over every reply: nothing that reads one item, or asks a grader.
4836
+ for (const m of Array.isArray(ev.run) ? ev.run : []) {
4837
+ const entry = isObj(m) ? METRICS[m.type] : undefined;
4838
+ if (entry?.graded) bad.push(`Whole run: ${entry.label} is model-graded, so it cannot read a whole run`);
4839
+ else if (entry?.perItem) bad.push(`Whole run: ${entry.label} reads one item at a time, so it cannot read a whole run`);
4840
+ }
4841
+ }
4771
4842
  if (bad.length) return bad;
4772
4843
 
4773
4844
  const graded = ev.cases || [];
@@ -4915,7 +4986,7 @@ registerMetrics({ registerKinds });
4915
4986
  export {
4916
4987
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
4917
4988
  loopReplyError, preparedSize,
4918
- termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4989
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
4919
4990
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
4920
4991
  CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
4921
4992
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
package/lab/run-evals.js CHANGED
@@ -296,8 +296,8 @@ function readDataset(file) {
296
296
  broken(`${file}: ${e.message}`);
297
297
  }
298
298
  if (isObject(doc) && doc.format === "evals-lab/dataset") {
299
- if (![1, 2, 3, 4, 5].includes(doc.version)) {
300
- broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 5`);
299
+ if (![1, 2, 3, 4, 5, 6, 7].includes(doc.version)) {
300
+ broken(`${file} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to 7`);
301
301
  }
302
302
  doc = isObject(doc.dataset) ? doc.dataset.body : null;
303
303
  }
package/lab/server.py CHANGED
@@ -1356,23 +1356,25 @@ class Sources:
1356
1356
  # A lab from before this held a one-time import's `meta` row saying it ran;
1357
1357
  # it is left where it is, and nothing reads it.
1358
1358
 
1359
- DATASET_FIELDS = ("version", "source", "cases")
1360
- # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 5 was
1361
- # told by its `source` alone, and earlier ones by neither.
1362
- DATASET_BODY_VERSION = 6
1359
+ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
1360
+ # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 7 is an
1361
+ # eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
1362
+ # 5 was told by its `source` alone, and earlier ones by neither.
1363
+ DATASET_BODY_VERSION = 7
1363
1364
  DATASET_NAME_MAX = 80
1364
- # The file forms Export writes and Import reads. Export writes version 6;
1365
- # Import reads it and versions 1 to 5, upgraded, and refuses anything else,
1365
+ # The file forms Export writes and Import reads. Export writes version 7;
1366
+ # Import reads it and versions 1 to 6, upgraded, and refuses anything else,
1366
1367
  # as a pipeline of another version is refused. Versions 1 to 3 carried a
1367
1368
  # prompt, which an import gives to the Prompt library.
1368
1369
  EXPORT_ONE = "evals-lab/dataset"
1369
1370
  EXPORT_ALL = "evals-lab/datasets"
1370
- EXPORT_VERSION = 6
1371
- IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
1371
+ EXPORT_VERSION = 7
1372
+ IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
1373
+ SCORING_MODES = ("all", "weighted")
1372
1374
 
1373
1375
 
1374
1376
  def blank_dataset() -> dict:
1375
- return {"version": DATASET_BODY_VERSION, "source": None, "cases": []}
1377
+ return group_of_v6({"source": None, "cases": []})
1376
1378
 
1377
1379
 
1378
1380
  # vocab: the names older versions gave a case's fields
@@ -1400,7 +1402,7 @@ def _term_in(items, term) -> bool:
1400
1402
  def case_metrics(c: dict) -> list:
1401
1403
  """A version-4 case's expectations as the metrics that say the same:
1402
1404
  evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
1403
- through fixtures/dataset-v6.json."""
1405
+ through fixtures/dataset-v7.json."""
1404
1406
  def strs(v):
1405
1407
  return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
1406
1408
  if c.get("discarded") is True:
@@ -1466,8 +1468,18 @@ def case_of_v5(c):
1466
1468
  else m for m in c["metrics"]]}
1467
1469
 
1468
1470
 
1471
+ def group_of_v6(body: dict) -> dict:
1472
+ """A version-6 body as a version-7 eval group: evals-core.ts's
1473
+ groupOfV6. Scored All, the lab's grader, and no metrics of its own for
1474
+ every item or the whole run -- what a Metrics eval naming the dataset with
1475
+ none of its own graded."""
1476
+ rest = {k: v for k, v in body.items() if k != "version"}
1477
+ return {"version": DATASET_BODY_VERSION, "source": None, "scoring": {"mode": "all", "threshold": None},
1478
+ "grader": None, "every": [], "run": [], **rest}
1479
+
1480
+
1469
1481
  def upgrade_body(body):
1470
- """An earlier body as today's (version 6): evals-core.ts's
1482
+ """An earlier body as today's (version 7): evals-core.ts's
1471
1483
  upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
1472
1484
  its `replays` and `conformance` go (fixtures/replays.json holds the
1473
1485
  parser's tests). Version 2's `rules` go -- they clean a job's answer, so
@@ -1477,19 +1489,25 @@ def upgrade_body(body):
1477
1489
  `note`, `traits` go, and the body names no Source yet. A caller that
1478
1490
  needs the rules or the prompt takes them first (`body_rules`,
1479
1491
  `body_prompt`). A body naming its Source is version 5, whose Contains
1480
- metrics each come to say Ignore case (`case_of_v5`). A body saying it is
1481
- version 6 comes back as it was; so does anything that is not a body."""
1492
+ metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
1493
+ group's scoring, grader, Every item and Whole run (`group_of_v6`). A body
1494
+ saying it is version 7 comes back as it was; so does anything that is
1495
+ not a body."""
1482
1496
  if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
1483
1497
  return body
1498
+ # A body saying any other version is one this lab does not read, and is
1499
+ # left for dataset_problem to refuse.
1500
+ if "version" in body:
1501
+ return group_of_v6(body) if body["version"] == 6 else body
1484
1502
  if "source" in body:
1485
- up = {"version": DATASET_BODY_VERSION, **body}
1503
+ up = dict(body)
1486
1504
  if isinstance(body.get("cases"), list):
1487
1505
  up["cases"] = [case_of_v5(c) for c in body["cases"]]
1488
- return up
1506
+ return group_of_v6(up)
1489
1507
  cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
1490
1508
  if not isinstance(cases, list):
1491
1509
  return body
1492
- return {"version": DATASET_BODY_VERSION, "source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}
1510
+ return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
1493
1511
 
1494
1512
 
1495
1513
  def body_prompt(body):
@@ -1518,10 +1536,25 @@ def dataset_problem(body) -> str:
1518
1536
  return f"a dataset's body is version {DATASET_BODY_VERSION}"
1519
1537
  if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
1520
1538
  return "cases has to be a list of cases"
1521
- src = body["source"]
1522
- if src is not None and not (isinstance(src, dict) and isinstance(src.get("id"), str)
1523
- and isinstance(src.get("name"), str)):
1539
+ def is_ref(v):
1540
+ return isinstance(v, dict) and isinstance(v.get("id"), str) and isinstance(v.get("name"), str)
1541
+ if body["source"] is not None and not is_ref(body["source"]):
1524
1542
  return "a dataset names its Source as { id, name }, or null"
1543
+ # A group's own sections: their metrics are evals-core.ts's validateEvals'
1544
+ # to judge, as a case's are; the shape is the server's.
1545
+ sc = body["scoring"]
1546
+ if not isinstance(sc, dict) or sc.get("mode") not in SCORING_MODES:
1547
+ return "a group is scored all or weighted"
1548
+ number = lambda v: type(v) in (int, float)
1549
+ if sc["mode"] == "weighted" and not number(sc.get("threshold")):
1550
+ return "a group scored in points needs Pass at: the points an item has to reach"
1551
+ if sc.get("threshold") is not None and not number(sc.get("threshold")):
1552
+ return "a group's Pass at has to be a number"
1553
+ if body["grader"] is not None and not is_ref(body["grader"]):
1554
+ return "a group names its grader as { id, name }, or null"
1555
+ for k, label in (("every", "Every item"), ("run", "Whole run")):
1556
+ if not isinstance(body[k], list) or not all(isinstance(m, dict) for m in body[k]):
1557
+ return f"{label} has to be a list of metrics"
1525
1558
  return ""
1526
1559
 
1527
1560
 
@@ -1926,6 +1959,10 @@ class Datasets:
1926
1959
  given = False
1927
1960
  for did, name, version, raw in rows:
1928
1961
  body = json.loads(raw)
1962
+ # A version-6 body is read as version 7 (`_doc`) and saved as
1963
+ # one at its next edit, never rewritten here (§17).
1964
+ if isinstance(body, dict) and body.get("version") == 6:
1965
+ continue
1929
1966
  up = upgrade_body(body)
1930
1967
  if up is not body:
1931
1968
  self._archive(db, did, body)
@@ -1956,8 +1993,9 @@ class Datasets:
1956
1993
 
1957
1994
  @staticmethod
1958
1995
  def _doc(r, body=True):
1959
- """A row as the API answers it: a DatasetSummary, and its body with it."""
1960
- parsed = json.loads(r["body"])
1996
+ """A row as the API answers it: a DatasetSummary, and its body with it,
1997
+ read as today's version."""
1998
+ parsed = upgrade_body(json.loads(r["body"]))
1961
1999
  out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
1962
2000
  "version": r["version"], "updated": r["updated_at"]}
1963
2001
  if body:
@@ -2101,7 +2139,7 @@ class Datasets:
2101
2139
  rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
2102
2140
  "ORDER BY name COLLATE NOCASE, id").fetchall()
2103
2141
  return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
2104
- "datasets": [{"name": r["name"], "body": json.loads(r["body"])} for r in rows]}
2142
+ "datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2105
2143
 
2106
2144
  def import_file(self, doc):
2107
2145
  """Either export's file, as new datasets: import always creates, ids
@@ -3323,11 +3361,22 @@ class Queue:
3323
3361
  "dataset TEXT)")
3324
3362
  # A store from before runs kept their dataset gains the column;
3325
3363
  # its rows have none, and are pinned when first they need one.
3326
- if "dataset" not in [c[1] for c in db.execute("PRAGMA table_info(queue)")]:
3364
+ cols = [c[1] for c in db.execute("PRAGMA table_info(queue)")]
3365
+ if "dataset" not in cols:
3327
3366
  db.execute("ALTER TABLE queue ADD COLUMN dataset TEXT")
3367
+ # The run a re-run was queued from (#245); every earlier row is
3368
+ # one of its own.
3369
+ if "rerun_of" not in cols:
3370
+ db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
3328
3371
 
3329
3372
  # ---- rows -----------------------------------------------------------
3330
3373
 
3374
+ # A row read with the submit time of the run it re-runs, if any: the
3375
+ # page names a run by that time (History's Run ID), so a "Re-run of"
3376
+ # note reads without fetching the original.
3377
+ SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
3378
+ "LEFT JOIN queue o ON o.id = q.rerun_of")
3379
+
3331
3380
  @staticmethod
3332
3381
  def _row(r):
3333
3382
  if r is None:
@@ -3338,6 +3387,8 @@ class Queue:
3338
3387
  "snapshot": json.loads(r[6]), "results": json.loads(r[7]),
3339
3388
  "progress": json.loads(r[8]), "totals": json.loads(r[9]),
3340
3389
  "error": r[10],
3390
+ "rerunOf": r[12] if len(r) > 12 else None,
3391
+ "rerunOfAt": r[13] if len(r) > 13 else None,
3341
3392
  }
3342
3393
 
3343
3394
  # A row from before run documents has no version, and nothing here can
@@ -3351,12 +3402,12 @@ class Queue:
3351
3402
  return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
3352
3403
 
3353
3404
  def _all(self, db):
3354
- return [row for row in (self._row(r) for r in db.execute("SELECT * FROM queue"))
3405
+ return [row for row in (self._row(r) for r in db.execute(self.SELECT))
3355
3406
  if self._readable(row)]
3356
3407
 
3357
3408
  def get(self, rid):
3358
3409
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3359
- row = self._row(db.execute("SELECT * FROM queue WHERE id = ?",
3410
+ row = self._row(db.execute(self.SELECT + " WHERE q.id = ?",
3360
3411
  (rid,)).fetchone())
3361
3412
  return row if self._readable(row) else None
3362
3413
 
@@ -3379,13 +3430,13 @@ class Queue:
3379
3430
 
3380
3431
  # ---- submit ---------------------------------------------------------
3381
3432
 
3382
- def submit(self, run: dict, dataset=None):
3433
+ def submit(self, run: dict, dataset=None, rerun_of=None):
3383
3434
  """
3384
3435
  A new queued run. `run` is the run document (docs/pipeline-model.md
3385
3436
  §5): the pipeline, the profiles it resolved to without their keys, its
3386
3437
  content's file list in order and with its repeats, and the dataset's
3387
- version; `dataset` is that version's body, for a graded run. Returns
3388
- the row. Its items are that list, or the one inline text, each through
3438
+ version; `dataset` is that version's body, for a graded run;
3439
+ `rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
3389
3440
  every scenario -- so the total is the list's length, repeats and all,
3390
3441
  the same count the runner and the page make.
3391
3442
  """
@@ -3394,11 +3445,12 @@ class Queue:
3394
3445
  total = len(run_items(run))
3395
3446
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3396
3447
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
3397
- "snapshot, results, progress, totals, dataset) VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?)",
3448
+ "snapshot, results, progress, totals, dataset, rerun_of) "
3449
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
3398
3450
  (rid, "queued", now, json.dumps(run), "[]",
3399
3451
  json.dumps({"current": None, "n": 0, "total": total}),
3400
3452
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
3401
- None if dataset is None else json.dumps(dataset)))
3453
+ None if dataset is None else json.dumps(dataset), rerun_of))
3402
3454
  # Its prompts' uses, in the same transaction: a run is in the
3403
3455
  # library the moment it is queued, or not queued at all.
3404
3456
  if self.prompts is not None:
@@ -3406,7 +3458,7 @@ class Queue:
3406
3458
  # Read back under the lock the worker dequeues under, so the
3407
3459
  # answer is the run as it was queued: an idle worker can take it
3408
3460
  # the moment the lock is let go.
3409
- return self._row(db.execute("SELECT * FROM queue WHERE id = ?", (rid,)).fetchone())
3461
+ return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
3410
3462
 
3411
3463
  def dataset(self, rid, raw=False):
3412
3464
  """The dataset body a readable run was submitted against, or None --
@@ -3525,6 +3577,46 @@ class Queue:
3525
3577
  shutil.rmtree(d, ignore_errors=True)
3526
3578
  return n
3527
3579
 
3580
+ def rerun(self, rid):
3581
+ """
3582
+ A new run from a finished one's document (#245): a new id and submit
3583
+ time, the same document -- the file revisions and the plugins it
3584
+ pinned -- and the dataset body it was graded by, so a re-run measures
3585
+ the model again rather than changed data. The original is left as it
3586
+ was; the new row names it in `rerunOf`. Refused, naming what is
3587
+ missing, when a pinned file or dataset version is no longer kept. A
3588
+ Target's key is read from its profile now, at dequeue, as for any
3589
+ run: none is ever stored in one.
3590
+ """
3591
+ run = self.get(rid)
3592
+ if run is None:
3593
+ return None, (404, "no such run")
3594
+ if run["status"] in ("queued", "running"):
3595
+ return None, (409, "a run still in progress cannot be re-run")
3596
+ snap = run["snapshot"]
3597
+ _, err = self._pinned_files(snap)
3598
+ if err:
3599
+ return None, (409, err)
3600
+ ref = evals_dataset(snap)
3601
+ body = None
3602
+ if ref is not None:
3603
+ body = self.dataset(rid, raw=True)
3604
+ if body is None:
3605
+ # A run that never started kept no body: the dataset's, if it
3606
+ # still reads as the version the run was submitted against.
3607
+ now = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
3608
+ if now is None or not ref.get("version") or now[1] != ref.get("version"):
3609
+ return None, (409, f"the version of the dataset {ref.get('name') or ref.get('id')!r} "
3610
+ "this run was submitted against is no longer kept")
3611
+ body = now[0]
3612
+ _, err = self._plugin_args(run)
3613
+ if err:
3614
+ return None, (409, err)
3615
+ _, err = worker_destinations(snap)
3616
+ if err:
3617
+ return None, (403, err)
3618
+ return self.submit(snap, body, rerun_of=rid), None
3619
+
3528
3620
  def rerun_item(self, rid, index):
3529
3621
  """
3530
3622
  One item against the run's snapshot, by its index, updating the row in
@@ -3671,34 +3763,48 @@ class Queue:
3671
3763
  has gone fails the run with the reason stated. Returns the run dir,
3672
3764
  or (None, error)."""
3673
3765
  snap = run["snapshot"]
3674
- content = content_of(snap) or {}
3766
+ files, err = self._pinned_files(snap)
3767
+ if err:
3768
+ return None, err
3675
3769
  rundir = self.dir / run["id"]
3676
3770
  files_dir = rundir / "files"
3677
3771
  shutil.rmtree(rundir, ignore_errors=True)
3678
3772
  files_dir.mkdir(parents=True)
3679
- if content.get("type") == "source":
3680
- ref = content.get("ref") or {}
3681
- label = ref.get("name") or ref.get("id")
3682
- if SOURCES.get(ref.get("id")) is None:
3683
- return None, f"the source {label!r} is gone"
3684
- revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
3685
- changed = []
3686
- for name in dict.fromkeys(content.get("files") or []):
3687
- path = SOURCES.file_path(ref["id"], name)
3688
- if path is None or not path.is_file():
3689
- return None, f"{name!r} is gone from the source {label!r}"
3690
- if revs is not None:
3691
- path = SOURCES.rev_path(ref["id"], name, revs.get(name))
3692
- if path is None:
3693
- changed.append(repr(name))
3694
- continue
3695
- shutil.copy2(path, files_dir / name)
3696
- if changed:
3697
- return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
3698
- f"changed in the source {label!r} since this run was queued")
3773
+ for name, path in files:
3774
+ shutil.copy2(path, files_dir / name)
3699
3775
  (rundir / "run.json").write_text(json.dumps(snap))
3700
3776
  return rundir, None
3701
3777
 
3778
+ @staticmethod
3779
+ def _pinned_files(snap):
3780
+ """Each file a run document reads, by name, and the path holding the
3781
+ bytes it pinned at submit -- or (None, why), naming the Source that
3782
+ has gone, or the files gone from it or changed since. A document
3783
+ with no Source reads no files."""
3784
+ content = content_of(snap) or {}
3785
+ if content.get("type") != "source":
3786
+ return [], None
3787
+ ref = content.get("ref") or {}
3788
+ label = ref.get("name") or ref.get("id")
3789
+ if SOURCES.get(ref.get("id")) is None:
3790
+ return None, f"the source {label!r} is gone"
3791
+ revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
3792
+ files, changed = [], []
3793
+ for name in dict.fromkeys(content.get("files") or []):
3794
+ path = SOURCES.file_path(ref["id"], name)
3795
+ if path is None or not path.is_file():
3796
+ return None, f"{name!r} is gone from the source {label!r}"
3797
+ if revs is not None:
3798
+ path = SOURCES.rev_path(ref["id"], name, revs.get(name))
3799
+ if path is None:
3800
+ changed.append(repr(name))
3801
+ continue
3802
+ files.append((name, path))
3803
+ if changed:
3804
+ return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
3805
+ f"changed in the source {label!r} since this run was queued")
3806
+ return files, None
3807
+
3702
3808
  def _execute(self, run):
3703
3809
  """One run, FIFO. Never more than one of these at a time: the worker
3704
3810
  is a single thread, so the queue is single-flight by construction."""
@@ -5199,6 +5305,9 @@ class Handler(BaseHTTPRequestHandler):
5199
5305
  return self._send(404, b"not found", "text/plain")
5200
5306
 
5201
5307
  def _queue_action(self, path):
5308
+ # Drained, so a keep-alive connection is not left holding the
5309
+ # page's `{}` in front of its next request.
5310
+ self._payload()
5202
5311
  parts = path.split("/")
5203
5312
  rid = parts[3]
5204
5313
  action = parts[4] if len(parts) > 4 else ""
@@ -5206,6 +5315,10 @@ class Handler(BaseHTTPRequestHandler):
5206
5315
  run, err = QUEUE.cancel(rid)
5207
5316
  elif action == "resume":
5208
5317
  run, err = QUEUE.resume(rid)
5318
+ elif action == "rerun" and len(parts) == 5:
5319
+ run, err = QUEUE.rerun(rid)
5320
+ if not err:
5321
+ return self._json(201, {"run": run})
5209
5322
  elif action == "items" and len(parts) == 6:
5210
5323
  run, err = QUEUE.rerun_item(rid, urllib.parse.unquote(parts[5]))
5211
5324
  elif action == "rescore" and len(parts) == 6: