evals-lab 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
24
24
 
25
25
  import base64
26
26
  import calendar
27
+ import gzip
27
28
  import hashlib
28
29
  import hmac
29
30
  import io
@@ -247,6 +248,35 @@ TYPES = {
247
248
  ".png": "image/png",
248
249
  }
249
250
 
251
+ # What is gzipped for a client that asks (#225): text, which compresses five
252
+ # to ten times, and nothing already compressed. A body under GZIP_MIN goes as
253
+ # it is, since the header and the gzip frame would outweigh the saving.
254
+ COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
255
+ GZIP_MIN = 1024
256
+ # The built page's bundles, gzipped once: a new build is new names.
257
+ GZIPPED: dict = {}
258
+
259
+
260
+ def accepts_gzip(header: str) -> bool:
261
+ """Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
262
+ above 0. `gzip;q=0` is a refusal, not a request."""
263
+ star = False
264
+ for part in (header or "").split(","):
265
+ name, _, params = part.partition(";")
266
+ name, q = name.strip().lower(), 1.0
267
+ for p in params.split(";"):
268
+ k, _, v = p.partition("=")
269
+ if k.strip().lower() == "q":
270
+ try:
271
+ q = float(v)
272
+ except ValueError:
273
+ q = 0.0
274
+ if name == "gzip":
275
+ return q > 0
276
+ if name == "*":
277
+ star = q > 0
278
+ return star
279
+
250
280
 
251
281
  # What a Source may accept and ever be served back as -- deliberately not
252
282
  # TYPES: an upload that could come back as text/html is stored XSS, and the
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
469
499
  # A key taken off this list is no longer served or written, and its rows stay
470
500
  # in the store: a document is not migrated or deleted because nothing reads it.
471
501
  # promptlab.cases, .rules and their .base copies went that way when a dataset
472
- # became a row of its own (#86).
502
+ # became a row of its own (#86), and promptlab.mappings once a dataset named
503
+ # the Source it grades (#199).
473
504
  SYNCED = ("promptlab.workflows", "promptlab.profiles",
474
- "promptlab.versions", "promptlab.tokens", "promptlab.mappings")
505
+ "promptlab.versions", "promptlab.tokens")
475
506
  MAX_DOC = 8 * 1024 * 1024
476
507
  RUNS_PAGE = 25
477
508
  # A dataset request's caps, read from Content-Length before the body is, as a
@@ -1306,7 +1337,7 @@ class Sources:
1306
1337
 
1307
1338
  # ---- Datasets ----------------------------------------------------------------
1308
1339
  #
1309
- # A dataset is data a graded test names by id: its cases. (The prompt a new
1340
+ # A dataset is data a graded eval names by id: its cases. (The prompt a new
1310
1341
  # scenario starts from is the Prompt library's Default, below; a dataset from
1311
1342
  # before the library held one, and gave it to the library once.) It is one row in the
1312
1343
  # store's SQLite, its body one JSON document with a version that goes up by one
@@ -1325,44 +1356,158 @@ class Sources:
1325
1356
  # A lab from before this held a one-time import's `meta` row saying it ran;
1326
1357
  # it is left where it is, and nothing reads it.
1327
1358
 
1328
- DATASET_FIELDS = ("cases",)
1359
+ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
1360
+ # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 7 is an
1361
+ # eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
1362
+ # 5 was told by its `source` alone, and earlier ones by neither.
1363
+ DATASET_BODY_VERSION = 7
1329
1364
  DATASET_NAME_MAX = 80
1330
- # The file forms Export writes and Import reads. Export writes version 4;
1331
- # Import reads it and versions 1 to 3, upgraded, and refuses anything else,
1365
+ # The file forms Export writes and Import reads. Export writes version 7;
1366
+ # Import reads it and versions 1 to 6, upgraded, and refuses anything else,
1332
1367
  # as a pipeline of another version is refused. Versions 1 to 3 carried a
1333
1368
  # prompt, which an import gives to the Prompt library.
1334
1369
  EXPORT_ONE = "evals-lab/dataset"
1335
1370
  EXPORT_ALL = "evals-lab/datasets"
1336
- EXPORT_VERSION = 4
1337
- IMPORT_VERSIONS = (1, 2, 3, 4)
1371
+ EXPORT_VERSION = 7
1372
+ IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
1373
+ SCORING_MODES = ("all", "weighted")
1338
1374
 
1339
1375
 
1340
1376
  def blank_dataset() -> dict:
1341
- return {"cases": []}
1377
+ return group_of_v6({"source": None, "cases": []})
1378
+
1379
+
1380
+ # vocab: the names older versions gave a case's fields
1381
+ CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
1382
+ CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
1383
+
1384
+
1385
+ def _words(s) -> list:
1386
+ """evals-core.ts's words: the letters and digits of [s], lowercased."""
1387
+ return re.findall(r"[^\W_]+", str(s or "").lower())
1388
+
1389
+
1390
+ def _term_in(items, term) -> bool:
1391
+ """evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
1392
+ t = _words(term)
1393
+ if not t:
1394
+ return False
1395
+ for item in items:
1396
+ w = _words(item)
1397
+ if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
1398
+ return True
1399
+ return False
1400
+
1401
+
1402
+ def case_metrics(c: dict) -> list:
1403
+ """A version-4 case's expectations as the metrics that say the same:
1404
+ evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
1405
+ through fixtures/dataset-v7.json."""
1406
+ def strs(v):
1407
+ return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
1408
+ if c.get("discarded") is True:
1409
+ return [{"type": "discarded"}]
1410
+ out = []
1411
+ expect, allow = strs(c.get("expect")), strs(c.get("allow"))
1412
+ if expect:
1413
+ out.append({"type": "contains-all", "values": "\n".join(expect)})
1414
+ for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
1415
+ if strs(g):
1416
+ out.append({"type": "contains-any", "values": "\n".join(strs(g))})
1417
+ for t in strs(c.get("forbid")):
1418
+ # An exception excuses only the forbidden term inside it.
1419
+ except_ = [a for a in allow if _term_in([a], t)]
1420
+ out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
1421
+ whole = lambda v: v if type(v) is int else None
1422
+ lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
1423
+ if lo is not None or hi is not None:
1424
+ out.append({"type": "item-count", "min": lo, "max": hi})
1425
+ for t in strs(c.get("watch")):
1426
+ out.append({"type": "contains-any", "values": t, "weight": 0})
1427
+ return out
1428
+
1429
+
1430
+ def case_of_v4(c):
1431
+ """One case of any earlier version as a version-5 one: evals-core.ts's
1432
+ caseOfV4, in Python."""
1433
+ if not isinstance(c, dict):
1434
+ return c
1435
+ was = {}
1436
+ for k, v in c.items():
1437
+ key = CASE_RENAMED.get(k, k)
1438
+ if key not in was or key == k:
1439
+ was[key] = v
1440
+ if isinstance(was.get("item"), str):
1441
+ was.pop("filename", None)
1442
+ out = {}
1443
+ if "id" in was:
1444
+ out["id"] = was["id"]
1445
+ out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
1446
+ out["todo"] = was.get("todo") is True
1447
+ out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
1448
+ out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
1449
+ for k, v in was.items():
1450
+ if k not in out and k not in CASE_V4:
1451
+ out[k] = v
1452
+ return out
1453
+
1454
+
1455
+ # The metrics whose Ignore case version 6 made mean what it says for a reply
1456
+ # read as a list: evals-core.ts's CASE_FOLDING.
1457
+ CASE_FOLDING = ("contains", "contains-all", "contains-any")
1458
+
1459
+
1460
+ def case_of_v5(c):
1461
+ """A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
1462
+ Python. Each Contains metric says Ignore case, as version 5 matched a
1463
+ list's items whatever it said."""
1464
+ if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
1465
+ return c
1466
+ return {**c, "metrics": [{**m, "ignoreCase": True}
1467
+ if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
1468
+ else m for m in c["metrics"]]}
1469
+
1470
+
1471
+ def group_of_v6(body: dict) -> dict:
1472
+ """A version-6 body as a version-7 eval group: evals-core.ts's
1473
+ groupOfV6. Scored All, the lab's grader, and no metrics of its own for
1474
+ every item or the whole run -- what a Metrics eval naming the dataset with
1475
+ none of its own graded."""
1476
+ rest = {k: v for k, v in body.items() if k != "version"}
1477
+ return {"version": DATASET_BODY_VERSION, "source": None, "scoring": {"mode": "all", "threshold": None},
1478
+ "grader": None, "every": [], "run": [], **rest}
1342
1479
 
1343
1480
 
1344
1481
  def upgrade_body(body):
1345
- """An earlier body as today's: evals-core.ts's upgradeDatasetBody, in
1346
- Python. Version 1's `imageCases` are `cases`, each case's
1347
- `minTags`/`maxTags` its `minCount`/`maxCount`, and its `replays` and
1348
- `conformance` go (fixtures/replays.json holds the parser's tests).
1349
- Version 2's `rules` go -- they clean a job's answer, so they are the
1350
- job's -- and the terms a case watches for are its `watch`. Version 3's
1351
- `prompt` goes: the Prompt library holds prompts now. A caller that needs
1352
- the rules or the prompt takes them first (`body_rules`, `body_prompt`).
1353
- Anything else comes back as it was."""
1354
- if not isinstance(body, dict):
1482
+ """An earlier body as today's (version 7): evals-core.ts's
1483
+ upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
1484
+ its `replays` and `conformance` go (fixtures/replays.json holds the
1485
+ parser's tests). Version 2's `rules` go -- they clean a job's answer, so
1486
+ they are the job's. Version 3's `prompt` goes: the Prompt library holds
1487
+ prompts now. Version 4's case named its item `filename` and said what a
1488
+ good answer is in expectations; each becomes its metric, `why` the
1489
+ `note`, `traits` go, and the body names no Source yet. A caller that
1490
+ needs the rules or the prompt takes them first (`body_rules`,
1491
+ `body_prompt`). A body naming its Source is version 5, whose Contains
1492
+ metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
1493
+ group's scoring, grader, Every item and Whole run (`group_of_v6`). A body
1494
+ saying it is version 7 comes back as it was; so does anything that is
1495
+ not a body."""
1496
+ if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
1355
1497
  return body
1356
- if "imageCases" not in body and "rules" not in body:
1357
- if "prompt" not in body:
1358
- return body
1359
- return {k: v for k, v in body.items() if k != "prompt"}
1360
- renamed = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch"} # vocab: older names
1498
+ # A body saying any other version is one this lab does not read, and is
1499
+ # left for dataset_problem to refuse.
1500
+ if "version" in body:
1501
+ return group_of_v6(body) if body["version"] == 6 else body
1502
+ if "source" in body:
1503
+ up = dict(body)
1504
+ if isinstance(body.get("cases"), list):
1505
+ up["cases"] = [case_of_v5(c) for c in body["cases"]]
1506
+ return group_of_v6(up)
1361
1507
  cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
1362
1508
  if not isinstance(cases, list):
1363
1509
  return body
1364
- cases = [{renamed.get(k, k): v for k, v in c.items()} if isinstance(c, dict) else c for c in cases]
1365
- return {"cases": canonical_cases(cases)}
1510
+ return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
1366
1511
 
1367
1512
 
1368
1513
  def body_prompt(body):
@@ -1377,21 +1522,6 @@ def body_rules(body):
1377
1522
  return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
1378
1523
 
1379
1524
 
1380
- def canonical_cases(cases: list) -> list:
1381
- """Every case naming its file as `filename`: evals-core.ts's
1382
- canonicalCases, in Python. `old` is the key a set re-synced from an
1383
- app's own repository arrives with; the key keeps its place, so only its
1384
- spelling changes."""
1385
- old = "photo" # vocab: the older spelling of filename
1386
- out = []
1387
- for c in cases:
1388
- if isinstance(c, dict) and old in c:
1389
- c = {("filename" if k == old else k): v for k, v in c.items()
1390
- if not (k == old and "filename" in c)}
1391
- out.append(c)
1392
- return out
1393
-
1394
-
1395
1525
  def dataset_problem(body) -> str:
1396
1526
  """Why [body] is not a dataset's body, in one sentence, or ""."""
1397
1527
  if not isinstance(body, dict):
@@ -1402,8 +1532,29 @@ def dataset_problem(body) -> str:
1402
1532
  for k in DATASET_FIELDS:
1403
1533
  if k not in body:
1404
1534
  return f"a dataset's body has no \"{k}\""
1535
+ if body["version"] != DATASET_BODY_VERSION:
1536
+ return f"a dataset's body is version {DATASET_BODY_VERSION}"
1405
1537
  if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
1406
1538
  return "cases has to be a list of cases"
1539
+ def is_ref(v):
1540
+ return isinstance(v, dict) and isinstance(v.get("id"), str) and isinstance(v.get("name"), str)
1541
+ if body["source"] is not None and not is_ref(body["source"]):
1542
+ return "a dataset names its Source as { id, name }, or null"
1543
+ # A group's own sections: their metrics are evals-core.ts's validateEvals'
1544
+ # to judge, as a case's are; the shape is the server's.
1545
+ sc = body["scoring"]
1546
+ if not isinstance(sc, dict) or sc.get("mode") not in SCORING_MODES:
1547
+ return "a group is scored all or weighted"
1548
+ number = lambda v: type(v) in (int, float)
1549
+ if sc["mode"] == "weighted" and not number(sc.get("threshold")):
1550
+ return "a group scored in points needs Pass at: the points an item has to reach"
1551
+ if sc.get("threshold") is not None and not number(sc.get("threshold")):
1552
+ return "a group's Pass at has to be a number"
1553
+ if body["grader"] is not None and not is_ref(body["grader"]):
1554
+ return "a group names its grader as { id, name }, or null"
1555
+ for k, label in (("every", "Every item"), ("run", "Whole run")):
1556
+ if not isinstance(body[k], list) or not all(isinstance(m, dict) for m in body[k]):
1557
+ return f"{label} has to be a list of metrics"
1407
1558
  return ""
1408
1559
 
1409
1560
 
@@ -1788,6 +1939,11 @@ class Datasets:
1788
1939
  # so nothing is lost if a pipeline was missed.
1789
1940
  db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
1790
1941
  "dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
1942
+ # Each row's body as it was before the conversion below rewrote
1943
+ # it (#199): a version-4 case's expectations became metrics, and
1944
+ # the body it was typed as is kept, as the rules were.
1945
+ db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
1946
+ "dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
1791
1947
  # Rows from an earlier version are converted once, in place: a
1792
1948
  # dataset is typed in by hand and costly to re-enter, so it is
1793
1949
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -1803,9 +1959,15 @@ class Datasets:
1803
1959
  given = False
1804
1960
  for did, name, version, raw in rows:
1805
1961
  body = json.loads(raw)
1962
+ # A version-6 body is read as version 7 (`_doc`) and saved as
1963
+ # one at its next edit, never rewritten here (§17).
1964
+ if isinstance(body, dict) and body.get("version") == 6:
1965
+ continue
1806
1966
  up = upgrade_body(body)
1807
1967
  if up is not body:
1808
1968
  self._archive(db, did, body)
1969
+ db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
1970
+ (did, raw, self._now()))
1809
1971
  if prompts is not None:
1810
1972
  given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
1811
1973
  db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
@@ -1831,8 +1993,9 @@ class Datasets:
1831
1993
 
1832
1994
  @staticmethod
1833
1995
  def _doc(r, body=True):
1834
- """A row as the API answers it: a DatasetSummary, and its body with it."""
1835
- parsed = json.loads(r["body"])
1996
+ """A row as the API answers it: a DatasetSummary, and its body with it,
1997
+ read as today's version."""
1998
+ parsed = upgrade_body(json.loads(r["body"]))
1836
1999
  out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
1837
2000
  "version": r["version"], "updated": r["updated_at"]}
1838
2001
  if body:
@@ -1874,7 +2037,6 @@ class Datasets:
1874
2037
  def _insert(self, db, name, body):
1875
2038
  did = secrets.token_hex(6)
1876
2039
  now = self._now()
1877
- body = {**body, "cases": canonical_cases(body["cases"])}
1878
2040
  db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
1879
2041
  "VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
1880
2042
  return did
@@ -1885,7 +2047,7 @@ class Datasets:
1885
2047
  name, why = dataset_name(name)
1886
2048
  if why:
1887
2049
  return None, (400, why)
1888
- body = blank_dataset() if body is None else body
2050
+ body = blank_dataset() if body is None else upgrade_body(body)
1889
2051
  why = dataset_problem(body)
1890
2052
  if why:
1891
2053
  return None, (400, why)
@@ -1914,10 +2076,12 @@ class Datasets:
1914
2076
  the current row."""
1915
2077
  if type(version) is not int:
1916
2078
  return None, (400, "a save names the version it began from")
2079
+ # A body of an earlier version -- from a page loaded before this one --
2080
+ # is read as today's, as an import is.
2081
+ body = upgrade_body(body)
1917
2082
  why = dataset_problem(body)
1918
2083
  if why:
1919
2084
  return None, (400, why)
1920
- body = {**body, "cases": canonical_cases(body["cases"])}
1921
2085
  with self.store.lock, self._connect() as db, db:
1922
2086
  r = self._live(db, did)
1923
2087
  if r is None:
@@ -1975,7 +2139,7 @@ class Datasets:
1975
2139
  rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
1976
2140
  "ORDER BY name COLLATE NOCASE, id").fetchall()
1977
2141
  return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
1978
- "datasets": [{"name": r["name"], "body": json.loads(r["body"])} for r in rows]}
2142
+ "datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
1979
2143
 
1980
2144
  def import_file(self, doc):
1981
2145
  """Either export's file, as new datasets: import always creates, ids
@@ -2328,6 +2492,13 @@ class Packs:
2328
2492
  ds_ids = {}
2329
2493
  cuts = [] # (kind, id, the document as it was), kept before it changes
2330
2494
  for key, name, body, raw in pack["datasets"]:
2495
+ # The Source a dataset grades may be the pack's own, named by
2496
+ # its folder: pointed at the Source the pack made here.
2497
+ ref = body.get("source")
2498
+ if isinstance(ref, dict):
2499
+ sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
2500
+ if sid:
2501
+ body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
2331
2502
  did = owned.get(("dataset", key))
2332
2503
  current = DATASETS.get(did) if did else None
2333
2504
  if current:
@@ -2349,8 +2520,10 @@ class Packs:
2349
2520
  new_work = []
2350
2521
  for key, doc in pack["pipelines"]:
2351
2522
  doc = json.loads(json.dumps(doc))
2352
- tests = doc.get("tests")
2353
- for t in tests if isinstance(tests, list) else [tests]:
2523
+ # A pack written before version 11 spells its evals `tests`;
2524
+ # the page upgrades the pipeline as it reads it.
2525
+ evals = doc.get("evals", doc.get("tests"))
2526
+ for t in evals if isinstance(evals, list) else [evals]:
2354
2527
  ref = t.get("dataset") if isinstance(t, dict) else None
2355
2528
  if isinstance(ref, dict):
2356
2529
  did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
@@ -2487,7 +2660,7 @@ class Packs:
2487
2660
  #
2488
2661
  # A plugin is code, installed like a pack (docs/packs.md): a zip of a
2489
2662
  # manifest and the compiled JavaScript that registers what the lab lacks -- a
2490
- # kind of answer, a modifier, a test type, a connection type. Its code runs in
2663
+ # kind of answer, a modifier, an eval type, a connection type. Its code runs in
2491
2664
  # the page and in the runner, where the registries live; this server never
2492
2665
  # runs it. It reads the manifest's `registers` as data, so it can refuse two
2493
2666
  # plugins registering one id, and so a connection type's settings, chat path
@@ -2505,7 +2678,19 @@ PLUGIN_VERSIONS = (1,)
2505
2678
  PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
2506
2679
  PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
2507
2680
  PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
2508
- REGISTRIES = ("outputKinds", "modifiers", "testTypes", "connectionTypes")
2681
+ REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
2682
+ # evalTypes as a manifest written before pipeline version 11 spells it: a
2683
+ # plugin's own file, which the lab cannot upgrade, so it is read for good.
2684
+ OLD_REGISTRIES = {"testTypes": "evalTypes"}
2685
+
2686
+
2687
+ def plugin_registers(manifest: dict) -> dict:
2688
+ """A manifest's registers under today's names: testTypes read as evalTypes."""
2689
+ reg = dict(manifest.get("registers") or {})
2690
+ for old, now in OLD_REGISTRIES.items():
2691
+ if old in reg:
2692
+ reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
2693
+ return reg
2509
2694
  AUTH_WAYS = ("bearer", "x-api-key", "none")
2510
2695
 
2511
2696
 
@@ -2545,9 +2730,9 @@ def read_plugin(data: bytes):
2545
2730
  if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
2546
2731
  return None, "the plugin's entry names no JavaScript file it holds"
2547
2732
  reg = m.get("registers") or {}
2548
- if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
2733
+ if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
2549
2734
  return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
2550
- for k in ("outputKinds", "modifiers", "testTypes"):
2735
+ for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
2551
2736
  if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
2552
2737
  return None, f"registers.{k} is a list of ids"
2553
2738
  conns = reg.get("connectionTypes", [])
@@ -2570,8 +2755,8 @@ def read_plugin(data: bytes):
2570
2755
 
2571
2756
  def registered_ids(manifest: dict) -> set:
2572
2757
  """(registry, id) for everything a plugin's manifest says it registers."""
2573
- reg = manifest.get("registers") or {}
2574
- out = {(k, x) for k in ("outputKinds", "modifiers", "testTypes") for x in reg.get(k, [])}
2758
+ reg = plugin_registers(manifest)
2759
+ out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
2575
2760
  return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
2576
2761
 
2577
2762
 
@@ -2924,7 +3109,7 @@ class Plugins:
2924
3109
  return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
2925
3110
  "entry": json.loads(r["manifest"])["entry"],
2926
3111
  "description": json.loads(r["manifest"]).get("description") or "",
2927
- "registers": json.loads(r["manifest"]).get("registers") or {},
3112
+ "registers": plugin_registers(json.loads(r["manifest"])),
2928
3113
  "installed": r["installed_at"]} for r in self._rows()]
2929
3114
 
2930
3115
  def stamp(self) -> list:
@@ -3100,6 +3285,55 @@ def worker_refusal(code, stderr, env):
3100
3285
  return text if len(text) <= 2000 else text[:2000] + "…"
3101
3286
 
3102
3287
 
3288
+
3289
+ # A list of runs is read for its figures -- History's table, Home's recent
3290
+ # runs, Runs' progress -- and a run's results are mostly what Results alone
3291
+ # shows: every reply, every stage's request and answer, every item a score
3292
+ # found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
3293
+ # list carries a brief copy of its results, `brief` says so, and one run
3294
+ # (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
3295
+ # whether it ran; a brief score keeps its verdict and counts in place of its
3296
+ # lists. The replies go too: an eval over the whole run reads them, and the
3297
+ # page asks for that run whole rather than every list carrying them.
3298
+ BRIEF_SCORE = ("pass", "score", "points", "skipped")
3299
+ COUNTED = ("found", "missed", "invented")
3300
+
3301
+
3302
+ def brief_score(score):
3303
+ if not isinstance(score, dict):
3304
+ return score
3305
+ out = {k: score[k] for k in BRIEF_SCORE if k in score}
3306
+ for k in COUNTED:
3307
+ if isinstance(score.get(k), list):
3308
+ out[k] = len(score[k])
3309
+ return out
3310
+
3311
+
3312
+ def brief_cell(cell):
3313
+ if not isinstance(cell, dict):
3314
+ return cell
3315
+ out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
3316
+ if isinstance(cell.get("res"), dict):
3317
+ out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
3318
+ if isinstance(cell.get("scores"), dict):
3319
+ out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
3320
+ elif "scores" in cell:
3321
+ out["scores"] = cell["scores"]
3322
+ # A run from before version 6 kept its one test's score as `score`, which
3323
+ # the page reads as `scores.t1`: kept under its own name for that.
3324
+ if "score" in cell:
3325
+ out["score"] = brief_score(cell["score"])
3326
+ return out
3327
+
3328
+
3329
+ def brief_row(row):
3330
+ def item(it):
3331
+ if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
3332
+ return it
3333
+ return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
3334
+ return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
3335
+
3336
+
3103
3337
  class Queue:
3104
3338
  STATUS = ("queued", "running", "done", "incomplete", "cancelled",
3105
3339
  "failed", "interrupted")
@@ -3127,11 +3361,22 @@ class Queue:
3127
3361
  "dataset TEXT)")
3128
3362
  # A store from before runs kept their dataset gains the column;
3129
3363
  # its rows have none, and are pinned when first they need one.
3130
- if "dataset" not in [c[1] for c in db.execute("PRAGMA table_info(queue)")]:
3364
+ cols = [c[1] for c in db.execute("PRAGMA table_info(queue)")]
3365
+ if "dataset" not in cols:
3131
3366
  db.execute("ALTER TABLE queue ADD COLUMN dataset TEXT")
3367
+ # The run a re-run was queued from (#245); every earlier row is
3368
+ # one of its own.
3369
+ if "rerun_of" not in cols:
3370
+ db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
3132
3371
 
3133
3372
  # ---- rows -----------------------------------------------------------
3134
3373
 
3374
+ # A row read with the submit time of the run it re-runs, if any: the
3375
+ # page names a run by that time (History's Run ID), so a "Re-run of"
3376
+ # note reads without fetching the original.
3377
+ SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
3378
+ "LEFT JOIN queue o ON o.id = q.rerun_of")
3379
+
3135
3380
  @staticmethod
3136
3381
  def _row(r):
3137
3382
  if r is None:
@@ -3142,6 +3387,8 @@ class Queue:
3142
3387
  "snapshot": json.loads(r[6]), "results": json.loads(r[7]),
3143
3388
  "progress": json.loads(r[8]), "totals": json.loads(r[9]),
3144
3389
  "error": r[10],
3390
+ "rerunOf": r[12] if len(r) > 12 else None,
3391
+ "rerunOfAt": r[13] if len(r) > 13 else None,
3145
3392
  }
3146
3393
 
3147
3394
  # A row from before run documents has no version, and nothing here can
@@ -3155,24 +3402,26 @@ class Queue:
3155
3402
  return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
3156
3403
 
3157
3404
  def _all(self, db):
3158
- return [row for row in (self._row(r) for r in db.execute("SELECT * FROM queue"))
3405
+ return [row for row in (self._row(r) for r in db.execute(self.SELECT))
3159
3406
  if self._readable(row)]
3160
3407
 
3161
3408
  def get(self, rid):
3162
3409
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3163
- row = self._row(db.execute("SELECT * FROM queue WHERE id = ?",
3410
+ row = self._row(db.execute(self.SELECT + " WHERE q.id = ?",
3164
3411
  (rid,)).fetchone())
3165
3412
  return row if self._readable(row) else None
3166
3413
 
3167
- def list(self, limit=RUNS_PAGE, before=None):
3414
+ def list(self, limit=RUNS_PAGE, before=None, full=False):
3168
3415
  """Runs, newest first, and whether more follow. `before` is a
3169
3416
  `submittedAt` the page of runs stops at, so History can page through
3170
- them the way it pages the runs store."""
3417
+ them the way it pages the runs store. Each row is brief_row's unless
3418
+ `full` asks for the whole of it."""
3171
3419
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3172
3420
  rows = self._all(db)
3173
3421
  rows = [r for r in rows if before is None or r["submittedAt"] < before]
3174
3422
  rows.sort(key=lambda r: r["submittedAt"], reverse=True)
3175
- return rows[:limit], len(rows) > limit
3423
+ page = rows[:limit]
3424
+ return (page if full else [brief_row(r) for r in page]), len(rows) > limit
3176
3425
 
3177
3426
  def _set(self, rid, **fields):
3178
3427
  sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
@@ -3181,13 +3430,13 @@ class Queue:
3181
3430
 
3182
3431
  # ---- submit ---------------------------------------------------------
3183
3432
 
3184
- def submit(self, run: dict, dataset=None):
3433
+ def submit(self, run: dict, dataset=None, rerun_of=None):
3185
3434
  """
3186
3435
  A new queued run. `run` is the run document (docs/pipeline-model.md
3187
3436
  §5): the pipeline, the profiles it resolved to without their keys, its
3188
3437
  content's file list in order and with its repeats, and the dataset's
3189
- version; `dataset` is that version's body, for a graded run. Returns
3190
- the row. Its items are that list, or the one inline text, each through
3438
+ version; `dataset` is that version's body, for a graded run;
3439
+ `rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
3191
3440
  every scenario -- so the total is the list's length, repeats and all,
3192
3441
  the same count the runner and the page make.
3193
3442
  """
@@ -3196,11 +3445,12 @@ class Queue:
3196
3445
  total = len(run_items(run))
3197
3446
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3198
3447
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
3199
- "snapshot, results, progress, totals, dataset) VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?)",
3448
+ "snapshot, results, progress, totals, dataset, rerun_of) "
3449
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
3200
3450
  (rid, "queued", now, json.dumps(run), "[]",
3201
3451
  json.dumps({"current": None, "n": 0, "total": total}),
3202
3452
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
3203
- None if dataset is None else json.dumps(dataset)))
3453
+ None if dataset is None else json.dumps(dataset), rerun_of))
3204
3454
  # Its prompts' uses, in the same transaction: a run is in the
3205
3455
  # library the moment it is queued, or not queued at all.
3206
3456
  if self.prompts is not None:
@@ -3208,7 +3458,7 @@ class Queue:
3208
3458
  # Read back under the lock the worker dequeues under, so the
3209
3459
  # answer is the run as it was queued: an idle worker can take it
3210
3460
  # the moment the lock is let go.
3211
- return self._row(db.execute("SELECT * FROM queue WHERE id = ?", (rid,)).fetchone())
3461
+ return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
3212
3462
 
3213
3463
  def dataset(self, rid, raw=False):
3214
3464
  """The dataset body a readable run was submitted against, or None --
@@ -3241,7 +3491,7 @@ class Queue:
3241
3491
  written beside the run document. A row queued before runs kept their
3242
3492
  dataset has none, and is pinned to the dataset as it reads now, once,
3243
3493
  so every later pass over it agrees. Returns (args, None) or (None, why)."""
3244
- ref = tests_dataset(run["snapshot"])
3494
+ ref = evals_dataset(run["snapshot"])
3245
3495
  if ref is None:
3246
3496
  return [], None
3247
3497
  body = self.dataset(run["id"], raw=True)
@@ -3327,6 +3577,46 @@ class Queue:
3327
3577
  shutil.rmtree(d, ignore_errors=True)
3328
3578
  return n
3329
3579
 
3580
+ def rerun(self, rid):
3581
+ """
3582
+ A new run from a finished one's document (#245): a new id and submit
3583
+ time, the same document -- the file revisions and the plugins it
3584
+ pinned -- and the dataset body it was graded by, so a re-run measures
3585
+ the model again rather than changed data. The original is left as it
3586
+ was; the new row names it in `rerunOf`. Refused, naming what is
3587
+ missing, when a pinned file or dataset version is no longer kept. A
3588
+ Target's key is read from its profile now, at dequeue, as for any
3589
+ run: none is ever stored in one.
3590
+ """
3591
+ run = self.get(rid)
3592
+ if run is None:
3593
+ return None, (404, "no such run")
3594
+ if run["status"] in ("queued", "running"):
3595
+ return None, (409, "a run still in progress cannot be re-run")
3596
+ snap = run["snapshot"]
3597
+ _, err = self._pinned_files(snap)
3598
+ if err:
3599
+ return None, (409, err)
3600
+ ref = evals_dataset(snap)
3601
+ body = None
3602
+ if ref is not None:
3603
+ body = self.dataset(rid, raw=True)
3604
+ if body is None:
3605
+ # A run that never started kept no body: the dataset's, if it
3606
+ # still reads as the version the run was submitted against.
3607
+ now = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
3608
+ if now is None or not ref.get("version") or now[1] != ref.get("version"):
3609
+ return None, (409, f"the version of the dataset {ref.get('name') or ref.get('id')!r} "
3610
+ "this run was submitted against is no longer kept")
3611
+ body = now[0]
3612
+ _, err = self._plugin_args(run)
3613
+ if err:
3614
+ return None, (409, err)
3615
+ _, err = worker_destinations(snap)
3616
+ if err:
3617
+ return None, (403, err)
3618
+ return self.submit(snap, body, rerun_of=rid), None
3619
+
3330
3620
  def rerun_item(self, rid, index):
3331
3621
  """
3332
3622
  One item against the run's snapshot, by its index, updating the row in
@@ -3473,34 +3763,48 @@ class Queue:
3473
3763
  has gone fails the run with the reason stated. Returns the run dir,
3474
3764
  or (None, error)."""
3475
3765
  snap = run["snapshot"]
3476
- content = content_of(snap) or {}
3766
+ files, err = self._pinned_files(snap)
3767
+ if err:
3768
+ return None, err
3477
3769
  rundir = self.dir / run["id"]
3478
3770
  files_dir = rundir / "files"
3479
3771
  shutil.rmtree(rundir, ignore_errors=True)
3480
3772
  files_dir.mkdir(parents=True)
3481
- if content.get("type") == "source":
3482
- ref = content.get("ref") or {}
3483
- label = ref.get("name") or ref.get("id")
3484
- if SOURCES.get(ref.get("id")) is None:
3485
- return None, f"the source {label!r} is gone"
3486
- revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
3487
- changed = []
3488
- for name in dict.fromkeys(content.get("files") or []):
3489
- path = SOURCES.file_path(ref["id"], name)
3490
- if path is None or not path.is_file():
3491
- return None, f"{name!r} is gone from the source {label!r}"
3492
- if revs is not None:
3493
- path = SOURCES.rev_path(ref["id"], name, revs.get(name))
3494
- if path is None:
3495
- changed.append(repr(name))
3496
- continue
3497
- shutil.copy2(path, files_dir / name)
3498
- if changed:
3499
- return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
3500
- f"changed in the source {label!r} since this run was queued")
3773
+ for name, path in files:
3774
+ shutil.copy2(path, files_dir / name)
3501
3775
  (rundir / "run.json").write_text(json.dumps(snap))
3502
3776
  return rundir, None
3503
3777
 
3778
+ @staticmethod
3779
+ def _pinned_files(snap):
3780
+ """Each file a run document reads, by name, and the path holding the
3781
+ bytes it pinned at submit -- or (None, why), naming the Source that
3782
+ has gone, or the files gone from it or changed since. A document
3783
+ with no Source reads no files."""
3784
+ content = content_of(snap) or {}
3785
+ if content.get("type") != "source":
3786
+ return [], None
3787
+ ref = content.get("ref") or {}
3788
+ label = ref.get("name") or ref.get("id")
3789
+ if SOURCES.get(ref.get("id")) is None:
3790
+ return None, f"the source {label!r} is gone"
3791
+ revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
3792
+ files, changed = [], []
3793
+ for name in dict.fromkeys(content.get("files") or []):
3794
+ path = SOURCES.file_path(ref["id"], name)
3795
+ if path is None or not path.is_file():
3796
+ return None, f"{name!r} is gone from the source {label!r}"
3797
+ if revs is not None:
3798
+ path = SOURCES.rev_path(ref["id"], name, revs.get(name))
3799
+ if path is None:
3800
+ changed.append(repr(name))
3801
+ continue
3802
+ files.append((name, path))
3803
+ if changed:
3804
+ return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
3805
+ f"changed in the source {label!r} since this run was queued")
3806
+ return files, None
3807
+
3504
3808
  def _execute(self, run):
3505
3809
  """One run, FIFO. Never more than one of these at a time: the worker
3506
3810
  is a single thread, so the queue is single-flight by construction."""
@@ -3748,7 +4052,7 @@ def worker_destinations(run: dict):
3748
4052
 
3749
4053
  # A run document's fields, checked before it is accepted -- the rules
3750
4054
  # docs/pipeline-model.md §6 gives the server, in Python because the server is
3751
- # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or a test
4055
+ # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
3752
4056
  # means is the runner's to judge, and it fails the run with a sentence if it
3753
4057
  # cannot; what is here is what the server itself depends on: a version it
3754
4058
  # reads, its own cap, content it can count and copy, and connections that
@@ -3756,21 +4060,23 @@ def worker_destinations(run: dict):
3756
4060
  # 5: the pipeline and every chain have an id; version 4's scenarios keep
3757
4061
  # theirs, and an older run's chains are read as they are, by position.
3758
4062
  # 6: tests are an ordered list; a stored run's one test (or null) is read as
3759
- # a list of one (tests_dataset).
4063
+ # a list of one (evals_dataset).
3760
4064
  # 7: chains are jobs: the field is `jobs` and each job's type is "job".
3761
4065
  # 8: a job is its steps; the content is job 1's Attach Content step.
3762
- # 9: a test is Metrics; a Single Test or a Graded set is read converted.
4066
+ # 9: an eval is Metrics; a Single Test or a Graded set is read converted.
3763
4067
  # 10: a job's steps are its stages, and each scenario is a target whose own
3764
4068
  # step in each job is what it sends there (docs/pipeline-model.md §16).
3765
- PIPELINE_VERSION = 10
4069
+ # 11: `tests` are `evals`; nothing in an eval changes.
4070
+ # 12: a Contains metric's Ignore case holds item by item too, kept as written.
4071
+ PIPELINE_VERSION = 12
3766
4072
  # What a stored run may be: the current version, and the ones evals-core.ts's
3767
4073
  # upgradePipeline reads. A new submission is upgraded to the current one.
3768
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
4074
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
3769
4075
  TARGET_CAP = 4
3770
4076
  # Target steps whose words the Prompt library does not record as a use: they
3771
4077
  # ask no model (evals-core.ts's STEP_TYPES.echo).
3772
4078
  UNRECORDED_STEPS = {"echo"}
3773
- RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "tests", "profiles", "comment", "plugins")
4079
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
3774
4080
 
3775
4081
 
3776
4082
  def content_of(doc):
@@ -3927,14 +4233,14 @@ def connection_problems(pid, conn, at, bad):
3927
4233
  bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
3928
4234
 
3929
4235
 
3930
- def tests_dataset(doc):
3931
- """The dataset reference a document's tests grade against, or None: the
3932
- first test that names one. A run grades against one dataset (the core's
3933
- validatePipeline says so). Reads a stored document of either shape --
3934
- version 5's list, or the one test before it -- since rows keep the
3935
- document they were submitted with."""
3936
- tests = doc.get("tests") if isinstance(doc, dict) else None
3937
- for t in tests if isinstance(tests, list) else [tests]:
4236
+ def evals_dataset(doc):
4237
+ """The dataset reference a document's evals grade against, or None: the
4238
+ first eval that names one. A run grades against one dataset (the core's
4239
+ validatePipeline says so). Reads a stored document of any shape --
4240
+ version 11's `evals`, the `tests` before it, version 5's list or the one
4241
+ test before that -- since rows keep the document they were submitted with."""
4242
+ evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
4243
+ for t in evals if isinstance(evals, list) else [evals]:
3938
4244
  if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
3939
4245
  return t["dataset"]
3940
4246
  return None
@@ -4079,9 +4385,9 @@ def run_problems(run):
4079
4385
  pass
4080
4386
  else:
4081
4387
  bad.append("a run needs a Source's files, some inline text, or Prompt only")
4082
- tests = run.get("tests")
4083
- if not isinstance(tests, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in tests):
4084
- bad.append("tests has to be a list of tests, each naming its type")
4388
+ evals = run.get("evals")
4389
+ if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
4390
+ bad.append("evals has to be a list of evals, each naming its type")
4085
4391
  if run.get("comment") is not None and not isinstance(run["comment"], str):
4086
4392
  bad.append("comment has to be text")
4087
4393
  return bad
@@ -4146,11 +4452,25 @@ class Handler(BaseHTTPRequestHandler):
4146
4452
 
4147
4453
  # ---- helpers --------------------------------------------------------
4148
4454
 
4149
- def _send(self, code, body: bytes, ctype="application/json", headers=()):
4455
+ def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
4456
+ """`packed` keys a body that never changes under it -- a hashed
4457
+ bundle -- so it is gzipped once and kept, not on every request."""
4150
4458
  self.send_response(code)
4151
4459
  self.send_header("Content-Type", ctype)
4152
4460
  for k, v in headers:
4153
4461
  self.send_header(k, v)
4462
+ if ctype.startswith(COMPRESSIBLE):
4463
+ # Said whether or not this answer is compressed, so a cache
4464
+ # between here and the browser keys on it either way.
4465
+ self.send_header("Vary", "Accept-Encoding")
4466
+ if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
4467
+ if packed is None:
4468
+ body = gzip.compress(body, 6, mtime=0)
4469
+ else:
4470
+ if packed not in GZIPPED:
4471
+ GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
4472
+ body = GZIPPED[packed]
4473
+ self.send_header("Content-Encoding", "gzip")
4154
4474
  self.send_header("Content-Length", str(len(body)))
4155
4475
  self.end_headers()
4156
4476
  self.wfile.write(body)
@@ -4235,7 +4555,7 @@ class Handler(BaseHTTPRequestHandler):
4235
4555
  return
4236
4556
  path = self.path.split("?", 1)[0]
4237
4557
  # The lab is one page: a Connection and an Input make a scenario,
4238
- # Content and Tests are shared, and one to four scenarios run over the
4558
+ # Content and Evals are shared, and one to four scenarios run over the
4239
4559
  # content. It replaced the A/B page it was prototyped beside once it
4240
4560
  # carried everything that page did -- the graded set, the
4241
4561
  # Configuration group, the history -- rather than being left to rot
@@ -4255,7 +4575,7 @@ class Handler(BaseHTTPRequestHandler):
4255
4575
  if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
4256
4576
  return self._send(404, b"not found", "text/plain")
4257
4577
  return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
4258
- (("Cache-Control", "public, max-age=31536000, immutable"),))
4578
+ (("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
4259
4579
  if path == "/api/config":
4260
4580
  # The lab's own limits, for the page to disable rather than
4261
4581
  # hard-code: how many scenarios a run may have.
@@ -4279,14 +4599,14 @@ class Handler(BaseHTTPRequestHandler):
4279
4599
  except ValueError:
4280
4600
  return self._json(400, {"error": "limit has to be a number"})
4281
4601
  before = (query.get("before") or [None])[0]
4282
- runs, more = QUEUE.list(limit, before)
4602
+ runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
4283
4603
  return self._json(200, {"runs": runs, "more": more})
4284
4604
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
4285
4605
  # The dataset body a graded run was submitted against, which is
4286
4606
  # what its verdicts were graded by; the list never carries it.
4287
4607
  #
4288
4608
  # A run that is there but kept no copy -- one submitted before the
4289
- # lab kept them, or one with no graded test -- answers null, not
4609
+ # lab kept them, or one with no graded eval -- answers null, not
4290
4610
  # 404. It is not an error: the page reads it as "use the dataset
4291
4611
  # as it is now", and a 404 put a red line in the console every
4292
4612
  # time such a run was opened, which is the console people are
@@ -4820,14 +5140,14 @@ class Handler(BaseHTTPRequestHandler):
4820
5140
  # A graded run keeps the body of the dataset it names, as it reads
4821
5141
  # now, and records that body's fingerprint on the reference it
4822
5142
  # belongs to: the worker grades against that and nothing else.
4823
- ref = tests_dataset(run)
5143
+ ref = evals_dataset(run)
4824
5144
  body = None
4825
5145
  if ref is not None:
4826
5146
  snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
4827
5147
  if snap is None:
4828
- return self._json(400, {"error": "the run's graded test names no dataset this lab has"})
5148
+ return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
4829
5149
  body, version = snap
4830
- for t in run["tests"]:
5150
+ for t in run["evals"]:
4831
5151
  if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
4832
5152
  t["dataset"]["version"] = version
4833
5153
  return self._json(201, {"run": QUEUE.submit(run, body)})
@@ -4985,6 +5305,9 @@ class Handler(BaseHTTPRequestHandler):
4985
5305
  return self._send(404, b"not found", "text/plain")
4986
5306
 
4987
5307
  def _queue_action(self, path):
5308
+ # Drained, so a keep-alive connection is not left holding the
5309
+ # page's `{}` in front of its next request.
5310
+ self._payload()
4988
5311
  parts = path.split("/")
4989
5312
  rid = parts[3]
4990
5313
  action = parts[4] if len(parts) > 4 else ""
@@ -4992,6 +5315,10 @@ class Handler(BaseHTTPRequestHandler):
4992
5315
  run, err = QUEUE.cancel(rid)
4993
5316
  elif action == "resume":
4994
5317
  run, err = QUEUE.resume(rid)
5318
+ elif action == "rerun" and len(parts) == 5:
5319
+ run, err = QUEUE.rerun(rid)
5320
+ if not err:
5321
+ return self._json(201, {"run": run})
4995
5322
  elif action == "items" and len(parts) == 6:
4996
5323
  run, err = QUEUE.rerun_item(rid, urllib.parse.unquote(parts[5]))
4997
5324
  elif action == "rescore" and len(parts) == 6: