evals-lab 0.4.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -560,6 +560,48 @@ QUEUE_WAIT = 1.0
560
560
  WATCH_EVERY = 5.0
561
561
 
562
562
 
563
+ # A Target profile's key is written from the page and never read back by it
564
+ # (#258): a client holding the lab's password -- CI's runners and their logs
565
+ # among them -- is handed no key. Where a profile holds one, what is served
566
+ # (/api/state, the page's carried copy, a refused write's current copy) holds
567
+ # KEY_HELD and the profile's id instead, and a write that brings that back
568
+ # keeps the key the store holds for the id: the profile's own, or the one it
569
+ # was cloned or restored from. KEY_HELD starts with a character no key can
570
+ # (header_safe), so a pasted key is never taken for it.
571
+ KEY_HELD = "\u2022held:"
572
+
573
+
574
+ def held(key) -> bool:
575
+ return isinstance(key, str) and key.startswith(KEY_HELD)
576
+
577
+
578
+ def profiles_in(name: str, body):
579
+ """Every profile a synced document holds: the profiles store's list, and
580
+ each profile version's copy."""
581
+ if not isinstance(body, dict):
582
+ return
583
+ if name == "promptlab.profiles":
584
+ for p in body.get("list") or []:
585
+ if isinstance(p, dict):
586
+ yield p
587
+ elif name == "promptlab.versions":
588
+ for kept in (body.get("profile") or {}).values() if isinstance(body.get("profile"), dict) else ():
589
+ for v in kept if isinstance(kept, list) else ():
590
+ if isinstance(v, dict) and isinstance(v.get("doc"), dict):
591
+ yield v["doc"]
592
+
593
+
594
+ def hide_keys(docs: dict) -> dict:
595
+ out = {}
596
+ for name, d in docs.items():
597
+ d = json.loads(json.dumps(d))
598
+ for p in profiles_in(name, d.get("body")):
599
+ if isinstance(p.get("key"), str) and p["key"] and not held(p["key"]):
600
+ p["key"] = KEY_HELD + str(p.get("id") or "")
601
+ out[name] = d
602
+ return out
603
+
604
+
563
605
  class Store:
564
606
  """
565
607
  Documents by name, each with a version that goes up by one per write.
@@ -579,6 +621,11 @@ class Store:
579
621
  db.execute("CREATE TABLE IF NOT EXISTS docs (name TEXT PRIMARY KEY, "
580
622
  "version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
581
623
  db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
624
+ # The key of a profile a write took out, by id, for TRASH_SECONDS:
625
+ # Undo puts the profile back holding KEY_HELD, and this is what
626
+ # it holds.
627
+ db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
628
+ "key TEXT NOT NULL, at REAL NOT NULL)")
582
629
  # History was a document, capped at what a browser could hold. The
583
630
  # first start with the table moves what that document had into it,
584
631
  # once, and drops the document so the page stops carrying it.
@@ -600,8 +647,30 @@ class Store:
600
647
 
601
648
  def served(self) -> dict:
602
649
  """What a browser is handed: the SYNCED documents only, so a retired
603
- key's rows stay in the store without reaching a page again."""
604
- return {n: d for n, d in self.all().items() if n in SYNCED}
650
+ key's rows stay in the store without reaching a page again, and no
651
+ profile's key (KEY_HELD)."""
652
+ return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
653
+
654
+ def _keyring(self, db, now: dict) -> dict:
655
+ """Each profile id's key as the store holds it: its profile's, else a
656
+ dropped one's, else its newest version's."""
657
+ ring = {}
658
+ for p in profiles_in("promptlab.versions", now.get("promptlab.versions", {}).get("body")):
659
+ pid, key = str(p.get("id") or ""), p.get("key")
660
+ if pid not in ring and isinstance(key, str) and key and not held(key):
661
+ ring[pid] = key
662
+ db.execute("DELETE FROM dropped_keys WHERE at < ?", (time.time() - TRASH_SECONDS,))
663
+ ring.update(db.execute("SELECT id, key FROM dropped_keys").fetchall())
664
+ for p in profiles_in("promptlab.profiles", now.get("promptlab.profiles", {}).get("body")):
665
+ key = p.get("key")
666
+ if isinstance(key, str) and key and not held(key):
667
+ ring[str(p.get("id") or "")] = key
668
+ return ring
669
+
670
+ def held_key(self, key: str) -> str:
671
+ """The key KEY_HELD stands for, or "" when the store holds none."""
672
+ with self.lock, closing(sqlite3.connect(self.path)) as db, db:
673
+ return self._keyring(db, self._rows(db)).get(key[len(KEY_HELD):], "")
605
674
 
606
675
  def write(self, docs: dict):
607
676
  """
@@ -614,9 +683,23 @@ class Store:
614
683
  have = lambda n: now.get(n, {"version": 0, "body": None})
615
684
  stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
616
685
  if stale:
617
- return None, stale
686
+ return None, hide_keys(stale)
618
687
  at = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
619
688
  with db:
689
+ ring = self._keyring(db, now)
690
+ for n, d in docs.items():
691
+ for p in profiles_in(n, d["body"]):
692
+ if held(p.get("key")):
693
+ p["key"] = ring.get(p["key"][len(KEY_HELD):], "")
694
+ if "promptlab.profiles" in docs:
695
+ kept = {str(p.get("id") or "") for p in profiles_in(
696
+ "promptlab.profiles", docs["promptlab.profiles"]["body"])}
697
+ db.executemany("DELETE FROM dropped_keys WHERE id = ?", [(i,) for i in kept])
698
+ db.executemany(
699
+ "INSERT OR REPLACE INTO dropped_keys (id, key, at) VALUES (?, ?, ?)",
700
+ [(str(p.get("id") or ""), p["key"], time.time())
701
+ for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
702
+ if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
620
703
  for n, d in docs.items():
621
704
  db.execute(
622
705
  "INSERT INTO docs (name, version, body, updated_at) VALUES (?, ?, ?, ?) "
@@ -1177,6 +1260,18 @@ class Sources:
1177
1260
  return None, (500, "the files could not be copied")
1178
1261
  return self.get(sid), None
1179
1262
 
1263
+ def item_paths(self, sid):
1264
+ """Every file of a Source, as (name, path on disk) in its own order,
1265
+ or None when there is no such Source."""
1266
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1267
+ r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1268
+ if r is None:
1269
+ return None
1270
+ if r[0]:
1271
+ return [(f["name"], SAMPLES / f["name"]) for f in self._sample_files()]
1272
+ return [(row[0], self.dir / sid / row[0]) for row in db.execute(
1273
+ "SELECT name FROM source_files WHERE source = ? ORDER BY name", (sid,))]
1274
+
1180
1275
  def zip_files(self, sid, names):
1181
1276
  """The bytes of a stdlib zipfile holding exactly the named files, in
1182
1277
  the order named, or an error. Read under the store's lock so the file
@@ -1362,15 +1457,23 @@ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cas
1362
1457
  # 5 was told by its `source` alone, and earlier ones by neither.
1363
1458
  DATASET_BODY_VERSION = 7
1364
1459
  DATASET_NAME_MAX = 80
1365
- # The file forms Export writes and Import reads. Export writes version 7;
1366
- # Import reads it and versions 1 to 6, upgraded, and refuses anything else,
1367
- # as a pipeline of another version is refused. Versions 1 to 3 carried a
1368
- # prompt, which an import gives to the Prompt library.
1369
- EXPORT_ONE = "evals-lab/dataset"
1370
- EXPORT_ALL = "evals-lab/datasets"
1460
+ # The file forms Export writes and Import reads. Export writes an eval group
1461
+ # at version 7 (docs/pipeline-model.md §17); Import reads that, and a dataset
1462
+ # file of versions 1 to 7, upgraded, and refuses anything else, as a pipeline
1463
+ # of another version is refused. Versions 1 to 3 carried a prompt, which an
1464
+ # import gives to the Prompt library.
1465
+ EXPORT_ONE = "evals-lab/eval-group"
1466
+ EXPORT_ALL = "evals-lab/eval-groups"
1467
+ DATASET_ONE = "evals-lab/dataset"
1468
+ DATASET_ALL = "evals-lab/datasets"
1371
1469
  EXPORT_VERSION = 7
1372
1470
  IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
1471
+ # Each file form, and the key its one entry or its list sits under.
1472
+ EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
1373
1473
  SCORING_MODES = ("all", "weighted")
1474
+ # A group's newest version is edited in place by a save within this long of
1475
+ # the last, as a prompt's is (PROMPT_IDLE_SECONDS).
1476
+ GROUP_IDLE_SECONDS = float(os.environ.get("GROUP_IDLE_SECONDS", "30"))
1374
1477
 
1375
1478
 
1376
1479
  def blank_dataset() -> dict:
@@ -1574,6 +1677,30 @@ def fingerprint(body: dict) -> str:
1574
1677
  return hashlib.sha256(text.encode("utf-8")).hexdigest()[:CASE_SET_LEN]
1575
1678
 
1576
1679
 
1680
+ def pins_in(workflows) -> set:
1681
+ """(group id, version) for every link a stored pipeline pins: the
1682
+ promptlab.workflows body, each pipeline as its `work`. Only version 13
1683
+ links pin, and a pipeline of an earlier version has none."""
1684
+ out = set()
1685
+ listed = workflows.get("list") if isinstance(workflows, dict) else None
1686
+ for w in listed if isinstance(listed, list) else []:
1687
+ work = w.get("work") if isinstance(w, dict) else None
1688
+ evals = work.get("evals") if isinstance(work, dict) else None
1689
+ for t in evals if isinstance(evals, list) else []:
1690
+ if (isinstance(t, dict) and t.get("type") == "group" and isinstance(t.get("group"), dict)
1691
+ and isinstance(t["group"].get("id"), str) and type(t.get("pin")) is int):
1692
+ out.add((t["group"]["id"], t["pin"]))
1693
+ return out
1694
+
1695
+
1696
+ def group_key(ref) -> str:
1697
+ """The key a run keeps a group's body under: `<id>@<n>`, or the id alone
1698
+ for a reference from before the lab numbered versions."""
1699
+ n = ref.get("n") if isinstance(ref, dict) else None
1700
+ gid = ref.get("id") if isinstance(ref, dict) else None
1701
+ return f"{gid}@{n}" if type(n) is int else str(gid)
1702
+
1703
+
1577
1704
  def unique_dataset_name(name: str, taken: set) -> str:
1578
1705
  """`Receipts` again becomes `Receipts (2)`, compared without case."""
1579
1706
  lower = {t.lower() for t in taken}
@@ -1944,6 +2071,16 @@ class Datasets:
1944
2071
  # the body it was typed as is kept, as the rules were.
1945
2072
  db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
1946
2073
  "dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
2074
+ # Every version of each eval group, numbered from 1, for a link to
2075
+ # pin and a run to name (docs/pipeline-model.md §17). `ran` says a
2076
+ # run graded with it, which freezes it for good; a pin freezes it
2077
+ # while a stored pipeline holds the pin. Version 1 is added from
2078
+ # the row's body at its first save, pin or run, so a group nobody
2079
+ # touches holds what it held.
2080
+ db.execute("CREATE TABLE IF NOT EXISTS eval_group_versions ("
2081
+ "group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
2082
+ "created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
2083
+ "ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
1947
2084
  # Rows from an earlier version are converted once, in place: a
1948
2085
  # dataset is typed in by hand and costly to re-enter, so it is
1949
2086
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -1992,16 +2129,26 @@ class Datasets:
1992
2129
  return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
1993
2130
 
1994
2131
  @staticmethod
1995
- def _doc(r, body=True):
2132
+ def _doc(r, body=True, versions=1):
1996
2133
  """A row as the API answers it: a DatasetSummary, and its body with it,
1997
- read as today's version."""
2134
+ read as today's version. `version` is the save counter a write is
2135
+ arbitrated by; `versions` is how many group versions it has -- its
2136
+ newest one's number, which a pin may name, and 1 before any is kept,
2137
+ the row's body being version 1 in waiting."""
1998
2138
  parsed = upgrade_body(json.loads(r["body"]))
1999
2139
  out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
2000
- "version": r["version"], "updated": r["updated_at"]}
2140
+ "version": r["version"], "versions": versions, "updated": r["updated_at"]}
2001
2141
  if body:
2002
2142
  out["body"] = parsed
2003
2143
  return out
2004
2144
 
2145
+ @staticmethod
2146
+ def _counts(db) -> dict:
2147
+ return dict(db.execute("SELECT group_id, MAX(n) FROM eval_group_versions GROUP BY group_id"))
2148
+
2149
+ def _row_doc(self, db, r, body=True):
2150
+ return self._doc(r, body, self._counts(db).get(r["id"], 1))
2151
+
2005
2152
  def _connect(self):
2006
2153
  db = sqlite3.connect(self.store.path)
2007
2154
  db.row_factory = sqlite3.Row
@@ -2016,13 +2163,142 @@ class Datasets:
2016
2163
 
2017
2164
  def list(self) -> list:
2018
2165
  with self.store.lock, self._connect() as db:
2019
- return [self._doc(r, body=False) for r in db.execute(
2166
+ counts = self._counts(db)
2167
+ return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
2020
2168
  "SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id")]
2021
2169
 
2022
2170
  def get(self, did):
2023
2171
  with self.store.lock, self._connect() as db:
2024
2172
  r = self._live(db, did)
2025
- return self._doc(r) if r else None
2173
+ return self._row_doc(db, r) if r else None
2174
+
2175
+ # ---- a group's versions (docs/pipeline-model.md §17) -----------------
2176
+
2177
+ @staticmethod
2178
+ def _head(db, did):
2179
+ return db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? "
2180
+ "ORDER BY n DESC LIMIT 1", (did,)).fetchone()
2181
+
2182
+ def _mint(self, db, r):
2183
+ """The newest version of row [r], adding version 1 from its body
2184
+ first if it has none: edited when the row last was, so a group made a
2185
+ moment ago goes on being edited in place."""
2186
+ head = self._head(db, r["id"])
2187
+ if head is not None:
2188
+ return head
2189
+ try:
2190
+ edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
2191
+ except ValueError:
2192
+ # A stamp nothing here wrote is no recent edit.
2193
+ edited = 0
2194
+ db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
2195
+ "VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
2196
+ return self._head(db, r["id"])
2197
+
2198
+ @staticmethod
2199
+ def _pins(db) -> set:
2200
+ """Every (group, version) a stored pipeline pins, read in the caller's
2201
+ transaction from the docs table the page writes them to."""
2202
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'").fetchone()
2203
+ return pins_in(json.loads(row[0])) if row and row[0] else set()
2204
+
2205
+ def _cut(self, db, did, text):
2206
+ """A new newest version of [did] holding [text]; returns its number."""
2207
+ n = self._head(db, did)["n"] + 1
2208
+ db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
2209
+ "VALUES (?, ?, ?, ?, ?)", (did, n, text, self._now(), time.time()))
2210
+ return n
2211
+
2212
+ def _keep(self, db, r, text):
2213
+ """[text] as row [r]'s newest version: the newest edited in place while
2214
+ no run has graded with it, no link pins it and it was edited in the
2215
+ last GROUP_IDLE_SECONDS, and a new version otherwise."""
2216
+ head = self._mint(db, r)
2217
+ if text == head["body"]:
2218
+ return
2219
+ fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
2220
+ if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
2221
+ db.execute("UPDATE eval_group_versions SET body = ?, edited_at = ? WHERE group_id = ? AND n = ?",
2222
+ (text, time.time(), r["id"], head["n"]))
2223
+ else:
2224
+ self._cut(db, r["id"], text)
2225
+
2226
+ def versions(self, did):
2227
+ """A group's versions, newest first, without their bodies -- version 1
2228
+ alone, read from the row, before any is kept -- or None."""
2229
+ with self.store.lock, self._connect() as db:
2230
+ r = self._live(db, did)
2231
+ if r is None:
2232
+ return None
2233
+ pins = self._pins(db)
2234
+ rows = db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? ORDER BY n DESC",
2235
+ (did,)).fetchall()
2236
+ if not rows:
2237
+ return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (did, 1) in pins,
2238
+ "fingerprint": fingerprint(json.loads(r["body"]))}]
2239
+ return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
2240
+ "pinned": (did, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
2241
+ for v in rows]
2242
+
2243
+ def version_body(self, did, n):
2244
+ """Version [n]'s body, read as today's, or None."""
2245
+ with self.store.lock, self._connect() as db:
2246
+ r = self._live(db, did)
2247
+ if r is None:
2248
+ return None
2249
+ v = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
2250
+ (did, n)).fetchone()
2251
+ if v is None and n == 1 and self._head(db, did) is None:
2252
+ v = (r["body"],)
2253
+ return upgrade_body(json.loads(v[0])) if v else None
2254
+
2255
+ def restore_version(self, did, n):
2256
+ """An older version's body as the newest version, and the row's: as a
2257
+ prompt's Restore, nothing is rewritten, so a run or a pin naming any
2258
+ version still reads what it named. Returns (DatasetDoc, None)."""
2259
+ with self.store.lock, self._connect() as db, db:
2260
+ r = self._live(db, did)
2261
+ if r is None:
2262
+ return None, (404, "no such eval group")
2263
+ head = self._mint(db, r)
2264
+ old = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
2265
+ (did, n)).fetchone()
2266
+ if old is None:
2267
+ return None, (404, "no such version")
2268
+ if old["body"] != head["body"]:
2269
+ self._cut(db, did, old["body"])
2270
+ db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
2271
+ (old["body"], r["version"] + 1, self._now(), did))
2272
+ return self._row_doc(db, self._live(db, did)), None
2273
+
2274
+ def mint_pinned(self, pins):
2275
+ """Version 1 of each pinned group that has none yet: a pin names a
2276
+ version, so the version has to be kept from the moment it does."""
2277
+ with self.store.lock, self._connect() as db, db:
2278
+ for did, _ in pins:
2279
+ r = self._live(db, did)
2280
+ if r is not None:
2281
+ self._mint(db, r)
2282
+
2283
+ def resolve(self, did, pin=None):
2284
+ """The version a run submitted now grades with -- [pin], or the
2285
+ newest -- marked as graded with, in the same transaction, so no save
2286
+ can edit it in place between this and the run keeping its body.
2287
+ Returns (n, body), (None, why) for a pin the group has no version of,
2288
+ or None for a group the lab does not have. A submit refused after
2289
+ this leaves the version frozen, which costs only a new version at the
2290
+ next save."""
2291
+ with self.store.lock, self._connect() as db, db:
2292
+ r = self._live(db, did) if isinstance(did, str) else None
2293
+ if r is None:
2294
+ return None
2295
+ head = self._mint(db, r)
2296
+ v = head if pin is None else db.execute(
2297
+ "SELECT * FROM eval_group_versions WHERE group_id = ? AND n = ?", (did, pin)).fetchone()
2298
+ if v is None:
2299
+ return None, f"{r['name']} has no version {pin}"
2300
+ db.execute("UPDATE eval_group_versions SET ran = 1 WHERE group_id = ? AND n = ?", (did, v["n"]))
2301
+ return v["n"], json.loads(v["body"])
2026
2302
 
2027
2303
  def snapshot(self, did):
2028
2304
  """The body a run submitted now grades against, and its fingerprint;
@@ -2087,9 +2363,11 @@ class Datasets:
2087
2363
  if r is None:
2088
2364
  return None, (404, "no such dataset")
2089
2365
  if r["version"] != version:
2090
- return None, (409, {"current": self._doc(r)})
2366
+ return None, (409, {"current": self._row_doc(db, r)})
2367
+ text = json.dumps(body)
2368
+ self._keep(db, r, text)
2091
2369
  db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
2092
- (json.dumps(body), version + 1, self._now(), did))
2370
+ (text, version + 1, self._now(), did))
2093
2371
  return {"version": version + 1}, None
2094
2372
 
2095
2373
  def remove(self, did):
@@ -2115,58 +2393,69 @@ class Datasets:
2115
2393
  did = r["id"]
2116
2394
  return self.get(did), None
2117
2395
 
2396
+ @staticmethod
2397
+ def _purge(db, where, args):
2398
+ # A group's versions go with it: the trash held them, and Undo is
2399
+ # what would have brought them back.
2400
+ db.execute(f"DELETE FROM eval_group_versions WHERE group_id IN (SELECT id FROM datasets WHERE {where})", args)
2401
+ db.execute(f"DELETE FROM datasets WHERE {where}", args)
2402
+
2118
2403
  def lazy_trash(self):
2119
2404
  """A datasets request empties what has been trashed longer than
2120
2405
  TRASH_SECONDS, as a Sources request does."""
2121
2406
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2122
- db.execute("DELETE FROM datasets WHERE trash IS NOT NULL AND trashed_at < ?",
2123
- (time.time() - TRASH_SECONDS,))
2407
+ self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
2124
2408
 
2125
2409
  def empty_trash(self):
2126
2410
  """The startup sweep: a restart has nothing to undo."""
2127
2411
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2128
- db.execute("DELETE FROM datasets WHERE trash IS NOT NULL")
2412
+ self._purge(db, "trash IS NOT NULL", ())
2129
2413
 
2130
2414
  def export(self, did):
2131
2415
  d = self.get(did)
2132
2416
  if d is None:
2133
2417
  return None
2134
2418
  return {"format": EXPORT_ONE, "version": EXPORT_VERSION,
2135
- "dataset": {"name": d["name"], "body": d["body"]}}
2419
+ "group": {"name": d["name"], "body": d["body"]}}
2136
2420
 
2137
2421
  def export_all(self):
2138
2422
  with self.store.lock, self._connect() as db:
2139
2423
  rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
2140
2424
  "ORDER BY name COLLATE NOCASE, id").fetchall()
2141
2425
  return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
2142
- "datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2426
+ "groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2143
2427
 
2144
2428
  def import_file(self, doc):
2145
2429
  """Either export's file, as new datasets: import always creates, ids
2146
2430
  are this lab's, and a taken name gets ` (2)`. All of it or none.
2147
2431
  Returns ([DatasetSummary], None), or (None, (400, one sentence))."""
2148
2432
  if not isinstance(doc, dict) or "format" not in doc:
2149
- return None, (400, "that file is not a dataset export: it has no \"format\"")
2150
- if doc["format"] not in (EXPORT_ONE, EXPORT_ALL):
2433
+ return None, (400, "that file is not an eval group export: it has no \"format\"")
2434
+ if doc["format"] not in EXPORT_KEYS:
2151
2435
  return None, (400, f"that file's format is {json.dumps(doc['format'])}, and the lab "
2152
- f"imports {EXPORT_ONE} and {EXPORT_ALL}")
2436
+ f"imports {EXPORT_ONE}, {EXPORT_ALL}, {DATASET_ONE} and {DATASET_ALL}")
2153
2437
  version = doc.get("version")
2154
- # Exactly the integer: Python's True == 1.
2155
- if type(version) is not int or version not in IMPORT_VERSIONS:
2438
+ # Exactly the integer: Python's True == 1. An eval group's file began
2439
+ # at version 7; a dataset's reads back to version 1.
2440
+ grouped = doc["format"] in (EXPORT_ONE, EXPORT_ALL)
2441
+ if type(version) is not int or version not in ((EXPORT_VERSION,) if grouped else IMPORT_VERSIONS):
2156
2442
  return None, (400, f"that file is version {json.dumps(version)}, and the lab "
2157
- f"reads versions 1 to {EXPORT_VERSION}")
2158
- if doc["format"] == EXPORT_ONE:
2159
- items = [doc.get("dataset")]
2443
+ + (f"reads version {EXPORT_VERSION}" if grouped else f"reads versions 1 to {EXPORT_VERSION}"))
2444
+ one = doc["format"] in (EXPORT_ONE, DATASET_ONE)
2445
+ key = EXPORT_KEYS[doc["format"]]
2446
+ if one:
2447
+ items = [doc.get(key)]
2160
2448
  else:
2161
- items = doc.get("datasets")
2449
+ items = doc.get(key)
2162
2450
  if not isinstance(items, list):
2163
- return None, (400, "an export of every dataset holds them as a \"datasets\" list")
2451
+ return None, (400, f"an export of every {'group' if grouped else 'dataset'} holds them as a \"{key}\" list")
2164
2452
  for k in doc:
2165
- if k not in ("format", "version", "dataset" if doc["format"] == EXPORT_ONE else "datasets"):
2453
+ if k not in ("format", "version", key):
2166
2454
  return None, (400, f"that file has \"{k}\", which an export does not")
2167
2455
  ready = []
2456
+ what = "eval group" if grouped else "dataset"
2168
2457
  for i, item in enumerate(items):
2169
- at = "the dataset" if doc["format"] == EXPORT_ONE else f"dataset {i + 1}"
2458
+ at = f"the {what}" if one else f"{what} {i + 1}"
2170
2459
  if not isinstance(item, dict) or set(item) != {"name", "body"}:
2171
2460
  return None, (400, f"{at} has to be {{ \"name\", \"body\" }}")
2172
2461
  name, why = dataset_name(item["name"])
@@ -2339,13 +2628,17 @@ def read_pack(data: bytes):
2339
2628
  doc, why = load(name)
2340
2629
  if why:
2341
2630
  return None, why
2342
- if (not isinstance(doc, dict) or doc.get("format") != EXPORT_ONE
2343
- or doc.get("version") not in IMPORT_VERSIONS or not isinstance(doc.get("dataset"), dict)):
2344
- return None, f"{name} is not a dataset export the lab reads"
2345
- ds_name, why = dataset_name(doc["dataset"].get("name"))
2631
+ # A pack's datasets are eval groups, in either file form: the
2632
+ # group's (version 7) or the dataset's an older pack holds.
2633
+ fmt = doc.get("format") if isinstance(doc, dict) else None
2634
+ entry = doc.get(EXPORT_KEYS[fmt]) if fmt in (EXPORT_ONE, DATASET_ONE) else None
2635
+ if (not isinstance(entry, dict) or type(doc.get("version")) is not int
2636
+ or doc["version"] not in ((EXPORT_VERSION,) if fmt == EXPORT_ONE else IMPORT_VERSIONS)):
2637
+ return None, f"{name} is not an eval group export the lab reads"
2638
+ ds_name, why = dataset_name(entry.get("name"))
2346
2639
  if why:
2347
2640
  return None, f"{name}: {why}"
2348
- raw = doc["dataset"].get("body")
2641
+ raw = entry.get("body")
2349
2642
  body = upgrade_body(raw) if doc["version"] < EXPORT_VERSION else raw
2350
2643
  why = dataset_problem(body)
2351
2644
  if why:
@@ -2524,11 +2817,12 @@ class Packs:
2524
2817
  # the page upgrades the pipeline as it reads it.
2525
2818
  evals = doc.get("evals", doc.get("tests"))
2526
2819
  for t in evals if isinstance(evals, list) else [evals]:
2527
- ref = t.get("dataset") if isinstance(t, dict) else None
2528
- if isinstance(ref, dict):
2820
+ ref = eval_group_ref(t)
2821
+ if ref is not None:
2529
2822
  did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
2530
2823
  if did:
2531
- t["dataset"] = {"id": did, "name": DATASETS.get(did)["name"]}
2824
+ ref.clear()
2825
+ ref.update({"id": did, "name": DATASETS.get(did)["name"]})
2532
2826
  content = content_of(doc)
2533
2827
  if isinstance(content, dict) and isinstance(content.get("ref"), dict):
2534
2828
  sid = src_ids.get(content["ref"].get("id")) or src_ids.get(content["ref"].get("name"))
@@ -3251,13 +3545,15 @@ class Plugins:
3251
3545
  # is a marker the worker checks between items -- today's semantics, stopping
3252
3546
  # after the item in flight, never a process kill that loses its reply.
3253
3547
 
3254
- # A profile's key, as run-evals.js reads it: EVAL_API_KEY_<ID>, the id a run
3255
- # document keys its profiles table by, in the shell-safe spelling of itself.
3256
- # The id and not the name, so two profiles can share a name without sharing a
3257
- # key (docs/pipeline-model.md §5). One spelling on both sides -- evals-core.ts
3258
- # keyVar -- or the worker would never find the key the page never sent.
3259
- def key_var(profile_id: str) -> str:
3260
- return "EVAL_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(profile_id).upper()).strip("_")
3548
+ # A profile's key, as run-evals.js reads it: EVALSLAB_API_KEY_<SLUG>, the slug the
3549
+ # run document's connection carries (#253), or EVALSLAB_API_KEY_<ID> for one from
3550
+ # before slugs -- the id it keys its profiles table by -- in the shell-safe
3551
+ # spelling of itself. Never the name, so two profiles can share a name
3552
+ # without sharing a key (docs/pipeline-model.md §5). One spelling on both
3553
+ # sides -- evals-core.ts keyVar -- or the worker would never find the key the
3554
+ # page never sent.
3555
+ def key_var(profile_id: str, slug=None) -> str:
3556
+ return "EVALSLAB_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(slug or profile_id).upper()).strip("_")
3261
3557
 
3262
3558
 
3263
3559
  def run_items(run):
@@ -3331,7 +3627,33 @@ def brief_row(row):
3331
3627
  if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
3332
3628
  return it
3333
3629
  return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
3334
- return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
3630
+ # The run's verdict stays; each Target's eval by eval is the whole row's.
3631
+ brief = {k: v for k, v in row.items() if k != "verdicts"}
3632
+ return {**brief, "results": [item(it) for it in row.get("results") or []], "brief": True}
3633
+
3634
+
3635
+ # What of a worker's report a row keeps as its verdicts (#252): the run's --
3636
+ # pass, fail or incomplete, which the worker's exit code already said -- and
3637
+ # each Target's evals, by the eval's id, as `run.verdicts` names them. Only
3638
+ # these fields are carried, so nothing else the report holds lands in the row.
3639
+ # A status says whether the run finished; its verdict says whether it passed,
3640
+ # so a run that failed an eval is `done` with the verdict `fail`.
3641
+ VERDICTS = ("pass", "fail", "incomplete")
3642
+ VERDICT_FIELDS = ("name", "skipped", "verdict", "pass", "detail", "ran", "passed", "skippedItems")
3643
+
3644
+
3645
+ def report_verdicts(report):
3646
+ """(verdict, verdicts as JSON) from a worker's report, or (None, None)
3647
+ for one that carries none -- a worker from before #251."""
3648
+ verdict = report.get("verdict") if isinstance(report, dict) else None
3649
+ run = report.get("run") if verdict in VERDICTS else None
3650
+ targets = run.get("verdicts") if isinstance(run, dict) else None
3651
+ if not isinstance(targets, list):
3652
+ return None, None
3653
+ kept = [{eid: {k: e[k] for k in VERDICT_FIELDS if k in e}
3654
+ for eid, e in t.items() if isinstance(e, dict)}
3655
+ if isinstance(t, dict) else {} for t in targets]
3656
+ return verdict, json.dumps(kept)
3335
3657
 
3336
3658
 
3337
3659
  class Queue:
@@ -3368,13 +3690,27 @@ class Queue:
3368
3690
  # one of its own.
3369
3691
  if "rerun_of" not in cols:
3370
3692
  db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
3693
+ # The worker's verdicts (#252); a row from before them has none,
3694
+ # and is never given one after the fact.
3695
+ if "verdict" not in cols:
3696
+ db.execute("ALTER TABLE queue ADD COLUMN verdict TEXT")
3697
+ if "verdicts" not in cols:
3698
+ db.execute("ALTER TABLE queue ADD COLUMN verdicts TEXT")
3699
+ # The body of each eval group a run grades with, by `<id>@<n>`
3700
+ # (docs/pipeline-model.md §17). A row from before keeps its
3701
+ # `dataset` column, read as its one group.
3702
+ if "groups" not in cols:
3703
+ db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
3371
3704
 
3372
3705
  # ---- rows -----------------------------------------------------------
3373
3706
 
3374
3707
  # A row read with the submit time of the run it re-runs, if any: the
3375
3708
  # page names a run by that time (History's Run ID), so a "Re-run of"
3376
- # note reads without fetching the original.
3377
- SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
3709
+ # note reads without fetching the original. Its columns are named, since
3710
+ # the order ALTER TABLE added them in is no order _row can count on.
3711
+ SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
3712
+ "q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
3713
+ "q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
3378
3714
  "LEFT JOIN queue o ON o.id = q.rerun_of")
3379
3715
 
3380
3716
  @staticmethod
@@ -3387,8 +3723,9 @@ class Queue:
3387
3723
  "snapshot": json.loads(r[6]), "results": json.loads(r[7]),
3388
3724
  "progress": json.loads(r[8]), "totals": json.loads(r[9]),
3389
3725
  "error": r[10],
3390
- "rerunOf": r[12] if len(r) > 12 else None,
3391
- "rerunOfAt": r[13] if len(r) > 13 else None,
3726
+ "rerunOf": r[11], "rerunOfAt": r[14],
3727
+ "verdict": r[12],
3728
+ "verdicts": json.loads(r[13]) if r[13] else None,
3392
3729
  }
3393
3730
 
3394
3731
  # A row from before run documents has no version, and nothing here can
@@ -3411,15 +3748,19 @@ class Queue:
3411
3748
  (rid,)).fetchone())
3412
3749
  return row if self._readable(row) else None
3413
3750
 
3414
- def list(self, limit=RUNS_PAGE, before=None, full=False):
3415
- """Runs, newest first, and whether more follow. `before` is a
3416
- `submittedAt` the page of runs stops at, so History can page through
3417
- them the way it pages the runs store. Each row is brief_row's unless
3418
- `full` asks for the whole of it."""
3751
+ def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
3752
+ """Runs, newest first, and whether more follow. `before` and
3753
+ `before_id` are the `submittedAt` and id of the last run on the page
3754
+ before, so History can page through them the way it pages the runs
3755
+ store. `submittedAt` is to the second, so two runs can share one; the
3756
+ id breaks the tie, or a page ending between them would skip the
3757
+ second (#238). `before` alone stops at the second. Each row is
3758
+ brief_row's unless `full` asks for the whole of it."""
3419
3759
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3420
3760
  rows = self._all(db)
3421
- rows = [r for r in rows if before is None or r["submittedAt"] < before]
3422
- rows.sort(key=lambda r: r["submittedAt"], reverse=True)
3761
+ key = lambda r: (r["submittedAt"], r["id"])
3762
+ rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
3763
+ rows.sort(key=key, reverse=True)
3423
3764
  page = rows[:limit]
3424
3765
  return (page if full else [brief_row(r) for r in page]), len(rows) > limit
3425
3766
 
@@ -3430,27 +3771,30 @@ class Queue:
3430
3771
 
3431
3772
  # ---- submit ---------------------------------------------------------
3432
3773
 
3433
- def submit(self, run: dict, dataset=None, rerun_of=None):
3774
+ def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
3434
3775
  """
3435
3776
  A new queued run. `run` is the run document (docs/pipeline-model.md
3436
3777
  §5): the pipeline, the profiles it resolved to without their keys, its
3437
- content's file list in order and with its repeats, and the dataset's
3438
- version; `dataset` is that version's body, for a graded run;
3439
- `rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
3440
- every scenario -- so the total is the list's length, repeats and all,
3441
- the same count the runner and the page make.
3778
+ content's file list in order and with its repeats, and each eval
3779
+ group's version; `groups` is those versions' bodies, by `<id>@<n>`
3780
+ (§17), and `dataset` the one body a row from before them kept, which a
3781
+ re-run of one carries on; `rerun_of` is the run a re-run was queued
3782
+ from. Returns the row. Its items are that list, or the one inline
3783
+ text, each through every scenario -- so the total is the list's
3784
+ length, repeats and all, the same count the runner and the page make.
3442
3785
  """
3443
3786
  rid = secrets.token_hex(6)
3444
3787
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
3445
3788
  total = len(run_items(run))
3446
3789
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3447
3790
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
3448
- "snapshot, results, progress, totals, dataset, rerun_of) "
3449
- "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
3791
+ "snapshot, results, progress, totals, dataset, rerun_of, groups) "
3792
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
3450
3793
  (rid, "queued", now, json.dumps(run), "[]",
3451
3794
  json.dumps({"current": None, "n": 0, "total": total}),
3452
3795
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
3453
- None if dataset is None else json.dumps(dataset), rerun_of))
3796
+ None if dataset is None else json.dumps(dataset), rerun_of,
3797
+ None if groups is None else json.dumps(groups)))
3454
3798
  # Its prompts' uses, in the same transaction: a run is in the
3455
3799
  # library the moment it is queued, or not queued at all.
3456
3800
  if self.prompts is not None:
@@ -3460,19 +3804,43 @@ class Queue:
3460
3804
  # the moment the lock is let go.
3461
3805
  return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
3462
3806
 
3807
+ def groups(self, rid, raw=False):
3808
+ """The eval group bodies a readable run grades with, by `<id>@<n>`, or
3809
+ None for a run that is not there or kept none -- what its verdicts
3810
+ were graded by, whatever the groups hold now. A row from before runs
3811
+ kept a body per group answers its one `dataset` copy under its
3812
+ group's key. A copy kept at an earlier version reads as one of
3813
+ today's, unless [raw]: the worker is handed it as kept, since an
3814
+ earlier run's pipeline is upgraded under the rules that copy holds."""
3815
+ run = self.get(rid)
3816
+ if run is None:
3817
+ return None
3818
+ kept = self._kept(rid)
3819
+ if kept is None:
3820
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3821
+ old = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()[0]
3822
+ if not old:
3823
+ return None
3824
+ kept = {group_key(evals_dataset(run["snapshot"])): json.loads(old)}
3825
+ return kept if raw else {k: upgrade_body(b) for k, b in kept.items()}
3826
+
3827
+ def _kept(self, rid):
3828
+ """The row's own `groups` column, or None for a row from before it."""
3829
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3830
+ r = db.execute("SELECT groups FROM queue WHERE id = ?", (rid,)).fetchone()
3831
+ return json.loads(r[0]) if r and r[0] is not None else None
3832
+
3463
3833
  def dataset(self, rid, raw=False):
3464
- """The dataset body a readable run was submitted against, or None --
3465
- what its verdicts were graded by, whatever the dataset holds now. A
3466
- copy kept at an earlier version reads as one of today's, unless [raw]:
3467
- the worker is handed it as kept, since an earlier run's pipeline is
3468
- upgraded under the rules that copy holds."""
3834
+ """The body of the one Library group a readable run grades its cases
3835
+ against (evals_dataset), or None: the body the worker is handed as
3836
+ --dataset, and what the page reads a run's cases from."""
3469
3837
  if self.get(rid) is None:
3470
3838
  return None
3471
- with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3472
- r = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()
3473
- if not (r and r[0]):
3839
+ kept = self.groups(rid, raw=True)
3840
+ ref = evals_dataset(self.get(rid)["snapshot"])
3841
+ body = kept.get(group_key(ref)) if kept and ref is not None else None
3842
+ if body is None:
3474
3843
  return None
3475
- body = json.loads(r[0])
3476
3844
  return body if raw else upgrade_body(body)
3477
3845
 
3478
3846
  def _plugin_args(self, run):
@@ -3487,14 +3855,18 @@ class Queue:
3487
3855
  return ["--plugins", str(PLUGINS.dir)], None
3488
3856
 
3489
3857
  def _dataset_args(self, run, rundir):
3490
- """The worker's --dataset for a graded run: the body kept with the row,
3491
- written beside the run document. A row queued before runs kept their
3492
- dataset has none, and is pinned to the dataset as it reads now, once,
3493
- so every later pass over it agrees. Returns (args, None) or (None, why)."""
3858
+ """The worker's --dataset for a graded run: the body of its one
3859
+ Library group kept with the row, written beside the run document --
3860
+ one, until the worker reads a body per group (#233). A row queued
3861
+ before runs kept their dataset has none, and is pinned to the dataset
3862
+ as it reads now, once, so every later pass over it agrees. Returns
3863
+ (args, None) or (None, why)."""
3494
3864
  ref = evals_dataset(run["snapshot"])
3495
3865
  if ref is None:
3496
3866
  return [], None
3497
3867
  body = self.dataset(run["id"], raw=True)
3868
+ if body is None and self._kept(run["id"]) is not None:
3869
+ return None, f"the run kept no body of the eval group {ref.get('name') or ref.get('id')!r}"
3498
3870
  if body is None:
3499
3871
  snap = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
3500
3872
  if snap is None:
@@ -3547,7 +3919,8 @@ class Queue:
3547
3919
  if run["status"] not in ("cancelled", "interrupted", "incomplete"):
3548
3920
  return None, (409, "only a cancelled, interrupted or incomplete run can be resumed")
3549
3921
  (self.dir / rid / "cancel").unlink(missing_ok=True)
3550
- self._set(rid, status="queued", cancel=0, error=None)
3922
+ # What it reached is decided by the run it goes on to finish.
3923
+ self._set(rid, status="queued", cancel=0, error=None, verdict=None, verdicts=None)
3551
3924
  return self.get(rid), None
3552
3925
 
3553
3926
  def set_comment(self, rid, comment):
@@ -3597,9 +3970,12 @@ class Queue:
3597
3970
  _, err = self._pinned_files(snap)
3598
3971
  if err:
3599
3972
  return None, (409, err)
3973
+ # The group bodies the original kept: a re-run grades with exactly
3974
+ # them, whatever the groups or the pins read now.
3975
+ kept = self._kept(rid)
3600
3976
  ref = evals_dataset(snap)
3601
3977
  body = None
3602
- if ref is not None:
3978
+ if ref is not None and kept is None:
3603
3979
  body = self.dataset(rid, raw=True)
3604
3980
  if body is None:
3605
3981
  # A run that never started kept no body: the dataset's, if it
@@ -3615,7 +3991,7 @@ class Queue:
3615
3991
  _, err = worker_destinations(snap)
3616
3992
  if err:
3617
3993
  return None, (403, err)
3618
- return self.submit(snap, body, rerun_of=rid), None
3994
+ return self.submit(snap, body, rerun_of=rid, groups=kept), None
3619
3995
 
3620
3996
  def rerun_item(self, rid, index):
3621
3997
  """
@@ -3677,9 +4053,12 @@ class Queue:
3677
4053
  while len(results) <= index:
3678
4054
  results.append(None)
3679
4055
  results[index] = next((it for it in items if it and it.get("item") == index), items[0])
4056
+ # The worker read every item to reach its verdicts, so they are the
4057
+ # row's as it now stands.
4058
+ verdict, verdicts = report_verdicts(report)
3680
4059
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3681
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3682
- (json.dumps(results), None, run["id"]))
4060
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4061
+ "WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
3683
4062
  return self.get(run["id"]), None
3684
4063
 
3685
4064
  def rescore_item(self, rid, index):
@@ -3742,9 +4121,12 @@ class Queue:
3742
4121
  while len(results) <= index:
3743
4122
  results.append(None)
3744
4123
  results[index] = next((it for it in items if it and it.get("item") == index), items[0])
4124
+ # The worker read every item to reach its verdicts, so they are the
4125
+ # row's as it now stands.
4126
+ verdict, verdicts = report_verdicts(report)
3745
4127
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3746
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3747
- (json.dumps(results), None, run["id"]))
4128
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4129
+ "WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
3748
4130
  return self.get(run["id"]), None
3749
4131
 
3750
4132
  # ---- the run --------------------------------------------------------
@@ -3855,10 +4237,11 @@ class Queue:
3855
4237
  report = self._read_report(rid)
3856
4238
  items = report["run"]["items"] if report and isinstance(report.get("run"), dict) else None
3857
4239
  if isinstance(items, list):
4240
+ verdict, verdicts = report_verdicts(report)
3858
4241
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3859
4242
  if db.execute("SELECT 1 FROM queue WHERE id = ?", (rid,)).fetchone() is not None:
3860
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3861
- (json.dumps(items), None, rid))
4243
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4244
+ "WHERE id = ?", (json.dumps(items), None, verdict, verdicts, rid))
3862
4245
  run = self.get(rid)
3863
4246
  # The watchdog may have failed the run while it was on the wire; a
3864
4247
  # row that already left `running` is not this worker's to re-label.
@@ -3968,8 +4351,8 @@ class Queue:
3968
4351
  """The oldest queued run this server can read. One it cannot is never
3969
4352
  started: nothing here would know what it asks for."""
3970
4353
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3971
- for r in db.execute("SELECT * FROM queue WHERE status = 'queued' "
3972
- "ORDER BY submitted_at, rowid").fetchall():
4354
+ for r in db.execute(self.SELECT + " WHERE q.status = 'queued' "
4355
+ "ORDER BY q.submitted_at, q.rowid").fetchall():
3973
4356
  if not self._readable(self._row(r)):
3974
4357
  continue
3975
4358
  db.execute("UPDATE queue SET status = 'running', started_at = ? "
@@ -4015,10 +4398,89 @@ class Queue:
4015
4398
  proc.kill()
4016
4399
 
4017
4400
 
4401
+ # The key a relayed request is sent with: the one it carries, or -- a page
4402
+ # that was handed KEY_HELD in place of a profile's key -- the one the store
4403
+ # holds for that profile.
4404
+ def relay_key(payload: dict) -> str:
4405
+ key = str(payload.get("key") or "").strip()
4406
+ if held(key):
4407
+ return STORE.held_key(key).strip() if STORE is not None else ""
4408
+ return key
4409
+
4410
+
4018
4411
  # The worker-destination rules, at submit and again at dequeue: a run's
4019
4412
  # profiles take their keys from the profiles store, by id, so the request
4020
4413
  # carries none, and the relay's rules bind where they go. Returns (env_vars,
4021
4414
  # None) or (None, a named refusal).
4415
+ # Export for CI (#254): a pipeline as a bundle a repository keeps and
4416
+ # `evals-lab run` runs with no lab. The page writes its text -- the core's
4417
+ # exportBundle: pipeline.yaml, profiles.yaml by slug, datasets/<slug>.json --
4418
+ # and the server adds what only it holds: every installed plugin, as a run
4419
+ # stamps them all, and, when asked, the Source's files under items/. A key is
4420
+ # refused rather than zipped: no Setup key, no token, no field named like
4421
+ # one, in any of it (the core's bundleProblems asks the same of the page's).
4422
+ BUNDLE_TEXT = re.compile(r"pipeline\.yaml|profiles\.yaml|datasets/[a-z0-9]+(?:-[a-z0-9]+)*\.json")
4423
+ BUNDLE_KEY_FIELD = re.compile(
4424
+ r'^[\s-]*"?((?:api[-_]?)?key|authorization|bearer|secret|password|token)"?\s*:', re.I | re.M)
4425
+
4426
+
4427
+ def bundle_problems(files, keys=()):
4428
+ """Why [files] -- path to text -- cannot leave the lab, or ""."""
4429
+ for path, text in files.items():
4430
+ for key in keys:
4431
+ key = str(key or "").strip()
4432
+ if len(key) >= 4 and key in text:
4433
+ return f"{path} holds a Target profile's key, and a key never leaves the lab"
4434
+ if TOKEN_SHAPE.search(text):
4435
+ return f"{path} holds a token, and a key never leaves the lab"
4436
+ m = BUNDLE_KEY_FIELD.search(text)
4437
+ if m:
4438
+ return f"{path} holds a field named {m.group(1)}, and a key never leaves the lab"
4439
+ return ""
4440
+
4441
+
4442
+ def build_bundle(payload):
4443
+ """The zip Export for CI downloads, from the page's [payload]:
4444
+ { files: {path: text}, source: id or null, items: bool }. Returns
4445
+ (bytes, None) or (None, (status, one sentence))."""
4446
+ files = payload.get("files")
4447
+ if not isinstance(files, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in files.items()):
4448
+ return None, (400, "a bundle's files are text, by path")
4449
+ for path in files:
4450
+ if not BUNDLE_TEXT.fullmatch(path):
4451
+ return None, (400, f"{path!r} is not a file a bundle holds")
4452
+ if "pipeline.yaml" not in files or "profiles.yaml" not in files:
4453
+ return None, (400, "a bundle holds pipeline.yaml and profiles.yaml")
4454
+ stored = []
4455
+ if STORE is not None:
4456
+ stored = ((STORE.all().get("promptlab.profiles") or {}).get("body") or {}).get("list") or []
4457
+ why = bundle_problems(files, [p.get("key") for p in stored if isinstance(p, dict)])
4458
+ if why:
4459
+ return None, (400, why)
4460
+ items = []
4461
+ if payload.get("items"):
4462
+ sid = payload.get("source")
4463
+ found = SOURCES.item_paths(sid) if isinstance(sid, str) and SOURCES is not None else None
4464
+ if found is None:
4465
+ return None, (404, "no such source")
4466
+ items = found
4467
+ buf = io.BytesIO()
4468
+ with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
4469
+ for path, text in sorted(files.items()):
4470
+ zf.writestr(f"evals/{path}", text)
4471
+ if PLUGINS is not None:
4472
+ for p in PLUGINS.stamp():
4473
+ root = PLUGINS.dir / p["id"] / p["version"]
4474
+ for f in sorted(root.rglob("*")):
4475
+ if f.is_file():
4476
+ zf.write(f, f"evals/plugins/{p['id']}/{p['version']}/{f.relative_to(root).as_posix()}")
4477
+ for name, path in items:
4478
+ if not path.is_file():
4479
+ return None, (404, f"{name!r} is not in that Source")
4480
+ zf.write(path, f"evals/items/{name}")
4481
+ return buf.getvalue(), None
4482
+
4483
+
4022
4484
  def worker_destinations(run: dict):
4023
4485
  table = run.get("profiles") if isinstance(run, dict) else None
4024
4486
  if not isinstance(table, dict):
@@ -4046,7 +4508,7 @@ def worker_destinations(run: dict):
4046
4508
  why = allowed(base, key)
4047
4509
  if why:
4048
4510
  return None, f"Target profile {name}: {why}"
4049
- env[key_var(pid)] = key
4511
+ env[key_var(pid, conn.get("slug"))] = key
4050
4512
  return env, None
4051
4513
 
4052
4514
 
@@ -4068,15 +4530,17 @@ def worker_destinations(run: dict):
4068
4530
  # step in each job is what it sends there (docs/pipeline-model.md §16).
4069
4531
  # 11: `tests` are `evals`; nothing in an eval changes.
4070
4532
  # 12: a Contains metric's Ignore case holds item by item too, kept as written.
4071
- PIPELINE_VERSION = 12
4533
+ # 13: an eval is a link to an eval group, or a group of the pipeline's own,
4534
+ # and the document has an overall pass rule (docs/pipeline-model.md §17).
4535
+ PIPELINE_VERSION = 13
4072
4536
  # What a stored run may be: the current version, and the ones evals-core.ts's
4073
4537
  # upgradePipeline reads. A new submission is upgraded to the current one.
4074
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
4538
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
4075
4539
  TARGET_CAP = 4
4076
4540
  # Target steps whose words the Prompt library does not record as a use: they
4077
4541
  # ask no model (evals-core.ts's STEP_TYPES.echo).
4078
4542
  UNRECORDED_STEPS = {"echo"}
4079
- RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
4543
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "pass", "profiles", "comment", "plugins")
4080
4544
 
4081
4545
 
4082
4546
  def content_of(doc):
@@ -4122,14 +4586,17 @@ BUILTIN_CONNECTION_TYPES = dict(CONNECTION_TYPES)
4122
4586
  BUILTIN_LOCAL = set(LOCAL_CONNECTIONS)
4123
4587
  BUILTIN_CHAT_PATHS = dict(CONNECTION_CHAT_PATHS)
4124
4588
  BUILTIN_AUTH = dict(CONNECTION_AUTH)
4125
- CONNECTION_FIELDS = ("name", "url", "model", "type", "temperature", "px", "format",
4589
+ CONNECTION_FIELDS = ("name", "slug", "url", "model", "type", "temperature", "px", "format",
4126
4590
  "quality", "options")
4127
4591
  LOOKS_LIKE_A_KEY = re.compile(r"^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$",
4128
4592
  re.IGNORECASE)
4129
4593
  PROFILE_ID = re.compile(r"[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?")
4594
+ # evals-core.ts's SLUG and SLUG_MAX: what a key's variable is spelt from.
4595
+ PROFILE_SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*")
4596
+ SLUG_MAX = 32
4130
4597
 
4131
4598
 
4132
- def _fields(obj, at, allowed_fields, bad, key_from="EVAL_API_KEY_<ID>"):
4599
+ def _fields(obj, at, allowed_fields, bad, key_from="EVALSLAB_API_KEY_<ID>"):
4133
4600
  for k in obj:
4134
4601
  if k in allowed_fields:
4135
4602
  continue
@@ -4200,7 +4667,12 @@ CONNECTION_LISTS = {"http": http_endpoint_test}
4200
4667
 
4201
4668
 
4202
4669
  def connection_problems(pid, conn, at, bad):
4203
- _fields(conn, at, CONNECTION_FIELDS, bad, key_var(pid))
4670
+ slug = conn.get("slug")
4671
+ if slug is not None and not (isinstance(slug, str) and PROFILE_SLUG.fullmatch(slug) and len(slug) <= SLUG_MAX):
4672
+ bad.append(f"{at}: slug has to be lowercase letters and digits, with hyphens between, at most {SLUG_MAX}")
4673
+ slug = None
4674
+ var = key_var(pid, slug)
4675
+ _fields(conn, at, CONNECTION_FIELDS, bad, var)
4204
4676
  ctype = conn.get("type")
4205
4677
  if not isinstance(ctype, str) or ctype not in CONNECTION_TYPES:
4206
4678
  bad.append(f"{at}: type has to be one of {', '.join(CONNECTION_TYPES)}")
@@ -4211,7 +4683,7 @@ def connection_problems(pid, conn, at, bad):
4211
4683
  # Which settings there are is the type's to say; a key among them
4212
4684
  # is the server's.
4213
4685
  allowed = list(CONNECTION_TYPES.get(ctype, ())) if ctype else list(options)
4214
- _fields(options, f"{at}'s options", allowed, bad, key_var(pid))
4686
+ _fields(options, f"{at}'s options", allowed, bad, var)
4215
4687
  else:
4216
4688
  bad.append(f"{at}: options has to be an object")
4217
4689
  url = conn.get("url") or ""
@@ -4222,7 +4694,7 @@ def connection_problems(pid, conn, at, bad):
4222
4694
  parts = urllib.parse.urlsplit(url if "://" in url else "http://" + url)
4223
4695
  if parts.username or parts.password:
4224
4696
  bad.append(f"{at} has a key in its address, and a key never goes in a pipeline "
4225
- f"— ${key_var(pid)} supplies it")
4697
+ f"— ${var} supplies it")
4226
4698
  for why in CONNECTION_CHECKS.get(ctype, lambda _c: [])(conn):
4227
4699
  bad.append(f"{at} {why}")
4228
4700
  # llama.cpp runs its own llama-server: a hosted address is refused, as the
@@ -4233,16 +4705,33 @@ def connection_problems(pid, conn, at, bad):
4233
4705
  bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
4234
4706
 
4235
4707
 
4708
+ def eval_group_ref(t):
4709
+ """The Library group an eval reads the cases of, as the dict itself (so a
4710
+ caller may stamp it in place), or None: version 13's link (`group`) or a
4711
+ private group's `casesFrom`, or an earlier version's `dataset`
4712
+ (evals-core.ts casesRef)."""
4713
+ if not isinstance(t, dict):
4714
+ return None
4715
+ if t.get("type") == "group":
4716
+ if isinstance(t.get("group"), dict):
4717
+ return t["group"]
4718
+ own = t.get("own")
4719
+ return own["casesFrom"] if isinstance(own, dict) and isinstance(own.get("casesFrom"), dict) else None
4720
+ return t["dataset"] if isinstance(t.get("dataset"), dict) else None
4721
+
4722
+
4236
4723
  def evals_dataset(doc):
4237
- """The dataset reference a document's evals grade against, or None: the
4238
- first eval that names one. A run grades against one dataset (the core's
4724
+ """The Library group a document's evals read the cases of, or None: the
4725
+ first eval that names one. A run grades against one (the core's
4239
4726
  validatePipeline says so). Reads a stored document of any shape --
4240
- version 11's `evals`, the `tests` before it, version 5's list or the one
4241
- test before that -- since rows keep the document they were submitted with."""
4727
+ version 13's links, version 11's `evals`, the `tests` before it, version
4728
+ 5's list or the one test before that -- since rows keep the document they
4729
+ were submitted with."""
4242
4730
  evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
4243
4731
  for t in evals if isinstance(evals, list) else [evals]:
4244
- if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
4245
- return t["dataset"]
4732
+ ref = eval_group_ref(t)
4733
+ if ref is not None:
4734
+ return ref
4246
4735
  return None
4247
4736
 
4248
4737
 
@@ -4325,6 +4814,7 @@ def run_problems(run):
4325
4814
  table = run.get("profiles")
4326
4815
  if not isinstance(table, dict):
4327
4816
  return bad + ["profiles has to be an object of id → connection"]
4817
+ spelt = {}
4328
4818
  for pid, conn in table.items():
4329
4819
  at = f"profile {pid}"
4330
4820
  if not PROFILE_ID.fullmatch(str(pid)):
@@ -4333,6 +4823,11 @@ def run_problems(run):
4333
4823
  if not isinstance(conn, dict):
4334
4824
  bad.append(f"{at} has to be an object")
4335
4825
  continue
4826
+ # Two profiles spelling one variable would hand one the other's key.
4827
+ var = key_var(pid, conn.get("slug") if isinstance(conn.get("slug"), str) else None)
4828
+ if var in spelt:
4829
+ bad.append(f"profiles {spelt[var]} and {pid} both take their key from ${var}")
4830
+ spelt[var] = pid
4336
4831
  connection_problems(pid, conn, at, bad)
4337
4832
  targets = run.get("targets")
4338
4833
  if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
@@ -4550,9 +5045,23 @@ class Handler(BaseHTTPRequestHandler):
4550
5045
 
4551
5046
  # ---- routes ---------------------------------------------------------
4552
5047
 
5048
+ # The eval groups' routes (docs/pipeline-model.md §17) are the datasets'
5049
+ # under their new name: /api/datasets goes on answering the same rows, so
5050
+ # dataset-diff.js and a script written against it keep working. A list
5051
+ # asked for by the new name is `groups`.
5052
+ GROUPS_ROUTE = "/api/eval-groups"
5053
+ grouped = False
5054
+
5055
+ def _alias(self):
5056
+ self.grouped = self.path == self.GROUPS_ROUTE or self.path.startswith((self.GROUPS_ROUTE + "/",
5057
+ self.GROUPS_ROUTE + "?"))
5058
+ if self.grouped:
5059
+ self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
5060
+
4553
5061
  def do_GET(self):
4554
5062
  if not self._authorised():
4555
5063
  return
5064
+ self._alias()
4556
5065
  path = self.path.split("?", 1)[0]
4557
5066
  # The lab is one page: a Connection and an Input make a scenario,
4558
5067
  # Content and Evals are shared, and one to four scenarios run over the
@@ -4599,8 +5108,16 @@ class Handler(BaseHTTPRequestHandler):
4599
5108
  except ValueError:
4600
5109
  return self._json(400, {"error": "limit has to be a number"})
4601
5110
  before = (query.get("before") or [None])[0]
4602
- runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
5111
+ before_id = (query.get("beforeId") or [None])[0]
5112
+ runs, more = QUEUE.list(limit, before, before_id, full=(query.get("full") or [""])[0] == "1")
4603
5113
  return self._json(200, {"runs": runs, "more": more})
5114
+ if path.startswith("/api/queue/") and path.endswith("/groups") and path.count("/") == 4:
5115
+ # The bodies of the eval groups a run grades with, by `<id>@<n>`
5116
+ # (§17); null for a run that kept none, as /dataset answers.
5117
+ run_id = path.split("/")[3]
5118
+ if QUEUE is None or QUEUE.get(run_id) is None:
5119
+ return self._send(404, b"not found", "text/plain")
5120
+ return self._json(200, QUEUE.groups(run_id))
4604
5121
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
4605
5122
  # The dataset body a graded run was submitted against, which is
4606
5123
  # what its verdicts were graded by; the list never carries it.
@@ -4762,6 +5279,7 @@ class Handler(BaseHTTPRequestHandler):
4762
5279
  def do_DELETE(self):
4763
5280
  if not self._authorised() or not self._from_this_page():
4764
5281
  return
5282
+ self._alias()
4765
5283
  path = self.path.split("?", 1)[0]
4766
5284
  if path == "/api/connections/google":
4767
5285
  if CONNECTIONS is None:
@@ -4829,6 +5347,7 @@ class Handler(BaseHTTPRequestHandler):
4829
5347
  def do_PATCH(self):
4830
5348
  if not self._authorised() or not self._from_this_page():
4831
5349
  return
5350
+ self._alias()
4832
5351
  parts = self.path.split("?", 1)[0].split("/")
4833
5352
  if len(parts) != 4 or parts[1] != "api":
4834
5353
  return self._send(404, b"not found", "text/plain")
@@ -4859,6 +5378,7 @@ class Handler(BaseHTTPRequestHandler):
4859
5378
  def do_PUT(self):
4860
5379
  if not self._authorised() or not self._from_this_page():
4861
5380
  return
5381
+ self._alias()
4862
5382
  parts = self.path.split("?", 1)[0].split("/")
4863
5383
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
4864
5384
  return self._prompts_put(parts[3])
@@ -4921,6 +5441,7 @@ class Handler(BaseHTTPRequestHandler):
4921
5441
  return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
4922
5442
 
4923
5443
  def _post(self):
5444
+ self._alias()
4924
5445
  path = self.path.split("?", 1)[0]
4925
5446
  if path == "/api/state":
4926
5447
  return self._state_write()
@@ -4940,6 +5461,12 @@ class Handler(BaseHTTPRequestHandler):
4940
5461
  return self._prompts_post(path)
4941
5462
  if path.startswith("/api/sources"):
4942
5463
  return self._sources_post(path)
5464
+ if path == "/api/bundle":
5465
+ data, err = build_bundle(self._payload() or {})
5466
+ if err:
5467
+ return self._json(err[0], {"error": err[1]})
5468
+ return self._send(200, data, "application/zip",
5469
+ (("Content-Disposition", 'attachment; filename="evals.zip"'),))
4943
5470
  payload = self._payload()
4944
5471
  if payload is None:
4945
5472
  return self._json(400, {"error": "bad body"})
@@ -5012,7 +5539,7 @@ class Handler(BaseHTTPRequestHandler):
5012
5539
 
5013
5540
  def _models_list(self, payload):
5014
5541
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
5015
- key = str(payload.get("key") or "").strip()
5542
+ key = relay_key(payload)
5016
5543
  ctype = str(payload.get("type") or "")
5017
5544
  # A type that lists some other way (an HTTP endpoint: its own test
5018
5545
  # path, key header and headers) says where and how.
@@ -5063,7 +5590,7 @@ class Handler(BaseHTTPRequestHandler):
5063
5590
  # and how the key travels, per the type -- the same mirror the models
5064
5591
  # list uses, so a client cannot point the relay at a path of its own.
5065
5592
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
5066
- key = str(payload.get("key") or "").strip()
5593
+ key = relay_key(payload)
5067
5594
  if not header_safe(key):
5068
5595
  return self._json(400, {"error": "That key has characters that "
5069
5596
  "cannot be sent in a header, so "
@@ -5099,6 +5626,10 @@ class Handler(BaseHTTPRequestHandler):
5099
5626
  for n, d in docs.items()})
5100
5627
  if stale is not None:
5101
5628
  return self._json(409, {"stale": stale})
5629
+ # A pin names a group's version, so the group keeps that version from
5630
+ # the moment a pipeline pins it (§17).
5631
+ if "promptlab.workflows" in docs and DATASETS is not None:
5632
+ DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
5102
5633
  return self._json(200, {"versions": versions})
5103
5634
 
5104
5635
  # ---- The run queue (#530) ----------------------------------------------
@@ -5137,20 +5668,30 @@ class Handler(BaseHTTPRequestHandler):
5137
5668
  # The plugins it runs under, as installed now: the server's to say,
5138
5669
  # whatever the document claimed.
5139
5670
  run["plugins"] = PLUGINS.stamp() if PLUGINS is not None else []
5140
- # A graded run keeps the body of the dataset it names, as it reads
5141
- # now, and records that body's fingerprint on the reference it
5142
- # belongs to: the worker grades against that and nothing else.
5143
- ref = evals_dataset(run)
5144
- body = None
5145
- if ref is not None:
5146
- snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
5147
- if snap is None:
5671
+ # Each Library group the evals read, resolved once (§17): a pinned
5672
+ # link to its pin, anything else to the group's newest version. The
5673
+ # reference records the version, `n`, and its body's fingerprint, and
5674
+ # the body is kept with the row under `<id>@<n>`: the worker grades
5675
+ # with that and nothing else, and the version is frozen from now on.
5676
+ groups = {}
5677
+ for t in run["evals"]:
5678
+ ref = eval_group_ref(t)
5679
+ if ref is None:
5680
+ continue
5681
+ pin = t.get("pin") if t.get("type") == "group" and t.get("group") is ref else None
5682
+ got = DATASETS.resolve(ref.get("id"), pin if type(pin) is int else None) if DATASETS else None
5683
+ if got is None:
5148
5684
  return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
5149
- body, version = snap
5150
- for t in run["evals"]:
5151
- if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
5152
- t["dataset"]["version"] = version
5153
- return self._json(201, {"run": QUEUE.submit(run, body)})
5685
+ n, body = got
5686
+ if n is None:
5687
+ return self._json(400, {"error": f"the run cannot be queued: {body}"})
5688
+ ref["n"], ref["version"] = n, fingerprint(body)
5689
+ groups[group_key(ref)] = body
5690
+ # The worker is handed one body until it reads one per group (#233).
5691
+ if len(groups) > 1:
5692
+ return self._json(400, {"error": "the run cannot be queued: its evals grade against "
5693
+ "one version of one Library eval group at a time"})
5694
+ return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
5154
5695
 
5155
5696
  # ---- Datasets ----------------------------------------------------------
5156
5697
  # Rows in the store (Datasets above). The page reads one by id; an export
@@ -5244,9 +5785,16 @@ class Handler(BaseHTTPRequestHandler):
5244
5785
  DATASETS.lazy_trash()
5245
5786
  parts = path.split("/")
5246
5787
  if len(parts) == 3:
5247
- return self._json(200, {"datasets": DATASETS.list()})
5788
+ return self._json(200, {"groups" if self.grouped else "datasets": DATASETS.list()})
5248
5789
  if len(parts) == 4 and parts[3] == "export":
5249
- return self._download(DATASETS.export_all(), "datasets", "all")
5790
+ return self._download(DATASETS.export_all(), "eval-groups", "all")
5791
+ if len(parts) == 5 and parts[4] == "versions":
5792
+ got = DATASETS.versions(parts[3])
5793
+ return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
5794
+ if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
5795
+ got = DATASETS.version_body(parts[3], int(parts[5]))
5796
+ return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
5797
+ else self._send(404, b"not found", "text/plain")
5250
5798
  if len(parts) == 4 and parts[3] == "archived-rules":
5251
5799
  return self._json(200, {"rules": DATASETS.archived_rules()})
5252
5800
  if len(parts) == 4:
@@ -5256,7 +5804,7 @@ class Handler(BaseHTTPRequestHandler):
5256
5804
  doc = DATASETS.export(parts[3])
5257
5805
  if doc is None:
5258
5806
  return self._send(404, b"not found", "text/plain")
5259
- return self._download(doc, "dataset", doc["dataset"]["name"])
5807
+ return self._download(doc, "eval-group", doc["group"]["name"])
5260
5808
  return self._send(404, b"not found", "text/plain")
5261
5809
 
5262
5810
  def _download(self, doc, kind, name):
@@ -5284,6 +5832,12 @@ class Handler(BaseHTTPRequestHandler):
5284
5832
  if err:
5285
5833
  return self._json(err[0], {"error": err[1]})
5286
5834
  return self._json(200, dataset)
5835
+ if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
5836
+ self._payload()
5837
+ dataset, err = DATASETS.restore_version(parts[3], int(parts[5]))
5838
+ if err:
5839
+ return self._json(err[0], {"error": err[1]})
5840
+ return self._json(200, dataset)
5287
5841
  if len(parts) == 4 and parts[3] == "import":
5288
5842
  doc, err = self._dataset_payload(DATASET_IMPORT_CAP)
5289
5843
  if err: