evals-lab 0.4.1 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -560,6 +560,48 @@ QUEUE_WAIT = 1.0
560
560
  WATCH_EVERY = 5.0
561
561
 
562
562
 
563
+ # A Target profile's key is written from the page and never read back by it
564
+ # (#258): a client holding the lab's password -- CI's runners and their logs
565
+ # among them -- is handed no key. Where a profile holds one, what is served
566
+ # (/api/state, the page's carried copy, a refused write's current copy) holds
567
+ # KEY_HELD and the profile's id instead, and a write that brings that back
568
+ # keeps the key the store holds for the id: the profile's own, or the one it
569
+ # was cloned or restored from. KEY_HELD starts with a character no key can
570
+ # (header_safe), so a pasted key is never taken for it.
571
+ KEY_HELD = "\u2022held:"
572
+
573
+
574
+ def held(key) -> bool:
575
+ return isinstance(key, str) and key.startswith(KEY_HELD)
576
+
577
+
578
+ def profiles_in(name: str, body):
579
+ """Every profile a synced document holds: the profiles store's list, and
580
+ each profile version's copy."""
581
+ if not isinstance(body, dict):
582
+ return
583
+ if name == "promptlab.profiles":
584
+ for p in body.get("list") or []:
585
+ if isinstance(p, dict):
586
+ yield p
587
+ elif name == "promptlab.versions":
588
+ for kept in (body.get("profile") or {}).values() if isinstance(body.get("profile"), dict) else ():
589
+ for v in kept if isinstance(kept, list) else ():
590
+ if isinstance(v, dict) and isinstance(v.get("doc"), dict):
591
+ yield v["doc"]
592
+
593
+
594
+ def hide_keys(docs: dict) -> dict:
595
+ out = {}
596
+ for name, d in docs.items():
597
+ d = json.loads(json.dumps(d))
598
+ for p in profiles_in(name, d.get("body")):
599
+ if isinstance(p.get("key"), str) and p["key"] and not held(p["key"]):
600
+ p["key"] = KEY_HELD + str(p.get("id") or "")
601
+ out[name] = d
602
+ return out
603
+
604
+
563
605
  class Store:
564
606
  """
565
607
  Documents by name, each with a version that goes up by one per write.
@@ -579,6 +621,11 @@ class Store:
579
621
  db.execute("CREATE TABLE IF NOT EXISTS docs (name TEXT PRIMARY KEY, "
580
622
  "version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
581
623
  db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
624
+ # The key of a profile a write took out, by id, for TRASH_SECONDS:
625
+ # Undo puts the profile back holding KEY_HELD, and this is what
626
+ # it holds.
627
+ db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
628
+ "key TEXT NOT NULL, at REAL NOT NULL)")
582
629
  # History was a document, capped at what a browser could hold. The
583
630
  # first start with the table moves what that document had into it,
584
631
  # once, and drops the document so the page stops carrying it.
@@ -600,8 +647,30 @@ class Store:
600
647
 
601
648
  def served(self) -> dict:
602
649
  """What a browser is handed: the SYNCED documents only, so a retired
603
- key's rows stay in the store without reaching a page again."""
604
- return {n: d for n, d in self.all().items() if n in SYNCED}
650
+ key's rows stay in the store without reaching a page again, and no
651
+ profile's key (KEY_HELD)."""
652
+ return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
653
+
654
+ def _keyring(self, db, now: dict) -> dict:
655
+ """Each profile id's key as the store holds it: its profile's, else a
656
+ dropped one's, else its newest version's."""
657
+ ring = {}
658
+ for p in profiles_in("promptlab.versions", now.get("promptlab.versions", {}).get("body")):
659
+ pid, key = str(p.get("id") or ""), p.get("key")
660
+ if pid not in ring and isinstance(key, str) and key and not held(key):
661
+ ring[pid] = key
662
+ db.execute("DELETE FROM dropped_keys WHERE at < ?", (time.time() - TRASH_SECONDS,))
663
+ ring.update(db.execute("SELECT id, key FROM dropped_keys").fetchall())
664
+ for p in profiles_in("promptlab.profiles", now.get("promptlab.profiles", {}).get("body")):
665
+ key = p.get("key")
666
+ if isinstance(key, str) and key and not held(key):
667
+ ring[str(p.get("id") or "")] = key
668
+ return ring
669
+
670
+ def held_key(self, key: str) -> str:
671
+ """The key KEY_HELD stands for, or "" when the store holds none."""
672
+ with self.lock, closing(sqlite3.connect(self.path)) as db, db:
673
+ return self._keyring(db, self._rows(db)).get(key[len(KEY_HELD):], "")
605
674
 
606
675
  def write(self, docs: dict):
607
676
  """
@@ -614,9 +683,23 @@ class Store:
614
683
  have = lambda n: now.get(n, {"version": 0, "body": None})
615
684
  stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
616
685
  if stale:
617
- return None, stale
686
+ return None, hide_keys(stale)
618
687
  at = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
619
688
  with db:
689
+ ring = self._keyring(db, now)
690
+ for n, d in docs.items():
691
+ for p in profiles_in(n, d["body"]):
692
+ if held(p.get("key")):
693
+ p["key"] = ring.get(p["key"][len(KEY_HELD):], "")
694
+ if "promptlab.profiles" in docs:
695
+ kept = {str(p.get("id") or "") for p in profiles_in(
696
+ "promptlab.profiles", docs["promptlab.profiles"]["body"])}
697
+ db.executemany("DELETE FROM dropped_keys WHERE id = ?", [(i,) for i in kept])
698
+ db.executemany(
699
+ "INSERT OR REPLACE INTO dropped_keys (id, key, at) VALUES (?, ?, ?)",
700
+ [(str(p.get("id") or ""), p["key"], time.time())
701
+ for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
702
+ if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
620
703
  for n, d in docs.items():
621
704
  db.execute(
622
705
  "INSERT INTO docs (name, version, body, updated_at) VALUES (?, ?, ?, ?) "
@@ -726,7 +809,7 @@ class Sources:
726
809
  out = []
727
810
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
728
811
  db.row_factory = sqlite3.Row
729
- for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
812
+ for r in db.execute("SELECT id, name, system, bytes, type, created_at FROM sources "
730
813
  "ORDER BY system DESC, name COLLATE NOCASE"):
731
814
  if r["system"]:
732
815
  files = self._sample_files()
@@ -734,10 +817,13 @@ class Sources:
734
817
  "type": r["type"],
735
818
  "files": len(files), "bytes": sum(f["bytes"] for f in files)})
736
819
  else:
737
- n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
738
- (r["id"],)).fetchone()[0]
820
+ n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
821
+ (r["id"],)).fetchone()
822
+ # When it last changed: made, or a file added -- what a
823
+ # picker orders its recent Sources by.
739
824
  out.append({"id": r["id"], "name": r["name"], "system": False,
740
- "type": r["type"], "files": n, "bytes": r["bytes"]})
825
+ "type": r["type"], "files": n, "bytes": r["bytes"],
826
+ "changed": max(filter(None, (r["created_at"], last)))})
741
827
  return out
742
828
 
743
829
  def get(self, sid) -> dict:
@@ -1177,6 +1263,18 @@ class Sources:
1177
1263
  return None, (500, "the files could not be copied")
1178
1264
  return self.get(sid), None
1179
1265
 
1266
+ def item_paths(self, sid):
1267
+ """Every file of a Source, as (name, path on disk) in its own order,
1268
+ or None when there is no such Source."""
1269
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1270
+ r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1271
+ if r is None:
1272
+ return None
1273
+ if r[0]:
1274
+ return [(f["name"], SAMPLES / f["name"]) for f in self._sample_files()]
1275
+ return [(row[0], self.dir / sid / row[0]) for row in db.execute(
1276
+ "SELECT name FROM source_files WHERE source = ? ORDER BY name", (sid,))]
1277
+
1180
1278
  def zip_files(self, sid, names):
1181
1279
  """The bytes of a stdlib zipfile holding exactly the named files, in
1182
1280
  the order named, or an error. Read under the store's lock so the file
@@ -1362,15 +1460,23 @@ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cas
1362
1460
  # 5 was told by its `source` alone, and earlier ones by neither.
1363
1461
  DATASET_BODY_VERSION = 7
1364
1462
  DATASET_NAME_MAX = 80
1365
- # The file forms Export writes and Import reads. Export writes version 7;
1366
- # Import reads it and versions 1 to 6, upgraded, and refuses anything else,
1367
- # as a pipeline of another version is refused. Versions 1 to 3 carried a
1368
- # prompt, which an import gives to the Prompt library.
1369
- EXPORT_ONE = "evals-lab/dataset"
1370
- EXPORT_ALL = "evals-lab/datasets"
1463
+ # The file forms Export writes and Import reads. Export writes an eval group
1464
+ # at version 7 (docs/pipeline-model.md §17); Import reads that, and a dataset
1465
+ # file of versions 1 to 7, upgraded, and refuses anything else, as a pipeline
1466
+ # of another version is refused. Versions 1 to 3 carried a prompt, which an
1467
+ # import gives to the Prompt library.
1468
+ EXPORT_ONE = "evals-lab/eval-group"
1469
+ EXPORT_ALL = "evals-lab/eval-groups"
1470
+ DATASET_ONE = "evals-lab/dataset"
1471
+ DATASET_ALL = "evals-lab/datasets"
1371
1472
  EXPORT_VERSION = 7
1372
1473
  IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
1474
+ # Each file form, and the key its one entry or its list sits under.
1475
+ EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
1373
1476
  SCORING_MODES = ("all", "weighted")
1477
+ # A group's newest version is edited in place by a save within this long of
1478
+ # the last, as a prompt's is (PROMPT_IDLE_SECONDS).
1479
+ GROUP_IDLE_SECONDS = float(os.environ.get("GROUP_IDLE_SECONDS", "30"))
1374
1480
 
1375
1481
 
1376
1482
  def blank_dataset() -> dict:
@@ -1574,6 +1680,30 @@ def fingerprint(body: dict) -> str:
1574
1680
  return hashlib.sha256(text.encode("utf-8")).hexdigest()[:CASE_SET_LEN]
1575
1681
 
1576
1682
 
1683
+ def pins_in(workflows) -> set:
1684
+ """(group id, version) for every link a stored pipeline pins: the
1685
+ promptlab.workflows body, each pipeline as its `work`. Only version 13
1686
+ links pin, and a pipeline of an earlier version has none."""
1687
+ out = set()
1688
+ listed = workflows.get("list") if isinstance(workflows, dict) else None
1689
+ for w in listed if isinstance(listed, list) else []:
1690
+ work = w.get("work") if isinstance(w, dict) else None
1691
+ evals = work.get("evals") if isinstance(work, dict) else None
1692
+ for t in evals if isinstance(evals, list) else []:
1693
+ if (isinstance(t, dict) and t.get("type") == "group" and isinstance(t.get("group"), dict)
1694
+ and isinstance(t["group"].get("id"), str) and type(t.get("pin")) is int):
1695
+ out.add((t["group"]["id"], t["pin"]))
1696
+ return out
1697
+
1698
+
1699
+ def group_key(ref) -> str:
1700
+ """The key a run keeps a group's body under: `<id>@<n>`, or the id alone
1701
+ for a reference from before the lab numbered versions."""
1702
+ n = ref.get("n") if isinstance(ref, dict) else None
1703
+ gid = ref.get("id") if isinstance(ref, dict) else None
1704
+ return f"{gid}@{n}" if type(n) is int else str(gid)
1705
+
1706
+
1577
1707
  def unique_dataset_name(name: str, taken: set) -> str:
1578
1708
  """`Receipts` again becomes `Receipts (2)`, compared without case."""
1579
1709
  lower = {t.lower() for t in taken}
@@ -1627,7 +1757,7 @@ def scenario_ref(sc, i):
1627
1757
  what version 4's upgrade gives one, by position."""
1628
1758
  sid = sc.get("id") if isinstance(sc, dict) else None
1629
1759
  name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
1630
- return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
1760
+ return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
1631
1761
 
1632
1762
 
1633
1763
  class Prompts:
@@ -1944,6 +2074,16 @@ class Datasets:
1944
2074
  # the body it was typed as is kept, as the rules were.
1945
2075
  db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
1946
2076
  "dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
2077
+ # Every version of each eval group, numbered from 1, for a link to
2078
+ # pin and a run to name (docs/pipeline-model.md §17). `ran` says a
2079
+ # run graded with it, which freezes it for good; a pin freezes it
2080
+ # while a stored pipeline holds the pin. Version 1 is added from
2081
+ # the row's body at its first save, pin or run, so a group nobody
2082
+ # touches holds what it held.
2083
+ db.execute("CREATE TABLE IF NOT EXISTS eval_group_versions ("
2084
+ "group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
2085
+ "created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
2086
+ "ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
1947
2087
  # Rows from an earlier version are converted once, in place: a
1948
2088
  # dataset is typed in by hand and costly to re-enter, so it is
1949
2089
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -1992,16 +2132,26 @@ class Datasets:
1992
2132
  return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
1993
2133
 
1994
2134
  @staticmethod
1995
- def _doc(r, body=True):
2135
+ def _doc(r, body=True, versions=1):
1996
2136
  """A row as the API answers it: a DatasetSummary, and its body with it,
1997
- read as today's version."""
2137
+ read as today's version. `version` is the save counter a write is
2138
+ arbitrated by; `versions` is how many group versions it has -- its
2139
+ newest one's number, which a pin may name, and 1 before any is kept,
2140
+ the row's body being version 1 in waiting."""
1998
2141
  parsed = upgrade_body(json.loads(r["body"]))
1999
2142
  out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
2000
- "version": r["version"], "updated": r["updated_at"]}
2143
+ "version": r["version"], "versions": versions, "updated": r["updated_at"]}
2001
2144
  if body:
2002
2145
  out["body"] = parsed
2003
2146
  return out
2004
2147
 
2148
+ @staticmethod
2149
+ def _counts(db) -> dict:
2150
+ return dict(db.execute("SELECT group_id, MAX(n) FROM eval_group_versions GROUP BY group_id"))
2151
+
2152
+ def _row_doc(self, db, r, body=True):
2153
+ return self._doc(r, body, self._counts(db).get(r["id"], 1))
2154
+
2005
2155
  def _connect(self):
2006
2156
  db = sqlite3.connect(self.store.path)
2007
2157
  db.row_factory = sqlite3.Row
@@ -2016,13 +2166,142 @@ class Datasets:
2016
2166
 
2017
2167
  def list(self) -> list:
2018
2168
  with self.store.lock, self._connect() as db:
2019
- return [self._doc(r, body=False) for r in db.execute(
2169
+ counts = self._counts(db)
2170
+ return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
2020
2171
  "SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id")]
2021
2172
 
2022
2173
  def get(self, did):
2023
2174
  with self.store.lock, self._connect() as db:
2024
2175
  r = self._live(db, did)
2025
- return self._doc(r) if r else None
2176
+ return self._row_doc(db, r) if r else None
2177
+
2178
+ # ---- a group's versions (docs/pipeline-model.md §17) -----------------
2179
+
2180
+ @staticmethod
2181
+ def _head(db, did):
2182
+ return db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? "
2183
+ "ORDER BY n DESC LIMIT 1", (did,)).fetchone()
2184
+
2185
+ def _mint(self, db, r):
2186
+ """The newest version of row [r], adding version 1 from its body
2187
+ first if it has none: edited when the row last was, so a group made a
2188
+ moment ago goes on being edited in place."""
2189
+ head = self._head(db, r["id"])
2190
+ if head is not None:
2191
+ return head
2192
+ try:
2193
+ edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
2194
+ except ValueError:
2195
+ # A stamp nothing here wrote is no recent edit.
2196
+ edited = 0
2197
+ db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
2198
+ "VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
2199
+ return self._head(db, r["id"])
2200
+
2201
+ @staticmethod
2202
+ def _pins(db) -> set:
2203
+ """Every (group, version) a stored pipeline pins, read in the caller's
2204
+ transaction from the docs table the page writes them to."""
2205
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'").fetchone()
2206
+ return pins_in(json.loads(row[0])) if row and row[0] else set()
2207
+
2208
+ def _cut(self, db, did, text):
2209
+ """A new newest version of [did] holding [text]; returns its number."""
2210
+ n = self._head(db, did)["n"] + 1
2211
+ db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
2212
+ "VALUES (?, ?, ?, ?, ?)", (did, n, text, self._now(), time.time()))
2213
+ return n
2214
+
2215
+ def _keep(self, db, r, text):
2216
+ """[text] as row [r]'s newest version: the newest edited in place while
2217
+ no run has graded with it, no link pins it and it was edited in the
2218
+ last GROUP_IDLE_SECONDS, and a new version otherwise."""
2219
+ head = self._mint(db, r)
2220
+ if text == head["body"]:
2221
+ return
2222
+ fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
2223
+ if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
2224
+ db.execute("UPDATE eval_group_versions SET body = ?, edited_at = ? WHERE group_id = ? AND n = ?",
2225
+ (text, time.time(), r["id"], head["n"]))
2226
+ else:
2227
+ self._cut(db, r["id"], text)
2228
+
2229
+ def versions(self, did):
2230
+ """A group's versions, newest first, without their bodies -- version 1
2231
+ alone, read from the row, before any is kept -- or None."""
2232
+ with self.store.lock, self._connect() as db:
2233
+ r = self._live(db, did)
2234
+ if r is None:
2235
+ return None
2236
+ pins = self._pins(db)
2237
+ rows = db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? ORDER BY n DESC",
2238
+ (did,)).fetchall()
2239
+ if not rows:
2240
+ return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (did, 1) in pins,
2241
+ "fingerprint": fingerprint(json.loads(r["body"]))}]
2242
+ return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
2243
+ "pinned": (did, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
2244
+ for v in rows]
2245
+
2246
+ def version_body(self, did, n):
2247
+ """Version [n]'s body, read as today's, or None."""
2248
+ with self.store.lock, self._connect() as db:
2249
+ r = self._live(db, did)
2250
+ if r is None:
2251
+ return None
2252
+ v = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
2253
+ (did, n)).fetchone()
2254
+ if v is None and n == 1 and self._head(db, did) is None:
2255
+ v = (r["body"],)
2256
+ return upgrade_body(json.loads(v[0])) if v else None
2257
+
2258
+ def restore_version(self, did, n):
2259
+ """An older version's body as the newest version, and the row's: as a
2260
+ prompt's Restore, nothing is rewritten, so a run or a pin naming any
2261
+ version still reads what it named. Returns (DatasetDoc, None)."""
2262
+ with self.store.lock, self._connect() as db, db:
2263
+ r = self._live(db, did)
2264
+ if r is None:
2265
+ return None, (404, "no such eval group")
2266
+ head = self._mint(db, r)
2267
+ old = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
2268
+ (did, n)).fetchone()
2269
+ if old is None:
2270
+ return None, (404, "no such version")
2271
+ if old["body"] != head["body"]:
2272
+ self._cut(db, did, old["body"])
2273
+ db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
2274
+ (old["body"], r["version"] + 1, self._now(), did))
2275
+ return self._row_doc(db, self._live(db, did)), None
2276
+
2277
+ def mint_pinned(self, pins):
2278
+ """Version 1 of each pinned group that has none yet: a pin names a
2279
+ version, so the version has to be kept from the moment it does."""
2280
+ with self.store.lock, self._connect() as db, db:
2281
+ for did, _ in pins:
2282
+ r = self._live(db, did)
2283
+ if r is not None:
2284
+ self._mint(db, r)
2285
+
2286
+ def resolve(self, did, pin=None):
2287
+ """The version a run submitted now grades with -- [pin], or the
2288
+ newest -- marked as graded with, in the same transaction, so no save
2289
+ can edit it in place between this and the run keeping its body.
2290
+ Returns (n, body), (None, why) for a pin the group has no version of,
2291
+ or None for a group the lab does not have. A submit refused after
2292
+ this leaves the version frozen, which costs only a new version at the
2293
+ next save."""
2294
+ with self.store.lock, self._connect() as db, db:
2295
+ r = self._live(db, did) if isinstance(did, str) else None
2296
+ if r is None:
2297
+ return None
2298
+ head = self._mint(db, r)
2299
+ v = head if pin is None else db.execute(
2300
+ "SELECT * FROM eval_group_versions WHERE group_id = ? AND n = ?", (did, pin)).fetchone()
2301
+ if v is None:
2302
+ return None, f"{r['name']} has no version {pin}"
2303
+ db.execute("UPDATE eval_group_versions SET ran = 1 WHERE group_id = ? AND n = ?", (did, v["n"]))
2304
+ return v["n"], json.loads(v["body"])
2026
2305
 
2027
2306
  def snapshot(self, did):
2028
2307
  """The body a run submitted now grades against, and its fingerprint;
@@ -2087,9 +2366,11 @@ class Datasets:
2087
2366
  if r is None:
2088
2367
  return None, (404, "no such dataset")
2089
2368
  if r["version"] != version:
2090
- return None, (409, {"current": self._doc(r)})
2369
+ return None, (409, {"current": self._row_doc(db, r)})
2370
+ text = json.dumps(body)
2371
+ self._keep(db, r, text)
2091
2372
  db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
2092
- (json.dumps(body), version + 1, self._now(), did))
2373
+ (text, version + 1, self._now(), did))
2093
2374
  return {"version": version + 1}, None
2094
2375
 
2095
2376
  def remove(self, did):
@@ -2115,58 +2396,69 @@ class Datasets:
2115
2396
  did = r["id"]
2116
2397
  return self.get(did), None
2117
2398
 
2399
+ @staticmethod
2400
+ def _purge(db, where, args):
2401
+ # A group's versions go with it: the trash held them, and Undo is
2402
+ # what would have brought them back.
2403
+ db.execute(f"DELETE FROM eval_group_versions WHERE group_id IN (SELECT id FROM datasets WHERE {where})", args)
2404
+ db.execute(f"DELETE FROM datasets WHERE {where}", args)
2405
+
2118
2406
  def lazy_trash(self):
2119
2407
  """A datasets request empties what has been trashed longer than
2120
2408
  TRASH_SECONDS, as a Sources request does."""
2121
2409
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2122
- db.execute("DELETE FROM datasets WHERE trash IS NOT NULL AND trashed_at < ?",
2123
- (time.time() - TRASH_SECONDS,))
2410
+ self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
2124
2411
 
2125
2412
  def empty_trash(self):
2126
2413
  """The startup sweep: a restart has nothing to undo."""
2127
2414
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2128
- db.execute("DELETE FROM datasets WHERE trash IS NOT NULL")
2415
+ self._purge(db, "trash IS NOT NULL", ())
2129
2416
 
2130
2417
  def export(self, did):
2131
2418
  d = self.get(did)
2132
2419
  if d is None:
2133
2420
  return None
2134
2421
  return {"format": EXPORT_ONE, "version": EXPORT_VERSION,
2135
- "dataset": {"name": d["name"], "body": d["body"]}}
2422
+ "group": {"name": d["name"], "body": d["body"]}}
2136
2423
 
2137
2424
  def export_all(self):
2138
2425
  with self.store.lock, self._connect() as db:
2139
2426
  rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
2140
2427
  "ORDER BY name COLLATE NOCASE, id").fetchall()
2141
2428
  return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
2142
- "datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2429
+ "groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2143
2430
 
2144
2431
  def import_file(self, doc):
2145
2432
  """Either export's file, as new datasets: import always creates, ids
2146
2433
  are this lab's, and a taken name gets ` (2)`. All of it or none.
2147
2434
  Returns ([DatasetSummary], None), or (None, (400, one sentence))."""
2148
2435
  if not isinstance(doc, dict) or "format" not in doc:
2149
- return None, (400, "that file is not a dataset export: it has no \"format\"")
2150
- if doc["format"] not in (EXPORT_ONE, EXPORT_ALL):
2436
+ return None, (400, "that file is not an eval group export: it has no \"format\"")
2437
+ if doc["format"] not in EXPORT_KEYS:
2151
2438
  return None, (400, f"that file's format is {json.dumps(doc['format'])}, and the lab "
2152
- f"imports {EXPORT_ONE} and {EXPORT_ALL}")
2439
+ f"imports {EXPORT_ONE}, {EXPORT_ALL}, {DATASET_ONE} and {DATASET_ALL}")
2153
2440
  version = doc.get("version")
2154
- # Exactly the integer: Python's True == 1.
2155
- if type(version) is not int or version not in IMPORT_VERSIONS:
2441
+ # Exactly the integer: Python's True == 1. An eval group's file began
2442
+ # at version 7; a dataset's reads back to version 1.
2443
+ grouped = doc["format"] in (EXPORT_ONE, EXPORT_ALL)
2444
+ if type(version) is not int or version not in ((EXPORT_VERSION,) if grouped else IMPORT_VERSIONS):
2156
2445
  return None, (400, f"that file is version {json.dumps(version)}, and the lab "
2157
- f"reads versions 1 to {EXPORT_VERSION}")
2158
- if doc["format"] == EXPORT_ONE:
2159
- items = [doc.get("dataset")]
2446
+ + (f"reads version {EXPORT_VERSION}" if grouped else f"reads versions 1 to {EXPORT_VERSION}"))
2447
+ one = doc["format"] in (EXPORT_ONE, DATASET_ONE)
2448
+ key = EXPORT_KEYS[doc["format"]]
2449
+ if one:
2450
+ items = [doc.get(key)]
2160
2451
  else:
2161
- items = doc.get("datasets")
2452
+ items = doc.get(key)
2162
2453
  if not isinstance(items, list):
2163
- return None, (400, "an export of every dataset holds them as a \"datasets\" list")
2454
+ return None, (400, f"an export of every {'group' if grouped else 'dataset'} holds them as a \"{key}\" list")
2164
2455
  for k in doc:
2165
- if k not in ("format", "version", "dataset" if doc["format"] == EXPORT_ONE else "datasets"):
2456
+ if k not in ("format", "version", key):
2166
2457
  return None, (400, f"that file has \"{k}\", which an export does not")
2167
2458
  ready = []
2459
+ what = "eval group" if grouped else "dataset"
2168
2460
  for i, item in enumerate(items):
2169
- at = "the dataset" if doc["format"] == EXPORT_ONE else f"dataset {i + 1}"
2461
+ at = f"the {what}" if one else f"{what} {i + 1}"
2170
2462
  if not isinstance(item, dict) or set(item) != {"name", "body"}:
2171
2463
  return None, (400, f"{at} has to be {{ \"name\", \"body\" }}")
2172
2464
  name, why = dataset_name(item["name"])
@@ -2339,13 +2631,17 @@ def read_pack(data: bytes):
2339
2631
  doc, why = load(name)
2340
2632
  if why:
2341
2633
  return None, why
2342
- if (not isinstance(doc, dict) or doc.get("format") != EXPORT_ONE
2343
- or doc.get("version") not in IMPORT_VERSIONS or not isinstance(doc.get("dataset"), dict)):
2344
- return None, f"{name} is not a dataset export the lab reads"
2345
- ds_name, why = dataset_name(doc["dataset"].get("name"))
2634
+ # A pack's datasets are eval groups, in either file form: the
2635
+ # group's (version 7) or the dataset's an older pack holds.
2636
+ fmt = doc.get("format") if isinstance(doc, dict) else None
2637
+ entry = doc.get(EXPORT_KEYS[fmt]) if fmt in (EXPORT_ONE, DATASET_ONE) else None
2638
+ if (not isinstance(entry, dict) or type(doc.get("version")) is not int
2639
+ or doc["version"] not in ((EXPORT_VERSION,) if fmt == EXPORT_ONE else IMPORT_VERSIONS)):
2640
+ return None, f"{name} is not an eval group export the lab reads"
2641
+ ds_name, why = dataset_name(entry.get("name"))
2346
2642
  if why:
2347
2643
  return None, f"{name}: {why}"
2348
- raw = doc["dataset"].get("body")
2644
+ raw = entry.get("body")
2349
2645
  body = upgrade_body(raw) if doc["version"] < EXPORT_VERSION else raw
2350
2646
  why = dataset_problem(body)
2351
2647
  if why:
@@ -2524,11 +2820,12 @@ class Packs:
2524
2820
  # the page upgrades the pipeline as it reads it.
2525
2821
  evals = doc.get("evals", doc.get("tests"))
2526
2822
  for t in evals if isinstance(evals, list) else [evals]:
2527
- ref = t.get("dataset") if isinstance(t, dict) else None
2528
- if isinstance(ref, dict):
2823
+ ref = eval_group_ref(t)
2824
+ if ref is not None:
2529
2825
  did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
2530
2826
  if did:
2531
- t["dataset"] = {"id": did, "name": DATASETS.get(did)["name"]}
2827
+ ref.clear()
2828
+ ref.update({"id": did, "name": DATASETS.get(did)["name"]})
2532
2829
  content = content_of(doc)
2533
2830
  if isinstance(content, dict) and isinstance(content.get("ref"), dict):
2534
2831
  sid = src_ids.get(content["ref"].get("id")) or src_ids.get(content["ref"].get("name"))
@@ -3251,13 +3548,15 @@ class Plugins:
3251
3548
  # is a marker the worker checks between items -- today's semantics, stopping
3252
3549
  # after the item in flight, never a process kill that loses its reply.
3253
3550
 
3254
- # A profile's key, as run-evals.js reads it: EVAL_API_KEY_<ID>, the id a run
3255
- # document keys its profiles table by, in the shell-safe spelling of itself.
3256
- # The id and not the name, so two profiles can share a name without sharing a
3257
- # key (docs/pipeline-model.md §5). One spelling on both sides -- evals-core.ts
3258
- # keyVar -- or the worker would never find the key the page never sent.
3259
- def key_var(profile_id: str) -> str:
3260
- return "EVAL_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(profile_id).upper()).strip("_")
3551
+ # A profile's key, as run-evals.js reads it: EVALSLAB_API_KEY_<SLUG>, the slug the
3552
+ # run document's connection carries (#253), or EVALSLAB_API_KEY_<ID> for one from
3553
+ # before slugs -- the id it keys its profiles table by -- in the shell-safe
3554
+ # spelling of itself. Never the name, so two profiles can share a name
3555
+ # without sharing a key (docs/pipeline-model.md §5). One spelling on both
3556
+ # sides -- evals-core.ts keyVar -- or the worker would never find the key the
3557
+ # page never sent.
3558
+ def key_var(profile_id: str, slug=None) -> str:
3559
+ return "EVALSLAB_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(slug or profile_id).upper()).strip("_")
3261
3560
 
3262
3561
 
3263
3562
  def run_items(run):
@@ -3331,7 +3630,33 @@ def brief_row(row):
3331
3630
  if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
3332
3631
  return it
3333
3632
  return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
3334
- return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
3633
+ # The run's verdict stays; each Target's eval by eval is the whole row's.
3634
+ brief = {k: v for k, v in row.items() if k != "verdicts"}
3635
+ return {**brief, "results": [item(it) for it in row.get("results") or []], "brief": True}
3636
+
3637
+
3638
+ # What of a worker's report a row keeps as its verdicts (#252): the run's --
3639
+ # pass, fail or incomplete, which the worker's exit code already said -- and
3640
+ # each Target's evals, by the eval's id, as `run.verdicts` names them. Only
3641
+ # these fields are carried, so nothing else the report holds lands in the row.
3642
+ # A status says whether the run finished; its verdict says whether it passed,
3643
+ # so a run that failed an eval is `done` with the verdict `fail`.
3644
+ VERDICTS = ("pass", "fail", "incomplete")
3645
+ VERDICT_FIELDS = ("name", "skipped", "verdict", "pass", "detail", "ran", "passed", "skippedItems")
3646
+
3647
+
3648
+ def report_verdicts(report):
3649
+ """(verdict, verdicts as JSON) from a worker's report, or (None, None)
3650
+ for one that carries none -- a worker from before #251."""
3651
+ verdict = report.get("verdict") if isinstance(report, dict) else None
3652
+ run = report.get("run") if verdict in VERDICTS else None
3653
+ targets = run.get("verdicts") if isinstance(run, dict) else None
3654
+ if not isinstance(targets, list):
3655
+ return None, None
3656
+ kept = [{eid: {k: e[k] for k in VERDICT_FIELDS if k in e}
3657
+ for eid, e in t.items() if isinstance(e, dict)}
3658
+ if isinstance(t, dict) else {} for t in targets]
3659
+ return verdict, json.dumps(kept)
3335
3660
 
3336
3661
 
3337
3662
  class Queue:
@@ -3368,13 +3693,27 @@ class Queue:
3368
3693
  # one of its own.
3369
3694
  if "rerun_of" not in cols:
3370
3695
  db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
3696
+ # The worker's verdicts (#252); a row from before them has none,
3697
+ # and is never given one after the fact.
3698
+ if "verdict" not in cols:
3699
+ db.execute("ALTER TABLE queue ADD COLUMN verdict TEXT")
3700
+ if "verdicts" not in cols:
3701
+ db.execute("ALTER TABLE queue ADD COLUMN verdicts TEXT")
3702
+ # The body of each eval group a run grades with, by `<id>@<n>`
3703
+ # (docs/pipeline-model.md §17). A row from before keeps its
3704
+ # `dataset` column, read as its one group.
3705
+ if "groups" not in cols:
3706
+ db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
3371
3707
 
3372
3708
  # ---- rows -----------------------------------------------------------
3373
3709
 
3374
3710
  # A row read with the submit time of the run it re-runs, if any: the
3375
3711
  # page names a run by that time (History's Run ID), so a "Re-run of"
3376
- # note reads without fetching the original.
3377
- SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
3712
+ # note reads without fetching the original. Its columns are named, since
3713
+ # the order ALTER TABLE added them in is no order _row can count on.
3714
+ SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
3715
+ "q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
3716
+ "q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
3378
3717
  "LEFT JOIN queue o ON o.id = q.rerun_of")
3379
3718
 
3380
3719
  @staticmethod
@@ -3387,8 +3726,9 @@ class Queue:
3387
3726
  "snapshot": json.loads(r[6]), "results": json.loads(r[7]),
3388
3727
  "progress": json.loads(r[8]), "totals": json.loads(r[9]),
3389
3728
  "error": r[10],
3390
- "rerunOf": r[12] if len(r) > 12 else None,
3391
- "rerunOfAt": r[13] if len(r) > 13 else None,
3729
+ "rerunOf": r[11], "rerunOfAt": r[14],
3730
+ "verdict": r[12],
3731
+ "verdicts": json.loads(r[13]) if r[13] else None,
3392
3732
  }
3393
3733
 
3394
3734
  # A row from before run documents has no version, and nothing here can
@@ -3411,15 +3751,19 @@ class Queue:
3411
3751
  (rid,)).fetchone())
3412
3752
  return row if self._readable(row) else None
3413
3753
 
3414
- def list(self, limit=RUNS_PAGE, before=None, full=False):
3415
- """Runs, newest first, and whether more follow. `before` is a
3416
- `submittedAt` the page of runs stops at, so History can page through
3417
- them the way it pages the runs store. Each row is brief_row's unless
3418
- `full` asks for the whole of it."""
3754
+ def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
3755
+ """Runs, newest first, and whether more follow. `before` and
3756
+ `before_id` are the `submittedAt` and id of the last run on the page
3757
+ before, so History can page through them the way it pages the runs
3758
+ store. `submittedAt` is to the second, so two runs can share one; the
3759
+ id breaks the tie, or a page ending between them would skip the
3760
+ second (#238). `before` alone stops at the second. Each row is
3761
+ brief_row's unless `full` asks for the whole of it."""
3419
3762
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3420
3763
  rows = self._all(db)
3421
- rows = [r for r in rows if before is None or r["submittedAt"] < before]
3422
- rows.sort(key=lambda r: r["submittedAt"], reverse=True)
3764
+ key = lambda r: (r["submittedAt"], r["id"])
3765
+ rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
3766
+ rows.sort(key=key, reverse=True)
3423
3767
  page = rows[:limit]
3424
3768
  return (page if full else [brief_row(r) for r in page]), len(rows) > limit
3425
3769
 
@@ -3430,27 +3774,30 @@ class Queue:
3430
3774
 
3431
3775
  # ---- submit ---------------------------------------------------------
3432
3776
 
3433
- def submit(self, run: dict, dataset=None, rerun_of=None):
3777
+ def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
3434
3778
  """
3435
3779
  A new queued run. `run` is the run document (docs/pipeline-model.md
3436
3780
  §5): the pipeline, the profiles it resolved to without their keys, its
3437
- content's file list in order and with its repeats, and the dataset's
3438
- version; `dataset` is that version's body, for a graded run;
3439
- `rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
3440
- every scenario -- so the total is the list's length, repeats and all,
3441
- the same count the runner and the page make.
3781
+ content's file list in order and with its repeats, and each eval
3782
+ group's version; `groups` is those versions' bodies, by `<id>@<n>`
3783
+ (§17), and `dataset` the one body a row from before them kept, which a
3784
+ re-run of one carries on; `rerun_of` is the run a re-run was queued
3785
+ from. Returns the row. Its items are that list, or the one inline
3786
+ text, each through every scenario -- so the total is the list's
3787
+ length, repeats and all, the same count the runner and the page make.
3442
3788
  """
3443
3789
  rid = secrets.token_hex(6)
3444
3790
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
3445
3791
  total = len(run_items(run))
3446
3792
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3447
3793
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
3448
- "snapshot, results, progress, totals, dataset, rerun_of) "
3449
- "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
3794
+ "snapshot, results, progress, totals, dataset, rerun_of, groups) "
3795
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
3450
3796
  (rid, "queued", now, json.dumps(run), "[]",
3451
3797
  json.dumps({"current": None, "n": 0, "total": total}),
3452
3798
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
3453
- None if dataset is None else json.dumps(dataset), rerun_of))
3799
+ None if dataset is None else json.dumps(dataset), rerun_of,
3800
+ None if groups is None else json.dumps(groups)))
3454
3801
  # Its prompts' uses, in the same transaction: a run is in the
3455
3802
  # library the moment it is queued, or not queued at all.
3456
3803
  if self.prompts is not None:
@@ -3460,19 +3807,43 @@ class Queue:
3460
3807
  # the moment the lock is let go.
3461
3808
  return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
3462
3809
 
3810
+ def groups(self, rid, raw=False):
3811
+ """The eval group bodies a readable run grades with, by `<id>@<n>`, or
3812
+ None for a run that is not there or kept none -- what its verdicts
3813
+ were graded by, whatever the groups hold now. A row from before runs
3814
+ kept a body per group answers its one `dataset` copy under its
3815
+ group's key. A copy kept at an earlier version reads as one of
3816
+ today's, unless [raw]: the worker is handed it as kept, since an
3817
+ earlier run's pipeline is upgraded under the rules that copy holds."""
3818
+ run = self.get(rid)
3819
+ if run is None:
3820
+ return None
3821
+ kept = self._kept(rid)
3822
+ if kept is None:
3823
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3824
+ old = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()[0]
3825
+ if not old:
3826
+ return None
3827
+ kept = {group_key(evals_dataset(run["snapshot"])): json.loads(old)}
3828
+ return kept if raw else {k: upgrade_body(b) for k, b in kept.items()}
3829
+
3830
+ def _kept(self, rid):
3831
+ """The row's own `groups` column, or None for a row from before it."""
3832
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3833
+ r = db.execute("SELECT groups FROM queue WHERE id = ?", (rid,)).fetchone()
3834
+ return json.loads(r[0]) if r and r[0] is not None else None
3835
+
3463
3836
  def dataset(self, rid, raw=False):
3464
- """The dataset body a readable run was submitted against, or None --
3465
- what its verdicts were graded by, whatever the dataset holds now. A
3466
- copy kept at an earlier version reads as one of today's, unless [raw]:
3467
- the worker is handed it as kept, since an earlier run's pipeline is
3468
- upgraded under the rules that copy holds."""
3837
+ """The body of the one Library group a readable run grades its cases
3838
+ against (evals_dataset), or None: the body the worker is handed as
3839
+ --dataset, and what the page reads a run's cases from."""
3469
3840
  if self.get(rid) is None:
3470
3841
  return None
3471
- with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3472
- r = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()
3473
- if not (r and r[0]):
3842
+ kept = self.groups(rid, raw=True)
3843
+ ref = evals_dataset(self.get(rid)["snapshot"])
3844
+ body = kept.get(group_key(ref)) if kept and ref is not None else None
3845
+ if body is None:
3474
3846
  return None
3475
- body = json.loads(r[0])
3476
3847
  return body if raw else upgrade_body(body)
3477
3848
 
3478
3849
  def _plugin_args(self, run):
@@ -3487,14 +3858,18 @@ class Queue:
3487
3858
  return ["--plugins", str(PLUGINS.dir)], None
3488
3859
 
3489
3860
  def _dataset_args(self, run, rundir):
3490
- """The worker's --dataset for a graded run: the body kept with the row,
3491
- written beside the run document. A row queued before runs kept their
3492
- dataset has none, and is pinned to the dataset as it reads now, once,
3493
- so every later pass over it agrees. Returns (args, None) or (None, why)."""
3861
+ """The worker's --dataset for a graded run: the body of its one
3862
+ Library group kept with the row, written beside the run document --
3863
+ one, until the worker reads a body per group (#233). A row queued
3864
+ before runs kept their dataset has none, and is pinned to the dataset
3865
+ as it reads now, once, so every later pass over it agrees. Returns
3866
+ (args, None) or (None, why)."""
3494
3867
  ref = evals_dataset(run["snapshot"])
3495
3868
  if ref is None:
3496
3869
  return [], None
3497
3870
  body = self.dataset(run["id"], raw=True)
3871
+ if body is None and self._kept(run["id"]) is not None:
3872
+ return None, f"the run kept no body of the eval group {ref.get('name') or ref.get('id')!r}"
3498
3873
  if body is None:
3499
3874
  snap = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
3500
3875
  if snap is None:
@@ -3510,6 +3885,27 @@ class Queue:
3510
3885
  (rundir / "dataset.json").write_text(json.dumps(body))
3511
3886
  return ["--dataset", str(rundir / "dataset.json")], None
3512
3887
 
3888
+ def _grading_args(self, run, rundir):
3889
+ """The worker's eval group bodies: --dataset for a run that grades
3890
+ against one group, which keeps its single-body path and old-row
3891
+ pinning, and --groups for a run that links several (#233) -- every body
3892
+ it kept, by `<id>@<n>`, written beside the run document. Returns
3893
+ (args, None) or (None, why)."""
3894
+ refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
3895
+ keys = {group_key(r) for r in refs if r is not None}
3896
+ if len(keys) <= 1:
3897
+ return self._dataset_args(run, rundir)
3898
+ kept = self.groups(run["id"], raw=True)
3899
+ if kept is None:
3900
+ return None, "the run kept no eval group bodies"
3901
+ missing = sorted({(r.get("name") or r.get("id")) for r in refs
3902
+ if r is not None and group_key(r) not in kept})
3903
+ if missing:
3904
+ return None, f"the run kept no body of the eval group {', '.join(missing)}"
3905
+ rundir.mkdir(parents=True, exist_ok=True)
3906
+ (rundir / "groups.json").write_text(json.dumps(kept))
3907
+ return ["--groups", str(rundir / "groups.json")], None
3908
+
3513
3909
  def _behind(self, rid):
3514
3910
  """How many submissions stand between this one and the worker, by
3515
3911
  submit time -- what a waiting form names when it says what it is
@@ -3547,7 +3943,8 @@ class Queue:
3547
3943
  if run["status"] not in ("cancelled", "interrupted", "incomplete"):
3548
3944
  return None, (409, "only a cancelled, interrupted or incomplete run can be resumed")
3549
3945
  (self.dir / rid / "cancel").unlink(missing_ok=True)
3550
- self._set(rid, status="queued", cancel=0, error=None)
3946
+ # What it reached is decided by the run it goes on to finish.
3947
+ self._set(rid, status="queued", cancel=0, error=None, verdict=None, verdicts=None)
3551
3948
  return self.get(rid), None
3552
3949
 
3553
3950
  def set_comment(self, rid, comment):
@@ -3597,9 +3994,12 @@ class Queue:
3597
3994
  _, err = self._pinned_files(snap)
3598
3995
  if err:
3599
3996
  return None, (409, err)
3997
+ # The group bodies the original kept: a re-run grades with exactly
3998
+ # them, whatever the groups or the pins read now.
3999
+ kept = self._kept(rid)
3600
4000
  ref = evals_dataset(snap)
3601
4001
  body = None
3602
- if ref is not None:
4002
+ if ref is not None and kept is None:
3603
4003
  body = self.dataset(rid, raw=True)
3604
4004
  if body is None:
3605
4005
  # A run that never started kept no body: the dataset's, if it
@@ -3615,7 +4015,7 @@ class Queue:
3615
4015
  _, err = worker_destinations(snap)
3616
4016
  if err:
3617
4017
  return None, (403, err)
3618
- return self.submit(snap, body, rerun_of=rid), None
4018
+ return self.submit(snap, body, rerun_of=rid, groups=kept), None
3619
4019
 
3620
4020
  def rerun_item(self, rid, index):
3621
4021
  """
@@ -3644,7 +4044,7 @@ class Queue:
3644
4044
  return None, (403, err)
3645
4045
  if NODE is None:
3646
4046
  return None, (500, "node is not installed, so nothing can run")
3647
- dataset, err = self._dataset_args(run, rundir)
4047
+ dataset, err = self._grading_args(run, rundir)
3648
4048
  if err:
3649
4049
  return None, (409, err)
3650
4050
  plugins, err = self._plugin_args(run)
@@ -3677,9 +4077,12 @@ class Queue:
3677
4077
  while len(results) <= index:
3678
4078
  results.append(None)
3679
4079
  results[index] = next((it for it in items if it and it.get("item") == index), items[0])
4080
+ # The worker read every item to reach its verdicts, so they are the
4081
+ # row's as it now stands.
4082
+ verdict, verdicts = report_verdicts(report)
3680
4083
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3681
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3682
- (json.dumps(results), None, run["id"]))
4084
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4085
+ "WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
3683
4086
  return self.get(run["id"]), None
3684
4087
 
3685
4088
  def rescore_item(self, rid, index):
@@ -3707,7 +4110,7 @@ class Queue:
3707
4110
  return None, (500, "node is not installed, so nothing can run")
3708
4111
  rundir = self.dir / run["id"]
3709
4112
  rundir.mkdir(parents=True, exist_ok=True)
3710
- dataset, err = self._dataset_args(run, rundir)
4113
+ dataset, err = self._grading_args(run, rundir)
3711
4114
  if err:
3712
4115
  return None, (409, err)
3713
4116
  plugins, err = self._plugin_args(run)
@@ -3742,9 +4145,12 @@ class Queue:
3742
4145
  while len(results) <= index:
3743
4146
  results.append(None)
3744
4147
  results[index] = next((it for it in items if it and it.get("item") == index), items[0])
4148
+ # The worker read every item to reach its verdicts, so they are the
4149
+ # row's as it now stands.
4150
+ verdict, verdicts = report_verdicts(report)
3745
4151
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3746
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3747
- (json.dumps(results), None, run["id"]))
4152
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4153
+ "WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
3748
4154
  return self.get(run["id"]), None
3749
4155
 
3750
4156
  # ---- the run --------------------------------------------------------
@@ -3820,7 +4226,7 @@ class Queue:
3820
4226
  return self._finish(rid, "failed", error=err)
3821
4227
  if NODE is None:
3822
4228
  return self._finish(rid, "failed", error="node is not installed, so nothing can run")
3823
- dataset, err = self._dataset_args(run, rundir)
4229
+ dataset, err = self._grading_args(run, rundir)
3824
4230
  if err:
3825
4231
  return self._finish(rid, "failed", error=err)
3826
4232
  plugins, err = self._plugin_args(run)
@@ -3855,10 +4261,11 @@ class Queue:
3855
4261
  report = self._read_report(rid)
3856
4262
  items = report["run"]["items"] if report and isinstance(report.get("run"), dict) else None
3857
4263
  if isinstance(items, list):
4264
+ verdict, verdicts = report_verdicts(report)
3858
4265
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3859
4266
  if db.execute("SELECT 1 FROM queue WHERE id = ?", (rid,)).fetchone() is not None:
3860
- db.execute("UPDATE queue SET results = ?, error = ? WHERE id = ?",
3861
- (json.dumps(items), None, rid))
4267
+ db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
4268
+ "WHERE id = ?", (json.dumps(items), None, verdict, verdicts, rid))
3862
4269
  run = self.get(rid)
3863
4270
  # The watchdog may have failed the run while it was on the wire; a
3864
4271
  # row that already left `running` is not this worker's to re-label.
@@ -3968,8 +4375,8 @@ class Queue:
3968
4375
  """The oldest queued run this server can read. One it cannot is never
3969
4376
  started: nothing here would know what it asks for."""
3970
4377
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3971
- for r in db.execute("SELECT * FROM queue WHERE status = 'queued' "
3972
- "ORDER BY submitted_at, rowid").fetchall():
4378
+ for r in db.execute(self.SELECT + " WHERE q.status = 'queued' "
4379
+ "ORDER BY q.submitted_at, q.rowid").fetchall():
3973
4380
  if not self._readable(self._row(r)):
3974
4381
  continue
3975
4382
  db.execute("UPDATE queue SET status = 'running', started_at = ? "
@@ -4015,10 +4422,89 @@ class Queue:
4015
4422
  proc.kill()
4016
4423
 
4017
4424
 
4425
+ # The key a relayed request is sent with: the one it carries, or -- a page
4426
+ # that was handed KEY_HELD in place of a profile's key -- the one the store
4427
+ # holds for that profile.
4428
+ def relay_key(payload: dict) -> str:
4429
+ key = str(payload.get("key") or "").strip()
4430
+ if held(key):
4431
+ return STORE.held_key(key).strip() if STORE is not None else ""
4432
+ return key
4433
+
4434
+
4018
4435
  # The worker-destination rules, at submit and again at dequeue: a run's
4019
4436
  # profiles take their keys from the profiles store, by id, so the request
4020
4437
  # carries none, and the relay's rules bind where they go. Returns (env_vars,
4021
4438
  # None) or (None, a named refusal).
4439
+ # Export for CI (#254): a pipeline as a bundle a repository keeps and
4440
+ # `evals-lab run` runs with no lab. The page writes its text -- the core's
4441
+ # exportBundle: pipeline.yaml, profiles.yaml by slug, datasets/<slug>.json --
4442
+ # and the server adds what only it holds: every installed plugin, as a run
4443
+ # stamps them all, and, when asked, the Source's files under items/. A key is
4444
+ # refused rather than zipped: no Setup key, no token, no field named like
4445
+ # one, in any of it (the core's bundleProblems asks the same of the page's).
4446
+ BUNDLE_TEXT = re.compile(r"pipeline\.yaml|profiles\.yaml|datasets/[a-z0-9]+(?:-[a-z0-9]+)*\.json")
4447
+ BUNDLE_KEY_FIELD = re.compile(
4448
+ r'^[\s-]*"?((?:api[-_]?)?key|authorization|bearer|secret|password|token)"?\s*:', re.I | re.M)
4449
+
4450
+
4451
+ def bundle_problems(files, keys=()):
4452
+ """Why [files] -- path to text -- cannot leave the lab, or ""."""
4453
+ for path, text in files.items():
4454
+ for key in keys:
4455
+ key = str(key or "").strip()
4456
+ if len(key) >= 4 and key in text:
4457
+ return f"{path} holds a Target profile's key, and a key never leaves the lab"
4458
+ if TOKEN_SHAPE.search(text):
4459
+ return f"{path} holds a token, and a key never leaves the lab"
4460
+ m = BUNDLE_KEY_FIELD.search(text)
4461
+ if m:
4462
+ return f"{path} holds a field named {m.group(1)}, and a key never leaves the lab"
4463
+ return ""
4464
+
4465
+
4466
+ def build_bundle(payload):
4467
+ """The zip Export for CI downloads, from the page's [payload]:
4468
+ { files: {path: text}, source: id or null, items: bool }. Returns
4469
+ (bytes, None) or (None, (status, one sentence))."""
4470
+ files = payload.get("files")
4471
+ if not isinstance(files, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in files.items()):
4472
+ return None, (400, "a bundle's files are text, by path")
4473
+ for path in files:
4474
+ if not BUNDLE_TEXT.fullmatch(path):
4475
+ return None, (400, f"{path!r} is not a file a bundle holds")
4476
+ if "pipeline.yaml" not in files or "profiles.yaml" not in files:
4477
+ return None, (400, "a bundle holds pipeline.yaml and profiles.yaml")
4478
+ stored = []
4479
+ if STORE is not None:
4480
+ stored = ((STORE.all().get("promptlab.profiles") or {}).get("body") or {}).get("list") or []
4481
+ why = bundle_problems(files, [p.get("key") for p in stored if isinstance(p, dict)])
4482
+ if why:
4483
+ return None, (400, why)
4484
+ items = []
4485
+ if payload.get("items"):
4486
+ sid = payload.get("source")
4487
+ found = SOURCES.item_paths(sid) if isinstance(sid, str) and SOURCES is not None else None
4488
+ if found is None:
4489
+ return None, (404, "no such source")
4490
+ items = found
4491
+ buf = io.BytesIO()
4492
+ with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
4493
+ for path, text in sorted(files.items()):
4494
+ zf.writestr(f"evals/{path}", text)
4495
+ if PLUGINS is not None:
4496
+ for p in PLUGINS.stamp():
4497
+ root = PLUGINS.dir / p["id"] / p["version"]
4498
+ for f in sorted(root.rglob("*")):
4499
+ if f.is_file():
4500
+ zf.write(f, f"evals/plugins/{p['id']}/{p['version']}/{f.relative_to(root).as_posix()}")
4501
+ for name, path in items:
4502
+ if not path.is_file():
4503
+ return None, (404, f"{name!r} is not in that Source")
4504
+ zf.write(path, f"evals/items/{name}")
4505
+ return buf.getvalue(), None
4506
+
4507
+
4022
4508
  def worker_destinations(run: dict):
4023
4509
  table = run.get("profiles") if isinstance(run, dict) else None
4024
4510
  if not isinstance(table, dict):
@@ -4046,7 +4532,7 @@ def worker_destinations(run: dict):
4046
4532
  why = allowed(base, key)
4047
4533
  if why:
4048
4534
  return None, f"Target profile {name}: {why}"
4049
- env[key_var(pid)] = key
4535
+ env[key_var(pid, conn.get("slug"))] = key
4050
4536
  return env, None
4051
4537
 
4052
4538
 
@@ -4068,15 +4554,17 @@ def worker_destinations(run: dict):
4068
4554
  # step in each job is what it sends there (docs/pipeline-model.md §16).
4069
4555
  # 11: `tests` are `evals`; nothing in an eval changes.
4070
4556
  # 12: a Contains metric's Ignore case holds item by item too, kept as written.
4071
- PIPELINE_VERSION = 12
4557
+ # 13: an eval is a link to an eval group, or a group of the pipeline's own,
4558
+ # and the document has an overall pass rule (docs/pipeline-model.md §17).
4559
+ PIPELINE_VERSION = 13
4072
4560
  # What a stored run may be: the current version, and the ones evals-core.ts's
4073
4561
  # upgradePipeline reads. A new submission is upgraded to the current one.
4074
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
4562
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
4075
4563
  TARGET_CAP = 4
4076
4564
  # Target steps whose words the Prompt library does not record as a use: they
4077
4565
  # ask no model (evals-core.ts's STEP_TYPES.echo).
4078
4566
  UNRECORDED_STEPS = {"echo"}
4079
- RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
4567
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "pass", "profiles", "comment", "plugins")
4080
4568
 
4081
4569
 
4082
4570
  def content_of(doc):
@@ -4122,14 +4610,17 @@ BUILTIN_CONNECTION_TYPES = dict(CONNECTION_TYPES)
4122
4610
  BUILTIN_LOCAL = set(LOCAL_CONNECTIONS)
4123
4611
  BUILTIN_CHAT_PATHS = dict(CONNECTION_CHAT_PATHS)
4124
4612
  BUILTIN_AUTH = dict(CONNECTION_AUTH)
4125
- CONNECTION_FIELDS = ("name", "url", "model", "type", "temperature", "px", "format",
4613
+ CONNECTION_FIELDS = ("name", "slug", "url", "model", "type", "temperature", "px", "format",
4126
4614
  "quality", "options")
4127
4615
  LOOKS_LIKE_A_KEY = re.compile(r"^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$",
4128
4616
  re.IGNORECASE)
4129
4617
  PROFILE_ID = re.compile(r"[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?")
4618
+ # evals-core.ts's SLUG and SLUG_MAX: what a key's variable is spelt from.
4619
+ PROFILE_SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*")
4620
+ SLUG_MAX = 32
4130
4621
 
4131
4622
 
4132
- def _fields(obj, at, allowed_fields, bad, key_from="EVAL_API_KEY_<ID>"):
4623
+ def _fields(obj, at, allowed_fields, bad, key_from="EVALSLAB_API_KEY_<ID>"):
4133
4624
  for k in obj:
4134
4625
  if k in allowed_fields:
4135
4626
  continue
@@ -4200,7 +4691,12 @@ CONNECTION_LISTS = {"http": http_endpoint_test}
4200
4691
 
4201
4692
 
4202
4693
  def connection_problems(pid, conn, at, bad):
4203
- _fields(conn, at, CONNECTION_FIELDS, bad, key_var(pid))
4694
+ slug = conn.get("slug")
4695
+ if slug is not None and not (isinstance(slug, str) and PROFILE_SLUG.fullmatch(slug) and len(slug) <= SLUG_MAX):
4696
+ bad.append(f"{at}: slug has to be lowercase letters and digits, with hyphens between, at most {SLUG_MAX}")
4697
+ slug = None
4698
+ var = key_var(pid, slug)
4699
+ _fields(conn, at, CONNECTION_FIELDS, bad, var)
4204
4700
  ctype = conn.get("type")
4205
4701
  if not isinstance(ctype, str) or ctype not in CONNECTION_TYPES:
4206
4702
  bad.append(f"{at}: type has to be one of {', '.join(CONNECTION_TYPES)}")
@@ -4211,7 +4707,7 @@ def connection_problems(pid, conn, at, bad):
4211
4707
  # Which settings there are is the type's to say; a key among them
4212
4708
  # is the server's.
4213
4709
  allowed = list(CONNECTION_TYPES.get(ctype, ())) if ctype else list(options)
4214
- _fields(options, f"{at}'s options", allowed, bad, key_var(pid))
4710
+ _fields(options, f"{at}'s options", allowed, bad, var)
4215
4711
  else:
4216
4712
  bad.append(f"{at}: options has to be an object")
4217
4713
  url = conn.get("url") or ""
@@ -4222,7 +4718,7 @@ def connection_problems(pid, conn, at, bad):
4222
4718
  parts = urllib.parse.urlsplit(url if "://" in url else "http://" + url)
4223
4719
  if parts.username or parts.password:
4224
4720
  bad.append(f"{at} has a key in its address, and a key never goes in a pipeline "
4225
- f"— ${key_var(pid)} supplies it")
4721
+ f"— ${var} supplies it")
4226
4722
  for why in CONNECTION_CHECKS.get(ctype, lambda _c: [])(conn):
4227
4723
  bad.append(f"{at} {why}")
4228
4724
  # llama.cpp runs its own llama-server: a hosted address is refused, as the
@@ -4233,16 +4729,33 @@ def connection_problems(pid, conn, at, bad):
4233
4729
  bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
4234
4730
 
4235
4731
 
4732
+ def eval_group_ref(t):
4733
+ """The Library group an eval reads the cases of, as the dict itself (so a
4734
+ caller may stamp it in place), or None: version 13's link (`group`) or a
4735
+ private group's `casesFrom`, or an earlier version's `dataset`
4736
+ (evals-core.ts casesRef)."""
4737
+ if not isinstance(t, dict):
4738
+ return None
4739
+ if t.get("type") == "group":
4740
+ if isinstance(t.get("group"), dict):
4741
+ return t["group"]
4742
+ own = t.get("own")
4743
+ return own["casesFrom"] if isinstance(own, dict) and isinstance(own.get("casesFrom"), dict) else None
4744
+ return t["dataset"] if isinstance(t.get("dataset"), dict) else None
4745
+
4746
+
4236
4747
  def evals_dataset(doc):
4237
- """The dataset reference a document's evals grade against, or None: the
4238
- first eval that names one. A run grades against one dataset (the core's
4748
+ """The Library group a document's evals read the cases of, or None: the
4749
+ first eval that names one. A run grades against one (the core's
4239
4750
  validatePipeline says so). Reads a stored document of any shape --
4240
- version 11's `evals`, the `tests` before it, version 5's list or the one
4241
- test before that -- since rows keep the document they were submitted with."""
4751
+ version 13's links, version 11's `evals`, the `tests` before it, version
4752
+ 5's list or the one test before that -- since rows keep the document they
4753
+ were submitted with."""
4242
4754
  evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
4243
4755
  for t in evals if isinstance(evals, list) else [evals]:
4244
- if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
4245
- return t["dataset"]
4756
+ ref = eval_group_ref(t)
4757
+ if ref is not None:
4758
+ return ref
4246
4759
  return None
4247
4760
 
4248
4761
 
@@ -4325,6 +4838,7 @@ def run_problems(run):
4325
4838
  table = run.get("profiles")
4326
4839
  if not isinstance(table, dict):
4327
4840
  return bad + ["profiles has to be an object of id → connection"]
4841
+ spelt = {}
4328
4842
  for pid, conn in table.items():
4329
4843
  at = f"profile {pid}"
4330
4844
  if not PROFILE_ID.fullmatch(str(pid)):
@@ -4333,6 +4847,11 @@ def run_problems(run):
4333
4847
  if not isinstance(conn, dict):
4334
4848
  bad.append(f"{at} has to be an object")
4335
4849
  continue
4850
+ # Two profiles spelling one variable would hand one the other's key.
4851
+ var = key_var(pid, conn.get("slug") if isinstance(conn.get("slug"), str) else None)
4852
+ if var in spelt:
4853
+ bad.append(f"profiles {spelt[var]} and {pid} both take their key from ${var}")
4854
+ spelt[var] = pid
4336
4855
  connection_problems(pid, conn, at, bad)
4337
4856
  targets = run.get("targets")
4338
4857
  if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
@@ -4550,9 +5069,23 @@ class Handler(BaseHTTPRequestHandler):
4550
5069
 
4551
5070
  # ---- routes ---------------------------------------------------------
4552
5071
 
5072
+ # The eval groups' routes (docs/pipeline-model.md §17) are the datasets'
5073
+ # under their new name: /api/datasets goes on answering the same rows, so
5074
+ # dataset-diff.js and a script written against it keep working. A list
5075
+ # asked for by the new name is `groups`.
5076
+ GROUPS_ROUTE = "/api/eval-groups"
5077
+ grouped = False
5078
+
5079
+ def _alias(self):
5080
+ self.grouped = self.path == self.GROUPS_ROUTE or self.path.startswith((self.GROUPS_ROUTE + "/",
5081
+ self.GROUPS_ROUTE + "?"))
5082
+ if self.grouped:
5083
+ self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
5084
+
4553
5085
  def do_GET(self):
4554
5086
  if not self._authorised():
4555
5087
  return
5088
+ self._alias()
4556
5089
  path = self.path.split("?", 1)[0]
4557
5090
  # The lab is one page: a Connection and an Input make a scenario,
4558
5091
  # Content and Evals are shared, and one to four scenarios run over the
@@ -4599,8 +5132,16 @@ class Handler(BaseHTTPRequestHandler):
4599
5132
  except ValueError:
4600
5133
  return self._json(400, {"error": "limit has to be a number"})
4601
5134
  before = (query.get("before") or [None])[0]
4602
- runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
5135
+ before_id = (query.get("beforeId") or [None])[0]
5136
+ runs, more = QUEUE.list(limit, before, before_id, full=(query.get("full") or [""])[0] == "1")
4603
5137
  return self._json(200, {"runs": runs, "more": more})
5138
+ if path.startswith("/api/queue/") and path.endswith("/groups") and path.count("/") == 4:
5139
+ # The bodies of the eval groups a run grades with, by `<id>@<n>`
5140
+ # (§17); null for a run that kept none, as /dataset answers.
5141
+ run_id = path.split("/")[3]
5142
+ if QUEUE is None or QUEUE.get(run_id) is None:
5143
+ return self._send(404, b"not found", "text/plain")
5144
+ return self._json(200, QUEUE.groups(run_id))
4604
5145
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
4605
5146
  # The dataset body a graded run was submitted against, which is
4606
5147
  # what its verdicts were graded by; the list never carries it.
@@ -4762,6 +5303,7 @@ class Handler(BaseHTTPRequestHandler):
4762
5303
  def do_DELETE(self):
4763
5304
  if not self._authorised() or not self._from_this_page():
4764
5305
  return
5306
+ self._alias()
4765
5307
  path = self.path.split("?", 1)[0]
4766
5308
  if path == "/api/connections/google":
4767
5309
  if CONNECTIONS is None:
@@ -4829,6 +5371,7 @@ class Handler(BaseHTTPRequestHandler):
4829
5371
  def do_PATCH(self):
4830
5372
  if not self._authorised() or not self._from_this_page():
4831
5373
  return
5374
+ self._alias()
4832
5375
  parts = self.path.split("?", 1)[0].split("/")
4833
5376
  if len(parts) != 4 or parts[1] != "api":
4834
5377
  return self._send(404, b"not found", "text/plain")
@@ -4859,6 +5402,7 @@ class Handler(BaseHTTPRequestHandler):
4859
5402
  def do_PUT(self):
4860
5403
  if not self._authorised() or not self._from_this_page():
4861
5404
  return
5405
+ self._alias()
4862
5406
  parts = self.path.split("?", 1)[0].split("/")
4863
5407
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
4864
5408
  return self._prompts_put(parts[3])
@@ -4921,6 +5465,7 @@ class Handler(BaseHTTPRequestHandler):
4921
5465
  return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
4922
5466
 
4923
5467
  def _post(self):
5468
+ self._alias()
4924
5469
  path = self.path.split("?", 1)[0]
4925
5470
  if path == "/api/state":
4926
5471
  return self._state_write()
@@ -4940,6 +5485,12 @@ class Handler(BaseHTTPRequestHandler):
4940
5485
  return self._prompts_post(path)
4941
5486
  if path.startswith("/api/sources"):
4942
5487
  return self._sources_post(path)
5488
+ if path == "/api/bundle":
5489
+ data, err = build_bundle(self._payload() or {})
5490
+ if err:
5491
+ return self._json(err[0], {"error": err[1]})
5492
+ return self._send(200, data, "application/zip",
5493
+ (("Content-Disposition", 'attachment; filename="evals.zip"'),))
4943
5494
  payload = self._payload()
4944
5495
  if payload is None:
4945
5496
  return self._json(400, {"error": "bad body"})
@@ -5012,7 +5563,7 @@ class Handler(BaseHTTPRequestHandler):
5012
5563
 
5013
5564
  def _models_list(self, payload):
5014
5565
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
5015
- key = str(payload.get("key") or "").strip()
5566
+ key = relay_key(payload)
5016
5567
  ctype = str(payload.get("type") or "")
5017
5568
  # A type that lists some other way (an HTTP endpoint: its own test
5018
5569
  # path, key header and headers) says where and how.
@@ -5063,7 +5614,7 @@ class Handler(BaseHTTPRequestHandler):
5063
5614
  # and how the key travels, per the type -- the same mirror the models
5064
5615
  # list uses, so a client cannot point the relay at a path of its own.
5065
5616
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
5066
- key = str(payload.get("key") or "").strip()
5617
+ key = relay_key(payload)
5067
5618
  if not header_safe(key):
5068
5619
  return self._json(400, {"error": "That key has characters that "
5069
5620
  "cannot be sent in a header, so "
@@ -5099,6 +5650,10 @@ class Handler(BaseHTTPRequestHandler):
5099
5650
  for n, d in docs.items()})
5100
5651
  if stale is not None:
5101
5652
  return self._json(409, {"stale": stale})
5653
+ # A pin names a group's version, so the group keeps that version from
5654
+ # the moment a pipeline pins it (§17).
5655
+ if "promptlab.workflows" in docs and DATASETS is not None:
5656
+ DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
5102
5657
  return self._json(200, {"versions": versions})
5103
5658
 
5104
5659
  # ---- The run queue (#530) ----------------------------------------------
@@ -5137,20 +5692,28 @@ class Handler(BaseHTTPRequestHandler):
5137
5692
  # The plugins it runs under, as installed now: the server's to say,
5138
5693
  # whatever the document claimed.
5139
5694
  run["plugins"] = PLUGINS.stamp() if PLUGINS is not None else []
5140
- # A graded run keeps the body of the dataset it names, as it reads
5141
- # now, and records that body's fingerprint on the reference it
5142
- # belongs to: the worker grades against that and nothing else.
5143
- ref = evals_dataset(run)
5144
- body = None
5145
- if ref is not None:
5146
- snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
5147
- if snap is None:
5695
+ # Each Library group the evals read, resolved once (§17): a pinned
5696
+ # link to its pin, anything else to the group's newest version. The
5697
+ # reference records the version, `n`, and its body's fingerprint, and
5698
+ # the body is kept with the row under `<id>@<n>`: the worker grades
5699
+ # with that and nothing else, and the version is frozen from now on.
5700
+ groups = {}
5701
+ for t in run["evals"]:
5702
+ ref = eval_group_ref(t)
5703
+ if ref is None:
5704
+ continue
5705
+ pin = t.get("pin") if t.get("type") == "group" and t.get("group") is ref else None
5706
+ got = DATASETS.resolve(ref.get("id"), pin if type(pin) is int else None) if DATASETS else None
5707
+ if got is None:
5148
5708
  return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
5149
- body, version = snap
5150
- for t in run["evals"]:
5151
- if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
5152
- t["dataset"]["version"] = version
5153
- return self._json(201, {"run": QUEUE.submit(run, body)})
5709
+ n, body = got
5710
+ if n is None:
5711
+ return self._json(400, {"error": f"the run cannot be queued: {body}"})
5712
+ ref["n"], ref["version"] = n, fingerprint(body)
5713
+ groups[group_key(ref)] = body
5714
+ # The worker reads a body per group now (#233), so a run may link
5715
+ # several; each body is kept with the row under `<id>@<n>`.
5716
+ return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
5154
5717
 
5155
5718
  # ---- Datasets ----------------------------------------------------------
5156
5719
  # Rows in the store (Datasets above). The page reads one by id; an export
@@ -5244,9 +5807,16 @@ class Handler(BaseHTTPRequestHandler):
5244
5807
  DATASETS.lazy_trash()
5245
5808
  parts = path.split("/")
5246
5809
  if len(parts) == 3:
5247
- return self._json(200, {"datasets": DATASETS.list()})
5810
+ return self._json(200, {"groups" if self.grouped else "datasets": DATASETS.list()})
5248
5811
  if len(parts) == 4 and parts[3] == "export":
5249
- return self._download(DATASETS.export_all(), "datasets", "all")
5812
+ return self._download(DATASETS.export_all(), "eval-groups", "all")
5813
+ if len(parts) == 5 and parts[4] == "versions":
5814
+ got = DATASETS.versions(parts[3])
5815
+ return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
5816
+ if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
5817
+ got = DATASETS.version_body(parts[3], int(parts[5]))
5818
+ return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
5819
+ else self._send(404, b"not found", "text/plain")
5250
5820
  if len(parts) == 4 and parts[3] == "archived-rules":
5251
5821
  return self._json(200, {"rules": DATASETS.archived_rules()})
5252
5822
  if len(parts) == 4:
@@ -5256,7 +5826,7 @@ class Handler(BaseHTTPRequestHandler):
5256
5826
  doc = DATASETS.export(parts[3])
5257
5827
  if doc is None:
5258
5828
  return self._send(404, b"not found", "text/plain")
5259
- return self._download(doc, "dataset", doc["dataset"]["name"])
5829
+ return self._download(doc, "eval-group", doc["group"]["name"])
5260
5830
  return self._send(404, b"not found", "text/plain")
5261
5831
 
5262
5832
  def _download(self, doc, kind, name):
@@ -5284,6 +5854,12 @@ class Handler(BaseHTTPRequestHandler):
5284
5854
  if err:
5285
5855
  return self._json(err[0], {"error": err[1]})
5286
5856
  return self._json(200, dataset)
5857
+ if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
5858
+ self._payload()
5859
+ dataset, err = DATASETS.restore_version(parts[3], int(parts[5]))
5860
+ if err:
5861
+ return self._json(err[0], {"error": err[1]})
5862
+ return self._json(200, dataset)
5287
5863
  if len(parts) == 4 and parts[3] == "import":
5288
5864
  doc, err = self._dataset_payload(DATASET_IMPORT_CAP)
5289
5865
  if err: