evals-lab 0.4.1 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +81 -0
- package/README.md +125 -28
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +542 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +735 -86
- package/lab/kinds/list.mjs +2 -1
- package/lab/run-evals.js +354 -99
- package/lab/server.py +711 -135
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +3 -0
- package/lab/web/dist/assets/main-DDoeU6hq.css +1 -0
- package/lab/web/dist/assets/main-nz6Q4jVm.js +21 -0
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +59 -0
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-BkZTEix2.js +0 -21
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/server.py
CHANGED
|
@@ -560,6 +560,48 @@ QUEUE_WAIT = 1.0
|
|
|
560
560
|
WATCH_EVERY = 5.0
|
|
561
561
|
|
|
562
562
|
|
|
563
|
+
# A Target profile's key is written from the page and never read back by it
|
|
564
|
+
# (#258): a client holding the lab's password -- CI's runners and their logs
|
|
565
|
+
# among them -- is handed no key. Where a profile holds one, what is served
|
|
566
|
+
# (/api/state, the page's carried copy, a refused write's current copy) holds
|
|
567
|
+
# KEY_HELD and the profile's id instead, and a write that brings that back
|
|
568
|
+
# keeps the key the store holds for the id: the profile's own, or the one it
|
|
569
|
+
# was cloned or restored from. KEY_HELD starts with a character no key can
|
|
570
|
+
# (header_safe), so a pasted key is never taken for it.
|
|
571
|
+
KEY_HELD = "\u2022held:"
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def held(key) -> bool:
|
|
575
|
+
return isinstance(key, str) and key.startswith(KEY_HELD)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def profiles_in(name: str, body):
|
|
579
|
+
"""Every profile a synced document holds: the profiles store's list, and
|
|
580
|
+
each profile version's copy."""
|
|
581
|
+
if not isinstance(body, dict):
|
|
582
|
+
return
|
|
583
|
+
if name == "promptlab.profiles":
|
|
584
|
+
for p in body.get("list") or []:
|
|
585
|
+
if isinstance(p, dict):
|
|
586
|
+
yield p
|
|
587
|
+
elif name == "promptlab.versions":
|
|
588
|
+
for kept in (body.get("profile") or {}).values() if isinstance(body.get("profile"), dict) else ():
|
|
589
|
+
for v in kept if isinstance(kept, list) else ():
|
|
590
|
+
if isinstance(v, dict) and isinstance(v.get("doc"), dict):
|
|
591
|
+
yield v["doc"]
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def hide_keys(docs: dict) -> dict:
|
|
595
|
+
out = {}
|
|
596
|
+
for name, d in docs.items():
|
|
597
|
+
d = json.loads(json.dumps(d))
|
|
598
|
+
for p in profiles_in(name, d.get("body")):
|
|
599
|
+
if isinstance(p.get("key"), str) and p["key"] and not held(p["key"]):
|
|
600
|
+
p["key"] = KEY_HELD + str(p.get("id") or "")
|
|
601
|
+
out[name] = d
|
|
602
|
+
return out
|
|
603
|
+
|
|
604
|
+
|
|
563
605
|
class Store:
|
|
564
606
|
"""
|
|
565
607
|
Documents by name, each with a version that goes up by one per write.
|
|
@@ -579,6 +621,11 @@ class Store:
|
|
|
579
621
|
db.execute("CREATE TABLE IF NOT EXISTS docs (name TEXT PRIMARY KEY, "
|
|
580
622
|
"version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
|
|
581
623
|
db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
|
|
624
|
+
# The key of a profile a write took out, by id, for TRASH_SECONDS:
|
|
625
|
+
# Undo puts the profile back holding KEY_HELD, and this is what
|
|
626
|
+
# it holds.
|
|
627
|
+
db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
|
|
628
|
+
"key TEXT NOT NULL, at REAL NOT NULL)")
|
|
582
629
|
# History was a document, capped at what a browser could hold. The
|
|
583
630
|
# first start with the table moves what that document had into it,
|
|
584
631
|
# once, and drops the document so the page stops carrying it.
|
|
@@ -600,8 +647,30 @@ class Store:
|
|
|
600
647
|
|
|
601
648
|
def served(self) -> dict:
|
|
602
649
|
"""What a browser is handed: the SYNCED documents only, so a retired
|
|
603
|
-
key's rows stay in the store without reaching a page again
|
|
604
|
-
|
|
650
|
+
key's rows stay in the store without reaching a page again, and no
|
|
651
|
+
profile's key (KEY_HELD)."""
|
|
652
|
+
return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
|
|
653
|
+
|
|
654
|
+
def _keyring(self, db, now: dict) -> dict:
|
|
655
|
+
"""Each profile id's key as the store holds it: its profile's, else a
|
|
656
|
+
dropped one's, else its newest version's."""
|
|
657
|
+
ring = {}
|
|
658
|
+
for p in profiles_in("promptlab.versions", now.get("promptlab.versions", {}).get("body")):
|
|
659
|
+
pid, key = str(p.get("id") or ""), p.get("key")
|
|
660
|
+
if pid not in ring and isinstance(key, str) and key and not held(key):
|
|
661
|
+
ring[pid] = key
|
|
662
|
+
db.execute("DELETE FROM dropped_keys WHERE at < ?", (time.time() - TRASH_SECONDS,))
|
|
663
|
+
ring.update(db.execute("SELECT id, key FROM dropped_keys").fetchall())
|
|
664
|
+
for p in profiles_in("promptlab.profiles", now.get("promptlab.profiles", {}).get("body")):
|
|
665
|
+
key = p.get("key")
|
|
666
|
+
if isinstance(key, str) and key and not held(key):
|
|
667
|
+
ring[str(p.get("id") or "")] = key
|
|
668
|
+
return ring
|
|
669
|
+
|
|
670
|
+
def held_key(self, key: str) -> str:
|
|
671
|
+
"""The key KEY_HELD stands for, or "" when the store holds none."""
|
|
672
|
+
with self.lock, closing(sqlite3.connect(self.path)) as db, db:
|
|
673
|
+
return self._keyring(db, self._rows(db)).get(key[len(KEY_HELD):], "")
|
|
605
674
|
|
|
606
675
|
def write(self, docs: dict):
|
|
607
676
|
"""
|
|
@@ -614,9 +683,23 @@ class Store:
|
|
|
614
683
|
have = lambda n: now.get(n, {"version": 0, "body": None})
|
|
615
684
|
stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
|
|
616
685
|
if stale:
|
|
617
|
-
return None, stale
|
|
686
|
+
return None, hide_keys(stale)
|
|
618
687
|
at = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
619
688
|
with db:
|
|
689
|
+
ring = self._keyring(db, now)
|
|
690
|
+
for n, d in docs.items():
|
|
691
|
+
for p in profiles_in(n, d["body"]):
|
|
692
|
+
if held(p.get("key")):
|
|
693
|
+
p["key"] = ring.get(p["key"][len(KEY_HELD):], "")
|
|
694
|
+
if "promptlab.profiles" in docs:
|
|
695
|
+
kept = {str(p.get("id") or "") for p in profiles_in(
|
|
696
|
+
"promptlab.profiles", docs["promptlab.profiles"]["body"])}
|
|
697
|
+
db.executemany("DELETE FROM dropped_keys WHERE id = ?", [(i,) for i in kept])
|
|
698
|
+
db.executemany(
|
|
699
|
+
"INSERT OR REPLACE INTO dropped_keys (id, key, at) VALUES (?, ?, ?)",
|
|
700
|
+
[(str(p.get("id") or ""), p["key"], time.time())
|
|
701
|
+
for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
|
|
702
|
+
if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
|
|
620
703
|
for n, d in docs.items():
|
|
621
704
|
db.execute(
|
|
622
705
|
"INSERT INTO docs (name, version, body, updated_at) VALUES (?, ?, ?, ?) "
|
|
@@ -726,7 +809,7 @@ class Sources:
|
|
|
726
809
|
out = []
|
|
727
810
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
728
811
|
db.row_factory = sqlite3.Row
|
|
729
|
-
for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
|
|
812
|
+
for r in db.execute("SELECT id, name, system, bytes, type, created_at FROM sources "
|
|
730
813
|
"ORDER BY system DESC, name COLLATE NOCASE"):
|
|
731
814
|
if r["system"]:
|
|
732
815
|
files = self._sample_files()
|
|
@@ -734,10 +817,13 @@ class Sources:
|
|
|
734
817
|
"type": r["type"],
|
|
735
818
|
"files": len(files), "bytes": sum(f["bytes"] for f in files)})
|
|
736
819
|
else:
|
|
737
|
-
n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
|
|
738
|
-
|
|
820
|
+
n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
|
|
821
|
+
(r["id"],)).fetchone()
|
|
822
|
+
# When it last changed: made, or a file added -- what a
|
|
823
|
+
# picker orders its recent Sources by.
|
|
739
824
|
out.append({"id": r["id"], "name": r["name"], "system": False,
|
|
740
|
-
"type": r["type"], "files": n, "bytes": r["bytes"]
|
|
825
|
+
"type": r["type"], "files": n, "bytes": r["bytes"],
|
|
826
|
+
"changed": max(filter(None, (r["created_at"], last)))})
|
|
741
827
|
return out
|
|
742
828
|
|
|
743
829
|
def get(self, sid) -> dict:
|
|
@@ -1177,6 +1263,18 @@ class Sources:
|
|
|
1177
1263
|
return None, (500, "the files could not be copied")
|
|
1178
1264
|
return self.get(sid), None
|
|
1179
1265
|
|
|
1266
|
+
def item_paths(self, sid):
|
|
1267
|
+
"""Every file of a Source, as (name, path on disk) in its own order,
|
|
1268
|
+
or None when there is no such Source."""
|
|
1269
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1270
|
+
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1271
|
+
if r is None:
|
|
1272
|
+
return None
|
|
1273
|
+
if r[0]:
|
|
1274
|
+
return [(f["name"], SAMPLES / f["name"]) for f in self._sample_files()]
|
|
1275
|
+
return [(row[0], self.dir / sid / row[0]) for row in db.execute(
|
|
1276
|
+
"SELECT name FROM source_files WHERE source = ? ORDER BY name", (sid,))]
|
|
1277
|
+
|
|
1180
1278
|
def zip_files(self, sid, names):
|
|
1181
1279
|
"""The bytes of a stdlib zipfile holding exactly the named files, in
|
|
1182
1280
|
the order named, or an error. Read under the store's lock so the file
|
|
@@ -1362,15 +1460,23 @@ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cas
|
|
|
1362
1460
|
# 5 was told by its `source` alone, and earlier ones by neither.
|
|
1363
1461
|
DATASET_BODY_VERSION = 7
|
|
1364
1462
|
DATASET_NAME_MAX = 80
|
|
1365
|
-
# The file forms Export writes and Import reads. Export writes
|
|
1366
|
-
#
|
|
1367
|
-
#
|
|
1368
|
-
#
|
|
1369
|
-
|
|
1370
|
-
|
|
1463
|
+
# The file forms Export writes and Import reads. Export writes an eval group
|
|
1464
|
+
# at version 7 (docs/pipeline-model.md §17); Import reads that, and a dataset
|
|
1465
|
+
# file of versions 1 to 7, upgraded, and refuses anything else, as a pipeline
|
|
1466
|
+
# of another version is refused. Versions 1 to 3 carried a prompt, which an
|
|
1467
|
+
# import gives to the Prompt library.
|
|
1468
|
+
EXPORT_ONE = "evals-lab/eval-group"
|
|
1469
|
+
EXPORT_ALL = "evals-lab/eval-groups"
|
|
1470
|
+
DATASET_ONE = "evals-lab/dataset"
|
|
1471
|
+
DATASET_ALL = "evals-lab/datasets"
|
|
1371
1472
|
EXPORT_VERSION = 7
|
|
1372
1473
|
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
|
|
1474
|
+
# Each file form, and the key its one entry or its list sits under.
|
|
1475
|
+
EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
|
|
1373
1476
|
SCORING_MODES = ("all", "weighted")
|
|
1477
|
+
# A group's newest version is edited in place by a save within this long of
|
|
1478
|
+
# the last, as a prompt's is (PROMPT_IDLE_SECONDS).
|
|
1479
|
+
GROUP_IDLE_SECONDS = float(os.environ.get("GROUP_IDLE_SECONDS", "30"))
|
|
1374
1480
|
|
|
1375
1481
|
|
|
1376
1482
|
def blank_dataset() -> dict:
|
|
@@ -1574,6 +1680,30 @@ def fingerprint(body: dict) -> str:
|
|
|
1574
1680
|
return hashlib.sha256(text.encode("utf-8")).hexdigest()[:CASE_SET_LEN]
|
|
1575
1681
|
|
|
1576
1682
|
|
|
1683
|
+
def pins_in(workflows) -> set:
|
|
1684
|
+
"""(group id, version) for every link a stored pipeline pins: the
|
|
1685
|
+
promptlab.workflows body, each pipeline as its `work`. Only version 13
|
|
1686
|
+
links pin, and a pipeline of an earlier version has none."""
|
|
1687
|
+
out = set()
|
|
1688
|
+
listed = workflows.get("list") if isinstance(workflows, dict) else None
|
|
1689
|
+
for w in listed if isinstance(listed, list) else []:
|
|
1690
|
+
work = w.get("work") if isinstance(w, dict) else None
|
|
1691
|
+
evals = work.get("evals") if isinstance(work, dict) else None
|
|
1692
|
+
for t in evals if isinstance(evals, list) else []:
|
|
1693
|
+
if (isinstance(t, dict) and t.get("type") == "group" and isinstance(t.get("group"), dict)
|
|
1694
|
+
and isinstance(t["group"].get("id"), str) and type(t.get("pin")) is int):
|
|
1695
|
+
out.add((t["group"]["id"], t["pin"]))
|
|
1696
|
+
return out
|
|
1697
|
+
|
|
1698
|
+
|
|
1699
|
+
def group_key(ref) -> str:
|
|
1700
|
+
"""The key a run keeps a group's body under: `<id>@<n>`, or the id alone
|
|
1701
|
+
for a reference from before the lab numbered versions."""
|
|
1702
|
+
n = ref.get("n") if isinstance(ref, dict) else None
|
|
1703
|
+
gid = ref.get("id") if isinstance(ref, dict) else None
|
|
1704
|
+
return f"{gid}@{n}" if type(n) is int else str(gid)
|
|
1705
|
+
|
|
1706
|
+
|
|
1577
1707
|
def unique_dataset_name(name: str, taken: set) -> str:
|
|
1578
1708
|
"""`Receipts` again becomes `Receipts (2)`, compared without case."""
|
|
1579
1709
|
lower = {t.lower() for t in taken}
|
|
@@ -1627,7 +1757,7 @@ def scenario_ref(sc, i):
|
|
|
1627
1757
|
what version 4's upgrade gives one, by position."""
|
|
1628
1758
|
sid = sc.get("id") if isinstance(sc, dict) else None
|
|
1629
1759
|
name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
|
|
1630
|
-
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
|
|
1760
|
+
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
|
|
1631
1761
|
|
|
1632
1762
|
|
|
1633
1763
|
class Prompts:
|
|
@@ -1944,6 +2074,16 @@ class Datasets:
|
|
|
1944
2074
|
# the body it was typed as is kept, as the rules were.
|
|
1945
2075
|
db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
|
|
1946
2076
|
"dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
2077
|
+
# Every version of each eval group, numbered from 1, for a link to
|
|
2078
|
+
# pin and a run to name (docs/pipeline-model.md §17). `ran` says a
|
|
2079
|
+
# run graded with it, which freezes it for good; a pin freezes it
|
|
2080
|
+
# while a stored pipeline holds the pin. Version 1 is added from
|
|
2081
|
+
# the row's body at its first save, pin or run, so a group nobody
|
|
2082
|
+
# touches holds what it held.
|
|
2083
|
+
db.execute("CREATE TABLE IF NOT EXISTS eval_group_versions ("
|
|
2084
|
+
"group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
|
|
2085
|
+
"created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
|
|
2086
|
+
"ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
|
|
1947
2087
|
# Rows from an earlier version are converted once, in place: a
|
|
1948
2088
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
1949
2089
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -1992,16 +2132,26 @@ class Datasets:
|
|
|
1992
2132
|
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
1993
2133
|
|
|
1994
2134
|
@staticmethod
|
|
1995
|
-
def _doc(r, body=True):
|
|
2135
|
+
def _doc(r, body=True, versions=1):
|
|
1996
2136
|
"""A row as the API answers it: a DatasetSummary, and its body with it,
|
|
1997
|
-
read as today's version.
|
|
2137
|
+
read as today's version. `version` is the save counter a write is
|
|
2138
|
+
arbitrated by; `versions` is how many group versions it has -- its
|
|
2139
|
+
newest one's number, which a pin may name, and 1 before any is kept,
|
|
2140
|
+
the row's body being version 1 in waiting."""
|
|
1998
2141
|
parsed = upgrade_body(json.loads(r["body"]))
|
|
1999
2142
|
out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
|
|
2000
|
-
"version": r["version"], "updated": r["updated_at"]}
|
|
2143
|
+
"version": r["version"], "versions": versions, "updated": r["updated_at"]}
|
|
2001
2144
|
if body:
|
|
2002
2145
|
out["body"] = parsed
|
|
2003
2146
|
return out
|
|
2004
2147
|
|
|
2148
|
+
@staticmethod
|
|
2149
|
+
def _counts(db) -> dict:
|
|
2150
|
+
return dict(db.execute("SELECT group_id, MAX(n) FROM eval_group_versions GROUP BY group_id"))
|
|
2151
|
+
|
|
2152
|
+
def _row_doc(self, db, r, body=True):
|
|
2153
|
+
return self._doc(r, body, self._counts(db).get(r["id"], 1))
|
|
2154
|
+
|
|
2005
2155
|
def _connect(self):
|
|
2006
2156
|
db = sqlite3.connect(self.store.path)
|
|
2007
2157
|
db.row_factory = sqlite3.Row
|
|
@@ -2016,13 +2166,142 @@ class Datasets:
|
|
|
2016
2166
|
|
|
2017
2167
|
def list(self) -> list:
|
|
2018
2168
|
with self.store.lock, self._connect() as db:
|
|
2019
|
-
|
|
2169
|
+
counts = self._counts(db)
|
|
2170
|
+
return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
|
|
2020
2171
|
"SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id")]
|
|
2021
2172
|
|
|
2022
2173
|
def get(self, did):
|
|
2023
2174
|
with self.store.lock, self._connect() as db:
|
|
2024
2175
|
r = self._live(db, did)
|
|
2025
|
-
return self.
|
|
2176
|
+
return self._row_doc(db, r) if r else None
|
|
2177
|
+
|
|
2178
|
+
# ---- a group's versions (docs/pipeline-model.md §17) -----------------
|
|
2179
|
+
|
|
2180
|
+
@staticmethod
|
|
2181
|
+
def _head(db, did):
|
|
2182
|
+
return db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? "
|
|
2183
|
+
"ORDER BY n DESC LIMIT 1", (did,)).fetchone()
|
|
2184
|
+
|
|
2185
|
+
def _mint(self, db, r):
|
|
2186
|
+
"""The newest version of row [r], adding version 1 from its body
|
|
2187
|
+
first if it has none: edited when the row last was, so a group made a
|
|
2188
|
+
moment ago goes on being edited in place."""
|
|
2189
|
+
head = self._head(db, r["id"])
|
|
2190
|
+
if head is not None:
|
|
2191
|
+
return head
|
|
2192
|
+
try:
|
|
2193
|
+
edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
|
|
2194
|
+
except ValueError:
|
|
2195
|
+
# A stamp nothing here wrote is no recent edit.
|
|
2196
|
+
edited = 0
|
|
2197
|
+
db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
|
|
2198
|
+
"VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
|
|
2199
|
+
return self._head(db, r["id"])
|
|
2200
|
+
|
|
2201
|
+
@staticmethod
|
|
2202
|
+
def _pins(db) -> set:
|
|
2203
|
+
"""Every (group, version) a stored pipeline pins, read in the caller's
|
|
2204
|
+
transaction from the docs table the page writes them to."""
|
|
2205
|
+
row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'").fetchone()
|
|
2206
|
+
return pins_in(json.loads(row[0])) if row and row[0] else set()
|
|
2207
|
+
|
|
2208
|
+
def _cut(self, db, did, text):
|
|
2209
|
+
"""A new newest version of [did] holding [text]; returns its number."""
|
|
2210
|
+
n = self._head(db, did)["n"] + 1
|
|
2211
|
+
db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
|
|
2212
|
+
"VALUES (?, ?, ?, ?, ?)", (did, n, text, self._now(), time.time()))
|
|
2213
|
+
return n
|
|
2214
|
+
|
|
2215
|
+
def _keep(self, db, r, text):
|
|
2216
|
+
"""[text] as row [r]'s newest version: the newest edited in place while
|
|
2217
|
+
no run has graded with it, no link pins it and it was edited in the
|
|
2218
|
+
last GROUP_IDLE_SECONDS, and a new version otherwise."""
|
|
2219
|
+
head = self._mint(db, r)
|
|
2220
|
+
if text == head["body"]:
|
|
2221
|
+
return
|
|
2222
|
+
fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
|
|
2223
|
+
if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
|
|
2224
|
+
db.execute("UPDATE eval_group_versions SET body = ?, edited_at = ? WHERE group_id = ? AND n = ?",
|
|
2225
|
+
(text, time.time(), r["id"], head["n"]))
|
|
2226
|
+
else:
|
|
2227
|
+
self._cut(db, r["id"], text)
|
|
2228
|
+
|
|
2229
|
+
def versions(self, did):
|
|
2230
|
+
"""A group's versions, newest first, without their bodies -- version 1
|
|
2231
|
+
alone, read from the row, before any is kept -- or None."""
|
|
2232
|
+
with self.store.lock, self._connect() as db:
|
|
2233
|
+
r = self._live(db, did)
|
|
2234
|
+
if r is None:
|
|
2235
|
+
return None
|
|
2236
|
+
pins = self._pins(db)
|
|
2237
|
+
rows = db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? ORDER BY n DESC",
|
|
2238
|
+
(did,)).fetchall()
|
|
2239
|
+
if not rows:
|
|
2240
|
+
return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (did, 1) in pins,
|
|
2241
|
+
"fingerprint": fingerprint(json.loads(r["body"]))}]
|
|
2242
|
+
return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
|
|
2243
|
+
"pinned": (did, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
|
|
2244
|
+
for v in rows]
|
|
2245
|
+
|
|
2246
|
+
def version_body(self, did, n):
|
|
2247
|
+
"""Version [n]'s body, read as today's, or None."""
|
|
2248
|
+
with self.store.lock, self._connect() as db:
|
|
2249
|
+
r = self._live(db, did)
|
|
2250
|
+
if r is None:
|
|
2251
|
+
return None
|
|
2252
|
+
v = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
|
|
2253
|
+
(did, n)).fetchone()
|
|
2254
|
+
if v is None and n == 1 and self._head(db, did) is None:
|
|
2255
|
+
v = (r["body"],)
|
|
2256
|
+
return upgrade_body(json.loads(v[0])) if v else None
|
|
2257
|
+
|
|
2258
|
+
def restore_version(self, did, n):
|
|
2259
|
+
"""An older version's body as the newest version, and the row's: as a
|
|
2260
|
+
prompt's Restore, nothing is rewritten, so a run or a pin naming any
|
|
2261
|
+
version still reads what it named. Returns (DatasetDoc, None)."""
|
|
2262
|
+
with self.store.lock, self._connect() as db, db:
|
|
2263
|
+
r = self._live(db, did)
|
|
2264
|
+
if r is None:
|
|
2265
|
+
return None, (404, "no such eval group")
|
|
2266
|
+
head = self._mint(db, r)
|
|
2267
|
+
old = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
|
|
2268
|
+
(did, n)).fetchone()
|
|
2269
|
+
if old is None:
|
|
2270
|
+
return None, (404, "no such version")
|
|
2271
|
+
if old["body"] != head["body"]:
|
|
2272
|
+
self._cut(db, did, old["body"])
|
|
2273
|
+
db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
|
|
2274
|
+
(old["body"], r["version"] + 1, self._now(), did))
|
|
2275
|
+
return self._row_doc(db, self._live(db, did)), None
|
|
2276
|
+
|
|
2277
|
+
def mint_pinned(self, pins):
|
|
2278
|
+
"""Version 1 of each pinned group that has none yet: a pin names a
|
|
2279
|
+
version, so the version has to be kept from the moment it does."""
|
|
2280
|
+
with self.store.lock, self._connect() as db, db:
|
|
2281
|
+
for did, _ in pins:
|
|
2282
|
+
r = self._live(db, did)
|
|
2283
|
+
if r is not None:
|
|
2284
|
+
self._mint(db, r)
|
|
2285
|
+
|
|
2286
|
+
def resolve(self, did, pin=None):
|
|
2287
|
+
"""The version a run submitted now grades with -- [pin], or the
|
|
2288
|
+
newest -- marked as graded with, in the same transaction, so no save
|
|
2289
|
+
can edit it in place between this and the run keeping its body.
|
|
2290
|
+
Returns (n, body), (None, why) for a pin the group has no version of,
|
|
2291
|
+
or None for a group the lab does not have. A submit refused after
|
|
2292
|
+
this leaves the version frozen, which costs only a new version at the
|
|
2293
|
+
next save."""
|
|
2294
|
+
with self.store.lock, self._connect() as db, db:
|
|
2295
|
+
r = self._live(db, did) if isinstance(did, str) else None
|
|
2296
|
+
if r is None:
|
|
2297
|
+
return None
|
|
2298
|
+
head = self._mint(db, r)
|
|
2299
|
+
v = head if pin is None else db.execute(
|
|
2300
|
+
"SELECT * FROM eval_group_versions WHERE group_id = ? AND n = ?", (did, pin)).fetchone()
|
|
2301
|
+
if v is None:
|
|
2302
|
+
return None, f"{r['name']} has no version {pin}"
|
|
2303
|
+
db.execute("UPDATE eval_group_versions SET ran = 1 WHERE group_id = ? AND n = ?", (did, v["n"]))
|
|
2304
|
+
return v["n"], json.loads(v["body"])
|
|
2026
2305
|
|
|
2027
2306
|
def snapshot(self, did):
|
|
2028
2307
|
"""The body a run submitted now grades against, and its fingerprint;
|
|
@@ -2087,9 +2366,11 @@ class Datasets:
|
|
|
2087
2366
|
if r is None:
|
|
2088
2367
|
return None, (404, "no such dataset")
|
|
2089
2368
|
if r["version"] != version:
|
|
2090
|
-
return None, (409, {"current": self.
|
|
2369
|
+
return None, (409, {"current": self._row_doc(db, r)})
|
|
2370
|
+
text = json.dumps(body)
|
|
2371
|
+
self._keep(db, r, text)
|
|
2091
2372
|
db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
|
|
2092
|
-
(
|
|
2373
|
+
(text, version + 1, self._now(), did))
|
|
2093
2374
|
return {"version": version + 1}, None
|
|
2094
2375
|
|
|
2095
2376
|
def remove(self, did):
|
|
@@ -2115,58 +2396,69 @@ class Datasets:
|
|
|
2115
2396
|
did = r["id"]
|
|
2116
2397
|
return self.get(did), None
|
|
2117
2398
|
|
|
2399
|
+
@staticmethod
|
|
2400
|
+
def _purge(db, where, args):
|
|
2401
|
+
# A group's versions go with it: the trash held them, and Undo is
|
|
2402
|
+
# what would have brought them back.
|
|
2403
|
+
db.execute(f"DELETE FROM eval_group_versions WHERE group_id IN (SELECT id FROM datasets WHERE {where})", args)
|
|
2404
|
+
db.execute(f"DELETE FROM datasets WHERE {where}", args)
|
|
2405
|
+
|
|
2118
2406
|
def lazy_trash(self):
|
|
2119
2407
|
"""A datasets request empties what has been trashed longer than
|
|
2120
2408
|
TRASH_SECONDS, as a Sources request does."""
|
|
2121
2409
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2122
|
-
|
|
2123
|
-
(time.time() - TRASH_SECONDS,))
|
|
2410
|
+
self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
|
|
2124
2411
|
|
|
2125
2412
|
def empty_trash(self):
|
|
2126
2413
|
"""The startup sweep: a restart has nothing to undo."""
|
|
2127
2414
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2128
|
-
|
|
2415
|
+
self._purge(db, "trash IS NOT NULL", ())
|
|
2129
2416
|
|
|
2130
2417
|
def export(self, did):
|
|
2131
2418
|
d = self.get(did)
|
|
2132
2419
|
if d is None:
|
|
2133
2420
|
return None
|
|
2134
2421
|
return {"format": EXPORT_ONE, "version": EXPORT_VERSION,
|
|
2135
|
-
"
|
|
2422
|
+
"group": {"name": d["name"], "body": d["body"]}}
|
|
2136
2423
|
|
|
2137
2424
|
def export_all(self):
|
|
2138
2425
|
with self.store.lock, self._connect() as db:
|
|
2139
2426
|
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
|
|
2140
2427
|
"ORDER BY name COLLATE NOCASE, id").fetchall()
|
|
2141
2428
|
return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
|
|
2142
|
-
"
|
|
2429
|
+
"groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
|
|
2143
2430
|
|
|
2144
2431
|
def import_file(self, doc):
|
|
2145
2432
|
"""Either export's file, as new datasets: import always creates, ids
|
|
2146
2433
|
are this lab's, and a taken name gets ` (2)`. All of it or none.
|
|
2147
2434
|
Returns ([DatasetSummary], None), or (None, (400, one sentence))."""
|
|
2148
2435
|
if not isinstance(doc, dict) or "format" not in doc:
|
|
2149
|
-
return None, (400, "that file is not
|
|
2150
|
-
if doc["format"] not in
|
|
2436
|
+
return None, (400, "that file is not an eval group export: it has no \"format\"")
|
|
2437
|
+
if doc["format"] not in EXPORT_KEYS:
|
|
2151
2438
|
return None, (400, f"that file's format is {json.dumps(doc['format'])}, and the lab "
|
|
2152
|
-
f"imports {EXPORT_ONE} and {
|
|
2439
|
+
f"imports {EXPORT_ONE}, {EXPORT_ALL}, {DATASET_ONE} and {DATASET_ALL}")
|
|
2153
2440
|
version = doc.get("version")
|
|
2154
|
-
# Exactly the integer: Python's True == 1.
|
|
2155
|
-
|
|
2441
|
+
# Exactly the integer: Python's True == 1. An eval group's file began
|
|
2442
|
+
# at version 7; a dataset's reads back to version 1.
|
|
2443
|
+
grouped = doc["format"] in (EXPORT_ONE, EXPORT_ALL)
|
|
2444
|
+
if type(version) is not int or version not in ((EXPORT_VERSION,) if grouped else IMPORT_VERSIONS):
|
|
2156
2445
|
return None, (400, f"that file is version {json.dumps(version)}, and the lab "
|
|
2157
|
-
f"reads versions 1 to {EXPORT_VERSION}")
|
|
2158
|
-
|
|
2159
|
-
|
|
2446
|
+
+ (f"reads version {EXPORT_VERSION}" if grouped else f"reads versions 1 to {EXPORT_VERSION}"))
|
|
2447
|
+
one = doc["format"] in (EXPORT_ONE, DATASET_ONE)
|
|
2448
|
+
key = EXPORT_KEYS[doc["format"]]
|
|
2449
|
+
if one:
|
|
2450
|
+
items = [doc.get(key)]
|
|
2160
2451
|
else:
|
|
2161
|
-
items = doc.get(
|
|
2452
|
+
items = doc.get(key)
|
|
2162
2453
|
if not isinstance(items, list):
|
|
2163
|
-
return None, (400, "an export of every dataset holds them as a \"
|
|
2454
|
+
return None, (400, f"an export of every {'group' if grouped else 'dataset'} holds them as a \"{key}\" list")
|
|
2164
2455
|
for k in doc:
|
|
2165
|
-
if k not in ("format", "version",
|
|
2456
|
+
if k not in ("format", "version", key):
|
|
2166
2457
|
return None, (400, f"that file has \"{k}\", which an export does not")
|
|
2167
2458
|
ready = []
|
|
2459
|
+
what = "eval group" if grouped else "dataset"
|
|
2168
2460
|
for i, item in enumerate(items):
|
|
2169
|
-
at = "the
|
|
2461
|
+
at = f"the {what}" if one else f"{what} {i + 1}"
|
|
2170
2462
|
if not isinstance(item, dict) or set(item) != {"name", "body"}:
|
|
2171
2463
|
return None, (400, f"{at} has to be {{ \"name\", \"body\" }}")
|
|
2172
2464
|
name, why = dataset_name(item["name"])
|
|
@@ -2339,13 +2631,17 @@ def read_pack(data: bytes):
|
|
|
2339
2631
|
doc, why = load(name)
|
|
2340
2632
|
if why:
|
|
2341
2633
|
return None, why
|
|
2342
|
-
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2634
|
+
# A pack's datasets are eval groups, in either file form: the
|
|
2635
|
+
# group's (version 7) or the dataset's an older pack holds.
|
|
2636
|
+
fmt = doc.get("format") if isinstance(doc, dict) else None
|
|
2637
|
+
entry = doc.get(EXPORT_KEYS[fmt]) if fmt in (EXPORT_ONE, DATASET_ONE) else None
|
|
2638
|
+
if (not isinstance(entry, dict) or type(doc.get("version")) is not int
|
|
2639
|
+
or doc["version"] not in ((EXPORT_VERSION,) if fmt == EXPORT_ONE else IMPORT_VERSIONS)):
|
|
2640
|
+
return None, f"{name} is not an eval group export the lab reads"
|
|
2641
|
+
ds_name, why = dataset_name(entry.get("name"))
|
|
2346
2642
|
if why:
|
|
2347
2643
|
return None, f"{name}: {why}"
|
|
2348
|
-
raw =
|
|
2644
|
+
raw = entry.get("body")
|
|
2349
2645
|
body = upgrade_body(raw) if doc["version"] < EXPORT_VERSION else raw
|
|
2350
2646
|
why = dataset_problem(body)
|
|
2351
2647
|
if why:
|
|
@@ -2524,11 +2820,12 @@ class Packs:
|
|
|
2524
2820
|
# the page upgrades the pipeline as it reads it.
|
|
2525
2821
|
evals = doc.get("evals", doc.get("tests"))
|
|
2526
2822
|
for t in evals if isinstance(evals, list) else [evals]:
|
|
2527
|
-
ref =
|
|
2528
|
-
if
|
|
2823
|
+
ref = eval_group_ref(t)
|
|
2824
|
+
if ref is not None:
|
|
2529
2825
|
did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
|
|
2530
2826
|
if did:
|
|
2531
|
-
|
|
2827
|
+
ref.clear()
|
|
2828
|
+
ref.update({"id": did, "name": DATASETS.get(did)["name"]})
|
|
2532
2829
|
content = content_of(doc)
|
|
2533
2830
|
if isinstance(content, dict) and isinstance(content.get("ref"), dict):
|
|
2534
2831
|
sid = src_ids.get(content["ref"].get("id")) or src_ids.get(content["ref"].get("name"))
|
|
@@ -3251,13 +3548,15 @@ class Plugins:
|
|
|
3251
3548
|
# is a marker the worker checks between items -- today's semantics, stopping
|
|
3252
3549
|
# after the item in flight, never a process kill that loses its reply.
|
|
3253
3550
|
|
|
3254
|
-
# A profile's key, as run-evals.js reads it:
|
|
3255
|
-
# document
|
|
3256
|
-
#
|
|
3257
|
-
#
|
|
3258
|
-
#
|
|
3259
|
-
|
|
3260
|
-
|
|
3551
|
+
# A profile's key, as run-evals.js reads it: EVALSLAB_API_KEY_<SLUG>, the slug the
|
|
3552
|
+
# run document's connection carries (#253), or EVALSLAB_API_KEY_<ID> for one from
|
|
3553
|
+
# before slugs -- the id it keys its profiles table by -- in the shell-safe
|
|
3554
|
+
# spelling of itself. Never the name, so two profiles can share a name
|
|
3555
|
+
# without sharing a key (docs/pipeline-model.md §5). One spelling on both
|
|
3556
|
+
# sides -- evals-core.ts keyVar -- or the worker would never find the key the
|
|
3557
|
+
# page never sent.
|
|
3558
|
+
def key_var(profile_id: str, slug=None) -> str:
|
|
3559
|
+
return "EVALSLAB_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(slug or profile_id).upper()).strip("_")
|
|
3261
3560
|
|
|
3262
3561
|
|
|
3263
3562
|
def run_items(run):
|
|
@@ -3331,7 +3630,33 @@ def brief_row(row):
|
|
|
3331
3630
|
if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
|
|
3332
3631
|
return it
|
|
3333
3632
|
return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
|
|
3334
|
-
|
|
3633
|
+
# The run's verdict stays; each Target's eval by eval is the whole row's.
|
|
3634
|
+
brief = {k: v for k, v in row.items() if k != "verdicts"}
|
|
3635
|
+
return {**brief, "results": [item(it) for it in row.get("results") or []], "brief": True}
|
|
3636
|
+
|
|
3637
|
+
|
|
3638
|
+
# What of a worker's report a row keeps as its verdicts (#252): the run's --
|
|
3639
|
+
# pass, fail or incomplete, which the worker's exit code already said -- and
|
|
3640
|
+
# each Target's evals, by the eval's id, as `run.verdicts` names them. Only
|
|
3641
|
+
# these fields are carried, so nothing else the report holds lands in the row.
|
|
3642
|
+
# A status says whether the run finished; its verdict says whether it passed,
|
|
3643
|
+
# so a run that failed an eval is `done` with the verdict `fail`.
|
|
3644
|
+
VERDICTS = ("pass", "fail", "incomplete")
|
|
3645
|
+
VERDICT_FIELDS = ("name", "skipped", "verdict", "pass", "detail", "ran", "passed", "skippedItems")
|
|
3646
|
+
|
|
3647
|
+
|
|
3648
|
+
def report_verdicts(report):
|
|
3649
|
+
"""(verdict, verdicts as JSON) from a worker's report, or (None, None)
|
|
3650
|
+
for one that carries none -- a worker from before #251."""
|
|
3651
|
+
verdict = report.get("verdict") if isinstance(report, dict) else None
|
|
3652
|
+
run = report.get("run") if verdict in VERDICTS else None
|
|
3653
|
+
targets = run.get("verdicts") if isinstance(run, dict) else None
|
|
3654
|
+
if not isinstance(targets, list):
|
|
3655
|
+
return None, None
|
|
3656
|
+
kept = [{eid: {k: e[k] for k in VERDICT_FIELDS if k in e}
|
|
3657
|
+
for eid, e in t.items() if isinstance(e, dict)}
|
|
3658
|
+
if isinstance(t, dict) else {} for t in targets]
|
|
3659
|
+
return verdict, json.dumps(kept)
|
|
3335
3660
|
|
|
3336
3661
|
|
|
3337
3662
|
class Queue:
|
|
@@ -3368,13 +3693,27 @@ class Queue:
|
|
|
3368
3693
|
# one of its own.
|
|
3369
3694
|
if "rerun_of" not in cols:
|
|
3370
3695
|
db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
|
|
3696
|
+
# The worker's verdicts (#252); a row from before them has none,
|
|
3697
|
+
# and is never given one after the fact.
|
|
3698
|
+
if "verdict" not in cols:
|
|
3699
|
+
db.execute("ALTER TABLE queue ADD COLUMN verdict TEXT")
|
|
3700
|
+
if "verdicts" not in cols:
|
|
3701
|
+
db.execute("ALTER TABLE queue ADD COLUMN verdicts TEXT")
|
|
3702
|
+
# The body of each eval group a run grades with, by `<id>@<n>`
|
|
3703
|
+
# (docs/pipeline-model.md §17). A row from before keeps its
|
|
3704
|
+
# `dataset` column, read as its one group.
|
|
3705
|
+
if "groups" not in cols:
|
|
3706
|
+
db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
|
|
3371
3707
|
|
|
3372
3708
|
# ---- rows -----------------------------------------------------------
|
|
3373
3709
|
|
|
3374
3710
|
# A row read with the submit time of the run it re-runs, if any: the
|
|
3375
3711
|
# page names a run by that time (History's Run ID), so a "Re-run of"
|
|
3376
|
-
# note reads without fetching the original.
|
|
3377
|
-
|
|
3712
|
+
# note reads without fetching the original. Its columns are named, since
|
|
3713
|
+
# the order ALTER TABLE added them in is no order _row can count on.
|
|
3714
|
+
SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
|
|
3715
|
+
"q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
|
|
3716
|
+
"q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
|
|
3378
3717
|
"LEFT JOIN queue o ON o.id = q.rerun_of")
|
|
3379
3718
|
|
|
3380
3719
|
@staticmethod
|
|
@@ -3387,8 +3726,9 @@ class Queue:
|
|
|
3387
3726
|
"snapshot": json.loads(r[6]), "results": json.loads(r[7]),
|
|
3388
3727
|
"progress": json.loads(r[8]), "totals": json.loads(r[9]),
|
|
3389
3728
|
"error": r[10],
|
|
3390
|
-
"rerunOf": r[
|
|
3391
|
-
"
|
|
3729
|
+
"rerunOf": r[11], "rerunOfAt": r[14],
|
|
3730
|
+
"verdict": r[12],
|
|
3731
|
+
"verdicts": json.loads(r[13]) if r[13] else None,
|
|
3392
3732
|
}
|
|
3393
3733
|
|
|
3394
3734
|
# A row from before run documents has no version, and nothing here can
|
|
@@ -3411,15 +3751,19 @@ class Queue:
|
|
|
3411
3751
|
(rid,)).fetchone())
|
|
3412
3752
|
return row if self._readable(row) else None
|
|
3413
3753
|
|
|
3414
|
-
def list(self, limit=RUNS_PAGE, before=None, full=False):
|
|
3415
|
-
"""Runs, newest first, and whether more follow. `before`
|
|
3416
|
-
`
|
|
3417
|
-
them the way it pages the runs
|
|
3418
|
-
`
|
|
3754
|
+
def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
|
|
3755
|
+
"""Runs, newest first, and whether more follow. `before` and
|
|
3756
|
+
`before_id` are the `submittedAt` and id of the last run on the page
|
|
3757
|
+
before, so History can page through them the way it pages the runs
|
|
3758
|
+
store. `submittedAt` is to the second, so two runs can share one; the
|
|
3759
|
+
id breaks the tie, or a page ending between them would skip the
|
|
3760
|
+
second (#238). `before` alone stops at the second. Each row is
|
|
3761
|
+
brief_row's unless `full` asks for the whole of it."""
|
|
3419
3762
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3420
3763
|
rows = self._all(db)
|
|
3421
|
-
|
|
3422
|
-
rows
|
|
3764
|
+
key = lambda r: (r["submittedAt"], r["id"])
|
|
3765
|
+
rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
|
|
3766
|
+
rows.sort(key=key, reverse=True)
|
|
3423
3767
|
page = rows[:limit]
|
|
3424
3768
|
return (page if full else [brief_row(r) for r in page]), len(rows) > limit
|
|
3425
3769
|
|
|
@@ -3430,27 +3774,30 @@ class Queue:
|
|
|
3430
3774
|
|
|
3431
3775
|
# ---- submit ---------------------------------------------------------
|
|
3432
3776
|
|
|
3433
|
-
def submit(self, run: dict, dataset=None, rerun_of=None):
|
|
3777
|
+
def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
|
|
3434
3778
|
"""
|
|
3435
3779
|
A new queued run. `run` is the run document (docs/pipeline-model.md
|
|
3436
3780
|
§5): the pipeline, the profiles it resolved to without their keys, its
|
|
3437
|
-
content's file list in order and with its repeats, and
|
|
3438
|
-
version; `
|
|
3439
|
-
`
|
|
3440
|
-
|
|
3441
|
-
the
|
|
3781
|
+
content's file list in order and with its repeats, and each eval
|
|
3782
|
+
group's version; `groups` is those versions' bodies, by `<id>@<n>`
|
|
3783
|
+
(§17), and `dataset` the one body a row from before them kept, which a
|
|
3784
|
+
re-run of one carries on; `rerun_of` is the run a re-run was queued
|
|
3785
|
+
from. Returns the row. Its items are that list, or the one inline
|
|
3786
|
+
text, each through every scenario -- so the total is the list's
|
|
3787
|
+
length, repeats and all, the same count the runner and the page make.
|
|
3442
3788
|
"""
|
|
3443
3789
|
rid = secrets.token_hex(6)
|
|
3444
3790
|
now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
3445
3791
|
total = len(run_items(run))
|
|
3446
3792
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3447
3793
|
db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
|
|
3448
|
-
"snapshot, results, progress, totals, dataset, rerun_of) "
|
|
3449
|
-
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
|
|
3794
|
+
"snapshot, results, progress, totals, dataset, rerun_of, groups) "
|
|
3795
|
+
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
3450
3796
|
(rid, "queued", now, json.dumps(run), "[]",
|
|
3451
3797
|
json.dumps({"current": None, "n": 0, "total": total}),
|
|
3452
3798
|
json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
|
|
3453
|
-
None if dataset is None else json.dumps(dataset), rerun_of
|
|
3799
|
+
None if dataset is None else json.dumps(dataset), rerun_of,
|
|
3800
|
+
None if groups is None else json.dumps(groups)))
|
|
3454
3801
|
# Its prompts' uses, in the same transaction: a run is in the
|
|
3455
3802
|
# library the moment it is queued, or not queued at all.
|
|
3456
3803
|
if self.prompts is not None:
|
|
@@ -3460,19 +3807,43 @@ class Queue:
|
|
|
3460
3807
|
# the moment the lock is let go.
|
|
3461
3808
|
return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
|
|
3462
3809
|
|
|
3810
|
+
def groups(self, rid, raw=False):
|
|
3811
|
+
"""The eval group bodies a readable run grades with, by `<id>@<n>`, or
|
|
3812
|
+
None for a run that is not there or kept none -- what its verdicts
|
|
3813
|
+
were graded by, whatever the groups hold now. A row from before runs
|
|
3814
|
+
kept a body per group answers its one `dataset` copy under its
|
|
3815
|
+
group's key. A copy kept at an earlier version reads as one of
|
|
3816
|
+
today's, unless [raw]: the worker is handed it as kept, since an
|
|
3817
|
+
earlier run's pipeline is upgraded under the rules that copy holds."""
|
|
3818
|
+
run = self.get(rid)
|
|
3819
|
+
if run is None:
|
|
3820
|
+
return None
|
|
3821
|
+
kept = self._kept(rid)
|
|
3822
|
+
if kept is None:
|
|
3823
|
+
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3824
|
+
old = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()[0]
|
|
3825
|
+
if not old:
|
|
3826
|
+
return None
|
|
3827
|
+
kept = {group_key(evals_dataset(run["snapshot"])): json.loads(old)}
|
|
3828
|
+
return kept if raw else {k: upgrade_body(b) for k, b in kept.items()}
|
|
3829
|
+
|
|
3830
|
+
def _kept(self, rid):
|
|
3831
|
+
"""The row's own `groups` column, or None for a row from before it."""
|
|
3832
|
+
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3833
|
+
r = db.execute("SELECT groups FROM queue WHERE id = ?", (rid,)).fetchone()
|
|
3834
|
+
return json.loads(r[0]) if r and r[0] is not None else None
|
|
3835
|
+
|
|
3463
3836
|
def dataset(self, rid, raw=False):
|
|
3464
|
-
"""The
|
|
3465
|
-
|
|
3466
|
-
|
|
3467
|
-
the worker is handed it as kept, since an earlier run's pipeline is
|
|
3468
|
-
upgraded under the rules that copy holds."""
|
|
3837
|
+
"""The body of the one Library group a readable run grades its cases
|
|
3838
|
+
against (evals_dataset), or None: the body the worker is handed as
|
|
3839
|
+
--dataset, and what the page reads a run's cases from."""
|
|
3469
3840
|
if self.get(rid) is None:
|
|
3470
3841
|
return None
|
|
3471
|
-
|
|
3472
|
-
|
|
3473
|
-
|
|
3842
|
+
kept = self.groups(rid, raw=True)
|
|
3843
|
+
ref = evals_dataset(self.get(rid)["snapshot"])
|
|
3844
|
+
body = kept.get(group_key(ref)) if kept and ref is not None else None
|
|
3845
|
+
if body is None:
|
|
3474
3846
|
return None
|
|
3475
|
-
body = json.loads(r[0])
|
|
3476
3847
|
return body if raw else upgrade_body(body)
|
|
3477
3848
|
|
|
3478
3849
|
def _plugin_args(self, run):
|
|
@@ -3487,14 +3858,18 @@ class Queue:
|
|
|
3487
3858
|
return ["--plugins", str(PLUGINS.dir)], None
|
|
3488
3859
|
|
|
3489
3860
|
def _dataset_args(self, run, rundir):
|
|
3490
|
-
"""The worker's --dataset for a graded run: the body
|
|
3491
|
-
|
|
3492
|
-
|
|
3493
|
-
|
|
3861
|
+
"""The worker's --dataset for a graded run: the body of its one
|
|
3862
|
+
Library group kept with the row, written beside the run document --
|
|
3863
|
+
one, until the worker reads a body per group (#233). A row queued
|
|
3864
|
+
before runs kept their dataset has none, and is pinned to the dataset
|
|
3865
|
+
as it reads now, once, so every later pass over it agrees. Returns
|
|
3866
|
+
(args, None) or (None, why)."""
|
|
3494
3867
|
ref = evals_dataset(run["snapshot"])
|
|
3495
3868
|
if ref is None:
|
|
3496
3869
|
return [], None
|
|
3497
3870
|
body = self.dataset(run["id"], raw=True)
|
|
3871
|
+
if body is None and self._kept(run["id"]) is not None:
|
|
3872
|
+
return None, f"the run kept no body of the eval group {ref.get('name') or ref.get('id')!r}"
|
|
3498
3873
|
if body is None:
|
|
3499
3874
|
snap = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
|
|
3500
3875
|
if snap is None:
|
|
@@ -3510,6 +3885,27 @@ class Queue:
|
|
|
3510
3885
|
(rundir / "dataset.json").write_text(json.dumps(body))
|
|
3511
3886
|
return ["--dataset", str(rundir / "dataset.json")], None
|
|
3512
3887
|
|
|
3888
|
+
def _grading_args(self, run, rundir):
|
|
3889
|
+
"""The worker's eval group bodies: --dataset for a run that grades
|
|
3890
|
+
against one group, which keeps its single-body path and old-row
|
|
3891
|
+
pinning, and --groups for a run that links several (#233) -- every body
|
|
3892
|
+
it kept, by `<id>@<n>`, written beside the run document. Returns
|
|
3893
|
+
(args, None) or (None, why)."""
|
|
3894
|
+
refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
|
|
3895
|
+
keys = {group_key(r) for r in refs if r is not None}
|
|
3896
|
+
if len(keys) <= 1:
|
|
3897
|
+
return self._dataset_args(run, rundir)
|
|
3898
|
+
kept = self.groups(run["id"], raw=True)
|
|
3899
|
+
if kept is None:
|
|
3900
|
+
return None, "the run kept no eval group bodies"
|
|
3901
|
+
missing = sorted({(r.get("name") or r.get("id")) for r in refs
|
|
3902
|
+
if r is not None and group_key(r) not in kept})
|
|
3903
|
+
if missing:
|
|
3904
|
+
return None, f"the run kept no body of the eval group {', '.join(missing)}"
|
|
3905
|
+
rundir.mkdir(parents=True, exist_ok=True)
|
|
3906
|
+
(rundir / "groups.json").write_text(json.dumps(kept))
|
|
3907
|
+
return ["--groups", str(rundir / "groups.json")], None
|
|
3908
|
+
|
|
3513
3909
|
def _behind(self, rid):
|
|
3514
3910
|
"""How many submissions stand between this one and the worker, by
|
|
3515
3911
|
submit time -- what a waiting form names when it says what it is
|
|
@@ -3547,7 +3943,8 @@ class Queue:
|
|
|
3547
3943
|
if run["status"] not in ("cancelled", "interrupted", "incomplete"):
|
|
3548
3944
|
return None, (409, "only a cancelled, interrupted or incomplete run can be resumed")
|
|
3549
3945
|
(self.dir / rid / "cancel").unlink(missing_ok=True)
|
|
3550
|
-
|
|
3946
|
+
# What it reached is decided by the run it goes on to finish.
|
|
3947
|
+
self._set(rid, status="queued", cancel=0, error=None, verdict=None, verdicts=None)
|
|
3551
3948
|
return self.get(rid), None
|
|
3552
3949
|
|
|
3553
3950
|
def set_comment(self, rid, comment):
|
|
@@ -3597,9 +3994,12 @@ class Queue:
|
|
|
3597
3994
|
_, err = self._pinned_files(snap)
|
|
3598
3995
|
if err:
|
|
3599
3996
|
return None, (409, err)
|
|
3997
|
+
# The group bodies the original kept: a re-run grades with exactly
|
|
3998
|
+
# them, whatever the groups or the pins read now.
|
|
3999
|
+
kept = self._kept(rid)
|
|
3600
4000
|
ref = evals_dataset(snap)
|
|
3601
4001
|
body = None
|
|
3602
|
-
if ref is not None:
|
|
4002
|
+
if ref is not None and kept is None:
|
|
3603
4003
|
body = self.dataset(rid, raw=True)
|
|
3604
4004
|
if body is None:
|
|
3605
4005
|
# A run that never started kept no body: the dataset's, if it
|
|
@@ -3615,7 +4015,7 @@ class Queue:
|
|
|
3615
4015
|
_, err = worker_destinations(snap)
|
|
3616
4016
|
if err:
|
|
3617
4017
|
return None, (403, err)
|
|
3618
|
-
return self.submit(snap, body, rerun_of=rid), None
|
|
4018
|
+
return self.submit(snap, body, rerun_of=rid, groups=kept), None
|
|
3619
4019
|
|
|
3620
4020
|
def rerun_item(self, rid, index):
|
|
3621
4021
|
"""
|
|
@@ -3644,7 +4044,7 @@ class Queue:
|
|
|
3644
4044
|
return None, (403, err)
|
|
3645
4045
|
if NODE is None:
|
|
3646
4046
|
return None, (500, "node is not installed, so nothing can run")
|
|
3647
|
-
dataset, err = self.
|
|
4047
|
+
dataset, err = self._grading_args(run, rundir)
|
|
3648
4048
|
if err:
|
|
3649
4049
|
return None, (409, err)
|
|
3650
4050
|
plugins, err = self._plugin_args(run)
|
|
@@ -3677,9 +4077,12 @@ class Queue:
|
|
|
3677
4077
|
while len(results) <= index:
|
|
3678
4078
|
results.append(None)
|
|
3679
4079
|
results[index] = next((it for it in items if it and it.get("item") == index), items[0])
|
|
4080
|
+
# The worker read every item to reach its verdicts, so they are the
|
|
4081
|
+
# row's as it now stands.
|
|
4082
|
+
verdict, verdicts = report_verdicts(report)
|
|
3680
4083
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3681
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3682
|
-
(json.dumps(results), None, run["id"]))
|
|
4084
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4085
|
+
"WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
|
|
3683
4086
|
return self.get(run["id"]), None
|
|
3684
4087
|
|
|
3685
4088
|
def rescore_item(self, rid, index):
|
|
@@ -3707,7 +4110,7 @@ class Queue:
|
|
|
3707
4110
|
return None, (500, "node is not installed, so nothing can run")
|
|
3708
4111
|
rundir = self.dir / run["id"]
|
|
3709
4112
|
rundir.mkdir(parents=True, exist_ok=True)
|
|
3710
|
-
dataset, err = self.
|
|
4113
|
+
dataset, err = self._grading_args(run, rundir)
|
|
3711
4114
|
if err:
|
|
3712
4115
|
return None, (409, err)
|
|
3713
4116
|
plugins, err = self._plugin_args(run)
|
|
@@ -3742,9 +4145,12 @@ class Queue:
|
|
|
3742
4145
|
while len(results) <= index:
|
|
3743
4146
|
results.append(None)
|
|
3744
4147
|
results[index] = next((it for it in items if it and it.get("item") == index), items[0])
|
|
4148
|
+
# The worker read every item to reach its verdicts, so they are the
|
|
4149
|
+
# row's as it now stands.
|
|
4150
|
+
verdict, verdicts = report_verdicts(report)
|
|
3745
4151
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3746
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3747
|
-
(json.dumps(results), None, run["id"]))
|
|
4152
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4153
|
+
"WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
|
|
3748
4154
|
return self.get(run["id"]), None
|
|
3749
4155
|
|
|
3750
4156
|
# ---- the run --------------------------------------------------------
|
|
@@ -3820,7 +4226,7 @@ class Queue:
|
|
|
3820
4226
|
return self._finish(rid, "failed", error=err)
|
|
3821
4227
|
if NODE is None:
|
|
3822
4228
|
return self._finish(rid, "failed", error="node is not installed, so nothing can run")
|
|
3823
|
-
dataset, err = self.
|
|
4229
|
+
dataset, err = self._grading_args(run, rundir)
|
|
3824
4230
|
if err:
|
|
3825
4231
|
return self._finish(rid, "failed", error=err)
|
|
3826
4232
|
plugins, err = self._plugin_args(run)
|
|
@@ -3855,10 +4261,11 @@ class Queue:
|
|
|
3855
4261
|
report = self._read_report(rid)
|
|
3856
4262
|
items = report["run"]["items"] if report and isinstance(report.get("run"), dict) else None
|
|
3857
4263
|
if isinstance(items, list):
|
|
4264
|
+
verdict, verdicts = report_verdicts(report)
|
|
3858
4265
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3859
4266
|
if db.execute("SELECT 1 FROM queue WHERE id = ?", (rid,)).fetchone() is not None:
|
|
3860
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3861
|
-
(json.dumps(items), None, rid))
|
|
4267
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4268
|
+
"WHERE id = ?", (json.dumps(items), None, verdict, verdicts, rid))
|
|
3862
4269
|
run = self.get(rid)
|
|
3863
4270
|
# The watchdog may have failed the run while it was on the wire; a
|
|
3864
4271
|
# row that already left `running` is not this worker's to re-label.
|
|
@@ -3968,8 +4375,8 @@ class Queue:
|
|
|
3968
4375
|
"""The oldest queued run this server can read. One it cannot is never
|
|
3969
4376
|
started: nothing here would know what it asks for."""
|
|
3970
4377
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3971
|
-
for r in db.execute(
|
|
3972
|
-
"ORDER BY submitted_at, rowid").fetchall():
|
|
4378
|
+
for r in db.execute(self.SELECT + " WHERE q.status = 'queued' "
|
|
4379
|
+
"ORDER BY q.submitted_at, q.rowid").fetchall():
|
|
3973
4380
|
if not self._readable(self._row(r)):
|
|
3974
4381
|
continue
|
|
3975
4382
|
db.execute("UPDATE queue SET status = 'running', started_at = ? "
|
|
@@ -4015,10 +4422,89 @@ class Queue:
|
|
|
4015
4422
|
proc.kill()
|
|
4016
4423
|
|
|
4017
4424
|
|
|
4425
|
+
# The key a relayed request is sent with: the one it carries, or -- a page
|
|
4426
|
+
# that was handed KEY_HELD in place of a profile's key -- the one the store
|
|
4427
|
+
# holds for that profile.
|
|
4428
|
+
def relay_key(payload: dict) -> str:
|
|
4429
|
+
key = str(payload.get("key") or "").strip()
|
|
4430
|
+
if held(key):
|
|
4431
|
+
return STORE.held_key(key).strip() if STORE is not None else ""
|
|
4432
|
+
return key
|
|
4433
|
+
|
|
4434
|
+
|
|
4018
4435
|
# The worker-destination rules, at submit and again at dequeue: a run's
|
|
4019
4436
|
# profiles take their keys from the profiles store, by id, so the request
|
|
4020
4437
|
# carries none, and the relay's rules bind where they go. Returns (env_vars,
|
|
4021
4438
|
# None) or (None, a named refusal).
|
|
4439
|
+
# Export for CI (#254): a pipeline as a bundle a repository keeps and
|
|
4440
|
+
# `evals-lab run` runs with no lab. The page writes its text -- the core's
|
|
4441
|
+
# exportBundle: pipeline.yaml, profiles.yaml by slug, datasets/<slug>.json --
|
|
4442
|
+
# and the server adds what only it holds: every installed plugin, as a run
|
|
4443
|
+
# stamps them all, and, when asked, the Source's files under items/. A key is
|
|
4444
|
+
# refused rather than zipped: no Setup key, no token, no field named like
|
|
4445
|
+
# one, in any of it (the core's bundleProblems asks the same of the page's).
|
|
4446
|
+
BUNDLE_TEXT = re.compile(r"pipeline\.yaml|profiles\.yaml|datasets/[a-z0-9]+(?:-[a-z0-9]+)*\.json")
|
|
4447
|
+
BUNDLE_KEY_FIELD = re.compile(
|
|
4448
|
+
r'^[\s-]*"?((?:api[-_]?)?key|authorization|bearer|secret|password|token)"?\s*:', re.I | re.M)
|
|
4449
|
+
|
|
4450
|
+
|
|
4451
|
+
def bundle_problems(files, keys=()):
|
|
4452
|
+
"""Why [files] -- path to text -- cannot leave the lab, or ""."""
|
|
4453
|
+
for path, text in files.items():
|
|
4454
|
+
for key in keys:
|
|
4455
|
+
key = str(key or "").strip()
|
|
4456
|
+
if len(key) >= 4 and key in text:
|
|
4457
|
+
return f"{path} holds a Target profile's key, and a key never leaves the lab"
|
|
4458
|
+
if TOKEN_SHAPE.search(text):
|
|
4459
|
+
return f"{path} holds a token, and a key never leaves the lab"
|
|
4460
|
+
m = BUNDLE_KEY_FIELD.search(text)
|
|
4461
|
+
if m:
|
|
4462
|
+
return f"{path} holds a field named {m.group(1)}, and a key never leaves the lab"
|
|
4463
|
+
return ""
|
|
4464
|
+
|
|
4465
|
+
|
|
4466
|
+
def build_bundle(payload):
|
|
4467
|
+
"""The zip Export for CI downloads, from the page's [payload]:
|
|
4468
|
+
{ files: {path: text}, source: id or null, items: bool }. Returns
|
|
4469
|
+
(bytes, None) or (None, (status, one sentence))."""
|
|
4470
|
+
files = payload.get("files")
|
|
4471
|
+
if not isinstance(files, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in files.items()):
|
|
4472
|
+
return None, (400, "a bundle's files are text, by path")
|
|
4473
|
+
for path in files:
|
|
4474
|
+
if not BUNDLE_TEXT.fullmatch(path):
|
|
4475
|
+
return None, (400, f"{path!r} is not a file a bundle holds")
|
|
4476
|
+
if "pipeline.yaml" not in files or "profiles.yaml" not in files:
|
|
4477
|
+
return None, (400, "a bundle holds pipeline.yaml and profiles.yaml")
|
|
4478
|
+
stored = []
|
|
4479
|
+
if STORE is not None:
|
|
4480
|
+
stored = ((STORE.all().get("promptlab.profiles") or {}).get("body") or {}).get("list") or []
|
|
4481
|
+
why = bundle_problems(files, [p.get("key") for p in stored if isinstance(p, dict)])
|
|
4482
|
+
if why:
|
|
4483
|
+
return None, (400, why)
|
|
4484
|
+
items = []
|
|
4485
|
+
if payload.get("items"):
|
|
4486
|
+
sid = payload.get("source")
|
|
4487
|
+
found = SOURCES.item_paths(sid) if isinstance(sid, str) and SOURCES is not None else None
|
|
4488
|
+
if found is None:
|
|
4489
|
+
return None, (404, "no such source")
|
|
4490
|
+
items = found
|
|
4491
|
+
buf = io.BytesIO()
|
|
4492
|
+
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
4493
|
+
for path, text in sorted(files.items()):
|
|
4494
|
+
zf.writestr(f"evals/{path}", text)
|
|
4495
|
+
if PLUGINS is not None:
|
|
4496
|
+
for p in PLUGINS.stamp():
|
|
4497
|
+
root = PLUGINS.dir / p["id"] / p["version"]
|
|
4498
|
+
for f in sorted(root.rglob("*")):
|
|
4499
|
+
if f.is_file():
|
|
4500
|
+
zf.write(f, f"evals/plugins/{p['id']}/{p['version']}/{f.relative_to(root).as_posix()}")
|
|
4501
|
+
for name, path in items:
|
|
4502
|
+
if not path.is_file():
|
|
4503
|
+
return None, (404, f"{name!r} is not in that Source")
|
|
4504
|
+
zf.write(path, f"evals/items/{name}")
|
|
4505
|
+
return buf.getvalue(), None
|
|
4506
|
+
|
|
4507
|
+
|
|
4022
4508
|
def worker_destinations(run: dict):
|
|
4023
4509
|
table = run.get("profiles") if isinstance(run, dict) else None
|
|
4024
4510
|
if not isinstance(table, dict):
|
|
@@ -4046,7 +4532,7 @@ def worker_destinations(run: dict):
|
|
|
4046
4532
|
why = allowed(base, key)
|
|
4047
4533
|
if why:
|
|
4048
4534
|
return None, f"Target profile {name}: {why}"
|
|
4049
|
-
env[key_var(pid)] = key
|
|
4535
|
+
env[key_var(pid, conn.get("slug"))] = key
|
|
4050
4536
|
return env, None
|
|
4051
4537
|
|
|
4052
4538
|
|
|
@@ -4068,15 +4554,17 @@ def worker_destinations(run: dict):
|
|
|
4068
4554
|
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
4069
4555
|
# 11: `tests` are `evals`; nothing in an eval changes.
|
|
4070
4556
|
# 12: a Contains metric's Ignore case holds item by item too, kept as written.
|
|
4071
|
-
|
|
4557
|
+
# 13: an eval is a link to an eval group, or a group of the pipeline's own,
|
|
4558
|
+
# and the document has an overall pass rule (docs/pipeline-model.md §17).
|
|
4559
|
+
PIPELINE_VERSION = 13
|
|
4072
4560
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
4073
4561
|
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
4074
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
|
4562
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
|
|
4075
4563
|
TARGET_CAP = 4
|
|
4076
4564
|
# Target steps whose words the Prompt library does not record as a use: they
|
|
4077
4565
|
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
4078
4566
|
UNRECORDED_STEPS = {"echo"}
|
|
4079
|
-
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
|
|
4567
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "pass", "profiles", "comment", "plugins")
|
|
4080
4568
|
|
|
4081
4569
|
|
|
4082
4570
|
def content_of(doc):
|
|
@@ -4122,14 +4610,17 @@ BUILTIN_CONNECTION_TYPES = dict(CONNECTION_TYPES)
|
|
|
4122
4610
|
BUILTIN_LOCAL = set(LOCAL_CONNECTIONS)
|
|
4123
4611
|
BUILTIN_CHAT_PATHS = dict(CONNECTION_CHAT_PATHS)
|
|
4124
4612
|
BUILTIN_AUTH = dict(CONNECTION_AUTH)
|
|
4125
|
-
CONNECTION_FIELDS = ("name", "url", "model", "type", "temperature", "px", "format",
|
|
4613
|
+
CONNECTION_FIELDS = ("name", "slug", "url", "model", "type", "temperature", "px", "format",
|
|
4126
4614
|
"quality", "options")
|
|
4127
4615
|
LOOKS_LIKE_A_KEY = re.compile(r"^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$",
|
|
4128
4616
|
re.IGNORECASE)
|
|
4129
4617
|
PROFILE_ID = re.compile(r"[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?")
|
|
4618
|
+
# evals-core.ts's SLUG and SLUG_MAX: what a key's variable is spelt from.
|
|
4619
|
+
PROFILE_SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*")
|
|
4620
|
+
SLUG_MAX = 32
|
|
4130
4621
|
|
|
4131
4622
|
|
|
4132
|
-
def _fields(obj, at, allowed_fields, bad, key_from="
|
|
4623
|
+
def _fields(obj, at, allowed_fields, bad, key_from="EVALSLAB_API_KEY_<ID>"):
|
|
4133
4624
|
for k in obj:
|
|
4134
4625
|
if k in allowed_fields:
|
|
4135
4626
|
continue
|
|
@@ -4200,7 +4691,12 @@ CONNECTION_LISTS = {"http": http_endpoint_test}
|
|
|
4200
4691
|
|
|
4201
4692
|
|
|
4202
4693
|
def connection_problems(pid, conn, at, bad):
|
|
4203
|
-
|
|
4694
|
+
slug = conn.get("slug")
|
|
4695
|
+
if slug is not None and not (isinstance(slug, str) and PROFILE_SLUG.fullmatch(slug) and len(slug) <= SLUG_MAX):
|
|
4696
|
+
bad.append(f"{at}: slug has to be lowercase letters and digits, with hyphens between, at most {SLUG_MAX}")
|
|
4697
|
+
slug = None
|
|
4698
|
+
var = key_var(pid, slug)
|
|
4699
|
+
_fields(conn, at, CONNECTION_FIELDS, bad, var)
|
|
4204
4700
|
ctype = conn.get("type")
|
|
4205
4701
|
if not isinstance(ctype, str) or ctype not in CONNECTION_TYPES:
|
|
4206
4702
|
bad.append(f"{at}: type has to be one of {', '.join(CONNECTION_TYPES)}")
|
|
@@ -4211,7 +4707,7 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4211
4707
|
# Which settings there are is the type's to say; a key among them
|
|
4212
4708
|
# is the server's.
|
|
4213
4709
|
allowed = list(CONNECTION_TYPES.get(ctype, ())) if ctype else list(options)
|
|
4214
|
-
_fields(options, f"{at}'s options", allowed, bad,
|
|
4710
|
+
_fields(options, f"{at}'s options", allowed, bad, var)
|
|
4215
4711
|
else:
|
|
4216
4712
|
bad.append(f"{at}: options has to be an object")
|
|
4217
4713
|
url = conn.get("url") or ""
|
|
@@ -4222,7 +4718,7 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4222
4718
|
parts = urllib.parse.urlsplit(url if "://" in url else "http://" + url)
|
|
4223
4719
|
if parts.username or parts.password:
|
|
4224
4720
|
bad.append(f"{at} has a key in its address, and a key never goes in a pipeline "
|
|
4225
|
-
f"— ${
|
|
4721
|
+
f"— ${var} supplies it")
|
|
4226
4722
|
for why in CONNECTION_CHECKS.get(ctype, lambda _c: [])(conn):
|
|
4227
4723
|
bad.append(f"{at} {why}")
|
|
4228
4724
|
# llama.cpp runs its own llama-server: a hosted address is refused, as the
|
|
@@ -4233,16 +4729,33 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4233
4729
|
bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
|
|
4234
4730
|
|
|
4235
4731
|
|
|
4732
|
+
def eval_group_ref(t):
|
|
4733
|
+
"""The Library group an eval reads the cases of, as the dict itself (so a
|
|
4734
|
+
caller may stamp it in place), or None: version 13's link (`group`) or a
|
|
4735
|
+
private group's `casesFrom`, or an earlier version's `dataset`
|
|
4736
|
+
(evals-core.ts casesRef)."""
|
|
4737
|
+
if not isinstance(t, dict):
|
|
4738
|
+
return None
|
|
4739
|
+
if t.get("type") == "group":
|
|
4740
|
+
if isinstance(t.get("group"), dict):
|
|
4741
|
+
return t["group"]
|
|
4742
|
+
own = t.get("own")
|
|
4743
|
+
return own["casesFrom"] if isinstance(own, dict) and isinstance(own.get("casesFrom"), dict) else None
|
|
4744
|
+
return t["dataset"] if isinstance(t.get("dataset"), dict) else None
|
|
4745
|
+
|
|
4746
|
+
|
|
4236
4747
|
def evals_dataset(doc):
|
|
4237
|
-
"""The
|
|
4238
|
-
first eval that names one. A run grades against one
|
|
4748
|
+
"""The Library group a document's evals read the cases of, or None: the
|
|
4749
|
+
first eval that names one. A run grades against one (the core's
|
|
4239
4750
|
validatePipeline says so). Reads a stored document of any shape --
|
|
4240
|
-
version 11's `evals`, the `tests` before it, version
|
|
4241
|
-
test before that -- since rows keep the document they
|
|
4751
|
+
version 13's links, version 11's `evals`, the `tests` before it, version
|
|
4752
|
+
5's list or the one test before that -- since rows keep the document they
|
|
4753
|
+
were submitted with."""
|
|
4242
4754
|
evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
|
|
4243
4755
|
for t in evals if isinstance(evals, list) else [evals]:
|
|
4244
|
-
|
|
4245
|
-
|
|
4756
|
+
ref = eval_group_ref(t)
|
|
4757
|
+
if ref is not None:
|
|
4758
|
+
return ref
|
|
4246
4759
|
return None
|
|
4247
4760
|
|
|
4248
4761
|
|
|
@@ -4325,6 +4838,7 @@ def run_problems(run):
|
|
|
4325
4838
|
table = run.get("profiles")
|
|
4326
4839
|
if not isinstance(table, dict):
|
|
4327
4840
|
return bad + ["profiles has to be an object of id → connection"]
|
|
4841
|
+
spelt = {}
|
|
4328
4842
|
for pid, conn in table.items():
|
|
4329
4843
|
at = f"profile {pid}"
|
|
4330
4844
|
if not PROFILE_ID.fullmatch(str(pid)):
|
|
@@ -4333,6 +4847,11 @@ def run_problems(run):
|
|
|
4333
4847
|
if not isinstance(conn, dict):
|
|
4334
4848
|
bad.append(f"{at} has to be an object")
|
|
4335
4849
|
continue
|
|
4850
|
+
# Two profiles spelling one variable would hand one the other's key.
|
|
4851
|
+
var = key_var(pid, conn.get("slug") if isinstance(conn.get("slug"), str) else None)
|
|
4852
|
+
if var in spelt:
|
|
4853
|
+
bad.append(f"profiles {spelt[var]} and {pid} both take their key from ${var}")
|
|
4854
|
+
spelt[var] = pid
|
|
4336
4855
|
connection_problems(pid, conn, at, bad)
|
|
4337
4856
|
targets = run.get("targets")
|
|
4338
4857
|
if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
|
|
@@ -4550,9 +5069,23 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4550
5069
|
|
|
4551
5070
|
# ---- routes ---------------------------------------------------------
|
|
4552
5071
|
|
|
5072
|
+
# The eval groups' routes (docs/pipeline-model.md §17) are the datasets'
|
|
5073
|
+
# under their new name: /api/datasets goes on answering the same rows, so
|
|
5074
|
+
# dataset-diff.js and a script written against it keep working. A list
|
|
5075
|
+
# asked for by the new name is `groups`.
|
|
5076
|
+
GROUPS_ROUTE = "/api/eval-groups"
|
|
5077
|
+
grouped = False
|
|
5078
|
+
|
|
5079
|
+
def _alias(self):
|
|
5080
|
+
self.grouped = self.path == self.GROUPS_ROUTE or self.path.startswith((self.GROUPS_ROUTE + "/",
|
|
5081
|
+
self.GROUPS_ROUTE + "?"))
|
|
5082
|
+
if self.grouped:
|
|
5083
|
+
self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
|
|
5084
|
+
|
|
4553
5085
|
def do_GET(self):
|
|
4554
5086
|
if not self._authorised():
|
|
4555
5087
|
return
|
|
5088
|
+
self._alias()
|
|
4556
5089
|
path = self.path.split("?", 1)[0]
|
|
4557
5090
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
4558
5091
|
# Content and Evals are shared, and one to four scenarios run over the
|
|
@@ -4599,8 +5132,16 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4599
5132
|
except ValueError:
|
|
4600
5133
|
return self._json(400, {"error": "limit has to be a number"})
|
|
4601
5134
|
before = (query.get("before") or [None])[0]
|
|
4602
|
-
|
|
5135
|
+
before_id = (query.get("beforeId") or [None])[0]
|
|
5136
|
+
runs, more = QUEUE.list(limit, before, before_id, full=(query.get("full") or [""])[0] == "1")
|
|
4603
5137
|
return self._json(200, {"runs": runs, "more": more})
|
|
5138
|
+
if path.startswith("/api/queue/") and path.endswith("/groups") and path.count("/") == 4:
|
|
5139
|
+
# The bodies of the eval groups a run grades with, by `<id>@<n>`
|
|
5140
|
+
# (§17); null for a run that kept none, as /dataset answers.
|
|
5141
|
+
run_id = path.split("/")[3]
|
|
5142
|
+
if QUEUE is None or QUEUE.get(run_id) is None:
|
|
5143
|
+
return self._send(404, b"not found", "text/plain")
|
|
5144
|
+
return self._json(200, QUEUE.groups(run_id))
|
|
4604
5145
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
4605
5146
|
# The dataset body a graded run was submitted against, which is
|
|
4606
5147
|
# what its verdicts were graded by; the list never carries it.
|
|
@@ -4762,6 +5303,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4762
5303
|
def do_DELETE(self):
|
|
4763
5304
|
if not self._authorised() or not self._from_this_page():
|
|
4764
5305
|
return
|
|
5306
|
+
self._alias()
|
|
4765
5307
|
path = self.path.split("?", 1)[0]
|
|
4766
5308
|
if path == "/api/connections/google":
|
|
4767
5309
|
if CONNECTIONS is None:
|
|
@@ -4829,6 +5371,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4829
5371
|
def do_PATCH(self):
|
|
4830
5372
|
if not self._authorised() or not self._from_this_page():
|
|
4831
5373
|
return
|
|
5374
|
+
self._alias()
|
|
4832
5375
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4833
5376
|
if len(parts) != 4 or parts[1] != "api":
|
|
4834
5377
|
return self._send(404, b"not found", "text/plain")
|
|
@@ -4859,6 +5402,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4859
5402
|
def do_PUT(self):
|
|
4860
5403
|
if not self._authorised() or not self._from_this_page():
|
|
4861
5404
|
return
|
|
5405
|
+
self._alias()
|
|
4862
5406
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4863
5407
|
if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
|
|
4864
5408
|
return self._prompts_put(parts[3])
|
|
@@ -4921,6 +5465,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4921
5465
|
return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
|
|
4922
5466
|
|
|
4923
5467
|
def _post(self):
|
|
5468
|
+
self._alias()
|
|
4924
5469
|
path = self.path.split("?", 1)[0]
|
|
4925
5470
|
if path == "/api/state":
|
|
4926
5471
|
return self._state_write()
|
|
@@ -4940,6 +5485,12 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4940
5485
|
return self._prompts_post(path)
|
|
4941
5486
|
if path.startswith("/api/sources"):
|
|
4942
5487
|
return self._sources_post(path)
|
|
5488
|
+
if path == "/api/bundle":
|
|
5489
|
+
data, err = build_bundle(self._payload() or {})
|
|
5490
|
+
if err:
|
|
5491
|
+
return self._json(err[0], {"error": err[1]})
|
|
5492
|
+
return self._send(200, data, "application/zip",
|
|
5493
|
+
(("Content-Disposition", 'attachment; filename="evals.zip"'),))
|
|
4943
5494
|
payload = self._payload()
|
|
4944
5495
|
if payload is None:
|
|
4945
5496
|
return self._json(400, {"error": "bad body"})
|
|
@@ -5012,7 +5563,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5012
5563
|
|
|
5013
5564
|
def _models_list(self, payload):
|
|
5014
5565
|
base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
|
|
5015
|
-
key =
|
|
5566
|
+
key = relay_key(payload)
|
|
5016
5567
|
ctype = str(payload.get("type") or "")
|
|
5017
5568
|
# A type that lists some other way (an HTTP endpoint: its own test
|
|
5018
5569
|
# path, key header and headers) says where and how.
|
|
@@ -5063,7 +5614,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5063
5614
|
# and how the key travels, per the type -- the same mirror the models
|
|
5064
5615
|
# list uses, so a client cannot point the relay at a path of its own.
|
|
5065
5616
|
base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
|
|
5066
|
-
key =
|
|
5617
|
+
key = relay_key(payload)
|
|
5067
5618
|
if not header_safe(key):
|
|
5068
5619
|
return self._json(400, {"error": "That key has characters that "
|
|
5069
5620
|
"cannot be sent in a header, so "
|
|
@@ -5099,6 +5650,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5099
5650
|
for n, d in docs.items()})
|
|
5100
5651
|
if stale is not None:
|
|
5101
5652
|
return self._json(409, {"stale": stale})
|
|
5653
|
+
# A pin names a group's version, so the group keeps that version from
|
|
5654
|
+
# the moment a pipeline pins it (§17).
|
|
5655
|
+
if "promptlab.workflows" in docs and DATASETS is not None:
|
|
5656
|
+
DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
|
|
5102
5657
|
return self._json(200, {"versions": versions})
|
|
5103
5658
|
|
|
5104
5659
|
# ---- The run queue (#530) ----------------------------------------------
|
|
@@ -5137,20 +5692,28 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5137
5692
|
# The plugins it runs under, as installed now: the server's to say,
|
|
5138
5693
|
# whatever the document claimed.
|
|
5139
5694
|
run["plugins"] = PLUGINS.stamp() if PLUGINS is not None else []
|
|
5140
|
-
#
|
|
5141
|
-
#
|
|
5142
|
-
#
|
|
5143
|
-
|
|
5144
|
-
|
|
5145
|
-
|
|
5146
|
-
|
|
5147
|
-
|
|
5695
|
+
# Each Library group the evals read, resolved once (§17): a pinned
|
|
5696
|
+
# link to its pin, anything else to the group's newest version. The
|
|
5697
|
+
# reference records the version, `n`, and its body's fingerprint, and
|
|
5698
|
+
# the body is kept with the row under `<id>@<n>`: the worker grades
|
|
5699
|
+
# with that and nothing else, and the version is frozen from now on.
|
|
5700
|
+
groups = {}
|
|
5701
|
+
for t in run["evals"]:
|
|
5702
|
+
ref = eval_group_ref(t)
|
|
5703
|
+
if ref is None:
|
|
5704
|
+
continue
|
|
5705
|
+
pin = t.get("pin") if t.get("type") == "group" and t.get("group") is ref else None
|
|
5706
|
+
got = DATASETS.resolve(ref.get("id"), pin if type(pin) is int else None) if DATASETS else None
|
|
5707
|
+
if got is None:
|
|
5148
5708
|
return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
|
|
5149
|
-
|
|
5150
|
-
|
|
5151
|
-
|
|
5152
|
-
|
|
5153
|
-
|
|
5709
|
+
n, body = got
|
|
5710
|
+
if n is None:
|
|
5711
|
+
return self._json(400, {"error": f"the run cannot be queued: {body}"})
|
|
5712
|
+
ref["n"], ref["version"] = n, fingerprint(body)
|
|
5713
|
+
groups[group_key(ref)] = body
|
|
5714
|
+
# The worker reads a body per group now (#233), so a run may link
|
|
5715
|
+
# several; each body is kept with the row under `<id>@<n>`.
|
|
5716
|
+
return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
|
|
5154
5717
|
|
|
5155
5718
|
# ---- Datasets ----------------------------------------------------------
|
|
5156
5719
|
# Rows in the store (Datasets above). The page reads one by id; an export
|
|
@@ -5244,9 +5807,16 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5244
5807
|
DATASETS.lazy_trash()
|
|
5245
5808
|
parts = path.split("/")
|
|
5246
5809
|
if len(parts) == 3:
|
|
5247
|
-
return self._json(200, {"datasets": DATASETS.list()})
|
|
5810
|
+
return self._json(200, {"groups" if self.grouped else "datasets": DATASETS.list()})
|
|
5248
5811
|
if len(parts) == 4 and parts[3] == "export":
|
|
5249
|
-
return self._download(DATASETS.export_all(), "
|
|
5812
|
+
return self._download(DATASETS.export_all(), "eval-groups", "all")
|
|
5813
|
+
if len(parts) == 5 and parts[4] == "versions":
|
|
5814
|
+
got = DATASETS.versions(parts[3])
|
|
5815
|
+
return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
|
|
5816
|
+
if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
|
|
5817
|
+
got = DATASETS.version_body(parts[3], int(parts[5]))
|
|
5818
|
+
return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
|
|
5819
|
+
else self._send(404, b"not found", "text/plain")
|
|
5250
5820
|
if len(parts) == 4 and parts[3] == "archived-rules":
|
|
5251
5821
|
return self._json(200, {"rules": DATASETS.archived_rules()})
|
|
5252
5822
|
if len(parts) == 4:
|
|
@@ -5256,7 +5826,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5256
5826
|
doc = DATASETS.export(parts[3])
|
|
5257
5827
|
if doc is None:
|
|
5258
5828
|
return self._send(404, b"not found", "text/plain")
|
|
5259
|
-
return self._download(doc, "
|
|
5829
|
+
return self._download(doc, "eval-group", doc["group"]["name"])
|
|
5260
5830
|
return self._send(404, b"not found", "text/plain")
|
|
5261
5831
|
|
|
5262
5832
|
def _download(self, doc, kind, name):
|
|
@@ -5284,6 +5854,12 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5284
5854
|
if err:
|
|
5285
5855
|
return self._json(err[0], {"error": err[1]})
|
|
5286
5856
|
return self._json(200, dataset)
|
|
5857
|
+
if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
|
|
5858
|
+
self._payload()
|
|
5859
|
+
dataset, err = DATASETS.restore_version(parts[3], int(parts[5]))
|
|
5860
|
+
if err:
|
|
5861
|
+
return self._json(err[0], {"error": err[1]})
|
|
5862
|
+
return self._json(200, dataset)
|
|
5287
5863
|
if len(parts) == 4 and parts[3] == "import":
|
|
5288
5864
|
doc, err = self._dataset_payload(DATASET_IMPORT_CAP)
|
|
5289
5865
|
if err:
|