evals-lab 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +169 -145
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +531 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +672 -61
- package/lab/run-evals.js +195 -57
- package/lab/server.py +681 -127
- package/lab/web/dist/assets/gallery-BFf9vis6.js +3 -0
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/main-DeeRLWnO.css +1 -0
- package/lab/web/dist/assets/main-LT0U2TYF.js +21 -0
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +59 -0
- package/lab/web/dist/assets/tokens-lq45aAPS.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/main-DyDG-V9N.js +0 -21
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/server.py
CHANGED
|
@@ -560,6 +560,48 @@ QUEUE_WAIT = 1.0
|
|
|
560
560
|
WATCH_EVERY = 5.0
|
|
561
561
|
|
|
562
562
|
|
|
563
|
+
# A Target profile's key is written from the page and never read back by it
|
|
564
|
+
# (#258): a client holding the lab's password -- CI's runners and their logs
|
|
565
|
+
# among them -- is handed no key. Where a profile holds one, what is served
|
|
566
|
+
# (/api/state, the page's carried copy, a refused write's current copy) holds
|
|
567
|
+
# KEY_HELD and the profile's id instead, and a write that brings that back
|
|
568
|
+
# keeps the key the store holds for the id: the profile's own, or the one it
|
|
569
|
+
# was cloned or restored from. KEY_HELD starts with a character no key can
|
|
570
|
+
# (header_safe), so a pasted key is never taken for it.
|
|
571
|
+
KEY_HELD = "\u2022held:"
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def held(key) -> bool:
|
|
575
|
+
return isinstance(key, str) and key.startswith(KEY_HELD)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def profiles_in(name: str, body):
|
|
579
|
+
"""Every profile a synced document holds: the profiles store's list, and
|
|
580
|
+
each profile version's copy."""
|
|
581
|
+
if not isinstance(body, dict):
|
|
582
|
+
return
|
|
583
|
+
if name == "promptlab.profiles":
|
|
584
|
+
for p in body.get("list") or []:
|
|
585
|
+
if isinstance(p, dict):
|
|
586
|
+
yield p
|
|
587
|
+
elif name == "promptlab.versions":
|
|
588
|
+
for kept in (body.get("profile") or {}).values() if isinstance(body.get("profile"), dict) else ():
|
|
589
|
+
for v in kept if isinstance(kept, list) else ():
|
|
590
|
+
if isinstance(v, dict) and isinstance(v.get("doc"), dict):
|
|
591
|
+
yield v["doc"]
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def hide_keys(docs: dict) -> dict:
|
|
595
|
+
out = {}
|
|
596
|
+
for name, d in docs.items():
|
|
597
|
+
d = json.loads(json.dumps(d))
|
|
598
|
+
for p in profiles_in(name, d.get("body")):
|
|
599
|
+
if isinstance(p.get("key"), str) and p["key"] and not held(p["key"]):
|
|
600
|
+
p["key"] = KEY_HELD + str(p.get("id") or "")
|
|
601
|
+
out[name] = d
|
|
602
|
+
return out
|
|
603
|
+
|
|
604
|
+
|
|
563
605
|
class Store:
|
|
564
606
|
"""
|
|
565
607
|
Documents by name, each with a version that goes up by one per write.
|
|
@@ -579,6 +621,11 @@ class Store:
|
|
|
579
621
|
db.execute("CREATE TABLE IF NOT EXISTS docs (name TEXT PRIMARY KEY, "
|
|
580
622
|
"version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
|
|
581
623
|
db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
|
|
624
|
+
# The key of a profile a write took out, by id, for TRASH_SECONDS:
|
|
625
|
+
# Undo puts the profile back holding KEY_HELD, and this is what
|
|
626
|
+
# it holds.
|
|
627
|
+
db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
|
|
628
|
+
"key TEXT NOT NULL, at REAL NOT NULL)")
|
|
582
629
|
# History was a document, capped at what a browser could hold. The
|
|
583
630
|
# first start with the table moves what that document had into it,
|
|
584
631
|
# once, and drops the document so the page stops carrying it.
|
|
@@ -600,8 +647,30 @@ class Store:
|
|
|
600
647
|
|
|
601
648
|
def served(self) -> dict:
|
|
602
649
|
"""What a browser is handed: the SYNCED documents only, so a retired
|
|
603
|
-
key's rows stay in the store without reaching a page again
|
|
604
|
-
|
|
650
|
+
key's rows stay in the store without reaching a page again, and no
|
|
651
|
+
profile's key (KEY_HELD)."""
|
|
652
|
+
return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
|
|
653
|
+
|
|
654
|
+
def _keyring(self, db, now: dict) -> dict:
|
|
655
|
+
"""Each profile id's key as the store holds it: its profile's, else a
|
|
656
|
+
dropped one's, else its newest version's."""
|
|
657
|
+
ring = {}
|
|
658
|
+
for p in profiles_in("promptlab.versions", now.get("promptlab.versions", {}).get("body")):
|
|
659
|
+
pid, key = str(p.get("id") or ""), p.get("key")
|
|
660
|
+
if pid not in ring and isinstance(key, str) and key and not held(key):
|
|
661
|
+
ring[pid] = key
|
|
662
|
+
db.execute("DELETE FROM dropped_keys WHERE at < ?", (time.time() - TRASH_SECONDS,))
|
|
663
|
+
ring.update(db.execute("SELECT id, key FROM dropped_keys").fetchall())
|
|
664
|
+
for p in profiles_in("promptlab.profiles", now.get("promptlab.profiles", {}).get("body")):
|
|
665
|
+
key = p.get("key")
|
|
666
|
+
if isinstance(key, str) and key and not held(key):
|
|
667
|
+
ring[str(p.get("id") or "")] = key
|
|
668
|
+
return ring
|
|
669
|
+
|
|
670
|
+
def held_key(self, key: str) -> str:
|
|
671
|
+
"""The key KEY_HELD stands for, or "" when the store holds none."""
|
|
672
|
+
with self.lock, closing(sqlite3.connect(self.path)) as db, db:
|
|
673
|
+
return self._keyring(db, self._rows(db)).get(key[len(KEY_HELD):], "")
|
|
605
674
|
|
|
606
675
|
def write(self, docs: dict):
|
|
607
676
|
"""
|
|
@@ -614,9 +683,23 @@ class Store:
|
|
|
614
683
|
have = lambda n: now.get(n, {"version": 0, "body": None})
|
|
615
684
|
stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
|
|
616
685
|
if stale:
|
|
617
|
-
return None, stale
|
|
686
|
+
return None, hide_keys(stale)
|
|
618
687
|
at = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
619
688
|
with db:
|
|
689
|
+
ring = self._keyring(db, now)
|
|
690
|
+
for n, d in docs.items():
|
|
691
|
+
for p in profiles_in(n, d["body"]):
|
|
692
|
+
if held(p.get("key")):
|
|
693
|
+
p["key"] = ring.get(p["key"][len(KEY_HELD):], "")
|
|
694
|
+
if "promptlab.profiles" in docs:
|
|
695
|
+
kept = {str(p.get("id") or "") for p in profiles_in(
|
|
696
|
+
"promptlab.profiles", docs["promptlab.profiles"]["body"])}
|
|
697
|
+
db.executemany("DELETE FROM dropped_keys WHERE id = ?", [(i,) for i in kept])
|
|
698
|
+
db.executemany(
|
|
699
|
+
"INSERT OR REPLACE INTO dropped_keys (id, key, at) VALUES (?, ?, ?)",
|
|
700
|
+
[(str(p.get("id") or ""), p["key"], time.time())
|
|
701
|
+
for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
|
|
702
|
+
if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
|
|
620
703
|
for n, d in docs.items():
|
|
621
704
|
db.execute(
|
|
622
705
|
"INSERT INTO docs (name, version, body, updated_at) VALUES (?, ?, ?, ?) "
|
|
@@ -1177,6 +1260,18 @@ class Sources:
|
|
|
1177
1260
|
return None, (500, "the files could not be copied")
|
|
1178
1261
|
return self.get(sid), None
|
|
1179
1262
|
|
|
1263
|
+
def item_paths(self, sid):
|
|
1264
|
+
"""Every file of a Source, as (name, path on disk) in its own order,
|
|
1265
|
+
or None when there is no such Source."""
|
|
1266
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1267
|
+
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1268
|
+
if r is None:
|
|
1269
|
+
return None
|
|
1270
|
+
if r[0]:
|
|
1271
|
+
return [(f["name"], SAMPLES / f["name"]) for f in self._sample_files()]
|
|
1272
|
+
return [(row[0], self.dir / sid / row[0]) for row in db.execute(
|
|
1273
|
+
"SELECT name FROM source_files WHERE source = ? ORDER BY name", (sid,))]
|
|
1274
|
+
|
|
1180
1275
|
def zip_files(self, sid, names):
|
|
1181
1276
|
"""The bytes of a stdlib zipfile holding exactly the named files, in
|
|
1182
1277
|
the order named, or an error. Read under the store's lock so the file
|
|
@@ -1362,15 +1457,23 @@ DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cas
|
|
|
1362
1457
|
# 5 was told by its `source` alone, and earlier ones by neither.
|
|
1363
1458
|
DATASET_BODY_VERSION = 7
|
|
1364
1459
|
DATASET_NAME_MAX = 80
|
|
1365
|
-
# The file forms Export writes and Import reads. Export writes
|
|
1366
|
-
#
|
|
1367
|
-
#
|
|
1368
|
-
#
|
|
1369
|
-
|
|
1370
|
-
|
|
1460
|
+
# The file forms Export writes and Import reads. Export writes an eval group
|
|
1461
|
+
# at version 7 (docs/pipeline-model.md §17); Import reads that, and a dataset
|
|
1462
|
+
# file of versions 1 to 7, upgraded, and refuses anything else, as a pipeline
|
|
1463
|
+
# of another version is refused. Versions 1 to 3 carried a prompt, which an
|
|
1464
|
+
# import gives to the Prompt library.
|
|
1465
|
+
EXPORT_ONE = "evals-lab/eval-group"
|
|
1466
|
+
EXPORT_ALL = "evals-lab/eval-groups"
|
|
1467
|
+
DATASET_ONE = "evals-lab/dataset"
|
|
1468
|
+
DATASET_ALL = "evals-lab/datasets"
|
|
1371
1469
|
EXPORT_VERSION = 7
|
|
1372
1470
|
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
|
|
1471
|
+
# Each file form, and the key its one entry or its list sits under.
|
|
1472
|
+
EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
|
|
1373
1473
|
SCORING_MODES = ("all", "weighted")
|
|
1474
|
+
# A group's newest version is edited in place by a save within this long of
|
|
1475
|
+
# the last, as a prompt's is (PROMPT_IDLE_SECONDS).
|
|
1476
|
+
GROUP_IDLE_SECONDS = float(os.environ.get("GROUP_IDLE_SECONDS", "30"))
|
|
1374
1477
|
|
|
1375
1478
|
|
|
1376
1479
|
def blank_dataset() -> dict:
|
|
@@ -1574,6 +1677,30 @@ def fingerprint(body: dict) -> str:
|
|
|
1574
1677
|
return hashlib.sha256(text.encode("utf-8")).hexdigest()[:CASE_SET_LEN]
|
|
1575
1678
|
|
|
1576
1679
|
|
|
1680
|
+
def pins_in(workflows) -> set:
|
|
1681
|
+
"""(group id, version) for every link a stored pipeline pins: the
|
|
1682
|
+
promptlab.workflows body, each pipeline as its `work`. Only version 13
|
|
1683
|
+
links pin, and a pipeline of an earlier version has none."""
|
|
1684
|
+
out = set()
|
|
1685
|
+
listed = workflows.get("list") if isinstance(workflows, dict) else None
|
|
1686
|
+
for w in listed if isinstance(listed, list) else []:
|
|
1687
|
+
work = w.get("work") if isinstance(w, dict) else None
|
|
1688
|
+
evals = work.get("evals") if isinstance(work, dict) else None
|
|
1689
|
+
for t in evals if isinstance(evals, list) else []:
|
|
1690
|
+
if (isinstance(t, dict) and t.get("type") == "group" and isinstance(t.get("group"), dict)
|
|
1691
|
+
and isinstance(t["group"].get("id"), str) and type(t.get("pin")) is int):
|
|
1692
|
+
out.add((t["group"]["id"], t["pin"]))
|
|
1693
|
+
return out
|
|
1694
|
+
|
|
1695
|
+
|
|
1696
|
+
def group_key(ref) -> str:
|
|
1697
|
+
"""The key a run keeps a group's body under: `<id>@<n>`, or the id alone
|
|
1698
|
+
for a reference from before the lab numbered versions."""
|
|
1699
|
+
n = ref.get("n") if isinstance(ref, dict) else None
|
|
1700
|
+
gid = ref.get("id") if isinstance(ref, dict) else None
|
|
1701
|
+
return f"{gid}@{n}" if type(n) is int else str(gid)
|
|
1702
|
+
|
|
1703
|
+
|
|
1577
1704
|
def unique_dataset_name(name: str, taken: set) -> str:
|
|
1578
1705
|
"""`Receipts` again becomes `Receipts (2)`, compared without case."""
|
|
1579
1706
|
lower = {t.lower() for t in taken}
|
|
@@ -1944,6 +2071,16 @@ class Datasets:
|
|
|
1944
2071
|
# the body it was typed as is kept, as the rules were.
|
|
1945
2072
|
db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
|
|
1946
2073
|
"dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
2074
|
+
# Every version of each eval group, numbered from 1, for a link to
|
|
2075
|
+
# pin and a run to name (docs/pipeline-model.md §17). `ran` says a
|
|
2076
|
+
# run graded with it, which freezes it for good; a pin freezes it
|
|
2077
|
+
# while a stored pipeline holds the pin. Version 1 is added from
|
|
2078
|
+
# the row's body at its first save, pin or run, so a group nobody
|
|
2079
|
+
# touches holds what it held.
|
|
2080
|
+
db.execute("CREATE TABLE IF NOT EXISTS eval_group_versions ("
|
|
2081
|
+
"group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
|
|
2082
|
+
"created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
|
|
2083
|
+
"ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
|
|
1947
2084
|
# Rows from an earlier version are converted once, in place: a
|
|
1948
2085
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
1949
2086
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -1992,16 +2129,26 @@ class Datasets:
|
|
|
1992
2129
|
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
1993
2130
|
|
|
1994
2131
|
@staticmethod
|
|
1995
|
-
def _doc(r, body=True):
|
|
2132
|
+
def _doc(r, body=True, versions=1):
|
|
1996
2133
|
"""A row as the API answers it: a DatasetSummary, and its body with it,
|
|
1997
|
-
read as today's version.
|
|
2134
|
+
read as today's version. `version` is the save counter a write is
|
|
2135
|
+
arbitrated by; `versions` is how many group versions it has -- its
|
|
2136
|
+
newest one's number, which a pin may name, and 1 before any is kept,
|
|
2137
|
+
the row's body being version 1 in waiting."""
|
|
1998
2138
|
parsed = upgrade_body(json.loads(r["body"]))
|
|
1999
2139
|
out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
|
|
2000
|
-
"version": r["version"], "updated": r["updated_at"]}
|
|
2140
|
+
"version": r["version"], "versions": versions, "updated": r["updated_at"]}
|
|
2001
2141
|
if body:
|
|
2002
2142
|
out["body"] = parsed
|
|
2003
2143
|
return out
|
|
2004
2144
|
|
|
2145
|
+
@staticmethod
|
|
2146
|
+
def _counts(db) -> dict:
|
|
2147
|
+
return dict(db.execute("SELECT group_id, MAX(n) FROM eval_group_versions GROUP BY group_id"))
|
|
2148
|
+
|
|
2149
|
+
def _row_doc(self, db, r, body=True):
|
|
2150
|
+
return self._doc(r, body, self._counts(db).get(r["id"], 1))
|
|
2151
|
+
|
|
2005
2152
|
def _connect(self):
|
|
2006
2153
|
db = sqlite3.connect(self.store.path)
|
|
2007
2154
|
db.row_factory = sqlite3.Row
|
|
@@ -2016,13 +2163,142 @@ class Datasets:
|
|
|
2016
2163
|
|
|
2017
2164
|
def list(self) -> list:
|
|
2018
2165
|
with self.store.lock, self._connect() as db:
|
|
2019
|
-
|
|
2166
|
+
counts = self._counts(db)
|
|
2167
|
+
return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
|
|
2020
2168
|
"SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id")]
|
|
2021
2169
|
|
|
2022
2170
|
def get(self, did):
|
|
2023
2171
|
with self.store.lock, self._connect() as db:
|
|
2024
2172
|
r = self._live(db, did)
|
|
2025
|
-
return self.
|
|
2173
|
+
return self._row_doc(db, r) if r else None
|
|
2174
|
+
|
|
2175
|
+
# ---- a group's versions (docs/pipeline-model.md §17) -----------------
|
|
2176
|
+
|
|
2177
|
+
@staticmethod
|
|
2178
|
+
def _head(db, did):
|
|
2179
|
+
return db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? "
|
|
2180
|
+
"ORDER BY n DESC LIMIT 1", (did,)).fetchone()
|
|
2181
|
+
|
|
2182
|
+
def _mint(self, db, r):
|
|
2183
|
+
"""The newest version of row [r], adding version 1 from its body
|
|
2184
|
+
first if it has none: edited when the row last was, so a group made a
|
|
2185
|
+
moment ago goes on being edited in place."""
|
|
2186
|
+
head = self._head(db, r["id"])
|
|
2187
|
+
if head is not None:
|
|
2188
|
+
return head
|
|
2189
|
+
try:
|
|
2190
|
+
edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
|
|
2191
|
+
except ValueError:
|
|
2192
|
+
# A stamp nothing here wrote is no recent edit.
|
|
2193
|
+
edited = 0
|
|
2194
|
+
db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
|
|
2195
|
+
"VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
|
|
2196
|
+
return self._head(db, r["id"])
|
|
2197
|
+
|
|
2198
|
+
@staticmethod
|
|
2199
|
+
def _pins(db) -> set:
|
|
2200
|
+
"""Every (group, version) a stored pipeline pins, read in the caller's
|
|
2201
|
+
transaction from the docs table the page writes them to."""
|
|
2202
|
+
row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'").fetchone()
|
|
2203
|
+
return pins_in(json.loads(row[0])) if row and row[0] else set()
|
|
2204
|
+
|
|
2205
|
+
def _cut(self, db, did, text):
|
|
2206
|
+
"""A new newest version of [did] holding [text]; returns its number."""
|
|
2207
|
+
n = self._head(db, did)["n"] + 1
|
|
2208
|
+
db.execute("INSERT INTO eval_group_versions (group_id, n, body, created_at, edited_at) "
|
|
2209
|
+
"VALUES (?, ?, ?, ?, ?)", (did, n, text, self._now(), time.time()))
|
|
2210
|
+
return n
|
|
2211
|
+
|
|
2212
|
+
def _keep(self, db, r, text):
|
|
2213
|
+
"""[text] as row [r]'s newest version: the newest edited in place while
|
|
2214
|
+
no run has graded with it, no link pins it and it was edited in the
|
|
2215
|
+
last GROUP_IDLE_SECONDS, and a new version otherwise."""
|
|
2216
|
+
head = self._mint(db, r)
|
|
2217
|
+
if text == head["body"]:
|
|
2218
|
+
return
|
|
2219
|
+
fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
|
|
2220
|
+
if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
|
|
2221
|
+
db.execute("UPDATE eval_group_versions SET body = ?, edited_at = ? WHERE group_id = ? AND n = ?",
|
|
2222
|
+
(text, time.time(), r["id"], head["n"]))
|
|
2223
|
+
else:
|
|
2224
|
+
self._cut(db, r["id"], text)
|
|
2225
|
+
|
|
2226
|
+
def versions(self, did):
|
|
2227
|
+
"""A group's versions, newest first, without their bodies -- version 1
|
|
2228
|
+
alone, read from the row, before any is kept -- or None."""
|
|
2229
|
+
with self.store.lock, self._connect() as db:
|
|
2230
|
+
r = self._live(db, did)
|
|
2231
|
+
if r is None:
|
|
2232
|
+
return None
|
|
2233
|
+
pins = self._pins(db)
|
|
2234
|
+
rows = db.execute("SELECT * FROM eval_group_versions WHERE group_id = ? ORDER BY n DESC",
|
|
2235
|
+
(did,)).fetchall()
|
|
2236
|
+
if not rows:
|
|
2237
|
+
return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (did, 1) in pins,
|
|
2238
|
+
"fingerprint": fingerprint(json.loads(r["body"]))}]
|
|
2239
|
+
return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
|
|
2240
|
+
"pinned": (did, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
|
|
2241
|
+
for v in rows]
|
|
2242
|
+
|
|
2243
|
+
def version_body(self, did, n):
|
|
2244
|
+
"""Version [n]'s body, read as today's, or None."""
|
|
2245
|
+
with self.store.lock, self._connect() as db:
|
|
2246
|
+
r = self._live(db, did)
|
|
2247
|
+
if r is None:
|
|
2248
|
+
return None
|
|
2249
|
+
v = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
|
|
2250
|
+
(did, n)).fetchone()
|
|
2251
|
+
if v is None and n == 1 and self._head(db, did) is None:
|
|
2252
|
+
v = (r["body"],)
|
|
2253
|
+
return upgrade_body(json.loads(v[0])) if v else None
|
|
2254
|
+
|
|
2255
|
+
def restore_version(self, did, n):
|
|
2256
|
+
"""An older version's body as the newest version, and the row's: as a
|
|
2257
|
+
prompt's Restore, nothing is rewritten, so a run or a pin naming any
|
|
2258
|
+
version still reads what it named. Returns (DatasetDoc, None)."""
|
|
2259
|
+
with self.store.lock, self._connect() as db, db:
|
|
2260
|
+
r = self._live(db, did)
|
|
2261
|
+
if r is None:
|
|
2262
|
+
return None, (404, "no such eval group")
|
|
2263
|
+
head = self._mint(db, r)
|
|
2264
|
+
old = db.execute("SELECT body FROM eval_group_versions WHERE group_id = ? AND n = ?",
|
|
2265
|
+
(did, n)).fetchone()
|
|
2266
|
+
if old is None:
|
|
2267
|
+
return None, (404, "no such version")
|
|
2268
|
+
if old["body"] != head["body"]:
|
|
2269
|
+
self._cut(db, did, old["body"])
|
|
2270
|
+
db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
|
|
2271
|
+
(old["body"], r["version"] + 1, self._now(), did))
|
|
2272
|
+
return self._row_doc(db, self._live(db, did)), None
|
|
2273
|
+
|
|
2274
|
+
def mint_pinned(self, pins):
|
|
2275
|
+
"""Version 1 of each pinned group that has none yet: a pin names a
|
|
2276
|
+
version, so the version has to be kept from the moment it does."""
|
|
2277
|
+
with self.store.lock, self._connect() as db, db:
|
|
2278
|
+
for did, _ in pins:
|
|
2279
|
+
r = self._live(db, did)
|
|
2280
|
+
if r is not None:
|
|
2281
|
+
self._mint(db, r)
|
|
2282
|
+
|
|
2283
|
+
def resolve(self, did, pin=None):
|
|
2284
|
+
"""The version a run submitted now grades with -- [pin], or the
|
|
2285
|
+
newest -- marked as graded with, in the same transaction, so no save
|
|
2286
|
+
can edit it in place between this and the run keeping its body.
|
|
2287
|
+
Returns (n, body), (None, why) for a pin the group has no version of,
|
|
2288
|
+
or None for a group the lab does not have. A submit refused after
|
|
2289
|
+
this leaves the version frozen, which costs only a new version at the
|
|
2290
|
+
next save."""
|
|
2291
|
+
with self.store.lock, self._connect() as db, db:
|
|
2292
|
+
r = self._live(db, did) if isinstance(did, str) else None
|
|
2293
|
+
if r is None:
|
|
2294
|
+
return None
|
|
2295
|
+
head = self._mint(db, r)
|
|
2296
|
+
v = head if pin is None else db.execute(
|
|
2297
|
+
"SELECT * FROM eval_group_versions WHERE group_id = ? AND n = ?", (did, pin)).fetchone()
|
|
2298
|
+
if v is None:
|
|
2299
|
+
return None, f"{r['name']} has no version {pin}"
|
|
2300
|
+
db.execute("UPDATE eval_group_versions SET ran = 1 WHERE group_id = ? AND n = ?", (did, v["n"]))
|
|
2301
|
+
return v["n"], json.loads(v["body"])
|
|
2026
2302
|
|
|
2027
2303
|
def snapshot(self, did):
|
|
2028
2304
|
"""The body a run submitted now grades against, and its fingerprint;
|
|
@@ -2087,9 +2363,11 @@ class Datasets:
|
|
|
2087
2363
|
if r is None:
|
|
2088
2364
|
return None, (404, "no such dataset")
|
|
2089
2365
|
if r["version"] != version:
|
|
2090
|
-
return None, (409, {"current": self.
|
|
2366
|
+
return None, (409, {"current": self._row_doc(db, r)})
|
|
2367
|
+
text = json.dumps(body)
|
|
2368
|
+
self._keep(db, r, text)
|
|
2091
2369
|
db.execute("UPDATE datasets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
|
|
2092
|
-
(
|
|
2370
|
+
(text, version + 1, self._now(), did))
|
|
2093
2371
|
return {"version": version + 1}, None
|
|
2094
2372
|
|
|
2095
2373
|
def remove(self, did):
|
|
@@ -2115,58 +2393,69 @@ class Datasets:
|
|
|
2115
2393
|
did = r["id"]
|
|
2116
2394
|
return self.get(did), None
|
|
2117
2395
|
|
|
2396
|
+
@staticmethod
|
|
2397
|
+
def _purge(db, where, args):
|
|
2398
|
+
# A group's versions go with it: the trash held them, and Undo is
|
|
2399
|
+
# what would have brought them back.
|
|
2400
|
+
db.execute(f"DELETE FROM eval_group_versions WHERE group_id IN (SELECT id FROM datasets WHERE {where})", args)
|
|
2401
|
+
db.execute(f"DELETE FROM datasets WHERE {where}", args)
|
|
2402
|
+
|
|
2118
2403
|
def lazy_trash(self):
|
|
2119
2404
|
"""A datasets request empties what has been trashed longer than
|
|
2120
2405
|
TRASH_SECONDS, as a Sources request does."""
|
|
2121
2406
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2122
|
-
|
|
2123
|
-
(time.time() - TRASH_SECONDS,))
|
|
2407
|
+
self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
|
|
2124
2408
|
|
|
2125
2409
|
def empty_trash(self):
|
|
2126
2410
|
"""The startup sweep: a restart has nothing to undo."""
|
|
2127
2411
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2128
|
-
|
|
2412
|
+
self._purge(db, "trash IS NOT NULL", ())
|
|
2129
2413
|
|
|
2130
2414
|
def export(self, did):
|
|
2131
2415
|
d = self.get(did)
|
|
2132
2416
|
if d is None:
|
|
2133
2417
|
return None
|
|
2134
2418
|
return {"format": EXPORT_ONE, "version": EXPORT_VERSION,
|
|
2135
|
-
"
|
|
2419
|
+
"group": {"name": d["name"], "body": d["body"]}}
|
|
2136
2420
|
|
|
2137
2421
|
def export_all(self):
|
|
2138
2422
|
with self.store.lock, self._connect() as db:
|
|
2139
2423
|
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
|
|
2140
2424
|
"ORDER BY name COLLATE NOCASE, id").fetchall()
|
|
2141
2425
|
return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
|
|
2142
|
-
"
|
|
2426
|
+
"groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
|
|
2143
2427
|
|
|
2144
2428
|
def import_file(self, doc):
|
|
2145
2429
|
"""Either export's file, as new datasets: import always creates, ids
|
|
2146
2430
|
are this lab's, and a taken name gets ` (2)`. All of it or none.
|
|
2147
2431
|
Returns ([DatasetSummary], None), or (None, (400, one sentence))."""
|
|
2148
2432
|
if not isinstance(doc, dict) or "format" not in doc:
|
|
2149
|
-
return None, (400, "that file is not
|
|
2150
|
-
if doc["format"] not in
|
|
2433
|
+
return None, (400, "that file is not an eval group export: it has no \"format\"")
|
|
2434
|
+
if doc["format"] not in EXPORT_KEYS:
|
|
2151
2435
|
return None, (400, f"that file's format is {json.dumps(doc['format'])}, and the lab "
|
|
2152
|
-
f"imports {EXPORT_ONE} and {
|
|
2436
|
+
f"imports {EXPORT_ONE}, {EXPORT_ALL}, {DATASET_ONE} and {DATASET_ALL}")
|
|
2153
2437
|
version = doc.get("version")
|
|
2154
|
-
# Exactly the integer: Python's True == 1.
|
|
2155
|
-
|
|
2438
|
+
# Exactly the integer: Python's True == 1. An eval group's file began
|
|
2439
|
+
# at version 7; a dataset's reads back to version 1.
|
|
2440
|
+
grouped = doc["format"] in (EXPORT_ONE, EXPORT_ALL)
|
|
2441
|
+
if type(version) is not int or version not in ((EXPORT_VERSION,) if grouped else IMPORT_VERSIONS):
|
|
2156
2442
|
return None, (400, f"that file is version {json.dumps(version)}, and the lab "
|
|
2157
|
-
f"reads versions 1 to {EXPORT_VERSION}")
|
|
2158
|
-
|
|
2159
|
-
|
|
2443
|
+
+ (f"reads version {EXPORT_VERSION}" if grouped else f"reads versions 1 to {EXPORT_VERSION}"))
|
|
2444
|
+
one = doc["format"] in (EXPORT_ONE, DATASET_ONE)
|
|
2445
|
+
key = EXPORT_KEYS[doc["format"]]
|
|
2446
|
+
if one:
|
|
2447
|
+
items = [doc.get(key)]
|
|
2160
2448
|
else:
|
|
2161
|
-
items = doc.get(
|
|
2449
|
+
items = doc.get(key)
|
|
2162
2450
|
if not isinstance(items, list):
|
|
2163
|
-
return None, (400, "an export of every dataset holds them as a \"
|
|
2451
|
+
return None, (400, f"an export of every {'group' if grouped else 'dataset'} holds them as a \"{key}\" list")
|
|
2164
2452
|
for k in doc:
|
|
2165
|
-
if k not in ("format", "version",
|
|
2453
|
+
if k not in ("format", "version", key):
|
|
2166
2454
|
return None, (400, f"that file has \"{k}\", which an export does not")
|
|
2167
2455
|
ready = []
|
|
2456
|
+
what = "eval group" if grouped else "dataset"
|
|
2168
2457
|
for i, item in enumerate(items):
|
|
2169
|
-
at = "the
|
|
2458
|
+
at = f"the {what}" if one else f"{what} {i + 1}"
|
|
2170
2459
|
if not isinstance(item, dict) or set(item) != {"name", "body"}:
|
|
2171
2460
|
return None, (400, f"{at} has to be {{ \"name\", \"body\" }}")
|
|
2172
2461
|
name, why = dataset_name(item["name"])
|
|
@@ -2339,13 +2628,17 @@ def read_pack(data: bytes):
|
|
|
2339
2628
|
doc, why = load(name)
|
|
2340
2629
|
if why:
|
|
2341
2630
|
return None, why
|
|
2342
|
-
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2631
|
+
# A pack's datasets are eval groups, in either file form: the
|
|
2632
|
+
# group's (version 7) or the dataset's an older pack holds.
|
|
2633
|
+
fmt = doc.get("format") if isinstance(doc, dict) else None
|
|
2634
|
+
entry = doc.get(EXPORT_KEYS[fmt]) if fmt in (EXPORT_ONE, DATASET_ONE) else None
|
|
2635
|
+
if (not isinstance(entry, dict) or type(doc.get("version")) is not int
|
|
2636
|
+
or doc["version"] not in ((EXPORT_VERSION,) if fmt == EXPORT_ONE else IMPORT_VERSIONS)):
|
|
2637
|
+
return None, f"{name} is not an eval group export the lab reads"
|
|
2638
|
+
ds_name, why = dataset_name(entry.get("name"))
|
|
2346
2639
|
if why:
|
|
2347
2640
|
return None, f"{name}: {why}"
|
|
2348
|
-
raw =
|
|
2641
|
+
raw = entry.get("body")
|
|
2349
2642
|
body = upgrade_body(raw) if doc["version"] < EXPORT_VERSION else raw
|
|
2350
2643
|
why = dataset_problem(body)
|
|
2351
2644
|
if why:
|
|
@@ -2524,11 +2817,12 @@ class Packs:
|
|
|
2524
2817
|
# the page upgrades the pipeline as it reads it.
|
|
2525
2818
|
evals = doc.get("evals", doc.get("tests"))
|
|
2526
2819
|
for t in evals if isinstance(evals, list) else [evals]:
|
|
2527
|
-
ref =
|
|
2528
|
-
if
|
|
2820
|
+
ref = eval_group_ref(t)
|
|
2821
|
+
if ref is not None:
|
|
2529
2822
|
did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
|
|
2530
2823
|
if did:
|
|
2531
|
-
|
|
2824
|
+
ref.clear()
|
|
2825
|
+
ref.update({"id": did, "name": DATASETS.get(did)["name"]})
|
|
2532
2826
|
content = content_of(doc)
|
|
2533
2827
|
if isinstance(content, dict) and isinstance(content.get("ref"), dict):
|
|
2534
2828
|
sid = src_ids.get(content["ref"].get("id")) or src_ids.get(content["ref"].get("name"))
|
|
@@ -3251,13 +3545,15 @@ class Plugins:
|
|
|
3251
3545
|
# is a marker the worker checks between items -- today's semantics, stopping
|
|
3252
3546
|
# after the item in flight, never a process kill that loses its reply.
|
|
3253
3547
|
|
|
3254
|
-
# A profile's key, as run-evals.js reads it:
|
|
3255
|
-
# document
|
|
3256
|
-
#
|
|
3257
|
-
#
|
|
3258
|
-
#
|
|
3259
|
-
|
|
3260
|
-
|
|
3548
|
+
# A profile's key, as run-evals.js reads it: EVALSLAB_API_KEY_<SLUG>, the slug the
|
|
3549
|
+
# run document's connection carries (#253), or EVALSLAB_API_KEY_<ID> for one from
|
|
3550
|
+
# before slugs -- the id it keys its profiles table by -- in the shell-safe
|
|
3551
|
+
# spelling of itself. Never the name, so two profiles can share a name
|
|
3552
|
+
# without sharing a key (docs/pipeline-model.md §5). One spelling on both
|
|
3553
|
+
# sides -- evals-core.ts keyVar -- or the worker would never find the key the
|
|
3554
|
+
# page never sent.
|
|
3555
|
+
def key_var(profile_id: str, slug=None) -> str:
|
|
3556
|
+
return "EVALSLAB_API_KEY_" + re.sub(r"[^A-Z0-9]+", "_", str(slug or profile_id).upper()).strip("_")
|
|
3261
3557
|
|
|
3262
3558
|
|
|
3263
3559
|
def run_items(run):
|
|
@@ -3331,7 +3627,33 @@ def brief_row(row):
|
|
|
3331
3627
|
if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
|
|
3332
3628
|
return it
|
|
3333
3629
|
return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
|
|
3334
|
-
|
|
3630
|
+
# The run's verdict stays; each Target's eval by eval is the whole row's.
|
|
3631
|
+
brief = {k: v for k, v in row.items() if k != "verdicts"}
|
|
3632
|
+
return {**brief, "results": [item(it) for it in row.get("results") or []], "brief": True}
|
|
3633
|
+
|
|
3634
|
+
|
|
3635
|
+
# What of a worker's report a row keeps as its verdicts (#252): the run's --
|
|
3636
|
+
# pass, fail or incomplete, which the worker's exit code already said -- and
|
|
3637
|
+
# each Target's evals, by the eval's id, as `run.verdicts` names them. Only
|
|
3638
|
+
# these fields are carried, so nothing else the report holds lands in the row.
|
|
3639
|
+
# A status says whether the run finished; its verdict says whether it passed,
|
|
3640
|
+
# so a run that failed an eval is `done` with the verdict `fail`.
|
|
3641
|
+
VERDICTS = ("pass", "fail", "incomplete")
|
|
3642
|
+
VERDICT_FIELDS = ("name", "skipped", "verdict", "pass", "detail", "ran", "passed", "skippedItems")
|
|
3643
|
+
|
|
3644
|
+
|
|
3645
|
+
def report_verdicts(report):
|
|
3646
|
+
"""(verdict, verdicts as JSON) from a worker's report, or (None, None)
|
|
3647
|
+
for one that carries none -- a worker from before #251."""
|
|
3648
|
+
verdict = report.get("verdict") if isinstance(report, dict) else None
|
|
3649
|
+
run = report.get("run") if verdict in VERDICTS else None
|
|
3650
|
+
targets = run.get("verdicts") if isinstance(run, dict) else None
|
|
3651
|
+
if not isinstance(targets, list):
|
|
3652
|
+
return None, None
|
|
3653
|
+
kept = [{eid: {k: e[k] for k in VERDICT_FIELDS if k in e}
|
|
3654
|
+
for eid, e in t.items() if isinstance(e, dict)}
|
|
3655
|
+
if isinstance(t, dict) else {} for t in targets]
|
|
3656
|
+
return verdict, json.dumps(kept)
|
|
3335
3657
|
|
|
3336
3658
|
|
|
3337
3659
|
class Queue:
|
|
@@ -3368,13 +3690,27 @@ class Queue:
|
|
|
3368
3690
|
# one of its own.
|
|
3369
3691
|
if "rerun_of" not in cols:
|
|
3370
3692
|
db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
|
|
3693
|
+
# The worker's verdicts (#252); a row from before them has none,
|
|
3694
|
+
# and is never given one after the fact.
|
|
3695
|
+
if "verdict" not in cols:
|
|
3696
|
+
db.execute("ALTER TABLE queue ADD COLUMN verdict TEXT")
|
|
3697
|
+
if "verdicts" not in cols:
|
|
3698
|
+
db.execute("ALTER TABLE queue ADD COLUMN verdicts TEXT")
|
|
3699
|
+
# The body of each eval group a run grades with, by `<id>@<n>`
|
|
3700
|
+
# (docs/pipeline-model.md §17). A row from before keeps its
|
|
3701
|
+
# `dataset` column, read as its one group.
|
|
3702
|
+
if "groups" not in cols:
|
|
3703
|
+
db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
|
|
3371
3704
|
|
|
3372
3705
|
# ---- rows -----------------------------------------------------------
|
|
3373
3706
|
|
|
3374
3707
|
# A row read with the submit time of the run it re-runs, if any: the
|
|
3375
3708
|
# page names a run by that time (History's Run ID), so a "Re-run of"
|
|
3376
|
-
# note reads without fetching the original.
|
|
3377
|
-
|
|
3709
|
+
# note reads without fetching the original. Its columns are named, since
|
|
3710
|
+
# the order ALTER TABLE added them in is no order _row can count on.
|
|
3711
|
+
SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
|
|
3712
|
+
"q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
|
|
3713
|
+
"q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
|
|
3378
3714
|
"LEFT JOIN queue o ON o.id = q.rerun_of")
|
|
3379
3715
|
|
|
3380
3716
|
@staticmethod
|
|
@@ -3387,8 +3723,9 @@ class Queue:
|
|
|
3387
3723
|
"snapshot": json.loads(r[6]), "results": json.loads(r[7]),
|
|
3388
3724
|
"progress": json.loads(r[8]), "totals": json.loads(r[9]),
|
|
3389
3725
|
"error": r[10],
|
|
3390
|
-
"rerunOf": r[
|
|
3391
|
-
"
|
|
3726
|
+
"rerunOf": r[11], "rerunOfAt": r[14],
|
|
3727
|
+
"verdict": r[12],
|
|
3728
|
+
"verdicts": json.loads(r[13]) if r[13] else None,
|
|
3392
3729
|
}
|
|
3393
3730
|
|
|
3394
3731
|
# A row from before run documents has no version, and nothing here can
|
|
@@ -3411,15 +3748,19 @@ class Queue:
|
|
|
3411
3748
|
(rid,)).fetchone())
|
|
3412
3749
|
return row if self._readable(row) else None
|
|
3413
3750
|
|
|
3414
|
-
def list(self, limit=RUNS_PAGE, before=None, full=False):
|
|
3415
|
-
"""Runs, newest first, and whether more follow. `before`
|
|
3416
|
-
`
|
|
3417
|
-
them the way it pages the runs
|
|
3418
|
-
`
|
|
3751
|
+
def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
|
|
3752
|
+
"""Runs, newest first, and whether more follow. `before` and
|
|
3753
|
+
`before_id` are the `submittedAt` and id of the last run on the page
|
|
3754
|
+
before, so History can page through them the way it pages the runs
|
|
3755
|
+
store. `submittedAt` is to the second, so two runs can share one; the
|
|
3756
|
+
id breaks the tie, or a page ending between them would skip the
|
|
3757
|
+
second (#238). `before` alone stops at the second. Each row is
|
|
3758
|
+
brief_row's unless `full` asks for the whole of it."""
|
|
3419
3759
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3420
3760
|
rows = self._all(db)
|
|
3421
|
-
|
|
3422
|
-
rows
|
|
3761
|
+
key = lambda r: (r["submittedAt"], r["id"])
|
|
3762
|
+
rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
|
|
3763
|
+
rows.sort(key=key, reverse=True)
|
|
3423
3764
|
page = rows[:limit]
|
|
3424
3765
|
return (page if full else [brief_row(r) for r in page]), len(rows) > limit
|
|
3425
3766
|
|
|
@@ -3430,27 +3771,30 @@ class Queue:
|
|
|
3430
3771
|
|
|
3431
3772
|
# ---- submit ---------------------------------------------------------
|
|
3432
3773
|
|
|
3433
|
-
def submit(self, run: dict, dataset=None, rerun_of=None):
|
|
3774
|
+
def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
|
|
3434
3775
|
"""
|
|
3435
3776
|
A new queued run. `run` is the run document (docs/pipeline-model.md
|
|
3436
3777
|
§5): the pipeline, the profiles it resolved to without their keys, its
|
|
3437
|
-
content's file list in order and with its repeats, and
|
|
3438
|
-
version; `
|
|
3439
|
-
`
|
|
3440
|
-
|
|
3441
|
-
the
|
|
3778
|
+
content's file list in order and with its repeats, and each eval
|
|
3779
|
+
group's version; `groups` is those versions' bodies, by `<id>@<n>`
|
|
3780
|
+
(§17), and `dataset` the one body a row from before them kept, which a
|
|
3781
|
+
re-run of one carries on; `rerun_of` is the run a re-run was queued
|
|
3782
|
+
from. Returns the row. Its items are that list, or the one inline
|
|
3783
|
+
text, each through every scenario -- so the total is the list's
|
|
3784
|
+
length, repeats and all, the same count the runner and the page make.
|
|
3442
3785
|
"""
|
|
3443
3786
|
rid = secrets.token_hex(6)
|
|
3444
3787
|
now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
3445
3788
|
total = len(run_items(run))
|
|
3446
3789
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3447
3790
|
db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
|
|
3448
|
-
"snapshot, results, progress, totals, dataset, rerun_of) "
|
|
3449
|
-
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
|
|
3791
|
+
"snapshot, results, progress, totals, dataset, rerun_of, groups) "
|
|
3792
|
+
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
3450
3793
|
(rid, "queued", now, json.dumps(run), "[]",
|
|
3451
3794
|
json.dumps({"current": None, "n": 0, "total": total}),
|
|
3452
3795
|
json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
|
|
3453
|
-
None if dataset is None else json.dumps(dataset), rerun_of
|
|
3796
|
+
None if dataset is None else json.dumps(dataset), rerun_of,
|
|
3797
|
+
None if groups is None else json.dumps(groups)))
|
|
3454
3798
|
# Its prompts' uses, in the same transaction: a run is in the
|
|
3455
3799
|
# library the moment it is queued, or not queued at all.
|
|
3456
3800
|
if self.prompts is not None:
|
|
@@ -3460,19 +3804,43 @@ class Queue:
|
|
|
3460
3804
|
# the moment the lock is let go.
|
|
3461
3805
|
return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
|
|
3462
3806
|
|
|
3807
|
+
def groups(self, rid, raw=False):
|
|
3808
|
+
"""The eval group bodies a readable run grades with, by `<id>@<n>`, or
|
|
3809
|
+
None for a run that is not there or kept none -- what its verdicts
|
|
3810
|
+
were graded by, whatever the groups hold now. A row from before runs
|
|
3811
|
+
kept a body per group answers its one `dataset` copy under its
|
|
3812
|
+
group's key. A copy kept at an earlier version reads as one of
|
|
3813
|
+
today's, unless [raw]: the worker is handed it as kept, since an
|
|
3814
|
+
earlier run's pipeline is upgraded under the rules that copy holds."""
|
|
3815
|
+
run = self.get(rid)
|
|
3816
|
+
if run is None:
|
|
3817
|
+
return None
|
|
3818
|
+
kept = self._kept(rid)
|
|
3819
|
+
if kept is None:
|
|
3820
|
+
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3821
|
+
old = db.execute("SELECT dataset FROM queue WHERE id = ?", (rid,)).fetchone()[0]
|
|
3822
|
+
if not old:
|
|
3823
|
+
return None
|
|
3824
|
+
kept = {group_key(evals_dataset(run["snapshot"])): json.loads(old)}
|
|
3825
|
+
return kept if raw else {k: upgrade_body(b) for k, b in kept.items()}
|
|
3826
|
+
|
|
3827
|
+
def _kept(self, rid):
|
|
3828
|
+
"""The row's own `groups` column, or None for a row from before it."""
|
|
3829
|
+
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3830
|
+
r = db.execute("SELECT groups FROM queue WHERE id = ?", (rid,)).fetchone()
|
|
3831
|
+
return json.loads(r[0]) if r and r[0] is not None else None
|
|
3832
|
+
|
|
3463
3833
|
def dataset(self, rid, raw=False):
|
|
3464
|
-
"""The
|
|
3465
|
-
|
|
3466
|
-
|
|
3467
|
-
the worker is handed it as kept, since an earlier run's pipeline is
|
|
3468
|
-
upgraded under the rules that copy holds."""
|
|
3834
|
+
"""The body of the one Library group a readable run grades its cases
|
|
3835
|
+
against (evals_dataset), or None: the body the worker is handed as
|
|
3836
|
+
--dataset, and what the page reads a run's cases from."""
|
|
3469
3837
|
if self.get(rid) is None:
|
|
3470
3838
|
return None
|
|
3471
|
-
|
|
3472
|
-
|
|
3473
|
-
|
|
3839
|
+
kept = self.groups(rid, raw=True)
|
|
3840
|
+
ref = evals_dataset(self.get(rid)["snapshot"])
|
|
3841
|
+
body = kept.get(group_key(ref)) if kept and ref is not None else None
|
|
3842
|
+
if body is None:
|
|
3474
3843
|
return None
|
|
3475
|
-
body = json.loads(r[0])
|
|
3476
3844
|
return body if raw else upgrade_body(body)
|
|
3477
3845
|
|
|
3478
3846
|
def _plugin_args(self, run):
|
|
@@ -3487,14 +3855,18 @@ class Queue:
|
|
|
3487
3855
|
return ["--plugins", str(PLUGINS.dir)], None
|
|
3488
3856
|
|
|
3489
3857
|
def _dataset_args(self, run, rundir):
|
|
3490
|
-
"""The worker's --dataset for a graded run: the body
|
|
3491
|
-
|
|
3492
|
-
|
|
3493
|
-
|
|
3858
|
+
"""The worker's --dataset for a graded run: the body of its one
|
|
3859
|
+
Library group kept with the row, written beside the run document --
|
|
3860
|
+
one, until the worker reads a body per group (#233). A row queued
|
|
3861
|
+
before runs kept their dataset has none, and is pinned to the dataset
|
|
3862
|
+
as it reads now, once, so every later pass over it agrees. Returns
|
|
3863
|
+
(args, None) or (None, why)."""
|
|
3494
3864
|
ref = evals_dataset(run["snapshot"])
|
|
3495
3865
|
if ref is None:
|
|
3496
3866
|
return [], None
|
|
3497
3867
|
body = self.dataset(run["id"], raw=True)
|
|
3868
|
+
if body is None and self._kept(run["id"]) is not None:
|
|
3869
|
+
return None, f"the run kept no body of the eval group {ref.get('name') or ref.get('id')!r}"
|
|
3498
3870
|
if body is None:
|
|
3499
3871
|
snap = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
|
|
3500
3872
|
if snap is None:
|
|
@@ -3547,7 +3919,8 @@ class Queue:
|
|
|
3547
3919
|
if run["status"] not in ("cancelled", "interrupted", "incomplete"):
|
|
3548
3920
|
return None, (409, "only a cancelled, interrupted or incomplete run can be resumed")
|
|
3549
3921
|
(self.dir / rid / "cancel").unlink(missing_ok=True)
|
|
3550
|
-
|
|
3922
|
+
# What it reached is decided by the run it goes on to finish.
|
|
3923
|
+
self._set(rid, status="queued", cancel=0, error=None, verdict=None, verdicts=None)
|
|
3551
3924
|
return self.get(rid), None
|
|
3552
3925
|
|
|
3553
3926
|
def set_comment(self, rid, comment):
|
|
@@ -3597,9 +3970,12 @@ class Queue:
|
|
|
3597
3970
|
_, err = self._pinned_files(snap)
|
|
3598
3971
|
if err:
|
|
3599
3972
|
return None, (409, err)
|
|
3973
|
+
# The group bodies the original kept: a re-run grades with exactly
|
|
3974
|
+
# them, whatever the groups or the pins read now.
|
|
3975
|
+
kept = self._kept(rid)
|
|
3600
3976
|
ref = evals_dataset(snap)
|
|
3601
3977
|
body = None
|
|
3602
|
-
if ref is not None:
|
|
3978
|
+
if ref is not None and kept is None:
|
|
3603
3979
|
body = self.dataset(rid, raw=True)
|
|
3604
3980
|
if body is None:
|
|
3605
3981
|
# A run that never started kept no body: the dataset's, if it
|
|
@@ -3615,7 +3991,7 @@ class Queue:
|
|
|
3615
3991
|
_, err = worker_destinations(snap)
|
|
3616
3992
|
if err:
|
|
3617
3993
|
return None, (403, err)
|
|
3618
|
-
return self.submit(snap, body, rerun_of=rid), None
|
|
3994
|
+
return self.submit(snap, body, rerun_of=rid, groups=kept), None
|
|
3619
3995
|
|
|
3620
3996
|
def rerun_item(self, rid, index):
|
|
3621
3997
|
"""
|
|
@@ -3677,9 +4053,12 @@ class Queue:
|
|
|
3677
4053
|
while len(results) <= index:
|
|
3678
4054
|
results.append(None)
|
|
3679
4055
|
results[index] = next((it for it in items if it and it.get("item") == index), items[0])
|
|
4056
|
+
# The worker read every item to reach its verdicts, so they are the
|
|
4057
|
+
# row's as it now stands.
|
|
4058
|
+
verdict, verdicts = report_verdicts(report)
|
|
3680
4059
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3681
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3682
|
-
(json.dumps(results), None, run["id"]))
|
|
4060
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4061
|
+
"WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
|
|
3683
4062
|
return self.get(run["id"]), None
|
|
3684
4063
|
|
|
3685
4064
|
def rescore_item(self, rid, index):
|
|
@@ -3742,9 +4121,12 @@ class Queue:
|
|
|
3742
4121
|
while len(results) <= index:
|
|
3743
4122
|
results.append(None)
|
|
3744
4123
|
results[index] = next((it for it in items if it and it.get("item") == index), items[0])
|
|
4124
|
+
# The worker read every item to reach its verdicts, so they are the
|
|
4125
|
+
# row's as it now stands.
|
|
4126
|
+
verdict, verdicts = report_verdicts(report)
|
|
3745
4127
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3746
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3747
|
-
(json.dumps(results), None, run["id"]))
|
|
4128
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4129
|
+
"WHERE id = ?", (json.dumps(results), None, verdict, verdicts, run["id"]))
|
|
3748
4130
|
return self.get(run["id"]), None
|
|
3749
4131
|
|
|
3750
4132
|
# ---- the run --------------------------------------------------------
|
|
@@ -3855,10 +4237,11 @@ class Queue:
|
|
|
3855
4237
|
report = self._read_report(rid)
|
|
3856
4238
|
items = report["run"]["items"] if report and isinstance(report.get("run"), dict) else None
|
|
3857
4239
|
if isinstance(items, list):
|
|
4240
|
+
verdict, verdicts = report_verdicts(report)
|
|
3858
4241
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3859
4242
|
if db.execute("SELECT 1 FROM queue WHERE id = ?", (rid,)).fetchone() is not None:
|
|
3860
|
-
db.execute("UPDATE queue SET results = ?, error =
|
|
3861
|
-
(json.dumps(items), None, rid))
|
|
4243
|
+
db.execute("UPDATE queue SET results = ?, error = ?, verdict = ?, verdicts = ? "
|
|
4244
|
+
"WHERE id = ?", (json.dumps(items), None, verdict, verdicts, rid))
|
|
3862
4245
|
run = self.get(rid)
|
|
3863
4246
|
# The watchdog may have failed the run while it was on the wire; a
|
|
3864
4247
|
# row that already left `running` is not this worker's to re-label.
|
|
@@ -3968,8 +4351,8 @@ class Queue:
|
|
|
3968
4351
|
"""The oldest queued run this server can read. One it cannot is never
|
|
3969
4352
|
started: nothing here would know what it asks for."""
|
|
3970
4353
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3971
|
-
for r in db.execute(
|
|
3972
|
-
"ORDER BY submitted_at, rowid").fetchall():
|
|
4354
|
+
for r in db.execute(self.SELECT + " WHERE q.status = 'queued' "
|
|
4355
|
+
"ORDER BY q.submitted_at, q.rowid").fetchall():
|
|
3973
4356
|
if not self._readable(self._row(r)):
|
|
3974
4357
|
continue
|
|
3975
4358
|
db.execute("UPDATE queue SET status = 'running', started_at = ? "
|
|
@@ -4015,10 +4398,89 @@ class Queue:
|
|
|
4015
4398
|
proc.kill()
|
|
4016
4399
|
|
|
4017
4400
|
|
|
4401
|
+
# The key a relayed request is sent with: the one it carries, or -- a page
|
|
4402
|
+
# that was handed KEY_HELD in place of a profile's key -- the one the store
|
|
4403
|
+
# holds for that profile.
|
|
4404
|
+
def relay_key(payload: dict) -> str:
|
|
4405
|
+
key = str(payload.get("key") or "").strip()
|
|
4406
|
+
if held(key):
|
|
4407
|
+
return STORE.held_key(key).strip() if STORE is not None else ""
|
|
4408
|
+
return key
|
|
4409
|
+
|
|
4410
|
+
|
|
4018
4411
|
# The worker-destination rules, at submit and again at dequeue: a run's
|
|
4019
4412
|
# profiles take their keys from the profiles store, by id, so the request
|
|
4020
4413
|
# carries none, and the relay's rules bind where they go. Returns (env_vars,
|
|
4021
4414
|
# None) or (None, a named refusal).
|
|
4415
|
+
# Export for CI (#254): a pipeline as a bundle a repository keeps and
|
|
4416
|
+
# `evals-lab run` runs with no lab. The page writes its text -- the core's
|
|
4417
|
+
# exportBundle: pipeline.yaml, profiles.yaml by slug, datasets/<slug>.json --
|
|
4418
|
+
# and the server adds what only it holds: every installed plugin, as a run
|
|
4419
|
+
# stamps them all, and, when asked, the Source's files under items/. A key is
|
|
4420
|
+
# refused rather than zipped: no Setup key, no token, no field named like
|
|
4421
|
+
# one, in any of it (the core's bundleProblems asks the same of the page's).
|
|
4422
|
+
BUNDLE_TEXT = re.compile(r"pipeline\.yaml|profiles\.yaml|datasets/[a-z0-9]+(?:-[a-z0-9]+)*\.json")
|
|
4423
|
+
BUNDLE_KEY_FIELD = re.compile(
|
|
4424
|
+
r'^[\s-]*"?((?:api[-_]?)?key|authorization|bearer|secret|password|token)"?\s*:', re.I | re.M)
|
|
4425
|
+
|
|
4426
|
+
|
|
4427
|
+
def bundle_problems(files, keys=()):
|
|
4428
|
+
"""Why [files] -- path to text -- cannot leave the lab, or ""."""
|
|
4429
|
+
for path, text in files.items():
|
|
4430
|
+
for key in keys:
|
|
4431
|
+
key = str(key or "").strip()
|
|
4432
|
+
if len(key) >= 4 and key in text:
|
|
4433
|
+
return f"{path} holds a Target profile's key, and a key never leaves the lab"
|
|
4434
|
+
if TOKEN_SHAPE.search(text):
|
|
4435
|
+
return f"{path} holds a token, and a key never leaves the lab"
|
|
4436
|
+
m = BUNDLE_KEY_FIELD.search(text)
|
|
4437
|
+
if m:
|
|
4438
|
+
return f"{path} holds a field named {m.group(1)}, and a key never leaves the lab"
|
|
4439
|
+
return ""
|
|
4440
|
+
|
|
4441
|
+
|
|
4442
|
+
def build_bundle(payload):
|
|
4443
|
+
"""The zip Export for CI downloads, from the page's [payload]:
|
|
4444
|
+
{ files: {path: text}, source: id or null, items: bool }. Returns
|
|
4445
|
+
(bytes, None) or (None, (status, one sentence))."""
|
|
4446
|
+
files = payload.get("files")
|
|
4447
|
+
if not isinstance(files, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in files.items()):
|
|
4448
|
+
return None, (400, "a bundle's files are text, by path")
|
|
4449
|
+
for path in files:
|
|
4450
|
+
if not BUNDLE_TEXT.fullmatch(path):
|
|
4451
|
+
return None, (400, f"{path!r} is not a file a bundle holds")
|
|
4452
|
+
if "pipeline.yaml" not in files or "profiles.yaml" not in files:
|
|
4453
|
+
return None, (400, "a bundle holds pipeline.yaml and profiles.yaml")
|
|
4454
|
+
stored = []
|
|
4455
|
+
if STORE is not None:
|
|
4456
|
+
stored = ((STORE.all().get("promptlab.profiles") or {}).get("body") or {}).get("list") or []
|
|
4457
|
+
why = bundle_problems(files, [p.get("key") for p in stored if isinstance(p, dict)])
|
|
4458
|
+
if why:
|
|
4459
|
+
return None, (400, why)
|
|
4460
|
+
items = []
|
|
4461
|
+
if payload.get("items"):
|
|
4462
|
+
sid = payload.get("source")
|
|
4463
|
+
found = SOURCES.item_paths(sid) if isinstance(sid, str) and SOURCES is not None else None
|
|
4464
|
+
if found is None:
|
|
4465
|
+
return None, (404, "no such source")
|
|
4466
|
+
items = found
|
|
4467
|
+
buf = io.BytesIO()
|
|
4468
|
+
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
4469
|
+
for path, text in sorted(files.items()):
|
|
4470
|
+
zf.writestr(f"evals/{path}", text)
|
|
4471
|
+
if PLUGINS is not None:
|
|
4472
|
+
for p in PLUGINS.stamp():
|
|
4473
|
+
root = PLUGINS.dir / p["id"] / p["version"]
|
|
4474
|
+
for f in sorted(root.rglob("*")):
|
|
4475
|
+
if f.is_file():
|
|
4476
|
+
zf.write(f, f"evals/plugins/{p['id']}/{p['version']}/{f.relative_to(root).as_posix()}")
|
|
4477
|
+
for name, path in items:
|
|
4478
|
+
if not path.is_file():
|
|
4479
|
+
return None, (404, f"{name!r} is not in that Source")
|
|
4480
|
+
zf.write(path, f"evals/items/{name}")
|
|
4481
|
+
return buf.getvalue(), None
|
|
4482
|
+
|
|
4483
|
+
|
|
4022
4484
|
def worker_destinations(run: dict):
|
|
4023
4485
|
table = run.get("profiles") if isinstance(run, dict) else None
|
|
4024
4486
|
if not isinstance(table, dict):
|
|
@@ -4046,7 +4508,7 @@ def worker_destinations(run: dict):
|
|
|
4046
4508
|
why = allowed(base, key)
|
|
4047
4509
|
if why:
|
|
4048
4510
|
return None, f"Target profile {name}: {why}"
|
|
4049
|
-
env[key_var(pid)] = key
|
|
4511
|
+
env[key_var(pid, conn.get("slug"))] = key
|
|
4050
4512
|
return env, None
|
|
4051
4513
|
|
|
4052
4514
|
|
|
@@ -4068,15 +4530,17 @@ def worker_destinations(run: dict):
|
|
|
4068
4530
|
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
4069
4531
|
# 11: `tests` are `evals`; nothing in an eval changes.
|
|
4070
4532
|
# 12: a Contains metric's Ignore case holds item by item too, kept as written.
|
|
4071
|
-
|
|
4533
|
+
# 13: an eval is a link to an eval group, or a group of the pipeline's own,
|
|
4534
|
+
# and the document has an overall pass rule (docs/pipeline-model.md §17).
|
|
4535
|
+
PIPELINE_VERSION = 13
|
|
4072
4536
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
4073
4537
|
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
4074
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
|
4538
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
|
|
4075
4539
|
TARGET_CAP = 4
|
|
4076
4540
|
# Target steps whose words the Prompt library does not record as a use: they
|
|
4077
4541
|
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
4078
4542
|
UNRECORDED_STEPS = {"echo"}
|
|
4079
|
-
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
|
|
4543
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "pass", "profiles", "comment", "plugins")
|
|
4080
4544
|
|
|
4081
4545
|
|
|
4082
4546
|
def content_of(doc):
|
|
@@ -4122,14 +4586,17 @@ BUILTIN_CONNECTION_TYPES = dict(CONNECTION_TYPES)
|
|
|
4122
4586
|
BUILTIN_LOCAL = set(LOCAL_CONNECTIONS)
|
|
4123
4587
|
BUILTIN_CHAT_PATHS = dict(CONNECTION_CHAT_PATHS)
|
|
4124
4588
|
BUILTIN_AUTH = dict(CONNECTION_AUTH)
|
|
4125
|
-
CONNECTION_FIELDS = ("name", "url", "model", "type", "temperature", "px", "format",
|
|
4589
|
+
CONNECTION_FIELDS = ("name", "slug", "url", "model", "type", "temperature", "px", "format",
|
|
4126
4590
|
"quality", "options")
|
|
4127
4591
|
LOOKS_LIKE_A_KEY = re.compile(r"^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$",
|
|
4128
4592
|
re.IGNORECASE)
|
|
4129
4593
|
PROFILE_ID = re.compile(r"[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?")
|
|
4594
|
+
# evals-core.ts's SLUG and SLUG_MAX: what a key's variable is spelt from.
|
|
4595
|
+
PROFILE_SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*")
|
|
4596
|
+
SLUG_MAX = 32
|
|
4130
4597
|
|
|
4131
4598
|
|
|
4132
|
-
def _fields(obj, at, allowed_fields, bad, key_from="
|
|
4599
|
+
def _fields(obj, at, allowed_fields, bad, key_from="EVALSLAB_API_KEY_<ID>"):
|
|
4133
4600
|
for k in obj:
|
|
4134
4601
|
if k in allowed_fields:
|
|
4135
4602
|
continue
|
|
@@ -4200,7 +4667,12 @@ CONNECTION_LISTS = {"http": http_endpoint_test}
|
|
|
4200
4667
|
|
|
4201
4668
|
|
|
4202
4669
|
def connection_problems(pid, conn, at, bad):
|
|
4203
|
-
|
|
4670
|
+
slug = conn.get("slug")
|
|
4671
|
+
if slug is not None and not (isinstance(slug, str) and PROFILE_SLUG.fullmatch(slug) and len(slug) <= SLUG_MAX):
|
|
4672
|
+
bad.append(f"{at}: slug has to be lowercase letters and digits, with hyphens between, at most {SLUG_MAX}")
|
|
4673
|
+
slug = None
|
|
4674
|
+
var = key_var(pid, slug)
|
|
4675
|
+
_fields(conn, at, CONNECTION_FIELDS, bad, var)
|
|
4204
4676
|
ctype = conn.get("type")
|
|
4205
4677
|
if not isinstance(ctype, str) or ctype not in CONNECTION_TYPES:
|
|
4206
4678
|
bad.append(f"{at}: type has to be one of {', '.join(CONNECTION_TYPES)}")
|
|
@@ -4211,7 +4683,7 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4211
4683
|
# Which settings there are is the type's to say; a key among them
|
|
4212
4684
|
# is the server's.
|
|
4213
4685
|
allowed = list(CONNECTION_TYPES.get(ctype, ())) if ctype else list(options)
|
|
4214
|
-
_fields(options, f"{at}'s options", allowed, bad,
|
|
4686
|
+
_fields(options, f"{at}'s options", allowed, bad, var)
|
|
4215
4687
|
else:
|
|
4216
4688
|
bad.append(f"{at}: options has to be an object")
|
|
4217
4689
|
url = conn.get("url") or ""
|
|
@@ -4222,7 +4694,7 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4222
4694
|
parts = urllib.parse.urlsplit(url if "://" in url else "http://" + url)
|
|
4223
4695
|
if parts.username or parts.password:
|
|
4224
4696
|
bad.append(f"{at} has a key in its address, and a key never goes in a pipeline "
|
|
4225
|
-
f"— ${
|
|
4697
|
+
f"— ${var} supplies it")
|
|
4226
4698
|
for why in CONNECTION_CHECKS.get(ctype, lambda _c: [])(conn):
|
|
4227
4699
|
bad.append(f"{at} {why}")
|
|
4228
4700
|
# llama.cpp runs its own llama-server: a hosted address is refused, as the
|
|
@@ -4233,16 +4705,33 @@ def connection_problems(pid, conn, at, bad):
|
|
|
4233
4705
|
bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
|
|
4234
4706
|
|
|
4235
4707
|
|
|
4708
|
+
def eval_group_ref(t):
|
|
4709
|
+
"""The Library group an eval reads the cases of, as the dict itself (so a
|
|
4710
|
+
caller may stamp it in place), or None: version 13's link (`group`) or a
|
|
4711
|
+
private group's `casesFrom`, or an earlier version's `dataset`
|
|
4712
|
+
(evals-core.ts casesRef)."""
|
|
4713
|
+
if not isinstance(t, dict):
|
|
4714
|
+
return None
|
|
4715
|
+
if t.get("type") == "group":
|
|
4716
|
+
if isinstance(t.get("group"), dict):
|
|
4717
|
+
return t["group"]
|
|
4718
|
+
own = t.get("own")
|
|
4719
|
+
return own["casesFrom"] if isinstance(own, dict) and isinstance(own.get("casesFrom"), dict) else None
|
|
4720
|
+
return t["dataset"] if isinstance(t.get("dataset"), dict) else None
|
|
4721
|
+
|
|
4722
|
+
|
|
4236
4723
|
def evals_dataset(doc):
|
|
4237
|
-
"""The
|
|
4238
|
-
first eval that names one. A run grades against one
|
|
4724
|
+
"""The Library group a document's evals read the cases of, or None: the
|
|
4725
|
+
first eval that names one. A run grades against one (the core's
|
|
4239
4726
|
validatePipeline says so). Reads a stored document of any shape --
|
|
4240
|
-
version 11's `evals`, the `tests` before it, version
|
|
4241
|
-
test before that -- since rows keep the document they
|
|
4727
|
+
version 13's links, version 11's `evals`, the `tests` before it, version
|
|
4728
|
+
5's list or the one test before that -- since rows keep the document they
|
|
4729
|
+
were submitted with."""
|
|
4242
4730
|
evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
|
|
4243
4731
|
for t in evals if isinstance(evals, list) else [evals]:
|
|
4244
|
-
|
|
4245
|
-
|
|
4732
|
+
ref = eval_group_ref(t)
|
|
4733
|
+
if ref is not None:
|
|
4734
|
+
return ref
|
|
4246
4735
|
return None
|
|
4247
4736
|
|
|
4248
4737
|
|
|
@@ -4325,6 +4814,7 @@ def run_problems(run):
|
|
|
4325
4814
|
table = run.get("profiles")
|
|
4326
4815
|
if not isinstance(table, dict):
|
|
4327
4816
|
return bad + ["profiles has to be an object of id → connection"]
|
|
4817
|
+
spelt = {}
|
|
4328
4818
|
for pid, conn in table.items():
|
|
4329
4819
|
at = f"profile {pid}"
|
|
4330
4820
|
if not PROFILE_ID.fullmatch(str(pid)):
|
|
@@ -4333,6 +4823,11 @@ def run_problems(run):
|
|
|
4333
4823
|
if not isinstance(conn, dict):
|
|
4334
4824
|
bad.append(f"{at} has to be an object")
|
|
4335
4825
|
continue
|
|
4826
|
+
# Two profiles spelling one variable would hand one the other's key.
|
|
4827
|
+
var = key_var(pid, conn.get("slug") if isinstance(conn.get("slug"), str) else None)
|
|
4828
|
+
if var in spelt:
|
|
4829
|
+
bad.append(f"profiles {spelt[var]} and {pid} both take their key from ${var}")
|
|
4830
|
+
spelt[var] = pid
|
|
4336
4831
|
connection_problems(pid, conn, at, bad)
|
|
4337
4832
|
targets = run.get("targets")
|
|
4338
4833
|
if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
|
|
@@ -4550,9 +5045,23 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4550
5045
|
|
|
4551
5046
|
# ---- routes ---------------------------------------------------------
|
|
4552
5047
|
|
|
5048
|
+
# The eval groups' routes (docs/pipeline-model.md §17) are the datasets'
|
|
5049
|
+
# under their new name: /api/datasets goes on answering the same rows, so
|
|
5050
|
+
# dataset-diff.js and a script written against it keep working. A list
|
|
5051
|
+
# asked for by the new name is `groups`.
|
|
5052
|
+
GROUPS_ROUTE = "/api/eval-groups"
|
|
5053
|
+
grouped = False
|
|
5054
|
+
|
|
5055
|
+
def _alias(self):
|
|
5056
|
+
self.grouped = self.path == self.GROUPS_ROUTE or self.path.startswith((self.GROUPS_ROUTE + "/",
|
|
5057
|
+
self.GROUPS_ROUTE + "?"))
|
|
5058
|
+
if self.grouped:
|
|
5059
|
+
self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
|
|
5060
|
+
|
|
4553
5061
|
def do_GET(self):
|
|
4554
5062
|
if not self._authorised():
|
|
4555
5063
|
return
|
|
5064
|
+
self._alias()
|
|
4556
5065
|
path = self.path.split("?", 1)[0]
|
|
4557
5066
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
4558
5067
|
# Content and Evals are shared, and one to four scenarios run over the
|
|
@@ -4599,8 +5108,16 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4599
5108
|
except ValueError:
|
|
4600
5109
|
return self._json(400, {"error": "limit has to be a number"})
|
|
4601
5110
|
before = (query.get("before") or [None])[0]
|
|
4602
|
-
|
|
5111
|
+
before_id = (query.get("beforeId") or [None])[0]
|
|
5112
|
+
runs, more = QUEUE.list(limit, before, before_id, full=(query.get("full") or [""])[0] == "1")
|
|
4603
5113
|
return self._json(200, {"runs": runs, "more": more})
|
|
5114
|
+
if path.startswith("/api/queue/") and path.endswith("/groups") and path.count("/") == 4:
|
|
5115
|
+
# The bodies of the eval groups a run grades with, by `<id>@<n>`
|
|
5116
|
+
# (§17); null for a run that kept none, as /dataset answers.
|
|
5117
|
+
run_id = path.split("/")[3]
|
|
5118
|
+
if QUEUE is None or QUEUE.get(run_id) is None:
|
|
5119
|
+
return self._send(404, b"not found", "text/plain")
|
|
5120
|
+
return self._json(200, QUEUE.groups(run_id))
|
|
4604
5121
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
4605
5122
|
# The dataset body a graded run was submitted against, which is
|
|
4606
5123
|
# what its verdicts were graded by; the list never carries it.
|
|
@@ -4762,6 +5279,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4762
5279
|
def do_DELETE(self):
|
|
4763
5280
|
if not self._authorised() or not self._from_this_page():
|
|
4764
5281
|
return
|
|
5282
|
+
self._alias()
|
|
4765
5283
|
path = self.path.split("?", 1)[0]
|
|
4766
5284
|
if path == "/api/connections/google":
|
|
4767
5285
|
if CONNECTIONS is None:
|
|
@@ -4829,6 +5347,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4829
5347
|
def do_PATCH(self):
|
|
4830
5348
|
if not self._authorised() or not self._from_this_page():
|
|
4831
5349
|
return
|
|
5350
|
+
self._alias()
|
|
4832
5351
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4833
5352
|
if len(parts) != 4 or parts[1] != "api":
|
|
4834
5353
|
return self._send(404, b"not found", "text/plain")
|
|
@@ -4859,6 +5378,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4859
5378
|
def do_PUT(self):
|
|
4860
5379
|
if not self._authorised() or not self._from_this_page():
|
|
4861
5380
|
return
|
|
5381
|
+
self._alias()
|
|
4862
5382
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4863
5383
|
if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
|
|
4864
5384
|
return self._prompts_put(parts[3])
|
|
@@ -4921,6 +5441,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4921
5441
|
return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
|
|
4922
5442
|
|
|
4923
5443
|
def _post(self):
|
|
5444
|
+
self._alias()
|
|
4924
5445
|
path = self.path.split("?", 1)[0]
|
|
4925
5446
|
if path == "/api/state":
|
|
4926
5447
|
return self._state_write()
|
|
@@ -4940,6 +5461,12 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4940
5461
|
return self._prompts_post(path)
|
|
4941
5462
|
if path.startswith("/api/sources"):
|
|
4942
5463
|
return self._sources_post(path)
|
|
5464
|
+
if path == "/api/bundle":
|
|
5465
|
+
data, err = build_bundle(self._payload() or {})
|
|
5466
|
+
if err:
|
|
5467
|
+
return self._json(err[0], {"error": err[1]})
|
|
5468
|
+
return self._send(200, data, "application/zip",
|
|
5469
|
+
(("Content-Disposition", 'attachment; filename="evals.zip"'),))
|
|
4943
5470
|
payload = self._payload()
|
|
4944
5471
|
if payload is None:
|
|
4945
5472
|
return self._json(400, {"error": "bad body"})
|
|
@@ -5012,7 +5539,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5012
5539
|
|
|
5013
5540
|
def _models_list(self, payload):
|
|
5014
5541
|
base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
|
|
5015
|
-
key =
|
|
5542
|
+
key = relay_key(payload)
|
|
5016
5543
|
ctype = str(payload.get("type") or "")
|
|
5017
5544
|
# A type that lists some other way (an HTTP endpoint: its own test
|
|
5018
5545
|
# path, key header and headers) says where and how.
|
|
@@ -5063,7 +5590,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5063
5590
|
# and how the key travels, per the type -- the same mirror the models
|
|
5064
5591
|
# list uses, so a client cannot point the relay at a path of its own.
|
|
5065
5592
|
base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
|
|
5066
|
-
key =
|
|
5593
|
+
key = relay_key(payload)
|
|
5067
5594
|
if not header_safe(key):
|
|
5068
5595
|
return self._json(400, {"error": "That key has characters that "
|
|
5069
5596
|
"cannot be sent in a header, so "
|
|
@@ -5099,6 +5626,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5099
5626
|
for n, d in docs.items()})
|
|
5100
5627
|
if stale is not None:
|
|
5101
5628
|
return self._json(409, {"stale": stale})
|
|
5629
|
+
# A pin names a group's version, so the group keeps that version from
|
|
5630
|
+
# the moment a pipeline pins it (§17).
|
|
5631
|
+
if "promptlab.workflows" in docs and DATASETS is not None:
|
|
5632
|
+
DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
|
|
5102
5633
|
return self._json(200, {"versions": versions})
|
|
5103
5634
|
|
|
5104
5635
|
# ---- The run queue (#530) ----------------------------------------------
|
|
@@ -5137,20 +5668,30 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5137
5668
|
# The plugins it runs under, as installed now: the server's to say,
|
|
5138
5669
|
# whatever the document claimed.
|
|
5139
5670
|
run["plugins"] = PLUGINS.stamp() if PLUGINS is not None else []
|
|
5140
|
-
#
|
|
5141
|
-
#
|
|
5142
|
-
#
|
|
5143
|
-
|
|
5144
|
-
|
|
5145
|
-
|
|
5146
|
-
|
|
5147
|
-
|
|
5671
|
+
# Each Library group the evals read, resolved once (§17): a pinned
|
|
5672
|
+
# link to its pin, anything else to the group's newest version. The
|
|
5673
|
+
# reference records the version, `n`, and its body's fingerprint, and
|
|
5674
|
+
# the body is kept with the row under `<id>@<n>`: the worker grades
|
|
5675
|
+
# with that and nothing else, and the version is frozen from now on.
|
|
5676
|
+
groups = {}
|
|
5677
|
+
for t in run["evals"]:
|
|
5678
|
+
ref = eval_group_ref(t)
|
|
5679
|
+
if ref is None:
|
|
5680
|
+
continue
|
|
5681
|
+
pin = t.get("pin") if t.get("type") == "group" and t.get("group") is ref else None
|
|
5682
|
+
got = DATASETS.resolve(ref.get("id"), pin if type(pin) is int else None) if DATASETS else None
|
|
5683
|
+
if got is None:
|
|
5148
5684
|
return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
|
|
5149
|
-
|
|
5150
|
-
|
|
5151
|
-
|
|
5152
|
-
|
|
5153
|
-
|
|
5685
|
+
n, body = got
|
|
5686
|
+
if n is None:
|
|
5687
|
+
return self._json(400, {"error": f"the run cannot be queued: {body}"})
|
|
5688
|
+
ref["n"], ref["version"] = n, fingerprint(body)
|
|
5689
|
+
groups[group_key(ref)] = body
|
|
5690
|
+
# The worker is handed one body until it reads one per group (#233).
|
|
5691
|
+
if len(groups) > 1:
|
|
5692
|
+
return self._json(400, {"error": "the run cannot be queued: its evals grade against "
|
|
5693
|
+
"one version of one Library eval group at a time"})
|
|
5694
|
+
return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
|
|
5154
5695
|
|
|
5155
5696
|
# ---- Datasets ----------------------------------------------------------
|
|
5156
5697
|
# Rows in the store (Datasets above). The page reads one by id; an export
|
|
@@ -5244,9 +5785,16 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5244
5785
|
DATASETS.lazy_trash()
|
|
5245
5786
|
parts = path.split("/")
|
|
5246
5787
|
if len(parts) == 3:
|
|
5247
|
-
return self._json(200, {"datasets": DATASETS.list()})
|
|
5788
|
+
return self._json(200, {"groups" if self.grouped else "datasets": DATASETS.list()})
|
|
5248
5789
|
if len(parts) == 4 and parts[3] == "export":
|
|
5249
|
-
return self._download(DATASETS.export_all(), "
|
|
5790
|
+
return self._download(DATASETS.export_all(), "eval-groups", "all")
|
|
5791
|
+
if len(parts) == 5 and parts[4] == "versions":
|
|
5792
|
+
got = DATASETS.versions(parts[3])
|
|
5793
|
+
return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
|
|
5794
|
+
if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
|
|
5795
|
+
got = DATASETS.version_body(parts[3], int(parts[5]))
|
|
5796
|
+
return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
|
|
5797
|
+
else self._send(404, b"not found", "text/plain")
|
|
5250
5798
|
if len(parts) == 4 and parts[3] == "archived-rules":
|
|
5251
5799
|
return self._json(200, {"rules": DATASETS.archived_rules()})
|
|
5252
5800
|
if len(parts) == 4:
|
|
@@ -5256,7 +5804,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5256
5804
|
doc = DATASETS.export(parts[3])
|
|
5257
5805
|
if doc is None:
|
|
5258
5806
|
return self._send(404, b"not found", "text/plain")
|
|
5259
|
-
return self._download(doc, "
|
|
5807
|
+
return self._download(doc, "eval-group", doc["group"]["name"])
|
|
5260
5808
|
return self._send(404, b"not found", "text/plain")
|
|
5261
5809
|
|
|
5262
5810
|
def _download(self, doc, kind, name):
|
|
@@ -5284,6 +5832,12 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5284
5832
|
if err:
|
|
5285
5833
|
return self._json(err[0], {"error": err[1]})
|
|
5286
5834
|
return self._json(200, dataset)
|
|
5835
|
+
if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
|
|
5836
|
+
self._payload()
|
|
5837
|
+
dataset, err = DATASETS.restore_version(parts[3], int(parts[5]))
|
|
5838
|
+
if err:
|
|
5839
|
+
return self._json(err[0], {"error": err[1]})
|
|
5840
|
+
return self._json(200, dataset)
|
|
5287
5841
|
if len(parts) == 4 and parts[3] == "import":
|
|
5288
5842
|
doc, err = self._dataset_payload(DATASET_IMPORT_CAP)
|
|
5289
5843
|
if err:
|