evals-lab 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +77 -0
- package/bin/run.js +14 -3
- package/lab/VERSION +1 -1
- package/lab/evals-core.mjs +353 -75
- package/lab/flows/flowApi.mjs +153 -0
- package/lab/flows/record.mjs +495 -0
- package/lab/kinds/list.mjs +2 -1
- package/lab/metrics/builtin.mjs +39 -34
- package/lab/run-evals.js +220 -83
- package/lab/server.py +617 -176
- package/lab/web/dist/assets/gallery-BsbUQC7Q.js +3 -0
- package/lab/web/dist/assets/main-BQL5j5oF.js +20 -0
- package/lab/web/dist/assets/main-Cza2gwQd.css +1 -0
- package/lab/web/dist/assets/tokens-C3kp9sWp.js +61 -0
- package/lab/web/dist/assets/tokens-CjqaFuXm.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-BFf9vis6.js +0 -3
- package/lab/web/dist/assets/main-DeeRLWnO.css +0 -1
- package/lab/web/dist/assets/main-LT0U2TYF.js +0 -21
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +0 -59
- package/lab/web/dist/assets/tokens-lq45aAPS.css +0 -1
package/lab/server.py
CHANGED
|
@@ -298,23 +298,53 @@ FILE_TYPES = {
|
|
|
298
298
|
}
|
|
299
299
|
|
|
300
300
|
# The kinds of Source, mirrored from evals-core.ts's SOURCE_TYPES: Python
|
|
301
|
-
# cannot load TypeScript, so the server keeps what it enforces
|
|
302
|
-
#
|
|
303
|
-
#
|
|
301
|
+
# cannot load TypeScript, so the server keeps what it enforces. A kind is a
|
|
302
|
+
# registry entry, never a branch on its id. The row's `type` is the kind; a
|
|
303
|
+
# workflow kind's enforcement comes from the platform its `config.platform`
|
|
304
|
+
# names (WORKFLOW_PLATFORMS below), a kind marked `platforms`.
|
|
304
305
|
SOURCE_TYPES = {
|
|
305
306
|
"files": {"label": "File Library", "uploads": FILE_TYPES},
|
|
306
|
-
|
|
307
|
-
# its records arrive with a later phase -- and a definition, kept as a
|
|
308
|
-
# versioned snapshot beside the row.
|
|
309
|
-
#
|
|
310
|
-
# `uploads` are the files it takes; `take`, when there is one, checks and
|
|
311
|
-
# rewrites each before it is kept; `signIn` is the sign-in it is made with,
|
|
312
|
-
# without which the lab cannot make one; `definition` means it keeps one.
|
|
313
|
-
"power-automate": {"label": "Power Automate workflow", "definition": True, "signIn": "microsoft",
|
|
314
|
-
"uploads": {".json": "application/json; charset=utf-8"}},
|
|
307
|
+
"workflow": {"label": "Workflow", "platforms": True},
|
|
315
308
|
}
|
|
316
309
|
DEFAULT_SOURCE_TYPE = "files"
|
|
317
310
|
|
|
311
|
+
# The workflow platforms, mirrored from evals-core.ts's WORKFLOW_PLATFORMS: the
|
|
312
|
+
# flow engine a workflow Source speaks to, named in its `config.platform`. A
|
|
313
|
+
# workflow kind's enforcement is the platform's, not the kind's.
|
|
314
|
+
#
|
|
315
|
+
# `uploads` are the files it takes; `take`, when there is one, checks and
|
|
316
|
+
# rewrites each before it is kept; `signIn` is the sign-in it is made with,
|
|
317
|
+
# without which the lab cannot make one; `definition` means it keeps one.
|
|
318
|
+
WORKFLOW_PLATFORMS = {
|
|
319
|
+
"power-automate": {"label": "Power Automate flow", "definition": True, "signIn": "microsoft",
|
|
320
|
+
"uploads": {".json": "application/json; charset=utf-8"}},
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _config_platform(config):
|
|
325
|
+
"""The platform a Source's stored config (a JSON string or None) names,
|
|
326
|
+
or None -- what the page labels a workflow by."""
|
|
327
|
+
try:
|
|
328
|
+
cfg = json.loads(config) if config else None
|
|
329
|
+
except ValueError:
|
|
330
|
+
return None
|
|
331
|
+
platform = cfg.get("platform") if isinstance(cfg, dict) else None
|
|
332
|
+
return platform if isinstance(platform, str) else None
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def source_entry(ptype, config):
|
|
336
|
+
"""The enforcement entry for a Source of type [ptype] and [config] (a
|
|
337
|
+
dict or None): its kind's, or, for a workflow kind, the platform its
|
|
338
|
+
config names. None for an unknown kind, or a workflow whose platform the
|
|
339
|
+
lab does not know. A reader asks the entry, never the id."""
|
|
340
|
+
kind = SOURCE_TYPES.get(ptype)
|
|
341
|
+
if kind is None:
|
|
342
|
+
return None
|
|
343
|
+
if kind.get("platforms"):
|
|
344
|
+
platform = config.get("platform") if isinstance(config, dict) else None
|
|
345
|
+
return WORKFLOW_PLATFORMS.get(platform)
|
|
346
|
+
return kind
|
|
347
|
+
|
|
318
348
|
# The Entra app the page signs in to Microsoft 365 with (docs/power-automate.md
|
|
319
349
|
# § Registering the app). A single-page app has no secret, so both are public
|
|
320
350
|
# and served to the page; the token it gets stays in the browser, and the
|
|
@@ -481,7 +511,7 @@ def take_record(name, data):
|
|
|
481
511
|
return json.dumps(kept, indent=2).encode("utf-8"), None
|
|
482
512
|
|
|
483
513
|
|
|
484
|
-
|
|
514
|
+
WORKFLOW_PLATFORMS["power-automate"]["take"] = take_record
|
|
485
515
|
|
|
486
516
|
# The sign-ins a Source type may be made with, and whether this lab has each:
|
|
487
517
|
# a type naming one this lab lacks cannot be made.
|
|
@@ -602,6 +632,45 @@ def hide_keys(docs: dict) -> dict:
|
|
|
602
632
|
return out
|
|
603
633
|
|
|
604
634
|
|
|
635
|
+
# ---- workspaces (docs/workspaces.md) -------------------------------------
|
|
636
|
+
# A workspace is the lab's first tenant dimension: a row in `workspaces` and a
|
|
637
|
+
# filter on every scopable table. Phase 1 is invisible -- a request that names
|
|
638
|
+
# no workspace resolves to the flagged default, so the existing page and CI
|
|
639
|
+
# keep working. It is isolation, not access control: until multi-user lands, a
|
|
640
|
+
# workspace is a filter, not a permission boundary (AGENTS.md).
|
|
641
|
+
#
|
|
642
|
+
# The workspace the current thread's db work is scoped to is held here, bound
|
|
643
|
+
# per request by the Handler and per run by the queue worker; unset, a scoped
|
|
644
|
+
# read falls back to the store's flagged default, which keeps direct callers
|
|
645
|
+
# (startup, a check's own calls) on the default workspace.
|
|
646
|
+
_WS = threading.local()
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def set_workspace(ws):
|
|
650
|
+
"""Bind the current thread to workspace `ws` for its db work; None clears
|
|
651
|
+
it, so scoped reads fall back to the store's default workspace."""
|
|
652
|
+
_WS.ws = ws
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def workspace_of(store) -> str:
|
|
656
|
+
ws = getattr(_WS, "ws", None)
|
|
657
|
+
return ws if ws else store.default_ws
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
# The `connection_shares` sentinels (phase 4 fills and enforces the table;
|
|
661
|
+
# phase 1 only creates it empty): every workspace, and future ones.
|
|
662
|
+
SHARE_ALL = "*all"
|
|
663
|
+
SHARE_NEW = "*new"
|
|
664
|
+
|
|
665
|
+
# How the `docs` table scopes by key (docs/workspaces.md): workflows are
|
|
666
|
+
# per-workspace, profiles/tokens global, and promptlab.versions is split --
|
|
667
|
+
# its pipeline/dataset slices per-workspace, its profile slice global, because
|
|
668
|
+
# profiles are global so their Restore must be too. served()/write() route
|
|
669
|
+
# each key by these; any other key is per-workspace.
|
|
670
|
+
DOC_GLOBAL = ("promptlab.profiles", "promptlab.tokens")
|
|
671
|
+
DOC_SPLIT = "promptlab.versions"
|
|
672
|
+
|
|
673
|
+
|
|
605
674
|
class Store:
|
|
606
675
|
"""
|
|
607
676
|
Documents by name, each with a version that goes up by one per write.
|
|
@@ -618,17 +687,36 @@ class Store:
|
|
|
618
687
|
self.path = path
|
|
619
688
|
self.lock = threading.Lock()
|
|
620
689
|
with self.lock, closing(sqlite3.connect(path)) as db, db:
|
|
621
|
-
|
|
622
|
-
|
|
690
|
+
# The workspace registry and the share table come first:
|
|
691
|
+
# everything else is scoped to a workspace, and the flagged default
|
|
692
|
+
# must exist before any backfill can assign rows to it.
|
|
693
|
+
db.execute("CREATE TABLE IF NOT EXISTS workspaces ("
|
|
694
|
+
"id TEXT PRIMARY KEY, slug TEXT UNIQUE, name TEXT NOT NULL, "
|
|
695
|
+
"archived INTEGER NOT NULL DEFAULT 0, is_default INTEGER NOT NULL DEFAULT 0, "
|
|
696
|
+
"last_used TEXT, created_at TEXT, trash TEXT, trashed_at REAL)")
|
|
697
|
+
# Shared-by-choice connections (profiles, Accounts): phase 4 fills
|
|
698
|
+
# and enforces this; phase 1 only creates it empty.
|
|
699
|
+
db.execute("CREATE TABLE IF NOT EXISTS connection_shares ("
|
|
700
|
+
"kind TEXT NOT NULL, id TEXT NOT NULL, workspace TEXT NOT NULL, "
|
|
701
|
+
"PRIMARY KEY (kind, id, workspace))")
|
|
702
|
+
self.default_ws = self._ensure_default(db)
|
|
623
703
|
db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
|
|
624
704
|
# The key of a profile a write took out, by id, for TRASH_SECONDS:
|
|
625
705
|
# Undo puts the profile back holding KEY_HELD, and this is what
|
|
626
|
-
# it holds.
|
|
706
|
+
# it holds. Profiles are global, so the ring is too.
|
|
627
707
|
db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
|
|
628
708
|
"key TEXT NOT NULL, at REAL NOT NULL)")
|
|
709
|
+
# The docs table keys documents by (name, workspace): a store from
|
|
710
|
+
# before workspaces keyed them by name alone and is rebuilt once,
|
|
711
|
+
# the global keys moved under NULL and promptlab.versions split.
|
|
712
|
+
# Idempotent: a later open finds the workspace column and leaves it.
|
|
713
|
+
cols = {r[1] for r in db.execute("PRAGMA table_info(docs)")}
|
|
714
|
+
if not cols:
|
|
715
|
+
self._create_docs(db)
|
|
629
716
|
# History was a document, capped at what a browser could hold. The
|
|
630
717
|
# first start with the table moves what that document had into it,
|
|
631
|
-
# once, and drops the document so the page stops carrying it.
|
|
718
|
+
# once, and drops the document so the page stops carrying it. It
|
|
719
|
+
# reads name/body, so it runs on either docs schema.
|
|
632
720
|
old = db.execute("SELECT body FROM docs WHERE name = 'promptlab.runs'").fetchone()
|
|
633
721
|
if old and not db.execute("SELECT 1 FROM runs LIMIT 1").fetchone():
|
|
634
722
|
for run in json.loads(old[0] or "null") or []:
|
|
@@ -636,20 +724,118 @@ class Store:
|
|
|
636
724
|
db.execute("INSERT OR IGNORE INTO runs (at, body) VALUES (?, ?)",
|
|
637
725
|
(run["at"], json.dumps(run)))
|
|
638
726
|
db.execute("DELETE FROM docs WHERE name = 'promptlab.runs'")
|
|
727
|
+
# The run history is the queue table now; this `runs` table holds
|
|
728
|
+
# only what the pre-queue history document migrated into it, and is
|
|
729
|
+
# served from nowhere. It carries the workspace column all the same,
|
|
730
|
+
# so the scopable-table set is whole and anything that ever reads it
|
|
731
|
+
# inherits the filter (docs/workspaces.md).
|
|
732
|
+
if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(runs)")}:
|
|
733
|
+
db.execute("ALTER TABLE runs ADD COLUMN workspace TEXT")
|
|
734
|
+
db.execute("UPDATE runs SET workspace = ? WHERE workspace IS NULL", (self.default_ws,))
|
|
735
|
+
if cols and "workspace" not in cols:
|
|
736
|
+
self._rebuild_docs(db)
|
|
737
|
+
|
|
738
|
+
def _ensure_default(self, db) -> str:
|
|
739
|
+
"""The flagged default workspace's id, seeding "Default" (/w/default,
|
|
740
|
+
is_default = 1) on a store that has none. Defined by the flag, not its
|
|
741
|
+
name or slug, so a later rename never moves it (docs/workspaces.md)."""
|
|
742
|
+
row = db.execute("SELECT id FROM workspaces WHERE is_default = 1").fetchone()
|
|
743
|
+
if row:
|
|
744
|
+
return row[0]
|
|
745
|
+
wid = secrets.token_hex(6)
|
|
746
|
+
now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
747
|
+
db.execute("INSERT INTO workspaces (id, slug, name, archived, is_default, last_used, created_at) "
|
|
748
|
+
"VALUES (?, 'default', 'Default', 0, 1, ?, ?)", (wid, now, now))
|
|
749
|
+
return wid
|
|
750
|
+
|
|
751
|
+
def workspace(self, slug: str) -> str:
|
|
752
|
+
"""The id of the workspace at /w/<slug>, or None. An unknown slug is
|
|
753
|
+
None; the Handler falls back to the default so a stale address still
|
|
754
|
+
reaches a working lab until the page's Gone state lands (phase 3)."""
|
|
755
|
+
with self.lock, closing(sqlite3.connect(self.path)) as db:
|
|
756
|
+
row = db.execute("SELECT id FROM workspaces WHERE slug = ?", (slug,)).fetchone()
|
|
757
|
+
return row[0] if row else None
|
|
639
758
|
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
759
|
+
@staticmethod
|
|
760
|
+
def _create_docs(db):
|
|
761
|
+
db.execute("CREATE TABLE docs (name TEXT NOT NULL, workspace TEXT, "
|
|
762
|
+
"version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
|
|
763
|
+
# NULLs are distinct in a UNIQUE index, so the global keys cannot rely
|
|
764
|
+
# on one PRIMARY KEY for their one-row-per-name rule: two partial
|
|
765
|
+
# indexes, scoped rows keyed by (name, workspace) and global rows by
|
|
766
|
+
# name, give each its own uniqueness and its own upsert target.
|
|
767
|
+
db.execute("CREATE UNIQUE INDEX docs_scoped ON docs(name, workspace) WHERE workspace IS NOT NULL")
|
|
768
|
+
db.execute("CREATE UNIQUE INDEX docs_global ON docs(name) WHERE workspace IS NULL")
|
|
769
|
+
|
|
770
|
+
def _rebuild_docs(self, db):
|
|
771
|
+
rows = db.execute("SELECT name, version, body, updated_at FROM docs").fetchall()
|
|
772
|
+
db.execute("ALTER TABLE docs RENAME TO docs_old")
|
|
773
|
+
self._create_docs(db)
|
|
774
|
+
put = lambda name, ws, version, body, at: db.execute(
|
|
775
|
+
"INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, ?, ?, ?, ?)",
|
|
776
|
+
(name, ws, version, body, at))
|
|
777
|
+
for name, version, body, at in rows:
|
|
778
|
+
if name in DOC_GLOBAL:
|
|
779
|
+
put(name, None, version, body, at)
|
|
780
|
+
elif name == DOC_SPLIT:
|
|
781
|
+
parsed = json.loads(body) if body else {}
|
|
782
|
+
put(name, self.default_ws, version, None if body is None else json.dumps(
|
|
783
|
+
{"pipeline": parsed.get("pipeline", {}), "dataset": parsed.get("dataset", {})}), at)
|
|
784
|
+
put(name, None, version, None if body is None else json.dumps(
|
|
785
|
+
{"profile": parsed.get("profile", {})}), at)
|
|
786
|
+
else:
|
|
787
|
+
put(name, self.default_ws, version, body, at)
|
|
788
|
+
db.execute("DROP TABLE docs_old")
|
|
789
|
+
|
|
790
|
+
def _doc_rows(self, db, ws: str) -> dict:
|
|
791
|
+
"""The logical document set a workspace reads: its scoped rows, the
|
|
792
|
+
global rows (profiles, tokens), and promptlab.versions reassembled from
|
|
793
|
+
its per-workspace pipeline/dataset slice and the global profile slice,
|
|
794
|
+
carrying the per-workspace slice's version so the page's one version
|
|
795
|
+
per key still arbitrates pipeline/dataset Restore."""
|
|
796
|
+
parse = lambda b: None if b is None else json.loads(b)
|
|
797
|
+
scoped = {name: (version, body) for name, version, body in db.execute(
|
|
798
|
+
"SELECT name, version, body FROM docs WHERE workspace = ?", (ws,))}
|
|
799
|
+
glob = {name: (version, body) for name, version, body in db.execute(
|
|
800
|
+
"SELECT name, version, body FROM docs WHERE workspace IS NULL")}
|
|
801
|
+
out = {name: {"version": v, "body": parse(b)}
|
|
802
|
+
for name, (v, b) in scoped.items() if name != DOC_SPLIT}
|
|
803
|
+
for name in DOC_GLOBAL:
|
|
804
|
+
if name in glob:
|
|
805
|
+
v, b = glob[name]
|
|
806
|
+
out[name] = {"version": v, "body": parse(b)}
|
|
807
|
+
sv, gv = scoped.get(DOC_SPLIT), glob.get(DOC_SPLIT)
|
|
808
|
+
if sv is not None or gv is not None:
|
|
809
|
+
sver, sbody = sv if sv is not None else (0, None)
|
|
810
|
+
sbody, gbody = parse(sbody) or {}, parse(gv[1]) if gv is not None else {}
|
|
811
|
+
out[DOC_SPLIT] = {"version": sver, "body": {
|
|
812
|
+
"pipeline": sbody.get("pipeline", {}), "dataset": sbody.get("dataset", {}),
|
|
813
|
+
"profile": (gbody or {}).get("profile", {})}}
|
|
814
|
+
return out
|
|
643
815
|
|
|
644
|
-
def all(self) -> dict:
|
|
816
|
+
def all(self, ws=None) -> dict:
|
|
645
817
|
with self.lock, closing(sqlite3.connect(self.path)) as db:
|
|
646
|
-
return self.
|
|
647
|
-
|
|
648
|
-
def served(self) -> dict:
|
|
649
|
-
"""What a browser is handed: the SYNCED documents
|
|
650
|
-
key's rows stay in the store without reaching a page
|
|
651
|
-
profile's key (KEY_HELD)."""
|
|
652
|
-
return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
|
|
818
|
+
return self._doc_rows(db, ws or workspace_of(self))
|
|
819
|
+
|
|
820
|
+
def served(self, ws=None) -> dict:
|
|
821
|
+
"""What a browser is handed for its workspace: the SYNCED documents
|
|
822
|
+
only, so a retired key's rows stay in the store without reaching a page
|
|
823
|
+
again, and no profile's key (KEY_HELD)."""
|
|
824
|
+
return hide_keys({n: d for n, d in self.all(ws).items() if n in SYNCED})
|
|
825
|
+
|
|
826
|
+
def _put(self, db, name, ws, version, body, at):
|
|
827
|
+
text = None if body is None else json.dumps(body)
|
|
828
|
+
if ws is None:
|
|
829
|
+
db.execute(
|
|
830
|
+
"INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, NULL, ?, ?, ?) "
|
|
831
|
+
"ON CONFLICT(name) WHERE workspace IS NULL DO UPDATE SET version = excluded.version, "
|
|
832
|
+
"body = excluded.body, updated_at = excluded.updated_at", (name, version, text, at))
|
|
833
|
+
else:
|
|
834
|
+
db.execute(
|
|
835
|
+
"INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, ?, ?, ?, ?) "
|
|
836
|
+
"ON CONFLICT(name, workspace) WHERE workspace IS NOT NULL DO UPDATE SET "
|
|
837
|
+
"version = excluded.version, body = excluded.body, updated_at = excluded.updated_at",
|
|
838
|
+
(name, ws, version, text, at))
|
|
653
839
|
|
|
654
840
|
def _keyring(self, db, now: dict) -> dict:
|
|
655
841
|
"""Each profile id's key as the store holds it: its profile's, else a
|
|
@@ -668,18 +854,24 @@ class Store:
|
|
|
668
854
|
return ring
|
|
669
855
|
|
|
670
856
|
def held_key(self, key: str) -> str:
|
|
671
|
-
"""The key KEY_HELD stands for, or "" when the store holds none.
|
|
857
|
+
"""The key KEY_HELD stands for, or "" when the store holds none.
|
|
858
|
+
Profiles are global, so the ring is read under the default workspace."""
|
|
672
859
|
with self.lock, closing(sqlite3.connect(self.path)) as db, db:
|
|
673
|
-
return self._keyring(db, self.
|
|
860
|
+
return self._keyring(db, self._doc_rows(db, self.default_ws)).get(key[len(KEY_HELD):], "")
|
|
674
861
|
|
|
675
|
-
def write(self, docs: dict):
|
|
862
|
+
def write(self, docs: dict, ws=None):
|
|
676
863
|
"""
|
|
677
|
-
`docs` is {name: {"version": the version it began from, "body": ...}}
|
|
678
|
-
|
|
679
|
-
|
|
864
|
+
`docs` is {name: {"version": the version it began from, "body": ...}},
|
|
865
|
+
written into workspace `ws` (the current thread's, by default). Each
|
|
866
|
+
key is routed by scope: workflows and the rest per-workspace, profiles
|
|
867
|
+
and tokens global, promptlab.versions split (pipeline/dataset
|
|
868
|
+
per-workspace, profile global). Returns ({name: new version}, None), or
|
|
869
|
+
(None, {name: current}) for every document that has moved on, with
|
|
870
|
+
nothing written.
|
|
680
871
|
"""
|
|
872
|
+
ws = ws or workspace_of(self)
|
|
681
873
|
with self.lock, closing(sqlite3.connect(self.path)) as db:
|
|
682
|
-
now = self.
|
|
874
|
+
now = self._doc_rows(db, ws)
|
|
683
875
|
have = lambda n: now.get(n, {"version": 0, "body": None})
|
|
684
876
|
stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
|
|
685
877
|
if stale:
|
|
@@ -701,11 +893,19 @@ class Store:
|
|
|
701
893
|
for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
|
|
702
894
|
if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
|
|
703
895
|
for n, d in docs.items():
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
896
|
+
version, body = have(n)["version"] + 1, d["body"]
|
|
897
|
+
if n == DOC_SPLIT:
|
|
898
|
+
parsed = body if isinstance(body, dict) else {}
|
|
899
|
+
self._put(db, n, ws, version, None if body is None else {
|
|
900
|
+
"pipeline": parsed.get("pipeline", {}), "dataset": parsed.get("dataset", {})}, at)
|
|
901
|
+
# The profile slice is global, on its own version
|
|
902
|
+
# counter: profiles are shared, so their Restore is too.
|
|
903
|
+
gv = db.execute("SELECT version FROM docs WHERE name = ? AND workspace IS NULL",
|
|
904
|
+
(n,)).fetchone()
|
|
905
|
+
self._put(db, n, None, (gv[0] if gv else 0) + 1,
|
|
906
|
+
None if body is None else {"profile": parsed.get("profile", {})}, at)
|
|
907
|
+
else:
|
|
908
|
+
self._put(db, n, None if n in DOC_GLOBAL else ws, version, body, at)
|
|
709
909
|
return {n: have(n)["version"] + 1 for n in docs}, None
|
|
710
910
|
|
|
711
911
|
|
|
@@ -781,6 +981,34 @@ class Sources:
|
|
|
781
981
|
f"DEFAULT '{DEFAULT_SOURCE_TYPE}'")
|
|
782
982
|
if "config" not in have:
|
|
783
983
|
db.execute("ALTER TABLE sources ADD COLUMN config TEXT")
|
|
984
|
+
# The workspace a Source belongs to (docs/workspaces.md). A store
|
|
985
|
+
# from before workspaces gains the column; every user Source with
|
|
986
|
+
# none -- a pre-workspace row, or one left unscoped by any edge --
|
|
987
|
+
# backfills to the default on open, idempotently, while the system
|
|
988
|
+
# samples row stays workspace-agnostic (NULL) so it shows in every
|
|
989
|
+
# workspace.
|
|
990
|
+
if "workspace" not in have:
|
|
991
|
+
db.execute("ALTER TABLE sources ADD COLUMN workspace TEXT")
|
|
992
|
+
db.execute("UPDATE sources SET workspace = ? WHERE workspace IS NULL AND system = 0",
|
|
993
|
+
(store.default_ws,))
|
|
994
|
+
# A Power Automate Source was its own kind once (type
|
|
995
|
+
# "power-automate", docs/power-automate.md); it is now the generic
|
|
996
|
+
# workflow kind, the platform named in its config
|
|
997
|
+
# (docs/sources-tab.md). Converted here, once and idempotently: a
|
|
998
|
+
# later open finds none left. remove()/restore_trash keep a row's
|
|
999
|
+
# type and config, so Undo brings a converted Source back exactly,
|
|
1000
|
+
# still speaking to Power Automate.
|
|
1001
|
+
for sid, config in db.execute(
|
|
1002
|
+
"SELECT id, config FROM sources WHERE type = 'power-automate'").fetchall():
|
|
1003
|
+
try:
|
|
1004
|
+
cfg = json.loads(config) if config else {}
|
|
1005
|
+
except ValueError:
|
|
1006
|
+
cfg = {}
|
|
1007
|
+
if not isinstance(cfg, dict):
|
|
1008
|
+
cfg = {}
|
|
1009
|
+
cfg["platform"] = "power-automate"
|
|
1010
|
+
db.execute("UPDATE sources SET type = 'workflow', config = ? WHERE id = ?",
|
|
1011
|
+
(json.dumps(cfg), sid))
|
|
784
1012
|
# A flow's definition, one row per version: a snapshot the page
|
|
785
1013
|
# took and redacted, redacted again here.
|
|
786
1014
|
db.execute("CREATE TABLE IF NOT EXISTS source_definitions ("
|
|
@@ -809,26 +1037,37 @@ class Sources:
|
|
|
809
1037
|
out = []
|
|
810
1038
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
811
1039
|
db.row_factory = sqlite3.Row
|
|
812
|
-
for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
|
|
813
|
-
"
|
|
1040
|
+
for r in db.execute("SELECT id, name, system, bytes, type, config, created_at FROM sources "
|
|
1041
|
+
"WHERE workspace = ? OR system = 1 "
|
|
1042
|
+
"ORDER BY system DESC, name COLLATE NOCASE", (workspace_of(self.store),)):
|
|
814
1043
|
if r["system"]:
|
|
815
1044
|
files = self._sample_files()
|
|
816
1045
|
out.append({"id": r["id"], "name": r["name"], "system": True,
|
|
817
1046
|
"type": r["type"],
|
|
818
1047
|
"files": len(files), "bytes": sum(f["bytes"] for f in files)})
|
|
819
1048
|
else:
|
|
820
|
-
n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
1049
|
+
n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
|
|
1050
|
+
(r["id"],)).fetchone()
|
|
1051
|
+
# When it last changed: made, or a file added -- what a
|
|
1052
|
+
# picker orders its recent Sources by.
|
|
1053
|
+
row = {"id": r["id"], "name": r["name"], "system": False,
|
|
1054
|
+
"type": r["type"], "files": n, "bytes": r["bytes"],
|
|
1055
|
+
"changed": max(filter(None, (r["created_at"], last)))}
|
|
1056
|
+
# A workflow's platform is how the page labels it; nothing
|
|
1057
|
+
# else of its config rides in the summary.
|
|
1058
|
+
platform = _config_platform(r["config"])
|
|
1059
|
+
if platform is not None:
|
|
1060
|
+
row["platform"] = platform
|
|
1061
|
+
out.append(row)
|
|
824
1062
|
return out
|
|
825
1063
|
|
|
826
1064
|
def get(self, sid) -> dict:
|
|
827
1065
|
"""One Source with its file list, or None."""
|
|
828
1066
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
829
1067
|
db.row_factory = sqlite3.Row
|
|
830
|
-
r = db.execute("SELECT id, name, system, bytes, type, config FROM sources
|
|
831
|
-
(
|
|
1068
|
+
r = db.execute("SELECT id, name, system, bytes, type, config FROM sources "
|
|
1069
|
+
"WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1070
|
+
(sid, workspace_of(self.store))).fetchone()
|
|
832
1071
|
if r is None:
|
|
833
1072
|
return None
|
|
834
1073
|
if r["system"]:
|
|
@@ -875,14 +1114,18 @@ class Sources:
|
|
|
875
1114
|
return None, (400, "that name is too long")
|
|
876
1115
|
if not isinstance(kind, str) or kind not in SOURCE_TYPES:
|
|
877
1116
|
return None, (400, f"the lab has no Source type {kind!r}")
|
|
1117
|
+
# A workflow kind needs a platform the lab knows, named in its config.
|
|
1118
|
+
if source_entry(kind, config) is None:
|
|
1119
|
+
return None, (400, "the lab has no such Source platform")
|
|
878
1120
|
kept, err = self._config(config)
|
|
879
1121
|
if err:
|
|
880
1122
|
return None, err
|
|
881
1123
|
sid = secrets.token_hex(6)
|
|
882
1124
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
883
|
-
db.execute("INSERT INTO sources (id, name, system, bytes, created_at, type, config) "
|
|
884
|
-
"VALUES (?, ?, 0, 0, ?, ?, ?)",
|
|
885
|
-
(sid, name, time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), kind, kept
|
|
1125
|
+
db.execute("INSERT INTO sources (id, name, system, bytes, created_at, type, config, workspace) "
|
|
1126
|
+
"VALUES (?, ?, 0, 0, ?, ?, ?, ?)",
|
|
1127
|
+
(sid, name, time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), kind, kept,
|
|
1128
|
+
workspace_of(self.store)))
|
|
886
1129
|
(self.dir / sid).mkdir(parents=True, exist_ok=True)
|
|
887
1130
|
return {"id": sid, "name": name, "system": False, "type": kind,
|
|
888
1131
|
"config": json.loads(kept) if kept else None, "files": [], "bytes": 0}, None
|
|
@@ -893,8 +1136,8 @@ class Sources:
|
|
|
893
1136
|
"""A Source's newest definition and the versions kept, or None for a
|
|
894
1137
|
Source that is not there or keeps none."""
|
|
895
1138
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
896
|
-
r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
897
|
-
if r is None or not (
|
|
1139
|
+
r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1140
|
+
if r is None or not (source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}).get("definition"):
|
|
898
1141
|
return None
|
|
899
1142
|
rows = db.execute("SELECT version, body, at FROM source_definitions WHERE source = ? "
|
|
900
1143
|
"ORDER BY version DESC", (sid,)).fetchall()
|
|
@@ -920,10 +1163,10 @@ class Sources:
|
|
|
920
1163
|
return None, (413, "that definition is over the cap")
|
|
921
1164
|
now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
922
1165
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
923
|
-
r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1166
|
+
r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
924
1167
|
if r is None:
|
|
925
1168
|
return None, (404, "no such source")
|
|
926
|
-
if not (
|
|
1169
|
+
if not (source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}).get("definition"):
|
|
927
1170
|
return None, (400, "a Source of that type keeps no definition")
|
|
928
1171
|
current = db.execute("SELECT MAX(version) FROM source_definitions WHERE source = ?",
|
|
929
1172
|
(sid,)).fetchone()[0] or 0
|
|
@@ -953,10 +1196,15 @@ class Sources:
|
|
|
953
1196
|
db.execute("DELETE FROM source_definitions WHERE source = ?", (sid,))
|
|
954
1197
|
|
|
955
1198
|
def type_of(self, sid):
|
|
956
|
-
"""A Source's
|
|
1199
|
+
"""A Source's enforcement entry -- its kind's, or its workflow
|
|
1200
|
+
platform's (source_entry); None for no such Source. {} for a known
|
|
1201
|
+
Source whose platform the lab does not know, so a reader asks the
|
|
1202
|
+
entry rather than the id."""
|
|
957
1203
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
958
|
-
r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
959
|
-
|
|
1204
|
+
r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1205
|
+
if r is None:
|
|
1206
|
+
return None
|
|
1207
|
+
return source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}
|
|
960
1208
|
|
|
961
1209
|
def uploads(self, sid):
|
|
962
1210
|
"""The files a Source takes, by extension, as its type declares them;
|
|
@@ -972,7 +1220,7 @@ class Sources:
|
|
|
972
1220
|
if len(name) > 80:
|
|
973
1221
|
return None, (400, "that name is too long")
|
|
974
1222
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
975
|
-
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1223
|
+
r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
976
1224
|
if r is None:
|
|
977
1225
|
return None, (404, "no such source")
|
|
978
1226
|
if r[0]:
|
|
@@ -985,8 +1233,9 @@ class Sources:
|
|
|
985
1233
|
Undo can restore it; the system Source is undeletable. Returns
|
|
986
1234
|
(token, None), or (None, error)."""
|
|
987
1235
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
988
|
-
r = db.execute("SELECT system, name, created_at, type, config FROM sources "
|
|
989
|
-
"WHERE id = ?
|
|
1236
|
+
r = db.execute("SELECT system, name, created_at, type, config, workspace FROM sources "
|
|
1237
|
+
"WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1238
|
+
(sid, workspace_of(self.store))).fetchone()
|
|
990
1239
|
if r is None:
|
|
991
1240
|
return None, (404, "no such source")
|
|
992
1241
|
if r[0]:
|
|
@@ -1001,7 +1250,7 @@ class Sources:
|
|
|
1001
1250
|
entry.mkdir(parents=True, exist_ok=True)
|
|
1002
1251
|
(entry / "manifest.json").write_text(json.dumps({
|
|
1003
1252
|
"kind": "source", "source": sid, "name": r[1], "created": r[2],
|
|
1004
|
-
"type": r[3], "config": r[4], "at": time.time(), "files": files}))
|
|
1253
|
+
"type": r[3], "config": r[4], "workspace": r[5], "at": time.time(), "files": files}))
|
|
1005
1254
|
if (self.dir / sid).is_dir():
|
|
1006
1255
|
shutil.move(str(self.dir / sid), str(entry / sid))
|
|
1007
1256
|
return token, None
|
|
@@ -1014,7 +1263,7 @@ class Sources:
|
|
|
1014
1263
|
return None, (400, "no files named")
|
|
1015
1264
|
names = list(dict.fromkeys(names))
|
|
1016
1265
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
1017
|
-
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1266
|
+
r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1018
1267
|
if r is None:
|
|
1019
1268
|
return None, (404, "no such source")
|
|
1020
1269
|
if r[0]:
|
|
@@ -1069,19 +1318,22 @@ class Sources:
|
|
|
1069
1318
|
sid, name = m.get("source"), m.get("name")
|
|
1070
1319
|
if not isinstance(sid, str) or not isinstance(name, str):
|
|
1071
1320
|
return None, (404, "no such trash entry")
|
|
1321
|
+
# Back into the workspace it was trashed from (an older trash entry
|
|
1322
|
+
# has none: the default). A name is unique within a workspace.
|
|
1323
|
+
ws = m.get("workspace") or self.store.default_ws
|
|
1072
1324
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
1073
1325
|
if db.execute("SELECT 1 FROM sources WHERE name = ? COLLATE NOCASE "
|
|
1074
|
-
"AND system = 0", (name,)).fetchone():
|
|
1326
|
+
"AND system = 0 AND workspace = ?", (name, ws)).fetchone():
|
|
1075
1327
|
return None, (409, f"{name!r} has been taken since, so nothing was restored")
|
|
1076
1328
|
with db:
|
|
1077
1329
|
kind = m.get("type")
|
|
1078
1330
|
db.execute("INSERT INTO sources (id, name, system, bytes, created_at, "
|
|
1079
|
-
"type, config) VALUES (?, ?, 0, ?, ?, ?, ?)",
|
|
1331
|
+
"type, config, workspace) VALUES (?, ?, 0, ?, ?, ?, ?, ?)",
|
|
1080
1332
|
(sid, name, sum(f.get("bytes", 0) for f in m["files"]),
|
|
1081
1333
|
m.get("created", time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())),
|
|
1082
1334
|
# As it was: a restore puts the row back, not a guess at it.
|
|
1083
1335
|
kind if isinstance(kind, str) else DEFAULT_SOURCE_TYPE,
|
|
1084
|
-
m.get("config") if isinstance(m.get("config"), str) else None))
|
|
1336
|
+
m.get("config") if isinstance(m.get("config"), str) else None, ws))
|
|
1085
1337
|
for f in m["files"]:
|
|
1086
1338
|
if isinstance(f, dict) and isinstance(f.get("name"), str):
|
|
1087
1339
|
db.execute("INSERT INTO source_files (source, name, bytes, at) "
|
|
@@ -1097,7 +1349,8 @@ class Sources:
|
|
|
1097
1349
|
return None, (404, "no such trash entry")
|
|
1098
1350
|
total = sum(f.get("bytes", 0) for f in files)
|
|
1099
1351
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
1100
|
-
r = db.execute("SELECT system, bytes FROM sources WHERE id = ?
|
|
1352
|
+
r = db.execute("SELECT system, bytes FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1353
|
+
(sid, workspace_of(self.store))).fetchone()
|
|
1101
1354
|
if r is None:
|
|
1102
1355
|
return None, (409, "the Source that held these files is gone, so nothing was restored")
|
|
1103
1356
|
if r[0]:
|
|
@@ -1168,14 +1421,15 @@ class Sources:
|
|
|
1168
1421
|
names = list(dict.fromkeys(names))
|
|
1169
1422
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1170
1423
|
db.row_factory = sqlite3.Row
|
|
1171
|
-
|
|
1172
|
-
|
|
1424
|
+
ws = workspace_of(self.store)
|
|
1425
|
+
dest = db.execute("SELECT system, bytes, type FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1426
|
+
(sid, ws)).fetchone()
|
|
1173
1427
|
if dest is None:
|
|
1174
1428
|
return None, (404, "no such source")
|
|
1175
1429
|
if dest["system"]:
|
|
1176
1430
|
return None, (403, "the sample library cannot be written to")
|
|
1177
|
-
src = db.execute("SELECT system, type FROM sources WHERE id = ?",
|
|
1178
|
-
(from_sid,)).fetchone()
|
|
1431
|
+
src = db.execute("SELECT system, type FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1432
|
+
(from_sid, ws)).fetchone()
|
|
1179
1433
|
if src is None:
|
|
1180
1434
|
return None, (404, "no such source")
|
|
1181
1435
|
# A file means what its Source's type says it means, so it only
|
|
@@ -1264,7 +1518,7 @@ class Sources:
|
|
|
1264
1518
|
"""Every file of a Source, as (name, path on disk) in its own order,
|
|
1265
1519
|
or None when there is no such Source."""
|
|
1266
1520
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1267
|
-
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1521
|
+
r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1268
1522
|
if r is None:
|
|
1269
1523
|
return None
|
|
1270
1524
|
if r[0]:
|
|
@@ -1281,7 +1535,7 @@ class Sources:
|
|
|
1281
1535
|
or not all(isinstance(n, str) for n in names):
|
|
1282
1536
|
return None, (400, "the names are needed")
|
|
1283
1537
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1284
|
-
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1538
|
+
r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1285
1539
|
if r is None:
|
|
1286
1540
|
return None, (404, "no such source")
|
|
1287
1541
|
if r[0]:
|
|
@@ -1347,7 +1601,8 @@ class Sources:
|
|
|
1347
1601
|
"""
|
|
1348
1602
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1349
1603
|
db.row_factory = sqlite3.Row
|
|
1350
|
-
r = db.execute("SELECT system, bytes FROM sources WHERE id = ?
|
|
1604
|
+
r = db.execute("SELECT system, bytes FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
|
|
1605
|
+
(sid, workspace_of(self.store))).fetchone()
|
|
1351
1606
|
if r is None:
|
|
1352
1607
|
return None, (404, "no such source")
|
|
1353
1608
|
if r["system"]:
|
|
@@ -1379,7 +1634,7 @@ class Sources:
|
|
|
1379
1634
|
def file_path(self, sid: str, name: str) -> Path:
|
|
1380
1635
|
"""The on-disk path of a stored file, or None. `name` is already clean."""
|
|
1381
1636
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
1382
|
-
r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
|
|
1637
|
+
r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
|
|
1383
1638
|
if r is None:
|
|
1384
1639
|
return None
|
|
1385
1640
|
if r[0]:
|
|
@@ -1452,22 +1707,23 @@ class Sources:
|
|
|
1452
1707
|
# it is left where it is, and nothing reads it.
|
|
1453
1708
|
|
|
1454
1709
|
DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
|
|
1455
|
-
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version
|
|
1710
|
+
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 8 reads
|
|
1711
|
+
# the recorded-reply metric ids under their new names (#299); version 7 is an
|
|
1456
1712
|
# eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
|
|
1457
1713
|
# 5 was told by its `source` alone, and earlier ones by neither.
|
|
1458
|
-
DATASET_BODY_VERSION =
|
|
1714
|
+
DATASET_BODY_VERSION = 8
|
|
1459
1715
|
DATASET_NAME_MAX = 80
|
|
1460
1716
|
# The file forms Export writes and Import reads. Export writes an eval group
|
|
1461
|
-
# at version
|
|
1462
|
-
# file of versions 1 to
|
|
1717
|
+
# at version 8 (docs/pipeline-model.md §17); Import reads that, and a dataset
|
|
1718
|
+
# file of versions 1 to 8, upgraded, and refuses anything else, as a pipeline
|
|
1463
1719
|
# of another version is refused. Versions 1 to 3 carried a prompt, which an
|
|
1464
1720
|
# import gives to the Prompt library.
|
|
1465
1721
|
EXPORT_ONE = "evals-lab/eval-group"
|
|
1466
1722
|
EXPORT_ALL = "evals-lab/eval-groups"
|
|
1467
1723
|
DATASET_ONE = "evals-lab/dataset"
|
|
1468
1724
|
DATASET_ALL = "evals-lab/datasets"
|
|
1469
|
-
EXPORT_VERSION =
|
|
1470
|
-
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
|
|
1725
|
+
EXPORT_VERSION = 8
|
|
1726
|
+
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8)
|
|
1471
1727
|
# Each file form, and the key its one entry or its list sits under.
|
|
1472
1728
|
EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
|
|
1473
1729
|
SCORING_MODES = ("all", "weighted")
|
|
@@ -1505,7 +1761,7 @@ def _term_in(items, term) -> bool:
|
|
|
1505
1761
|
def case_metrics(c: dict) -> list:
|
|
1506
1762
|
"""A version-4 case's expectations as the metrics that say the same:
|
|
1507
1763
|
evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
|
|
1508
|
-
through fixtures/dataset-
|
|
1764
|
+
through fixtures/dataset-v8.json."""
|
|
1509
1765
|
def strs(v):
|
|
1510
1766
|
return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
|
|
1511
1767
|
if c.get("discarded") is True:
|
|
@@ -1572,7 +1828,7 @@ def case_of_v5(c):
|
|
|
1572
1828
|
|
|
1573
1829
|
|
|
1574
1830
|
def group_of_v6(body: dict) -> dict:
|
|
1575
|
-
"""A version-6 body as a version-
|
|
1831
|
+
"""A version-6 body as a version-8 eval group: evals-core.ts's
|
|
1576
1832
|
groupOfV6. Scored All, the lab's grader, and no metrics of its own for
|
|
1577
1833
|
every item or the whole run -- what a Metrics eval naming the dataset with
|
|
1578
1834
|
none of its own graded."""
|
|
@@ -1581,8 +1837,43 @@ def group_of_v6(body: dict) -> dict:
|
|
|
1581
1837
|
"grader": None, "every": [], "run": [], **rest}
|
|
1582
1838
|
|
|
1583
1839
|
|
|
1840
|
+
# The recorded-reply metric ids renamed at version 8: evals-core.ts's
|
|
1841
|
+
# RECORDED_IDS. Stored tokens inside eval group bodies and a pipeline's private
|
|
1842
|
+
# group, so they are mapped wherever a body is read (#299).
|
|
1843
|
+
RECORDED_IDS = {
|
|
1844
|
+
"equals-production": "same-as-recorded",
|
|
1845
|
+
"fields-equal-production": "fields-equal-recorded",
|
|
1846
|
+
"same-parse-outcome": "same-parse-as-recorded",
|
|
1847
|
+
}
|
|
1848
|
+
|
|
1849
|
+
|
|
1850
|
+
def _rename_metric(m):
|
|
1851
|
+
return {**m, "type": RECORDED_IDS[m["type"]]} if isinstance(m, dict) and m.get("type") in RECORDED_IDS else m
|
|
1852
|
+
|
|
1853
|
+
|
|
1854
|
+
def _rename_metrics(lst):
|
|
1855
|
+
return [_rename_metric(m) for m in lst] if isinstance(lst, list) else lst
|
|
1856
|
+
|
|
1857
|
+
|
|
1858
|
+
def recorded_ids_v7(body):
|
|
1859
|
+
"""A version-7 body (or a freshly made version-8 one) with its
|
|
1860
|
+
recorded-reply metric ids read under their version-8 names, at version 8:
|
|
1861
|
+
evals-core.ts's recordedIdsV7."""
|
|
1862
|
+
out = dict(body)
|
|
1863
|
+
out["version"] = DATASET_BODY_VERSION
|
|
1864
|
+
if isinstance(body.get("every"), list):
|
|
1865
|
+
out["every"] = _rename_metrics(body["every"])
|
|
1866
|
+
if isinstance(body.get("run"), list):
|
|
1867
|
+
out["run"] = _rename_metrics(body["run"])
|
|
1868
|
+
if isinstance(body.get("cases"), list):
|
|
1869
|
+
out["cases"] = [{**c, "metrics": _rename_metrics(c["metrics"])}
|
|
1870
|
+
if isinstance(c, dict) and isinstance(c.get("metrics"), list) else c
|
|
1871
|
+
for c in body["cases"]]
|
|
1872
|
+
return out
|
|
1873
|
+
|
|
1874
|
+
|
|
1584
1875
|
def upgrade_body(body):
|
|
1585
|
-
"""An earlier body as today's (version
|
|
1876
|
+
"""An earlier body as today's (version 8): evals-core.ts's
|
|
1586
1877
|
upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
|
|
1587
1878
|
its `replays` and `conformance` go (fixtures/replays.json holds the
|
|
1588
1879
|
parser's tests). Version 2's `rules` go -- they clean a job's answer, so
|
|
@@ -1593,24 +1884,28 @@ def upgrade_body(body):
|
|
|
1593
1884
|
needs the rules or the prompt takes them first (`body_rules`,
|
|
1594
1885
|
`body_prompt`). A body naming its Source is version 5, whose Contains
|
|
1595
1886
|
metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
|
|
1596
|
-
group's scoring, grader, Every item and Whole run (`group_of_v6`)
|
|
1597
|
-
|
|
1598
|
-
|
|
1887
|
+
group's scoring, grader, Every item and Whole run (`group_of_v6`); version
|
|
1888
|
+
7 reads its recorded-reply metric ids under their version-8 names
|
|
1889
|
+
(`recorded_ids_v7`). A body saying it is version 8 comes back as it was;
|
|
1890
|
+
so does anything that is not a body."""
|
|
1599
1891
|
if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
|
|
1600
1892
|
return body
|
|
1601
|
-
# A body saying any other version is one this lab does not read, and is
|
|
1602
|
-
# left for dataset_problem to refuse.
|
|
1603
1893
|
if "version" in body:
|
|
1604
|
-
|
|
1894
|
+
# Version 7 reads its recorded-reply metric ids under their version-8
|
|
1895
|
+
# names; version 6 is its cases alone. Any other version is one this
|
|
1896
|
+
# lab does not read, left for dataset_problem to refuse.
|
|
1897
|
+
if body["version"] == 7:
|
|
1898
|
+
return recorded_ids_v7(body)
|
|
1899
|
+
return recorded_ids_v7(group_of_v6(body)) if body["version"] == 6 else body
|
|
1605
1900
|
if "source" in body:
|
|
1606
1901
|
up = dict(body)
|
|
1607
1902
|
if isinstance(body.get("cases"), list):
|
|
1608
1903
|
up["cases"] = [case_of_v5(c) for c in body["cases"]]
|
|
1609
|
-
return group_of_v6(up)
|
|
1904
|
+
return recorded_ids_v7(group_of_v6(up))
|
|
1610
1905
|
cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
|
|
1611
1906
|
if not isinstance(cases, list):
|
|
1612
1907
|
return body
|
|
1613
|
-
return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
|
|
1908
|
+
return recorded_ids_v7(group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}))
|
|
1614
1909
|
|
|
1615
1910
|
|
|
1616
1911
|
def body_prompt(body):
|
|
@@ -1754,7 +2049,7 @@ def scenario_ref(sc, i):
|
|
|
1754
2049
|
what version 4's upgrade gives one, by position."""
|
|
1755
2050
|
sid = sc.get("id") if isinstance(sc, dict) else None
|
|
1756
2051
|
name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
|
|
1757
|
-
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
|
|
2052
|
+
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
|
|
1758
2053
|
|
|
1759
2054
|
|
|
1760
2055
|
class Prompts:
|
|
@@ -1782,6 +2077,13 @@ class Prompts:
|
|
|
1782
2077
|
cols = {r[1] for r in db.execute("PRAGMA table_info(prompt_uses)")}
|
|
1783
2078
|
if "chain" in cols and "job" not in cols:
|
|
1784
2079
|
db.execute("ALTER TABLE prompt_uses RENAME COLUMN chain TO job")
|
|
2080
|
+
# The workspace a prompt belongs to (docs/workspaces.md): the
|
|
2081
|
+
# Default is per-workspace, so is_default is scoped too. Children
|
|
2082
|
+
# (versions, uses) co-scope through the prompt id. A store from
|
|
2083
|
+
# before workspaces backfills every prompt to the default.
|
|
2084
|
+
if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(prompts)")}:
|
|
2085
|
+
db.execute("ALTER TABLE prompts ADD COLUMN workspace TEXT")
|
|
2086
|
+
db.execute("UPDATE prompts SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
|
|
1785
2087
|
# One-off facts about the library: that the runs from before it
|
|
1786
2088
|
# have been read into it, so a restart does not read them again
|
|
1787
2089
|
# and bring back a prompt someone has since deleted.
|
|
@@ -1796,9 +2098,9 @@ class Prompts:
|
|
|
1796
2098
|
db.row_factory = sqlite3.Row
|
|
1797
2099
|
return closing(db)
|
|
1798
2100
|
|
|
1799
|
-
|
|
1800
|
-
|
|
1801
|
-
|
|
2101
|
+
def _live(self, db, pid):
|
|
2102
|
+
return db.execute("SELECT * FROM prompts WHERE id = ? AND trash IS NULL AND workspace = ?",
|
|
2103
|
+
(pid, workspace_of(self.store))).fetchone()
|
|
1802
2104
|
|
|
1803
2105
|
@staticmethod
|
|
1804
2106
|
def _head(db, pid):
|
|
@@ -1837,7 +2139,8 @@ class Prompts:
|
|
|
1837
2139
|
def list(self) -> list:
|
|
1838
2140
|
with self.store.lock, self._connect() as db:
|
|
1839
2141
|
return [self._summary(db, r) for r in db.execute(
|
|
1840
|
-
"SELECT * FROM prompts WHERE trash IS NULL ORDER BY updated_at DESC, id"
|
|
2142
|
+
"SELECT * FROM prompts WHERE trash IS NULL AND workspace = ? ORDER BY updated_at DESC, id",
|
|
2143
|
+
(workspace_of(self.store),))]
|
|
1841
2144
|
|
|
1842
2145
|
def get(self, pid):
|
|
1843
2146
|
with self.store.lock, self._connect() as db:
|
|
@@ -1847,7 +2150,8 @@ class Prompts:
|
|
|
1847
2150
|
def default(self):
|
|
1848
2151
|
"""The Default's newest version, {id, name, version, text}, or None."""
|
|
1849
2152
|
with self.store.lock, self._connect() as db:
|
|
1850
|
-
r = db.execute("SELECT * FROM prompts WHERE is_default = 1 AND trash IS NULL"
|
|
2153
|
+
r = db.execute("SELECT * FROM prompts WHERE is_default = 1 AND trash IS NULL AND workspace = ?",
|
|
2154
|
+
(workspace_of(self.store),)).fetchone()
|
|
1851
2155
|
if r is None:
|
|
1852
2156
|
return None
|
|
1853
2157
|
head = self._head(db, r["id"])
|
|
@@ -1859,10 +2163,11 @@ class Prompts:
|
|
|
1859
2163
|
def _insert(self, db, name, text, default=False):
|
|
1860
2164
|
pid = secrets.token_hex(6)
|
|
1861
2165
|
now = self._now()
|
|
2166
|
+
ws = workspace_of(self.store)
|
|
1862
2167
|
if default:
|
|
1863
|
-
db.execute("UPDATE prompts SET is_default = 0")
|
|
1864
|
-
db.execute("INSERT INTO prompts (id, name, is_default, created_at, updated_at)
|
|
1865
|
-
(pid, name, 1 if default else 0, now, now))
|
|
2168
|
+
db.execute("UPDATE prompts SET is_default = 0 WHERE workspace = ?", (ws,))
|
|
2169
|
+
db.execute("INSERT INTO prompts (id, name, is_default, created_at, updated_at, workspace) "
|
|
2170
|
+
"VALUES (?, ?, ?, ?, ?, ?)", (pid, name, 1 if default else 0, now, now, ws))
|
|
1866
2171
|
db.execute("INSERT INTO prompt_versions (prompt_id, version, text, created_at, edited_at) "
|
|
1867
2172
|
"VALUES (?, 1, ?, ?, ?)", (pid, text, now, time.time()))
|
|
1868
2173
|
return pid
|
|
@@ -1877,7 +2182,8 @@ class Prompts:
|
|
|
1877
2182
|
return version
|
|
1878
2183
|
|
|
1879
2184
|
def _has_default(self, db):
|
|
1880
|
-
return db.execute("SELECT 1 FROM prompts WHERE is_default = 1 AND trash IS NULL
|
|
2185
|
+
return db.execute("SELECT 1 FROM prompts WHERE is_default = 1 AND trash IS NULL AND workspace = ?",
|
|
2186
|
+
(workspace_of(self.store),)).fetchone() is not None
|
|
1881
2187
|
|
|
1882
2188
|
def adopt(self, db, text, name="", default=False):
|
|
1883
2189
|
"""A prompt reading [text]: the live one that already does, or a new
|
|
@@ -1893,13 +2199,13 @@ class Prompts:
|
|
|
1893
2199
|
db.execute("UPDATE prompts SET is_default = 1 WHERE id = ?", (pid,))
|
|
1894
2200
|
return pid
|
|
1895
2201
|
|
|
1896
|
-
|
|
1897
|
-
|
|
1898
|
-
|
|
2202
|
+
def _matching(self, db, text):
|
|
2203
|
+
"""The newest version of any live prompt in this workspace reading
|
|
2204
|
+
exactly [text]."""
|
|
1899
2205
|
return db.execute("SELECT v.prompt_id, v.version FROM prompt_versions v "
|
|
1900
|
-
"JOIN prompts p ON p.id = v.prompt_id AND p.trash IS NULL "
|
|
2206
|
+
"JOIN prompts p ON p.id = v.prompt_id AND p.trash IS NULL AND p.workspace = ? "
|
|
1901
2207
|
"WHERE v.text = ? ORDER BY v.created_at DESC, v.version DESC LIMIT 1",
|
|
1902
|
-
(text
|
|
2208
|
+
(workspace_of(self.store), text)).fetchone()
|
|
1903
2209
|
|
|
1904
2210
|
def record(self, db, rid, run, at):
|
|
1905
2211
|
"""A run's uses, one per scenario per job, linked by the rules in
|
|
@@ -1997,7 +2303,7 @@ class Prompts:
|
|
|
1997
2303
|
with self.store.lock, self._connect() as db, db:
|
|
1998
2304
|
if self._live(db, pid) is None:
|
|
1999
2305
|
return None, (404, "no such prompt")
|
|
2000
|
-
db.execute("UPDATE prompts SET is_default = 0")
|
|
2306
|
+
db.execute("UPDATE prompts SET is_default = 0 WHERE workspace = ?", (workspace_of(self.store),))
|
|
2001
2307
|
db.execute("UPDATE prompts SET is_default = 1 WHERE id = ?", (pid,))
|
|
2002
2308
|
return self._detail(db, self._live(db, pid)), None
|
|
2003
2309
|
|
|
@@ -2029,7 +2335,8 @@ class Prompts:
|
|
|
2029
2335
|
|
|
2030
2336
|
def restore(self, token):
|
|
2031
2337
|
with self.store.lock, self._connect() as db, db:
|
|
2032
|
-
r = db.execute("SELECT * FROM prompts WHERE trash = ?",
|
|
2338
|
+
r = db.execute("SELECT * FROM prompts WHERE trash = ? AND workspace = ?",
|
|
2339
|
+
(str(token), workspace_of(self.store))).fetchone()
|
|
2033
2340
|
if r is None:
|
|
2034
2341
|
return None, (404, "no such trash entry")
|
|
2035
2342
|
db.execute("UPDATE prompts SET trash = NULL, trashed_at = NULL WHERE id = ?", (r["id"],))
|
|
@@ -2081,6 +2388,12 @@ class Datasets:
|
|
|
2081
2388
|
"group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
|
|
2082
2389
|
"created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
|
|
2083
2390
|
"ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
|
|
2391
|
+
# The workspace a dataset belongs to (docs/workspaces.md); its
|
|
2392
|
+
# group versions and archives co-scope through the dataset id. A
|
|
2393
|
+
# store from before workspaces backfills every row to the default.
|
|
2394
|
+
if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(datasets)")}:
|
|
2395
|
+
db.execute("ALTER TABLE datasets ADD COLUMN workspace TEXT")
|
|
2396
|
+
db.execute("UPDATE datasets SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
|
|
2084
2397
|
# Rows from an earlier version are converted once, in place: a
|
|
2085
2398
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
2086
2399
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -2155,17 +2468,20 @@ class Datasets:
|
|
|
2155
2468
|
return closing(db)
|
|
2156
2469
|
|
|
2157
2470
|
def _live(self, db, did):
|
|
2158
|
-
return db.execute("SELECT * FROM datasets WHERE id = ? AND trash IS NULL",
|
|
2471
|
+
return db.execute("SELECT * FROM datasets WHERE id = ? AND trash IS NULL AND workspace = ?",
|
|
2472
|
+
(did, workspace_of(self.store))).fetchone()
|
|
2159
2473
|
|
|
2160
2474
|
def _names(self, db, but=None):
|
|
2161
|
-
return {r[0] for r in db.execute(
|
|
2162
|
-
|
|
2475
|
+
return {r[0] for r in db.execute(
|
|
2476
|
+
"SELECT name FROM datasets WHERE trash IS NULL AND id IS NOT ? AND workspace = ?",
|
|
2477
|
+
(but, workspace_of(self.store)))}
|
|
2163
2478
|
|
|
2164
2479
|
def list(self) -> list:
|
|
2165
2480
|
with self.store.lock, self._connect() as db:
|
|
2166
2481
|
counts = self._counts(db)
|
|
2167
2482
|
return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
|
|
2168
|
-
"SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id"
|
|
2483
|
+
"SELECT * FROM datasets WHERE trash IS NULL AND workspace = ? ORDER BY name COLLATE NOCASE, id",
|
|
2484
|
+
(workspace_of(self.store),))]
|
|
2169
2485
|
|
|
2170
2486
|
def get(self, did):
|
|
2171
2487
|
with self.store.lock, self._connect() as db:
|
|
@@ -2195,11 +2511,11 @@ class Datasets:
|
|
|
2195
2511
|
"VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
|
|
2196
2512
|
return self._head(db, r["id"])
|
|
2197
2513
|
|
|
2198
|
-
|
|
2199
|
-
def _pins(db) -> set:
|
|
2514
|
+
def _pins(self, db) -> set:
|
|
2200
2515
|
"""Every (group, version) a stored pipeline pins, read in the caller's
|
|
2201
|
-
transaction from
|
|
2202
|
-
row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'"
|
|
2516
|
+
transaction from this workspace's workflows document."""
|
|
2517
|
+
row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows' AND workspace = ?",
|
|
2518
|
+
(workspace_of(self.store),)).fetchone()
|
|
2203
2519
|
return pins_in(json.loads(row[0])) if row and row[0] else set()
|
|
2204
2520
|
|
|
2205
2521
|
def _cut(self, db, did, text):
|
|
@@ -2313,8 +2629,9 @@ class Datasets:
|
|
|
2313
2629
|
def _insert(self, db, name, body):
|
|
2314
2630
|
did = secrets.token_hex(6)
|
|
2315
2631
|
now = self._now()
|
|
2316
|
-
db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
|
|
2317
|
-
"VALUES (?, ?, 1, ?, ?, ?)",
|
|
2632
|
+
db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at, workspace) "
|
|
2633
|
+
"VALUES (?, ?, 1, ?, ?, ?, ?)",
|
|
2634
|
+
(did, name, json.dumps(body), now, now, workspace_of(self.store)))
|
|
2318
2635
|
return did
|
|
2319
2636
|
|
|
2320
2637
|
def create(self, name, body=None):
|
|
@@ -2384,7 +2701,8 @@ class Datasets:
|
|
|
2384
2701
|
"""A trashed dataset back, under a new ` (2)` name if its own has been
|
|
2385
2702
|
taken since. Returns (DatasetDoc, None)."""
|
|
2386
2703
|
with self.store.lock, self._connect() as db, db:
|
|
2387
|
-
r = db.execute("SELECT * FROM datasets WHERE trash = ?",
|
|
2704
|
+
r = db.execute("SELECT * FROM datasets WHERE trash = ? AND workspace = ?",
|
|
2705
|
+
(str(token), workspace_of(self.store))).fetchone()
|
|
2388
2706
|
if r is None:
|
|
2389
2707
|
return None, (404, "no such trash entry")
|
|
2390
2708
|
name = unique_dataset_name(r["name"], self._names(db, but=r["id"]))
|
|
@@ -2420,8 +2738,8 @@ class Datasets:
|
|
|
2420
2738
|
|
|
2421
2739
|
def export_all(self):
|
|
2422
2740
|
with self.store.lock, self._connect() as db:
|
|
2423
|
-
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
|
|
2424
|
-
"ORDER BY name COLLATE NOCASE, id").fetchall()
|
|
2741
|
+
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL AND workspace = ? "
|
|
2742
|
+
"ORDER BY name COLLATE NOCASE, id", (workspace_of(self.store),)).fetchall()
|
|
2425
2743
|
return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
|
|
2426
2744
|
"groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
|
|
2427
2745
|
|
|
@@ -2679,11 +2997,40 @@ class Packs:
|
|
|
2679
2997
|
self.trash = {}
|
|
2680
2998
|
with store.lock, closing(sqlite3.connect(store.path)) as db, db:
|
|
2681
2999
|
db.execute("CREATE TABLE IF NOT EXISTS packs ("
|
|
2682
|
-
"id TEXT
|
|
2683
|
-
"manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL
|
|
3000
|
+
"id TEXT NOT NULL, name TEXT NOT NULL, version TEXT NOT NULL, "
|
|
3001
|
+
"manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL, "
|
|
3002
|
+
"workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (id, workspace))")
|
|
2684
3003
|
db.execute("CREATE TABLE IF NOT EXISTS pack_items ("
|
|
2685
3004
|
"pack TEXT NOT NULL, kind TEXT NOT NULL, key TEXT NOT NULL, item TEXT NOT NULL, "
|
|
2686
|
-
"PRIMARY KEY (pack, kind, key))")
|
|
3005
|
+
"workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (pack, workspace, kind, key))")
|
|
3006
|
+
# A pack's id is its manifest's, not globally unique, so two
|
|
3007
|
+
# workspaces can hold the same pack: the workspace is part of both
|
|
3008
|
+
# keys (docs/workspaces.md). A store from before workspaces keyed
|
|
3009
|
+
# packs by id alone; it is rebuilt once, every row to the default.
|
|
3010
|
+
if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(packs)")}:
|
|
3011
|
+
self._rebuild(db, store.default_ws, "packs",
|
|
3012
|
+
"id, name, version, manifest, presets, installed_at",
|
|
3013
|
+
"id TEXT NOT NULL, name TEXT NOT NULL, version TEXT NOT NULL, "
|
|
3014
|
+
"manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL, "
|
|
3015
|
+
"workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (id, workspace)")
|
|
3016
|
+
if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(pack_items)")}:
|
|
3017
|
+
self._rebuild(db, store.default_ws, "pack_items", "pack, kind, key, item",
|
|
3018
|
+
"pack TEXT NOT NULL, kind TEXT NOT NULL, key TEXT NOT NULL, item TEXT NOT NULL, "
|
|
3019
|
+
"workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (pack, workspace, kind, key)")
|
|
3020
|
+
|
|
3021
|
+
@staticmethod
|
|
3022
|
+
def _rebuild(db, ws, table, cols, schema):
|
|
3023
|
+
"""A table keyed without a workspace, rebuilt with one its old rows
|
|
3024
|
+
take: SQLite cannot add to a PRIMARY KEY, so the rows move through a
|
|
3025
|
+
fresh table (the sources `type`/`config` migration's shape)."""
|
|
3026
|
+
rows = db.execute(f"SELECT {cols} FROM {table}").fetchall()
|
|
3027
|
+
db.execute(f"ALTER TABLE {table} RENAME TO {table}_old")
|
|
3028
|
+
db.execute(f"CREATE TABLE {table} ({schema})")
|
|
3029
|
+
names = cols.split(", ")
|
|
3030
|
+
ph = ", ".join("?" * (len(names) + 1))
|
|
3031
|
+
for r in rows:
|
|
3032
|
+
db.execute(f"INSERT INTO {table} ({cols}, workspace) VALUES ({ph})", (*r, ws))
|
|
3033
|
+
db.execute(f"DROP TABLE {table}_old")
|
|
2687
3034
|
|
|
2688
3035
|
def _connect(self):
|
|
2689
3036
|
db = sqlite3.connect(self.store.path)
|
|
@@ -2693,12 +3040,15 @@ class Packs:
|
|
|
2693
3040
|
def _items(self, pid):
|
|
2694
3041
|
with self.store.lock, self._connect() as db:
|
|
2695
3042
|
return {(r["kind"], r["key"]): r["item"]
|
|
2696
|
-
for r in db.execute("SELECT kind, key, item FROM pack_items WHERE pack = ?",
|
|
3043
|
+
for r in db.execute("SELECT kind, key, item FROM pack_items WHERE pack = ? AND workspace = ?",
|
|
3044
|
+
(pid, workspace_of(self.store)))}
|
|
2697
3045
|
|
|
2698
3046
|
def list(self) -> list:
|
|
2699
3047
|
with self.store.lock, self._connect() as db:
|
|
2700
|
-
|
|
2701
|
-
|
|
3048
|
+
ws = workspace_of(self.store)
|
|
3049
|
+
packs = db.execute("SELECT * FROM packs WHERE workspace = ? ORDER BY name COLLATE NOCASE",
|
|
3050
|
+
(ws,)).fetchall()
|
|
3051
|
+
items = db.execute("SELECT pack, kind, item FROM pack_items WHERE workspace = ?", (ws,)).fetchall()
|
|
2702
3052
|
workflows = {w.get("id"): w.get("name") for w in self._doc("promptlab.workflows")[1].get("list", [])}
|
|
2703
3053
|
out = []
|
|
2704
3054
|
for p in packs:
|
|
@@ -2718,7 +3068,8 @@ class Packs:
|
|
|
2718
3068
|
def presets(self) -> list:
|
|
2719
3069
|
"""Every installed pack's Setup presets, each marked with its pack."""
|
|
2720
3070
|
with self.store.lock, self._connect() as db:
|
|
2721
|
-
rows = db.execute("SELECT id, presets FROM packs ORDER BY name COLLATE NOCASE"
|
|
3071
|
+
rows = db.execute("SELECT id, presets FROM packs WHERE workspace = ? ORDER BY name COLLATE NOCASE",
|
|
3072
|
+
(workspace_of(self.store),)).fetchall()
|
|
2722
3073
|
return [{**p, "pack": r["id"]} for r in rows for p in json.loads(r["presets"])]
|
|
2723
3074
|
|
|
2724
3075
|
# The page's documents are the server's to write here too, through the
|
|
@@ -2757,9 +3108,11 @@ class Packs:
|
|
|
2757
3108
|
why = pack_requires_problem(m, set(plugins))
|
|
2758
3109
|
if why:
|
|
2759
3110
|
return None, (400, why)
|
|
3111
|
+
ws = workspace_of(self.store)
|
|
2760
3112
|
with self.lock:
|
|
2761
3113
|
with self.store.lock, self._connect() as db:
|
|
2762
|
-
have = db.execute("SELECT version FROM packs WHERE id = ?
|
|
3114
|
+
have = db.execute("SELECT version FROM packs WHERE id = ? AND workspace = ?",
|
|
3115
|
+
(m["id"], ws)).fetchone()
|
|
2763
3116
|
if have and have["version"] == m["packVersion"]:
|
|
2764
3117
|
return {"pack": m["id"], "installed": False}, None
|
|
2765
3118
|
owned = self._items(m["id"])
|
|
@@ -2854,22 +3207,23 @@ class Packs:
|
|
|
2854
3207
|
self._write_doc("promptlab.versions", keep)
|
|
2855
3208
|
now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
2856
3209
|
with self.store.lock, self._connect() as db, db:
|
|
2857
|
-
db.execute("INSERT INTO packs (id, name, version, manifest, presets, installed_at) "
|
|
2858
|
-
"VALUES (?, ?, ?, ?, ?, ?) ON CONFLICT(id) DO UPDATE SET
|
|
2859
|
-
"version = excluded.version, manifest = excluded.manifest, "
|
|
3210
|
+
db.execute("INSERT INTO packs (id, name, version, manifest, presets, installed_at, workspace) "
|
|
3211
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?) ON CONFLICT(id, workspace) DO UPDATE SET "
|
|
3212
|
+
"name = excluded.name, version = excluded.version, manifest = excluded.manifest, "
|
|
2860
3213
|
"presets = excluded.presets, installed_at = excluded.installed_at",
|
|
2861
3214
|
(m["id"], m["name"].strip(), m["packVersion"], json.dumps(m),
|
|
2862
|
-
json.dumps(pack["presets"]), now))
|
|
3215
|
+
json.dumps(pack["presets"]), now, ws))
|
|
2863
3216
|
for (kind, key), item in made.items():
|
|
2864
|
-
db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item)
|
|
2865
|
-
(m["id"], kind, key, item))
|
|
3217
|
+
db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item, workspace) "
|
|
3218
|
+
"VALUES (?, ?, ?, ?, ?)", (m["id"], kind, key, item, ws))
|
|
2866
3219
|
return {"pack": m["id"], "installed": True}, None
|
|
2867
3220
|
|
|
2868
3221
|
def remove(self, pid):
|
|
2869
3222
|
"""What the pack made, into the trash, at once. Returns (token, None)."""
|
|
3223
|
+
ws = workspace_of(self.store)
|
|
2870
3224
|
with self.lock:
|
|
2871
3225
|
with self.store.lock, self._connect() as db:
|
|
2872
|
-
row = db.execute("SELECT * FROM packs WHERE id = ?", (pid,)).fetchone()
|
|
3226
|
+
row = db.execute("SELECT * FROM packs WHERE id = ? AND workspace = ?", (pid, ws)).fetchone()
|
|
2873
3227
|
if row is None:
|
|
2874
3228
|
return None, (404, "no such pack")
|
|
2875
3229
|
items = self._items(pid)
|
|
@@ -2894,8 +3248,8 @@ class Packs:
|
|
|
2894
3248
|
if gone:
|
|
2895
3249
|
self._write_doc("promptlab.workflows", drop)
|
|
2896
3250
|
with self.store.lock, self._connect() as db, db:
|
|
2897
|
-
db.execute("DELETE FROM pack_items WHERE pack = ?", (pid,))
|
|
2898
|
-
db.execute("DELETE FROM packs WHERE id = ?", (pid,))
|
|
3251
|
+
db.execute("DELETE FROM pack_items WHERE pack = ? AND workspace = ?", (pid, ws))
|
|
3252
|
+
db.execute("DELETE FROM packs WHERE id = ? AND workspace = ?", (pid, ws))
|
|
2899
3253
|
token = secrets.token_hex(6)
|
|
2900
3254
|
self.trash[token] = entry
|
|
2901
3255
|
return token, None
|
|
@@ -2915,13 +3269,14 @@ class Packs:
|
|
|
2915
3269
|
self._write_doc("promptlab.workflows",
|
|
2916
3270
|
lambda body: {**body, "list": [*body.get("list", []), *entry["pipelines"]]})
|
|
2917
3271
|
r = entry["row"]
|
|
3272
|
+
ws = r.get("workspace", self.store.default_ws)
|
|
2918
3273
|
with self.store.lock, self._connect() as db, db:
|
|
2919
|
-
db.execute("INSERT OR REPLACE INTO packs (id, name, version, manifest, presets, installed_at) "
|
|
2920
|
-
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
2921
|
-
(r["id"], r["name"], r["version"], r["manifest"], r["presets"], r["installed_at"]))
|
|
3274
|
+
db.execute("INSERT OR REPLACE INTO packs (id, name, version, manifest, presets, installed_at, workspace) "
|
|
3275
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
3276
|
+
(r["id"], r["name"], r["version"], r["manifest"], r["presets"], r["installed_at"], ws))
|
|
2922
3277
|
for kind, key, item in entry["items"]:
|
|
2923
|
-
db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item)
|
|
2924
|
-
(r["id"], kind, key, item))
|
|
3278
|
+
db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item, workspace) "
|
|
3279
|
+
"VALUES (?, ?, ?, ?, ?)", (r["id"], kind, key, item, ws))
|
|
2925
3280
|
del self.trash[token]
|
|
2926
3281
|
return {"pack": r["id"]}, None
|
|
2927
3282
|
|
|
@@ -2932,7 +3287,8 @@ class Packs:
|
|
|
2932
3287
|
removed it and has datasets of its own is left alone. Returns what
|
|
2933
3288
|
happened, or None where nothing was tried."""
|
|
2934
3289
|
with self.store.lock, self._connect() as db:
|
|
2935
|
-
have = db.execute("SELECT version FROM packs WHERE id = 'demo'"
|
|
3290
|
+
have = db.execute("SELECT version FROM packs WHERE id = 'demo' AND workspace = ?",
|
|
3291
|
+
(workspace_of(self.store),)).fetchone()
|
|
2936
3292
|
if have is None and DATASETS.list():
|
|
2937
3293
|
return None
|
|
2938
3294
|
got, err = self.install(demo_pack())
|
|
@@ -3701,6 +4057,14 @@ class Queue:
|
|
|
3701
4057
|
# `dataset` column, read as its one group.
|
|
3702
4058
|
if "groups" not in cols:
|
|
3703
4059
|
db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
|
|
4060
|
+
# The workspace a run belongs to (docs/workspaces.md): History is
|
|
4061
|
+
# per-workspace, so the list and every id-keyed read filter by it.
|
|
4062
|
+
# The worker loop is the one global reader -- it grades every
|
|
4063
|
+
# workspace's runs by id. A store from before workspaces backfills
|
|
4064
|
+
# to the default.
|
|
4065
|
+
if "workspace" not in cols:
|
|
4066
|
+
db.execute("ALTER TABLE queue ADD COLUMN workspace TEXT")
|
|
4067
|
+
db.execute("UPDATE queue SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
|
|
3704
4068
|
|
|
3705
4069
|
# ---- rows -----------------------------------------------------------
|
|
3706
4070
|
|
|
@@ -3710,7 +4074,7 @@ class Queue:
|
|
|
3710
4074
|
# the order ALTER TABLE added them in is no order _row can count on.
|
|
3711
4075
|
SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
|
|
3712
4076
|
"q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
|
|
3713
|
-
"q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
|
|
4077
|
+
"q.rerun_of, q.verdict, q.verdicts, o.submitted_at, q.workspace FROM queue q "
|
|
3714
4078
|
"LEFT JOIN queue o ON o.id = q.rerun_of")
|
|
3715
4079
|
|
|
3716
4080
|
@staticmethod
|
|
@@ -3726,6 +4090,7 @@ class Queue:
|
|
|
3726
4090
|
"rerunOf": r[11], "rerunOfAt": r[14],
|
|
3727
4091
|
"verdict": r[12],
|
|
3728
4092
|
"verdicts": json.loads(r[13]) if r[13] else None,
|
|
4093
|
+
"workspace": r[15],
|
|
3729
4094
|
}
|
|
3730
4095
|
|
|
3731
4096
|
# A row from before run documents has no version, and nothing here can
|
|
@@ -3738,14 +4103,20 @@ class Queue:
|
|
|
3738
4103
|
def _readable(row):
|
|
3739
4104
|
return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
|
|
3740
4105
|
|
|
3741
|
-
def _all(self, db):
|
|
3742
|
-
|
|
4106
|
+
def _all(self, db, ws=None):
|
|
4107
|
+
"""Every readable run; a workspace's when `ws` is given, else the lot --
|
|
4108
|
+
the worker and startup's prompt backfill read across every workspace."""
|
|
4109
|
+
sql = self.SELECT + (" WHERE q.workspace = ?" if ws else "")
|
|
4110
|
+
return [row for row in (self._row(r) for r in db.execute(sql, (ws,) if ws else ()))
|
|
3743
4111
|
if self._readable(row)]
|
|
3744
4112
|
|
|
3745
|
-
def get(self, rid):
|
|
4113
|
+
def get(self, rid, ws=None):
|
|
4114
|
+
"""One run by id, or None. `ws` scopes the lookup so a workspace cannot
|
|
4115
|
+
read another's run by id (docs/workspaces.md); the worker reads
|
|
4116
|
+
unscoped, by the globally unique id."""
|
|
4117
|
+
sql = self.SELECT + " WHERE q.id = ?" + (" AND q.workspace = ?" if ws else "")
|
|
3746
4118
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3747
|
-
row = self._row(db.execute(
|
|
3748
|
-
(rid,)).fetchone())
|
|
4119
|
+
row = self._row(db.execute(sql, (rid, ws) if ws else (rid,)).fetchone())
|
|
3749
4120
|
return row if self._readable(row) else None
|
|
3750
4121
|
|
|
3751
4122
|
def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
|
|
@@ -3757,7 +4128,7 @@ class Queue:
|
|
|
3757
4128
|
second (#238). `before` alone stops at the second. Each row is
|
|
3758
4129
|
brief_row's unless `full` asks for the whole of it."""
|
|
3759
4130
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3760
|
-
rows = self._all(db)
|
|
4131
|
+
rows = self._all(db, workspace_of(self.store))
|
|
3761
4132
|
key = lambda r: (r["submittedAt"], r["id"])
|
|
3762
4133
|
rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
|
|
3763
4134
|
rows.sort(key=key, reverse=True)
|
|
@@ -3788,13 +4159,13 @@ class Queue:
|
|
|
3788
4159
|
total = len(run_items(run))
|
|
3789
4160
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3790
4161
|
db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
|
|
3791
|
-
"snapshot, results, progress, totals, dataset, rerun_of, groups) "
|
|
3792
|
-
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
4162
|
+
"snapshot, results, progress, totals, dataset, rerun_of, groups, workspace) "
|
|
4163
|
+
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
3793
4164
|
(rid, "queued", now, json.dumps(run), "[]",
|
|
3794
4165
|
json.dumps({"current": None, "n": 0, "total": total}),
|
|
3795
4166
|
json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
|
|
3796
4167
|
None if dataset is None else json.dumps(dataset), rerun_of,
|
|
3797
|
-
None if groups is None else json.dumps(groups)))
|
|
4168
|
+
None if groups is None else json.dumps(groups), workspace_of(self.store)))
|
|
3798
4169
|
# Its prompts' uses, in the same transaction: a run is in the
|
|
3799
4170
|
# library the moment it is queued, or not queued at all.
|
|
3800
4171
|
if self.prompts is not None:
|
|
@@ -3882,6 +4253,27 @@ class Queue:
|
|
|
3882
4253
|
(rundir / "dataset.json").write_text(json.dumps(body))
|
|
3883
4254
|
return ["--dataset", str(rundir / "dataset.json")], None
|
|
3884
4255
|
|
|
4256
|
+
def _grading_args(self, run, rundir):
|
|
4257
|
+
"""The worker's eval group bodies: --dataset for a run that grades
|
|
4258
|
+
against one group, which keeps its single-body path and old-row
|
|
4259
|
+
pinning, and --groups for a run that links several (#233) -- every body
|
|
4260
|
+
it kept, by `<id>@<n>`, written beside the run document. Returns
|
|
4261
|
+
(args, None) or (None, why)."""
|
|
4262
|
+
refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
|
|
4263
|
+
keys = {group_key(r) for r in refs if r is not None}
|
|
4264
|
+
if len(keys) <= 1:
|
|
4265
|
+
return self._dataset_args(run, rundir)
|
|
4266
|
+
kept = self.groups(run["id"], raw=True)
|
|
4267
|
+
if kept is None:
|
|
4268
|
+
return None, "the run kept no eval group bodies"
|
|
4269
|
+
missing = sorted({(r.get("name") or r.get("id")) for r in refs
|
|
4270
|
+
if r is not None and group_key(r) not in kept})
|
|
4271
|
+
if missing:
|
|
4272
|
+
return None, f"the run kept no body of the eval group {', '.join(missing)}"
|
|
4273
|
+
rundir.mkdir(parents=True, exist_ok=True)
|
|
4274
|
+
(rundir / "groups.json").write_text(json.dumps(kept))
|
|
4275
|
+
return ["--groups", str(rundir / "groups.json")], None
|
|
4276
|
+
|
|
3885
4277
|
def _behind(self, rid):
|
|
3886
4278
|
"""How many submissions stand between this one and the worker, by
|
|
3887
4279
|
submit time -- what a waiting form names when it says what it is
|
|
@@ -3939,12 +4331,15 @@ class Queue:
|
|
|
3939
4331
|
return self.get(rid), None
|
|
3940
4332
|
|
|
3941
4333
|
def clear(self):
|
|
3942
|
-
"""Empty
|
|
3943
|
-
item files go too, so a cleared run leaves nothing behind
|
|
3944
|
-
re-dequeued by a stray worker."""
|
|
4334
|
+
"""Empty this workspace's History: its runs, in every browser. The
|
|
4335
|
+
materialised item files go too, so a cleared run leaves nothing behind
|
|
4336
|
+
to be re-dequeued by a stray worker. Another workspace's runs stay."""
|
|
4337
|
+
ws = workspace_of(self.store)
|
|
3945
4338
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3946
|
-
|
|
3947
|
-
|
|
4339
|
+
gone = [r[0] for r in db.execute("SELECT id FROM queue WHERE workspace = ?", (ws,))]
|
|
4340
|
+
n = db.execute("DELETE FROM queue WHERE workspace = ?", (ws,)).rowcount
|
|
4341
|
+
for rid in gone:
|
|
4342
|
+
d = self.dir / rid
|
|
3948
4343
|
if d.is_dir():
|
|
3949
4344
|
(d / "cancel").unlink(missing_ok=True)
|
|
3950
4345
|
shutil.rmtree(d, ignore_errors=True)
|
|
@@ -4020,7 +4415,7 @@ class Queue:
|
|
|
4020
4415
|
return None, (403, err)
|
|
4021
4416
|
if NODE is None:
|
|
4022
4417
|
return None, (500, "node is not installed, so nothing can run")
|
|
4023
|
-
dataset, err = self.
|
|
4418
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4024
4419
|
if err:
|
|
4025
4420
|
return None, (409, err)
|
|
4026
4421
|
plugins, err = self._plugin_args(run)
|
|
@@ -4086,7 +4481,7 @@ class Queue:
|
|
|
4086
4481
|
return None, (500, "node is not installed, so nothing can run")
|
|
4087
4482
|
rundir = self.dir / run["id"]
|
|
4088
4483
|
rundir.mkdir(parents=True, exist_ok=True)
|
|
4089
|
-
dataset, err = self.
|
|
4484
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4090
4485
|
if err:
|
|
4091
4486
|
return None, (409, err)
|
|
4092
4487
|
plugins, err = self._plugin_args(run)
|
|
@@ -4202,7 +4597,7 @@ class Queue:
|
|
|
4202
4597
|
return self._finish(rid, "failed", error=err)
|
|
4203
4598
|
if NODE is None:
|
|
4204
4599
|
return self._finish(rid, "failed", error="node is not installed, so nothing can run")
|
|
4205
|
-
dataset, err = self.
|
|
4600
|
+
dataset, err = self._grading_args(run, rundir)
|
|
4206
4601
|
if err:
|
|
4207
4602
|
return self._finish(rid, "failed", error=err)
|
|
4208
4603
|
plugins, err = self._plugin_args(run)
|
|
@@ -4340,12 +4735,19 @@ class Queue:
|
|
|
4340
4735
|
if run is None:
|
|
4341
4736
|
self._stop.wait(QUEUE_WAIT)
|
|
4342
4737
|
continue
|
|
4738
|
+
# The worker grades every workspace's runs; while it grades this
|
|
4739
|
+
# one, the thread is bound to the run's workspace, so any dataset
|
|
4740
|
+
# or Source it resolves (a legacy row's pinned body) reads from
|
|
4741
|
+
# there, not the default (docs/workspaces.md).
|
|
4742
|
+
set_workspace(run.get("workspace"))
|
|
4343
4743
|
try:
|
|
4344
4744
|
self._execute(run)
|
|
4345
4745
|
except Exception as e:
|
|
4346
4746
|
# The type, never the message -- the relay's rule, for the
|
|
4347
4747
|
# relay's reason: an exception message can carry a key.
|
|
4348
4748
|
self._finish(run["id"], "failed", error=f"the worker failed: {type(e).__name__}")
|
|
4749
|
+
finally:
|
|
4750
|
+
set_workspace(None)
|
|
4349
4751
|
|
|
4350
4752
|
def _dequeue(self):
|
|
4351
4753
|
"""The oldest queued run this server can read. One it cannot is never
|
|
@@ -5034,7 +5436,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5034
5436
|
# Runs now execute on the server and nowhere else, so the page
|
|
5035
5437
|
# says what this one is missing, instead of a request failing
|
|
5036
5438
|
# oddly mid-run.
|
|
5037
|
-
head += carried("labstate", {"docs": STORE.served(),
|
|
5439
|
+
head += carried("labstate", {"docs": STORE.served(self.ws),
|
|
5038
5440
|
"runs": {"queue": QUEUE is not None, "node": NODE is not None,
|
|
5039
5441
|
"convert": CONVERT is not None}})
|
|
5040
5442
|
# The page marks where the carried data goes (web/index.html); the
|
|
@@ -5058,9 +5460,39 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5058
5460
|
if self.grouped:
|
|
5059
5461
|
self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
|
|
5060
5462
|
|
|
5463
|
+
# The workspace a request is in (docs/workspaces.md), the default when it
|
|
5464
|
+
# names none. Set it to the store's flagged default even with no store, so
|
|
5465
|
+
# the None-store guards below read a harmless value.
|
|
5466
|
+
ws = None
|
|
5467
|
+
|
|
5468
|
+
def _scope(self):
|
|
5469
|
+
"""Resolve and bind the request's workspace: the `/w/<slug>` path
|
|
5470
|
+
prefix, else the `X-Workspace` header, else the flagged default. The
|
|
5471
|
+
prefix is stripped from self.path so every route below is addressed the
|
|
5472
|
+
same way inside a workspace or out of it -- `/w/<slug>` becomes `/`,
|
|
5473
|
+
which serves the SPA shell, and `/w/<slug>/api/...` becomes `/api/...`.
|
|
5474
|
+
An unknown slug falls back to the default; the page's Gone state for a
|
|
5475
|
+
vanished workspace is phase 3. The slug is a bookmarkable address, not a
|
|
5476
|
+
secret: isolation here is a filter, not access control."""
|
|
5477
|
+
raw = self.path
|
|
5478
|
+
qpos = raw.find("?")
|
|
5479
|
+
path, query = (raw[:qpos], raw[qpos:]) if qpos >= 0 else (raw, "")
|
|
5480
|
+
slug = None
|
|
5481
|
+
if path == "/w" or path.startswith("/w/"):
|
|
5482
|
+
slug, sep, tail = path[3:].partition("/")
|
|
5483
|
+
self.path = ("/" + tail if sep or tail else "/") + query
|
|
5484
|
+
ws = None
|
|
5485
|
+
if STORE is not None:
|
|
5486
|
+
named = slug or self.headers.get("X-Workspace")
|
|
5487
|
+
ws = STORE.workspace(named) if named else None
|
|
5488
|
+
ws = ws or STORE.default_ws
|
|
5489
|
+
self.ws = ws
|
|
5490
|
+
set_workspace(ws)
|
|
5491
|
+
|
|
5061
5492
|
def do_GET(self):
|
|
5062
5493
|
if not self._authorised():
|
|
5063
5494
|
return
|
|
5495
|
+
self._scope()
|
|
5064
5496
|
self._alias()
|
|
5065
5497
|
path = self.path.split("?", 1)[0]
|
|
5066
5498
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
@@ -5096,7 +5528,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5096
5528
|
if path == "/api/state":
|
|
5097
5529
|
if STORE is None:
|
|
5098
5530
|
return self._send(404, b"not found", "text/plain")
|
|
5099
|
-
return self._json(200, {"docs": STORE.served()})
|
|
5531
|
+
return self._json(200, {"docs": STORE.served(self.ws)})
|
|
5100
5532
|
# The run queue (#530): a run is a row the server owns, and History
|
|
5101
5533
|
# reads the server. One run, or the list.
|
|
5102
5534
|
if path == "/api/queue":
|
|
@@ -5115,7 +5547,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5115
5547
|
# The bodies of the eval groups a run grades with, by `<id>@<n>`
|
|
5116
5548
|
# (§17); null for a run that kept none, as /dataset answers.
|
|
5117
5549
|
run_id = path.split("/")[3]
|
|
5118
|
-
if QUEUE is None or QUEUE.get(run_id) is None:
|
|
5550
|
+
if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
|
|
5119
5551
|
return self._send(404, b"not found", "text/plain")
|
|
5120
5552
|
return self._json(200, QUEUE.groups(run_id))
|
|
5121
5553
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
@@ -5129,13 +5561,13 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5129
5561
|
# time such a run was opened, which is the console people are
|
|
5130
5562
|
# told to watch for real faults. A run that is not there is 404.
|
|
5131
5563
|
run_id = path.split("/")[3]
|
|
5132
|
-
if QUEUE is None or QUEUE.get(run_id) is None:
|
|
5564
|
+
if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
|
|
5133
5565
|
return self._send(404, b"not found", "text/plain")
|
|
5134
5566
|
return self._json(200, QUEUE.dataset(run_id))
|
|
5135
5567
|
if path.startswith("/api/queue/"):
|
|
5136
5568
|
if QUEUE is None:
|
|
5137
5569
|
return self._send(404, b"not found", "text/plain")
|
|
5138
|
-
run = QUEUE.get(path[len("/api/queue/"):])
|
|
5570
|
+
run = QUEUE.get(path[len("/api/queue/"):], self.ws)
|
|
5139
5571
|
if run is None:
|
|
5140
5572
|
return self._send(404, b"not found", "text/plain")
|
|
5141
5573
|
# How far back in the queue this submission sits, for the form's
|
|
@@ -5279,6 +5711,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5279
5711
|
def do_DELETE(self):
|
|
5280
5712
|
if not self._authorised() or not self._from_this_page():
|
|
5281
5713
|
return
|
|
5714
|
+
self._scope()
|
|
5282
5715
|
self._alias()
|
|
5283
5716
|
path = self.path.split("?", 1)[0]
|
|
5284
5717
|
if path == "/api/connections/google":
|
|
@@ -5347,6 +5780,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5347
5780
|
def do_PATCH(self):
|
|
5348
5781
|
if not self._authorised() or not self._from_this_page():
|
|
5349
5782
|
return
|
|
5783
|
+
self._scope()
|
|
5350
5784
|
self._alias()
|
|
5351
5785
|
parts = self.path.split("?", 1)[0].split("/")
|
|
5352
5786
|
if len(parts) != 4 or parts[1] != "api":
|
|
@@ -5368,6 +5802,8 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5368
5802
|
return self._json(200, source)
|
|
5369
5803
|
if parts[2] == "queue" and QUEUE is not None:
|
|
5370
5804
|
payload = self._payload() or {}
|
|
5805
|
+
if QUEUE.get(parts[3], self.ws) is None:
|
|
5806
|
+
return self._send(404, b"not found", "text/plain")
|
|
5371
5807
|
run, err = QUEUE.set_comment(parts[3], payload.get("comment"))
|
|
5372
5808
|
if err:
|
|
5373
5809
|
code, message = err
|
|
@@ -5378,6 +5814,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5378
5814
|
def do_PUT(self):
|
|
5379
5815
|
if not self._authorised() or not self._from_this_page():
|
|
5380
5816
|
return
|
|
5817
|
+
self._scope()
|
|
5381
5818
|
self._alias()
|
|
5382
5819
|
parts = self.path.split("?", 1)[0].split("/")
|
|
5383
5820
|
if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
|
|
@@ -5441,6 +5878,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5441
5878
|
return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
|
|
5442
5879
|
|
|
5443
5880
|
def _post(self):
|
|
5881
|
+
self._scope()
|
|
5444
5882
|
self._alias()
|
|
5445
5883
|
path = self.path.split("?", 1)[0]
|
|
5446
5884
|
if path == "/api/state":
|
|
@@ -5623,7 +6061,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5623
6061
|
if len(json.dumps(d.get("body"))) > MAX_DOC:
|
|
5624
6062
|
return self._json(413, {"error": f"{name} is too large"})
|
|
5625
6063
|
versions, stale = STORE.write({n: {"version": d["version"], "body": d.get("body")}
|
|
5626
|
-
for n, d in docs.items()})
|
|
6064
|
+
for n, d in docs.items()}, self.ws)
|
|
5627
6065
|
if stale is not None:
|
|
5628
6066
|
return self._json(409, {"stale": stale})
|
|
5629
6067
|
# A pin names a group's version, so the group keeps that version from
|
|
@@ -5687,10 +6125,8 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5687
6125
|
return self._json(400, {"error": f"the run cannot be queued: {body}"})
|
|
5688
6126
|
ref["n"], ref["version"] = n, fingerprint(body)
|
|
5689
6127
|
groups[group_key(ref)] = body
|
|
5690
|
-
# The worker
|
|
5691
|
-
|
|
5692
|
-
return self._json(400, {"error": "the run cannot be queued: its evals grade against "
|
|
5693
|
-
"one version of one Library eval group at a time"})
|
|
6128
|
+
# The worker reads a body per group now (#233), so a run may link
|
|
6129
|
+
# several; each body is kept with the row under `<id>@<n>`.
|
|
5694
6130
|
return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
|
|
5695
6131
|
|
|
5696
6132
|
# ---- Datasets ----------------------------------------------------------
|
|
@@ -5865,6 +6301,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5865
6301
|
parts = path.split("/")
|
|
5866
6302
|
rid = parts[3]
|
|
5867
6303
|
action = parts[4] if len(parts) > 4 else ""
|
|
6304
|
+
# A run is reachable only from its own workspace: an id from another is
|
|
6305
|
+
# a run this workspace does not have (docs/workspaces.md).
|
|
6306
|
+
if QUEUE.get(rid, self.ws) is None:
|
|
6307
|
+
return self._send(404, b"not found", "text/plain")
|
|
5868
6308
|
if action == "cancel":
|
|
5869
6309
|
run, err = QUEUE.cancel(rid)
|
|
5870
6310
|
elif action == "resume":
|
|
@@ -5989,9 +6429,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
5989
6429
|
if len(parts) == 3:
|
|
5990
6430
|
payload = self._payload() or {}
|
|
5991
6431
|
kind = payload.get("type", DEFAULT_SOURCE_TYPE)
|
|
5992
|
-
# A
|
|
5993
|
-
# made in a lab without that sign-in
|
|
5994
|
-
|
|
6432
|
+
# A platform made with a sign-in (Power Automate, with Microsoft's)
|
|
6433
|
+
# cannot be made in a lab without that sign-in; an unknown kind or
|
|
6434
|
+
# platform is refused by create() below with its own sentence.
|
|
6435
|
+
sign_in = (source_entry(kind, payload.get("config")) or {}).get("signIn")
|
|
5995
6436
|
if sign_in and not SIGN_INS.get(sign_in, lambda: False)():
|
|
5996
6437
|
return self._json(400, {"error": "this lab has no Microsoft app to sign in with (Setup › Connections)"
|
|
5997
6438
|
if sign_in == "microsoft" else f"this lab has no {sign_in} sign-in"})
|