evals-lab 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -298,23 +298,53 @@ FILE_TYPES = {
298
298
  }
299
299
 
300
300
  # The kinds of Source, mirrored from evals-core.ts's SOURCE_TYPES: Python
301
- # cannot load TypeScript, so the server keeps what it enforces -- the id a
302
- # row may carry and the files a Source of that type takes. A type is a
303
- # registry entry, never a branch on its id.
301
+ # cannot load TypeScript, so the server keeps what it enforces. A kind is a
302
+ # registry entry, never a branch on its id. The row's `type` is the kind; a
303
+ # workflow kind's enforcement comes from the platform its `config.platform`
304
+ # names (WORKFLOW_PLATFORMS below), a kind marked `platforms`.
304
305
  SOURCE_TYPES = {
305
306
  "files": {"label": "File Library", "uploads": FILE_TYPES},
306
- # A Power Automate cloud flow (docs/power-automate.md): no uploads yet --
307
- # its records arrive with a later phase -- and a definition, kept as a
308
- # versioned snapshot beside the row.
309
- #
310
- # `uploads` are the files it takes; `take`, when there is one, checks and
311
- # rewrites each before it is kept; `signIn` is the sign-in it is made with,
312
- # without which the lab cannot make one; `definition` means it keeps one.
313
- "power-automate": {"label": "Power Automate workflow", "definition": True, "signIn": "microsoft",
314
- "uploads": {".json": "application/json; charset=utf-8"}},
307
+ "workflow": {"label": "Workflow", "platforms": True},
315
308
  }
316
309
  DEFAULT_SOURCE_TYPE = "files"
317
310
 
311
+ # The workflow platforms, mirrored from evals-core.ts's WORKFLOW_PLATFORMS: the
312
+ # flow engine a workflow Source speaks to, named in its `config.platform`. A
313
+ # workflow kind's enforcement is the platform's, not the kind's.
314
+ #
315
+ # `uploads` are the files it takes; `take`, when there is one, checks and
316
+ # rewrites each before it is kept; `signIn` is the sign-in it is made with,
317
+ # without which the lab cannot make one; `definition` means it keeps one.
318
+ WORKFLOW_PLATFORMS = {
319
+ "power-automate": {"label": "Power Automate flow", "definition": True, "signIn": "microsoft",
320
+ "uploads": {".json": "application/json; charset=utf-8"}},
321
+ }
322
+
323
+
324
+ def _config_platform(config):
325
+ """The platform a Source's stored config (a JSON string or None) names,
326
+ or None -- what the page labels a workflow by."""
327
+ try:
328
+ cfg = json.loads(config) if config else None
329
+ except ValueError:
330
+ return None
331
+ platform = cfg.get("platform") if isinstance(cfg, dict) else None
332
+ return platform if isinstance(platform, str) else None
333
+
334
+
335
+ def source_entry(ptype, config):
336
+ """The enforcement entry for a Source of type [ptype] and [config] (a
337
+ dict or None): its kind's, or, for a workflow kind, the platform its
338
+ config names. None for an unknown kind, or a workflow whose platform the
339
+ lab does not know. A reader asks the entry, never the id."""
340
+ kind = SOURCE_TYPES.get(ptype)
341
+ if kind is None:
342
+ return None
343
+ if kind.get("platforms"):
344
+ platform = config.get("platform") if isinstance(config, dict) else None
345
+ return WORKFLOW_PLATFORMS.get(platform)
346
+ return kind
347
+
318
348
  # The Entra app the page signs in to Microsoft 365 with (docs/power-automate.md
319
349
  # § Registering the app). A single-page app has no secret, so both are public
320
350
  # and served to the page; the token it gets stays in the browser, and the
@@ -481,7 +511,7 @@ def take_record(name, data):
481
511
  return json.dumps(kept, indent=2).encode("utf-8"), None
482
512
 
483
513
 
484
- SOURCE_TYPES["power-automate"]["take"] = take_record
514
+ WORKFLOW_PLATFORMS["power-automate"]["take"] = take_record
485
515
 
486
516
  # The sign-ins a Source type may be made with, and whether this lab has each:
487
517
  # a type naming one this lab lacks cannot be made.
@@ -602,6 +632,45 @@ def hide_keys(docs: dict) -> dict:
602
632
  return out
603
633
 
604
634
 
635
+ # ---- workspaces (docs/workspaces.md) -------------------------------------
636
+ # A workspace is the lab's first tenant dimension: a row in `workspaces` and a
637
+ # filter on every scopable table. Phase 1 is invisible -- a request that names
638
+ # no workspace resolves to the flagged default, so the existing page and CI
639
+ # keep working. It is isolation, not access control: until multi-user lands, a
640
+ # workspace is a filter, not a permission boundary (AGENTS.md).
641
+ #
642
+ # The workspace the current thread's db work is scoped to is held here, bound
643
+ # per request by the Handler and per run by the queue worker; unset, a scoped
644
+ # read falls back to the store's flagged default, which keeps direct callers
645
+ # (startup, a check's own calls) on the default workspace.
646
+ _WS = threading.local()
647
+
648
+
649
+ def set_workspace(ws):
650
+ """Bind the current thread to workspace `ws` for its db work; None clears
651
+ it, so scoped reads fall back to the store's default workspace."""
652
+ _WS.ws = ws
653
+
654
+
655
+ def workspace_of(store) -> str:
656
+ ws = getattr(_WS, "ws", None)
657
+ return ws if ws else store.default_ws
658
+
659
+
660
+ # The `connection_shares` sentinels (phase 4 fills and enforces the table;
661
+ # phase 1 only creates it empty): every workspace, and future ones.
662
+ SHARE_ALL = "*all"
663
+ SHARE_NEW = "*new"
664
+
665
+ # How the `docs` table scopes by key (docs/workspaces.md): workflows are
666
+ # per-workspace, profiles/tokens global, and promptlab.versions is split --
667
+ # its pipeline/dataset slices per-workspace, its profile slice global, because
668
+ # profiles are global so their Restore must be too. served()/write() route
669
+ # each key by these; any other key is per-workspace.
670
+ DOC_GLOBAL = ("promptlab.profiles", "promptlab.tokens")
671
+ DOC_SPLIT = "promptlab.versions"
672
+
673
+
605
674
  class Store:
606
675
  """
607
676
  Documents by name, each with a version that goes up by one per write.
@@ -618,17 +687,36 @@ class Store:
618
687
  self.path = path
619
688
  self.lock = threading.Lock()
620
689
  with self.lock, closing(sqlite3.connect(path)) as db, db:
621
- db.execute("CREATE TABLE IF NOT EXISTS docs (name TEXT PRIMARY KEY, "
622
- "version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
690
+ # The workspace registry and the share table come first:
691
+ # everything else is scoped to a workspace, and the flagged default
692
+ # must exist before any backfill can assign rows to it.
693
+ db.execute("CREATE TABLE IF NOT EXISTS workspaces ("
694
+ "id TEXT PRIMARY KEY, slug TEXT UNIQUE, name TEXT NOT NULL, "
695
+ "archived INTEGER NOT NULL DEFAULT 0, is_default INTEGER NOT NULL DEFAULT 0, "
696
+ "last_used TEXT, created_at TEXT, trash TEXT, trashed_at REAL)")
697
+ # Shared-by-choice connections (profiles, Accounts): phase 4 fills
698
+ # and enforces this; phase 1 only creates it empty.
699
+ db.execute("CREATE TABLE IF NOT EXISTS connection_shares ("
700
+ "kind TEXT NOT NULL, id TEXT NOT NULL, workspace TEXT NOT NULL, "
701
+ "PRIMARY KEY (kind, id, workspace))")
702
+ self.default_ws = self._ensure_default(db)
623
703
  db.execute("CREATE TABLE IF NOT EXISTS runs (at TEXT PRIMARY KEY, body TEXT NOT NULL)")
624
704
  # The key of a profile a write took out, by id, for TRASH_SECONDS:
625
705
  # Undo puts the profile back holding KEY_HELD, and this is what
626
- # it holds.
706
+ # it holds. Profiles are global, so the ring is too.
627
707
  db.execute("CREATE TABLE IF NOT EXISTS dropped_keys (id TEXT PRIMARY KEY, "
628
708
  "key TEXT NOT NULL, at REAL NOT NULL)")
709
+ # The docs table keys documents by (name, workspace): a store from
710
+ # before workspaces keyed them by name alone and is rebuilt once,
711
+ # the global keys moved under NULL and promptlab.versions split.
712
+ # Idempotent: a later open finds the workspace column and leaves it.
713
+ cols = {r[1] for r in db.execute("PRAGMA table_info(docs)")}
714
+ if not cols:
715
+ self._create_docs(db)
629
716
  # History was a document, capped at what a browser could hold. The
630
717
  # first start with the table moves what that document had into it,
631
- # once, and drops the document so the page stops carrying it.
718
+ # once, and drops the document so the page stops carrying it. It
719
+ # reads name/body, so it runs on either docs schema.
632
720
  old = db.execute("SELECT body FROM docs WHERE name = 'promptlab.runs'").fetchone()
633
721
  if old and not db.execute("SELECT 1 FROM runs LIMIT 1").fetchone():
634
722
  for run in json.loads(old[0] or "null") or []:
@@ -636,20 +724,118 @@ class Store:
636
724
  db.execute("INSERT OR IGNORE INTO runs (at, body) VALUES (?, ?)",
637
725
  (run["at"], json.dumps(run)))
638
726
  db.execute("DELETE FROM docs WHERE name = 'promptlab.runs'")
727
+ # The run history is the queue table now; this `runs` table holds
728
+ # only what the pre-queue history document migrated into it, and is
729
+ # served from nowhere. It carries the workspace column all the same,
730
+ # so the scopable-table set is whole and anything that ever reads it
731
+ # inherits the filter (docs/workspaces.md).
732
+ if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(runs)")}:
733
+ db.execute("ALTER TABLE runs ADD COLUMN workspace TEXT")
734
+ db.execute("UPDATE runs SET workspace = ? WHERE workspace IS NULL", (self.default_ws,))
735
+ if cols and "workspace" not in cols:
736
+ self._rebuild_docs(db)
737
+
738
+ def _ensure_default(self, db) -> str:
739
+ """The flagged default workspace's id, seeding "Default" (/w/default,
740
+ is_default = 1) on a store that has none. Defined by the flag, not its
741
+ name or slug, so a later rename never moves it (docs/workspaces.md)."""
742
+ row = db.execute("SELECT id FROM workspaces WHERE is_default = 1").fetchone()
743
+ if row:
744
+ return row[0]
745
+ wid = secrets.token_hex(6)
746
+ now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
747
+ db.execute("INSERT INTO workspaces (id, slug, name, archived, is_default, last_used, created_at) "
748
+ "VALUES (?, 'default', 'Default', 0, 1, ?, ?)", (wid, now, now))
749
+ return wid
750
+
751
+ def workspace(self, slug: str) -> str:
752
+ """The id of the workspace at /w/<slug>, or None. An unknown slug is
753
+ None; the Handler falls back to the default so a stale address still
754
+ reaches a working lab until the page's Gone state lands (phase 3)."""
755
+ with self.lock, closing(sqlite3.connect(self.path)) as db:
756
+ row = db.execute("SELECT id FROM workspaces WHERE slug = ?", (slug,)).fetchone()
757
+ return row[0] if row else None
639
758
 
640
- def _rows(self, db):
641
- return {name: {"version": version, "body": None if body is None else json.loads(body)}
642
- for name, version, body in db.execute("SELECT name, version, body FROM docs")}
759
+ @staticmethod
760
+ def _create_docs(db):
761
+ db.execute("CREATE TABLE docs (name TEXT NOT NULL, workspace TEXT, "
762
+ "version INTEGER NOT NULL, body TEXT, updated_at TEXT NOT NULL)")
763
+ # NULLs are distinct in a UNIQUE index, so the global keys cannot rely
764
+ # on one PRIMARY KEY for their one-row-per-name rule: two partial
765
+ # indexes, scoped rows keyed by (name, workspace) and global rows by
766
+ # name, give each its own uniqueness and its own upsert target.
767
+ db.execute("CREATE UNIQUE INDEX docs_scoped ON docs(name, workspace) WHERE workspace IS NOT NULL")
768
+ db.execute("CREATE UNIQUE INDEX docs_global ON docs(name) WHERE workspace IS NULL")
769
+
770
+ def _rebuild_docs(self, db):
771
+ rows = db.execute("SELECT name, version, body, updated_at FROM docs").fetchall()
772
+ db.execute("ALTER TABLE docs RENAME TO docs_old")
773
+ self._create_docs(db)
774
+ put = lambda name, ws, version, body, at: db.execute(
775
+ "INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, ?, ?, ?, ?)",
776
+ (name, ws, version, body, at))
777
+ for name, version, body, at in rows:
778
+ if name in DOC_GLOBAL:
779
+ put(name, None, version, body, at)
780
+ elif name == DOC_SPLIT:
781
+ parsed = json.loads(body) if body else {}
782
+ put(name, self.default_ws, version, None if body is None else json.dumps(
783
+ {"pipeline": parsed.get("pipeline", {}), "dataset": parsed.get("dataset", {})}), at)
784
+ put(name, None, version, None if body is None else json.dumps(
785
+ {"profile": parsed.get("profile", {})}), at)
786
+ else:
787
+ put(name, self.default_ws, version, body, at)
788
+ db.execute("DROP TABLE docs_old")
789
+
790
+ def _doc_rows(self, db, ws: str) -> dict:
791
+ """The logical document set a workspace reads: its scoped rows, the
792
+ global rows (profiles, tokens), and promptlab.versions reassembled from
793
+ its per-workspace pipeline/dataset slice and the global profile slice,
794
+ carrying the per-workspace slice's version so the page's one version
795
+ per key still arbitrates pipeline/dataset Restore."""
796
+ parse = lambda b: None if b is None else json.loads(b)
797
+ scoped = {name: (version, body) for name, version, body in db.execute(
798
+ "SELECT name, version, body FROM docs WHERE workspace = ?", (ws,))}
799
+ glob = {name: (version, body) for name, version, body in db.execute(
800
+ "SELECT name, version, body FROM docs WHERE workspace IS NULL")}
801
+ out = {name: {"version": v, "body": parse(b)}
802
+ for name, (v, b) in scoped.items() if name != DOC_SPLIT}
803
+ for name in DOC_GLOBAL:
804
+ if name in glob:
805
+ v, b = glob[name]
806
+ out[name] = {"version": v, "body": parse(b)}
807
+ sv, gv = scoped.get(DOC_SPLIT), glob.get(DOC_SPLIT)
808
+ if sv is not None or gv is not None:
809
+ sver, sbody = sv if sv is not None else (0, None)
810
+ sbody, gbody = parse(sbody) or {}, parse(gv[1]) if gv is not None else {}
811
+ out[DOC_SPLIT] = {"version": sver, "body": {
812
+ "pipeline": sbody.get("pipeline", {}), "dataset": sbody.get("dataset", {}),
813
+ "profile": (gbody or {}).get("profile", {})}}
814
+ return out
643
815
 
644
- def all(self) -> dict:
816
+ def all(self, ws=None) -> dict:
645
817
  with self.lock, closing(sqlite3.connect(self.path)) as db:
646
- return self._rows(db)
647
-
648
- def served(self) -> dict:
649
- """What a browser is handed: the SYNCED documents only, so a retired
650
- key's rows stay in the store without reaching a page again, and no
651
- profile's key (KEY_HELD)."""
652
- return hide_keys({n: d for n, d in self.all().items() if n in SYNCED})
818
+ return self._doc_rows(db, ws or workspace_of(self))
819
+
820
+ def served(self, ws=None) -> dict:
821
+ """What a browser is handed for its workspace: the SYNCED documents
822
+ only, so a retired key's rows stay in the store without reaching a page
823
+ again, and no profile's key (KEY_HELD)."""
824
+ return hide_keys({n: d for n, d in self.all(ws).items() if n in SYNCED})
825
+
826
+ def _put(self, db, name, ws, version, body, at):
827
+ text = None if body is None else json.dumps(body)
828
+ if ws is None:
829
+ db.execute(
830
+ "INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, NULL, ?, ?, ?) "
831
+ "ON CONFLICT(name) WHERE workspace IS NULL DO UPDATE SET version = excluded.version, "
832
+ "body = excluded.body, updated_at = excluded.updated_at", (name, version, text, at))
833
+ else:
834
+ db.execute(
835
+ "INSERT INTO docs (name, workspace, version, body, updated_at) VALUES (?, ?, ?, ?, ?) "
836
+ "ON CONFLICT(name, workspace) WHERE workspace IS NOT NULL DO UPDATE SET "
837
+ "version = excluded.version, body = excluded.body, updated_at = excluded.updated_at",
838
+ (name, ws, version, text, at))
653
839
 
654
840
  def _keyring(self, db, now: dict) -> dict:
655
841
  """Each profile id's key as the store holds it: its profile's, else a
@@ -668,18 +854,24 @@ class Store:
668
854
  return ring
669
855
 
670
856
  def held_key(self, key: str) -> str:
671
- """The key KEY_HELD stands for, or "" when the store holds none."""
857
+ """The key KEY_HELD stands for, or "" when the store holds none.
858
+ Profiles are global, so the ring is read under the default workspace."""
672
859
  with self.lock, closing(sqlite3.connect(self.path)) as db, db:
673
- return self._keyring(db, self._rows(db)).get(key[len(KEY_HELD):], "")
860
+ return self._keyring(db, self._doc_rows(db, self.default_ws)).get(key[len(KEY_HELD):], "")
674
861
 
675
- def write(self, docs: dict):
862
+ def write(self, docs: dict, ws=None):
676
863
  """
677
- `docs` is {name: {"version": the version it began from, "body": ...}}.
678
- Returns ({name: new version}, None), or (None, {name: current}) for
679
- every document that has moved on, with nothing written.
864
+ `docs` is {name: {"version": the version it began from, "body": ...}},
865
+ written into workspace `ws` (the current thread's, by default). Each
866
+ key is routed by scope: workflows and the rest per-workspace, profiles
867
+ and tokens global, promptlab.versions split (pipeline/dataset
868
+ per-workspace, profile global). Returns ({name: new version}, None), or
869
+ (None, {name: current}) for every document that has moved on, with
870
+ nothing written.
680
871
  """
872
+ ws = ws or workspace_of(self)
681
873
  with self.lock, closing(sqlite3.connect(self.path)) as db:
682
- now = self._rows(db)
874
+ now = self._doc_rows(db, ws)
683
875
  have = lambda n: now.get(n, {"version": 0, "body": None})
684
876
  stale = {n: have(n) for n, d in docs.items() if d["version"] != have(n)["version"]}
685
877
  if stale:
@@ -701,11 +893,19 @@ class Store:
701
893
  for p in profiles_in("promptlab.profiles", have("promptlab.profiles")["body"])
702
894
  if str(p.get("id") or "") not in kept and isinstance(p.get("key"), str) and p["key"]])
703
895
  for n, d in docs.items():
704
- db.execute(
705
- "INSERT INTO docs (name, version, body, updated_at) VALUES (?, ?, ?, ?) "
706
- "ON CONFLICT(name) DO UPDATE SET version = excluded.version, "
707
- "body = excluded.body, updated_at = excluded.updated_at",
708
- (n, have(n)["version"] + 1, None if d["body"] is None else json.dumps(d["body"]), at))
896
+ version, body = have(n)["version"] + 1, d["body"]
897
+ if n == DOC_SPLIT:
898
+ parsed = body if isinstance(body, dict) else {}
899
+ self._put(db, n, ws, version, None if body is None else {
900
+ "pipeline": parsed.get("pipeline", {}), "dataset": parsed.get("dataset", {})}, at)
901
+ # The profile slice is global, on its own version
902
+ # counter: profiles are shared, so their Restore is too.
903
+ gv = db.execute("SELECT version FROM docs WHERE name = ? AND workspace IS NULL",
904
+ (n,)).fetchone()
905
+ self._put(db, n, None, (gv[0] if gv else 0) + 1,
906
+ None if body is None else {"profile": parsed.get("profile", {})}, at)
907
+ else:
908
+ self._put(db, n, None if n in DOC_GLOBAL else ws, version, body, at)
709
909
  return {n: have(n)["version"] + 1 for n in docs}, None
710
910
 
711
911
 
@@ -781,6 +981,34 @@ class Sources:
781
981
  f"DEFAULT '{DEFAULT_SOURCE_TYPE}'")
782
982
  if "config" not in have:
783
983
  db.execute("ALTER TABLE sources ADD COLUMN config TEXT")
984
+ # The workspace a Source belongs to (docs/workspaces.md). A store
985
+ # from before workspaces gains the column; every user Source with
986
+ # none -- a pre-workspace row, or one left unscoped by any edge --
987
+ # backfills to the default on open, idempotently, while the system
988
+ # samples row stays workspace-agnostic (NULL) so it shows in every
989
+ # workspace.
990
+ if "workspace" not in have:
991
+ db.execute("ALTER TABLE sources ADD COLUMN workspace TEXT")
992
+ db.execute("UPDATE sources SET workspace = ? WHERE workspace IS NULL AND system = 0",
993
+ (store.default_ws,))
994
+ # A Power Automate Source was its own kind once (type
995
+ # "power-automate", docs/power-automate.md); it is now the generic
996
+ # workflow kind, the platform named in its config
997
+ # (docs/sources-tab.md). Converted here, once and idempotently: a
998
+ # later open finds none left. remove()/restore_trash keep a row's
999
+ # type and config, so Undo brings a converted Source back exactly,
1000
+ # still speaking to Power Automate.
1001
+ for sid, config in db.execute(
1002
+ "SELECT id, config FROM sources WHERE type = 'power-automate'").fetchall():
1003
+ try:
1004
+ cfg = json.loads(config) if config else {}
1005
+ except ValueError:
1006
+ cfg = {}
1007
+ if not isinstance(cfg, dict):
1008
+ cfg = {}
1009
+ cfg["platform"] = "power-automate"
1010
+ db.execute("UPDATE sources SET type = 'workflow', config = ? WHERE id = ?",
1011
+ (json.dumps(cfg), sid))
784
1012
  # A flow's definition, one row per version: a snapshot the page
785
1013
  # took and redacted, redacted again here.
786
1014
  db.execute("CREATE TABLE IF NOT EXISTS source_definitions ("
@@ -809,26 +1037,37 @@ class Sources:
809
1037
  out = []
810
1038
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
811
1039
  db.row_factory = sqlite3.Row
812
- for r in db.execute("SELECT id, name, system, bytes, type FROM sources "
813
- "ORDER BY system DESC, name COLLATE NOCASE"):
1040
+ for r in db.execute("SELECT id, name, system, bytes, type, config, created_at FROM sources "
1041
+ "WHERE workspace = ? OR system = 1 "
1042
+ "ORDER BY system DESC, name COLLATE NOCASE", (workspace_of(self.store),)):
814
1043
  if r["system"]:
815
1044
  files = self._sample_files()
816
1045
  out.append({"id": r["id"], "name": r["name"], "system": True,
817
1046
  "type": r["type"],
818
1047
  "files": len(files), "bytes": sum(f["bytes"] for f in files)})
819
1048
  else:
820
- n = db.execute("SELECT COUNT(*) FROM source_files WHERE source = ?",
821
- (r["id"],)).fetchone()[0]
822
- out.append({"id": r["id"], "name": r["name"], "system": False,
823
- "type": r["type"], "files": n, "bytes": r["bytes"]})
1049
+ n, last = db.execute("SELECT COUNT(*), MAX(at) FROM source_files WHERE source = ?",
1050
+ (r["id"],)).fetchone()
1051
+ # When it last changed: made, or a file added -- what a
1052
+ # picker orders its recent Sources by.
1053
+ row = {"id": r["id"], "name": r["name"], "system": False,
1054
+ "type": r["type"], "files": n, "bytes": r["bytes"],
1055
+ "changed": max(filter(None, (r["created_at"], last)))}
1056
+ # A workflow's platform is how the page labels it; nothing
1057
+ # else of its config rides in the summary.
1058
+ platform = _config_platform(r["config"])
1059
+ if platform is not None:
1060
+ row["platform"] = platform
1061
+ out.append(row)
824
1062
  return out
825
1063
 
826
1064
  def get(self, sid) -> dict:
827
1065
  """One Source with its file list, or None."""
828
1066
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
829
1067
  db.row_factory = sqlite3.Row
830
- r = db.execute("SELECT id, name, system, bytes, type, config FROM sources WHERE id = ?",
831
- (sid,)).fetchone()
1068
+ r = db.execute("SELECT id, name, system, bytes, type, config FROM sources "
1069
+ "WHERE id = ? AND (workspace = ? OR system = 1)",
1070
+ (sid, workspace_of(self.store))).fetchone()
832
1071
  if r is None:
833
1072
  return None
834
1073
  if r["system"]:
@@ -875,14 +1114,18 @@ class Sources:
875
1114
  return None, (400, "that name is too long")
876
1115
  if not isinstance(kind, str) or kind not in SOURCE_TYPES:
877
1116
  return None, (400, f"the lab has no Source type {kind!r}")
1117
+ # A workflow kind needs a platform the lab knows, named in its config.
1118
+ if source_entry(kind, config) is None:
1119
+ return None, (400, "the lab has no such Source platform")
878
1120
  kept, err = self._config(config)
879
1121
  if err:
880
1122
  return None, err
881
1123
  sid = secrets.token_hex(6)
882
1124
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
883
- db.execute("INSERT INTO sources (id, name, system, bytes, created_at, type, config) "
884
- "VALUES (?, ?, 0, 0, ?, ?, ?)",
885
- (sid, name, time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), kind, kept))
1125
+ db.execute("INSERT INTO sources (id, name, system, bytes, created_at, type, config, workspace) "
1126
+ "VALUES (?, ?, 0, 0, ?, ?, ?, ?)",
1127
+ (sid, name, time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), kind, kept,
1128
+ workspace_of(self.store)))
886
1129
  (self.dir / sid).mkdir(parents=True, exist_ok=True)
887
1130
  return {"id": sid, "name": name, "system": False, "type": kind,
888
1131
  "config": json.loads(kept) if kept else None, "files": [], "bytes": 0}, None
@@ -893,8 +1136,8 @@ class Sources:
893
1136
  """A Source's newest definition and the versions kept, or None for a
894
1137
  Source that is not there or keeps none."""
895
1138
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
896
- r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
897
- if r is None or not (SOURCE_TYPES.get(r[0]) or {}).get("definition"):
1139
+ r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1140
+ if r is None or not (source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}).get("definition"):
898
1141
  return None
899
1142
  rows = db.execute("SELECT version, body, at FROM source_definitions WHERE source = ? "
900
1143
  "ORDER BY version DESC", (sid,)).fetchall()
@@ -920,10 +1163,10 @@ class Sources:
920
1163
  return None, (413, "that definition is over the cap")
921
1164
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
922
1165
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
923
- r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
1166
+ r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
924
1167
  if r is None:
925
1168
  return None, (404, "no such source")
926
- if not (SOURCE_TYPES.get(r[0]) or {}).get("definition"):
1169
+ if not (source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}).get("definition"):
927
1170
  return None, (400, "a Source of that type keeps no definition")
928
1171
  current = db.execute("SELECT MAX(version) FROM source_definitions WHERE source = ?",
929
1172
  (sid,)).fetchone()[0] or 0
@@ -953,10 +1196,15 @@ class Sources:
953
1196
  db.execute("DELETE FROM source_definitions WHERE source = ?", (sid,))
954
1197
 
955
1198
  def type_of(self, sid):
956
- """A Source's type's entry in SOURCE_TYPES; None for no such Source."""
1199
+ """A Source's enforcement entry -- its kind's, or its workflow
1200
+ platform's (source_entry); None for no such Source. {} for a known
1201
+ Source whose platform the lab does not know, so a reader asks the
1202
+ entry rather than the id."""
957
1203
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
958
- r = db.execute("SELECT type FROM sources WHERE id = ?", (sid,)).fetchone()
959
- return None if r is None else (SOURCE_TYPES.get(r[0]) or {})
1204
+ r = db.execute("SELECT type, config FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1205
+ if r is None:
1206
+ return None
1207
+ return source_entry(r[0], json.loads(r[1]) if r[1] else None) or {}
960
1208
 
961
1209
  def uploads(self, sid):
962
1210
  """The files a Source takes, by extension, as its type declares them;
@@ -972,7 +1220,7 @@ class Sources:
972
1220
  if len(name) > 80:
973
1221
  return None, (400, "that name is too long")
974
1222
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
975
- r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1223
+ r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
976
1224
  if r is None:
977
1225
  return None, (404, "no such source")
978
1226
  if r[0]:
@@ -985,8 +1233,9 @@ class Sources:
985
1233
  Undo can restore it; the system Source is undeletable. Returns
986
1234
  (token, None), or (None, error)."""
987
1235
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
988
- r = db.execute("SELECT system, name, created_at, type, config FROM sources "
989
- "WHERE id = ?", (sid,)).fetchone()
1236
+ r = db.execute("SELECT system, name, created_at, type, config, workspace FROM sources "
1237
+ "WHERE id = ? AND (workspace = ? OR system = 1)",
1238
+ (sid, workspace_of(self.store))).fetchone()
990
1239
  if r is None:
991
1240
  return None, (404, "no such source")
992
1241
  if r[0]:
@@ -1001,7 +1250,7 @@ class Sources:
1001
1250
  entry.mkdir(parents=True, exist_ok=True)
1002
1251
  (entry / "manifest.json").write_text(json.dumps({
1003
1252
  "kind": "source", "source": sid, "name": r[1], "created": r[2],
1004
- "type": r[3], "config": r[4], "at": time.time(), "files": files}))
1253
+ "type": r[3], "config": r[4], "workspace": r[5], "at": time.time(), "files": files}))
1005
1254
  if (self.dir / sid).is_dir():
1006
1255
  shutil.move(str(self.dir / sid), str(entry / sid))
1007
1256
  return token, None
@@ -1014,7 +1263,7 @@ class Sources:
1014
1263
  return None, (400, "no files named")
1015
1264
  names = list(dict.fromkeys(names))
1016
1265
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
1017
- r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1266
+ r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1018
1267
  if r is None:
1019
1268
  return None, (404, "no such source")
1020
1269
  if r[0]:
@@ -1069,19 +1318,22 @@ class Sources:
1069
1318
  sid, name = m.get("source"), m.get("name")
1070
1319
  if not isinstance(sid, str) or not isinstance(name, str):
1071
1320
  return None, (404, "no such trash entry")
1321
+ # Back into the workspace it was trashed from (an older trash entry
1322
+ # has none: the default). A name is unique within a workspace.
1323
+ ws = m.get("workspace") or self.store.default_ws
1072
1324
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
1073
1325
  if db.execute("SELECT 1 FROM sources WHERE name = ? COLLATE NOCASE "
1074
- "AND system = 0", (name,)).fetchone():
1326
+ "AND system = 0 AND workspace = ?", (name, ws)).fetchone():
1075
1327
  return None, (409, f"{name!r} has been taken since, so nothing was restored")
1076
1328
  with db:
1077
1329
  kind = m.get("type")
1078
1330
  db.execute("INSERT INTO sources (id, name, system, bytes, created_at, "
1079
- "type, config) VALUES (?, ?, 0, ?, ?, ?, ?)",
1331
+ "type, config, workspace) VALUES (?, ?, 0, ?, ?, ?, ?, ?)",
1080
1332
  (sid, name, sum(f.get("bytes", 0) for f in m["files"]),
1081
1333
  m.get("created", time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())),
1082
1334
  # As it was: a restore puts the row back, not a guess at it.
1083
1335
  kind if isinstance(kind, str) else DEFAULT_SOURCE_TYPE,
1084
- m.get("config") if isinstance(m.get("config"), str) else None))
1336
+ m.get("config") if isinstance(m.get("config"), str) else None, ws))
1085
1337
  for f in m["files"]:
1086
1338
  if isinstance(f, dict) and isinstance(f.get("name"), str):
1087
1339
  db.execute("INSERT INTO source_files (source, name, bytes, at) "
@@ -1097,7 +1349,8 @@ class Sources:
1097
1349
  return None, (404, "no such trash entry")
1098
1350
  total = sum(f.get("bytes", 0) for f in files)
1099
1351
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
1100
- r = db.execute("SELECT system, bytes FROM sources WHERE id = ?", (sid,)).fetchone()
1352
+ r = db.execute("SELECT system, bytes FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
1353
+ (sid, workspace_of(self.store))).fetchone()
1101
1354
  if r is None:
1102
1355
  return None, (409, "the Source that held these files is gone, so nothing was restored")
1103
1356
  if r[0]:
@@ -1168,14 +1421,15 @@ class Sources:
1168
1421
  names = list(dict.fromkeys(names))
1169
1422
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1170
1423
  db.row_factory = sqlite3.Row
1171
- dest = db.execute("SELECT system, bytes, type FROM sources WHERE id = ?",
1172
- (sid,)).fetchone()
1424
+ ws = workspace_of(self.store)
1425
+ dest = db.execute("SELECT system, bytes, type FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
1426
+ (sid, ws)).fetchone()
1173
1427
  if dest is None:
1174
1428
  return None, (404, "no such source")
1175
1429
  if dest["system"]:
1176
1430
  return None, (403, "the sample library cannot be written to")
1177
- src = db.execute("SELECT system, type FROM sources WHERE id = ?",
1178
- (from_sid,)).fetchone()
1431
+ src = db.execute("SELECT system, type FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
1432
+ (from_sid, ws)).fetchone()
1179
1433
  if src is None:
1180
1434
  return None, (404, "no such source")
1181
1435
  # A file means what its Source's type says it means, so it only
@@ -1264,7 +1518,7 @@ class Sources:
1264
1518
  """Every file of a Source, as (name, path on disk) in its own order,
1265
1519
  or None when there is no such Source."""
1266
1520
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1267
- r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1521
+ r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1268
1522
  if r is None:
1269
1523
  return None
1270
1524
  if r[0]:
@@ -1281,7 +1535,7 @@ class Sources:
1281
1535
  or not all(isinstance(n, str) for n in names):
1282
1536
  return None, (400, "the names are needed")
1283
1537
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1284
- r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1538
+ r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1285
1539
  if r is None:
1286
1540
  return None, (404, "no such source")
1287
1541
  if r[0]:
@@ -1347,7 +1601,8 @@ class Sources:
1347
1601
  """
1348
1602
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1349
1603
  db.row_factory = sqlite3.Row
1350
- r = db.execute("SELECT system, bytes FROM sources WHERE id = ?", (sid,)).fetchone()
1604
+ r = db.execute("SELECT system, bytes FROM sources WHERE id = ? AND (workspace = ? OR system = 1)",
1605
+ (sid, workspace_of(self.store))).fetchone()
1351
1606
  if r is None:
1352
1607
  return None, (404, "no such source")
1353
1608
  if r["system"]:
@@ -1379,7 +1634,7 @@ class Sources:
1379
1634
  def file_path(self, sid: str, name: str) -> Path:
1380
1635
  """The on-disk path of a stored file, or None. `name` is already clean."""
1381
1636
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
1382
- r = db.execute("SELECT system FROM sources WHERE id = ?", (sid,)).fetchone()
1637
+ r = db.execute("SELECT system FROM sources WHERE id = ? AND (workspace = ? OR system = 1)", (sid, workspace_of(self.store))).fetchone()
1383
1638
  if r is None:
1384
1639
  return None
1385
1640
  if r[0]:
@@ -1452,22 +1707,23 @@ class Sources:
1452
1707
  # it is left where it is, and nothing reads it.
1453
1708
 
1454
1709
  DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
1455
- # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 7 is an
1710
+ # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 8 reads
1711
+ # the recorded-reply metric ids under their new names (#299); version 7 is an
1456
1712
  # eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
1457
1713
  # 5 was told by its `source` alone, and earlier ones by neither.
1458
- DATASET_BODY_VERSION = 7
1714
+ DATASET_BODY_VERSION = 8
1459
1715
  DATASET_NAME_MAX = 80
1460
1716
  # The file forms Export writes and Import reads. Export writes an eval group
1461
- # at version 7 (docs/pipeline-model.md §17); Import reads that, and a dataset
1462
- # file of versions 1 to 7, upgraded, and refuses anything else, as a pipeline
1717
+ # at version 8 (docs/pipeline-model.md §17); Import reads that, and a dataset
1718
+ # file of versions 1 to 8, upgraded, and refuses anything else, as a pipeline
1463
1719
  # of another version is refused. Versions 1 to 3 carried a prompt, which an
1464
1720
  # import gives to the Prompt library.
1465
1721
  EXPORT_ONE = "evals-lab/eval-group"
1466
1722
  EXPORT_ALL = "evals-lab/eval-groups"
1467
1723
  DATASET_ONE = "evals-lab/dataset"
1468
1724
  DATASET_ALL = "evals-lab/datasets"
1469
- EXPORT_VERSION = 7
1470
- IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
1725
+ EXPORT_VERSION = 8
1726
+ IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8)
1471
1727
  # Each file form, and the key its one entry or its list sits under.
1472
1728
  EXPORT_KEYS = {EXPORT_ONE: "group", EXPORT_ALL: "groups", DATASET_ONE: "dataset", DATASET_ALL: "datasets"}
1473
1729
  SCORING_MODES = ("all", "weighted")
@@ -1505,7 +1761,7 @@ def _term_in(items, term) -> bool:
1505
1761
  def case_metrics(c: dict) -> list:
1506
1762
  """A version-4 case's expectations as the metrics that say the same:
1507
1763
  evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
1508
- through fixtures/dataset-v7.json."""
1764
+ through fixtures/dataset-v8.json."""
1509
1765
  def strs(v):
1510
1766
  return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
1511
1767
  if c.get("discarded") is True:
@@ -1572,7 +1828,7 @@ def case_of_v5(c):
1572
1828
 
1573
1829
 
1574
1830
  def group_of_v6(body: dict) -> dict:
1575
- """A version-6 body as a version-7 eval group: evals-core.ts's
1831
+ """A version-6 body as a version-8 eval group: evals-core.ts's
1576
1832
  groupOfV6. Scored All, the lab's grader, and no metrics of its own for
1577
1833
  every item or the whole run -- what a Metrics eval naming the dataset with
1578
1834
  none of its own graded."""
@@ -1581,8 +1837,43 @@ def group_of_v6(body: dict) -> dict:
1581
1837
  "grader": None, "every": [], "run": [], **rest}
1582
1838
 
1583
1839
 
1840
+ # The recorded-reply metric ids renamed at version 8: evals-core.ts's
1841
+ # RECORDED_IDS. Stored tokens inside eval group bodies and a pipeline's private
1842
+ # group, so they are mapped wherever a body is read (#299).
1843
+ RECORDED_IDS = {
1844
+ "equals-production": "same-as-recorded",
1845
+ "fields-equal-production": "fields-equal-recorded",
1846
+ "same-parse-outcome": "same-parse-as-recorded",
1847
+ }
1848
+
1849
+
1850
+ def _rename_metric(m):
1851
+ return {**m, "type": RECORDED_IDS[m["type"]]} if isinstance(m, dict) and m.get("type") in RECORDED_IDS else m
1852
+
1853
+
1854
+ def _rename_metrics(lst):
1855
+ return [_rename_metric(m) for m in lst] if isinstance(lst, list) else lst
1856
+
1857
+
1858
+ def recorded_ids_v7(body):
1859
+ """A version-7 body (or a freshly made version-8 one) with its
1860
+ recorded-reply metric ids read under their version-8 names, at version 8:
1861
+ evals-core.ts's recordedIdsV7."""
1862
+ out = dict(body)
1863
+ out["version"] = DATASET_BODY_VERSION
1864
+ if isinstance(body.get("every"), list):
1865
+ out["every"] = _rename_metrics(body["every"])
1866
+ if isinstance(body.get("run"), list):
1867
+ out["run"] = _rename_metrics(body["run"])
1868
+ if isinstance(body.get("cases"), list):
1869
+ out["cases"] = [{**c, "metrics": _rename_metrics(c["metrics"])}
1870
+ if isinstance(c, dict) and isinstance(c.get("metrics"), list) else c
1871
+ for c in body["cases"]]
1872
+ return out
1873
+
1874
+
1584
1875
  def upgrade_body(body):
1585
- """An earlier body as today's (version 7): evals-core.ts's
1876
+ """An earlier body as today's (version 8): evals-core.ts's
1586
1877
  upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
1587
1878
  its `replays` and `conformance` go (fixtures/replays.json holds the
1588
1879
  parser's tests). Version 2's `rules` go -- they clean a job's answer, so
@@ -1593,24 +1884,28 @@ def upgrade_body(body):
1593
1884
  needs the rules or the prompt takes them first (`body_rules`,
1594
1885
  `body_prompt`). A body naming its Source is version 5, whose Contains
1595
1886
  metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
1596
- group's scoring, grader, Every item and Whole run (`group_of_v6`). A body
1597
- saying it is version 7 comes back as it was; so does anything that is
1598
- not a body."""
1887
+ group's scoring, grader, Every item and Whole run (`group_of_v6`); version
1888
+ 7 reads its recorded-reply metric ids under their version-8 names
1889
+ (`recorded_ids_v7`). A body saying it is version 8 comes back as it was;
1890
+ so does anything that is not a body."""
1599
1891
  if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
1600
1892
  return body
1601
- # A body saying any other version is one this lab does not read, and is
1602
- # left for dataset_problem to refuse.
1603
1893
  if "version" in body:
1604
- return group_of_v6(body) if body["version"] == 6 else body
1894
+ # Version 7 reads its recorded-reply metric ids under their version-8
1895
+ # names; version 6 is its cases alone. Any other version is one this
1896
+ # lab does not read, left for dataset_problem to refuse.
1897
+ if body["version"] == 7:
1898
+ return recorded_ids_v7(body)
1899
+ return recorded_ids_v7(group_of_v6(body)) if body["version"] == 6 else body
1605
1900
  if "source" in body:
1606
1901
  up = dict(body)
1607
1902
  if isinstance(body.get("cases"), list):
1608
1903
  up["cases"] = [case_of_v5(c) for c in body["cases"]]
1609
- return group_of_v6(up)
1904
+ return recorded_ids_v7(group_of_v6(up))
1610
1905
  cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
1611
1906
  if not isinstance(cases, list):
1612
1907
  return body
1613
- return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
1908
+ return recorded_ids_v7(group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}))
1614
1909
 
1615
1910
 
1616
1911
  def body_prompt(body):
@@ -1754,7 +2049,7 @@ def scenario_ref(sc, i):
1754
2049
  what version 4's upgrade gives one, by position."""
1755
2050
  sid = sc.get("id") if isinstance(sc, dict) else None
1756
2051
  name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
1757
- return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
2052
+ return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {chr(65 + i) if i < 26 else i + 1}")
1758
2053
 
1759
2054
 
1760
2055
  class Prompts:
@@ -1782,6 +2077,13 @@ class Prompts:
1782
2077
  cols = {r[1] for r in db.execute("PRAGMA table_info(prompt_uses)")}
1783
2078
  if "chain" in cols and "job" not in cols:
1784
2079
  db.execute("ALTER TABLE prompt_uses RENAME COLUMN chain TO job")
2080
+ # The workspace a prompt belongs to (docs/workspaces.md): the
2081
+ # Default is per-workspace, so is_default is scoped too. Children
2082
+ # (versions, uses) co-scope through the prompt id. A store from
2083
+ # before workspaces backfills every prompt to the default.
2084
+ if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(prompts)")}:
2085
+ db.execute("ALTER TABLE prompts ADD COLUMN workspace TEXT")
2086
+ db.execute("UPDATE prompts SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
1785
2087
  # One-off facts about the library: that the runs from before it
1786
2088
  # have been read into it, so a restart does not read them again
1787
2089
  # and bring back a prompt someone has since deleted.
@@ -1796,9 +2098,9 @@ class Prompts:
1796
2098
  db.row_factory = sqlite3.Row
1797
2099
  return closing(db)
1798
2100
 
1799
- @staticmethod
1800
- def _live(db, pid):
1801
- return db.execute("SELECT * FROM prompts WHERE id = ? AND trash IS NULL", (pid,)).fetchone()
2101
+ def _live(self, db, pid):
2102
+ return db.execute("SELECT * FROM prompts WHERE id = ? AND trash IS NULL AND workspace = ?",
2103
+ (pid, workspace_of(self.store))).fetchone()
1802
2104
 
1803
2105
  @staticmethod
1804
2106
  def _head(db, pid):
@@ -1837,7 +2139,8 @@ class Prompts:
1837
2139
  def list(self) -> list:
1838
2140
  with self.store.lock, self._connect() as db:
1839
2141
  return [self._summary(db, r) for r in db.execute(
1840
- "SELECT * FROM prompts WHERE trash IS NULL ORDER BY updated_at DESC, id")]
2142
+ "SELECT * FROM prompts WHERE trash IS NULL AND workspace = ? ORDER BY updated_at DESC, id",
2143
+ (workspace_of(self.store),))]
1841
2144
 
1842
2145
  def get(self, pid):
1843
2146
  with self.store.lock, self._connect() as db:
@@ -1847,7 +2150,8 @@ class Prompts:
1847
2150
  def default(self):
1848
2151
  """The Default's newest version, {id, name, version, text}, or None."""
1849
2152
  with self.store.lock, self._connect() as db:
1850
- r = db.execute("SELECT * FROM prompts WHERE is_default = 1 AND trash IS NULL").fetchone()
2153
+ r = db.execute("SELECT * FROM prompts WHERE is_default = 1 AND trash IS NULL AND workspace = ?",
2154
+ (workspace_of(self.store),)).fetchone()
1851
2155
  if r is None:
1852
2156
  return None
1853
2157
  head = self._head(db, r["id"])
@@ -1859,10 +2163,11 @@ class Prompts:
1859
2163
  def _insert(self, db, name, text, default=False):
1860
2164
  pid = secrets.token_hex(6)
1861
2165
  now = self._now()
2166
+ ws = workspace_of(self.store)
1862
2167
  if default:
1863
- db.execute("UPDATE prompts SET is_default = 0")
1864
- db.execute("INSERT INTO prompts (id, name, is_default, created_at, updated_at) VALUES (?, ?, ?, ?, ?)",
1865
- (pid, name, 1 if default else 0, now, now))
2168
+ db.execute("UPDATE prompts SET is_default = 0 WHERE workspace = ?", (ws,))
2169
+ db.execute("INSERT INTO prompts (id, name, is_default, created_at, updated_at, workspace) "
2170
+ "VALUES (?, ?, ?, ?, ?, ?)", (pid, name, 1 if default else 0, now, now, ws))
1866
2171
  db.execute("INSERT INTO prompt_versions (prompt_id, version, text, created_at, edited_at) "
1867
2172
  "VALUES (?, 1, ?, ?, ?)", (pid, text, now, time.time()))
1868
2173
  return pid
@@ -1877,7 +2182,8 @@ class Prompts:
1877
2182
  return version
1878
2183
 
1879
2184
  def _has_default(self, db):
1880
- return db.execute("SELECT 1 FROM prompts WHERE is_default = 1 AND trash IS NULL").fetchone() is not None
2185
+ return db.execute("SELECT 1 FROM prompts WHERE is_default = 1 AND trash IS NULL AND workspace = ?",
2186
+ (workspace_of(self.store),)).fetchone() is not None
1881
2187
 
1882
2188
  def adopt(self, db, text, name="", default=False):
1883
2189
  """A prompt reading [text]: the live one that already does, or a new
@@ -1893,13 +2199,13 @@ class Prompts:
1893
2199
  db.execute("UPDATE prompts SET is_default = 1 WHERE id = ?", (pid,))
1894
2200
  return pid
1895
2201
 
1896
- @staticmethod
1897
- def _matching(db, text):
1898
- """The newest version of any live prompt reading exactly [text]."""
2202
+ def _matching(self, db, text):
2203
+ """The newest version of any live prompt in this workspace reading
2204
+ exactly [text]."""
1899
2205
  return db.execute("SELECT v.prompt_id, v.version FROM prompt_versions v "
1900
- "JOIN prompts p ON p.id = v.prompt_id AND p.trash IS NULL "
2206
+ "JOIN prompts p ON p.id = v.prompt_id AND p.trash IS NULL AND p.workspace = ? "
1901
2207
  "WHERE v.text = ? ORDER BY v.created_at DESC, v.version DESC LIMIT 1",
1902
- (text,)).fetchone()
2208
+ (workspace_of(self.store), text)).fetchone()
1903
2209
 
1904
2210
  def record(self, db, rid, run, at):
1905
2211
  """A run's uses, one per scenario per job, linked by the rules in
@@ -1997,7 +2303,7 @@ class Prompts:
1997
2303
  with self.store.lock, self._connect() as db, db:
1998
2304
  if self._live(db, pid) is None:
1999
2305
  return None, (404, "no such prompt")
2000
- db.execute("UPDATE prompts SET is_default = 0")
2306
+ db.execute("UPDATE prompts SET is_default = 0 WHERE workspace = ?", (workspace_of(self.store),))
2001
2307
  db.execute("UPDATE prompts SET is_default = 1 WHERE id = ?", (pid,))
2002
2308
  return self._detail(db, self._live(db, pid)), None
2003
2309
 
@@ -2029,7 +2335,8 @@ class Prompts:
2029
2335
 
2030
2336
  def restore(self, token):
2031
2337
  with self.store.lock, self._connect() as db, db:
2032
- r = db.execute("SELECT * FROM prompts WHERE trash = ?", (str(token),)).fetchone()
2338
+ r = db.execute("SELECT * FROM prompts WHERE trash = ? AND workspace = ?",
2339
+ (str(token), workspace_of(self.store))).fetchone()
2033
2340
  if r is None:
2034
2341
  return None, (404, "no such trash entry")
2035
2342
  db.execute("UPDATE prompts SET trash = NULL, trashed_at = NULL WHERE id = ?", (r["id"],))
@@ -2081,6 +2388,12 @@ class Datasets:
2081
2388
  "group_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
2082
2389
  "created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
2083
2390
  "ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (group_id, n))")
2391
+ # The workspace a dataset belongs to (docs/workspaces.md); its
2392
+ # group versions and archives co-scope through the dataset id. A
2393
+ # store from before workspaces backfills every row to the default.
2394
+ if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(datasets)")}:
2395
+ db.execute("ALTER TABLE datasets ADD COLUMN workspace TEXT")
2396
+ db.execute("UPDATE datasets SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
2084
2397
  # Rows from an earlier version are converted once, in place: a
2085
2398
  # dataset is typed in by hand and costly to re-enter, so it is
2086
2399
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -2155,17 +2468,20 @@ class Datasets:
2155
2468
  return closing(db)
2156
2469
 
2157
2470
  def _live(self, db, did):
2158
- return db.execute("SELECT * FROM datasets WHERE id = ? AND trash IS NULL", (did,)).fetchone()
2471
+ return db.execute("SELECT * FROM datasets WHERE id = ? AND trash IS NULL AND workspace = ?",
2472
+ (did, workspace_of(self.store))).fetchone()
2159
2473
 
2160
2474
  def _names(self, db, but=None):
2161
- return {r[0] for r in db.execute("SELECT name FROM datasets WHERE trash IS NULL AND id IS NOT ?",
2162
- (but,))}
2475
+ return {r[0] for r in db.execute(
2476
+ "SELECT name FROM datasets WHERE trash IS NULL AND id IS NOT ? AND workspace = ?",
2477
+ (but, workspace_of(self.store)))}
2163
2478
 
2164
2479
  def list(self) -> list:
2165
2480
  with self.store.lock, self._connect() as db:
2166
2481
  counts = self._counts(db)
2167
2482
  return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
2168
- "SELECT * FROM datasets WHERE trash IS NULL ORDER BY name COLLATE NOCASE, id")]
2483
+ "SELECT * FROM datasets WHERE trash IS NULL AND workspace = ? ORDER BY name COLLATE NOCASE, id",
2484
+ (workspace_of(self.store),))]
2169
2485
 
2170
2486
  def get(self, did):
2171
2487
  with self.store.lock, self._connect() as db:
@@ -2195,11 +2511,11 @@ class Datasets:
2195
2511
  "VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
2196
2512
  return self._head(db, r["id"])
2197
2513
 
2198
- @staticmethod
2199
- def _pins(db) -> set:
2514
+ def _pins(self, db) -> set:
2200
2515
  """Every (group, version) a stored pipeline pins, read in the caller's
2201
- transaction from the docs table the page writes them to."""
2202
- row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows'").fetchone()
2516
+ transaction from this workspace's workflows document."""
2517
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows' AND workspace = ?",
2518
+ (workspace_of(self.store),)).fetchone()
2203
2519
  return pins_in(json.loads(row[0])) if row and row[0] else set()
2204
2520
 
2205
2521
  def _cut(self, db, did, text):
@@ -2313,8 +2629,9 @@ class Datasets:
2313
2629
  def _insert(self, db, name, body):
2314
2630
  did = secrets.token_hex(6)
2315
2631
  now = self._now()
2316
- db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
2317
- "VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
2632
+ db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at, workspace) "
2633
+ "VALUES (?, ?, 1, ?, ?, ?, ?)",
2634
+ (did, name, json.dumps(body), now, now, workspace_of(self.store)))
2318
2635
  return did
2319
2636
 
2320
2637
  def create(self, name, body=None):
@@ -2384,7 +2701,8 @@ class Datasets:
2384
2701
  """A trashed dataset back, under a new ` (2)` name if its own has been
2385
2702
  taken since. Returns (DatasetDoc, None)."""
2386
2703
  with self.store.lock, self._connect() as db, db:
2387
- r = db.execute("SELECT * FROM datasets WHERE trash = ?", (str(token),)).fetchone()
2704
+ r = db.execute("SELECT * FROM datasets WHERE trash = ? AND workspace = ?",
2705
+ (str(token), workspace_of(self.store))).fetchone()
2388
2706
  if r is None:
2389
2707
  return None, (404, "no such trash entry")
2390
2708
  name = unique_dataset_name(r["name"], self._names(db, but=r["id"]))
@@ -2420,8 +2738,8 @@ class Datasets:
2420
2738
 
2421
2739
  def export_all(self):
2422
2740
  with self.store.lock, self._connect() as db:
2423
- rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
2424
- "ORDER BY name COLLATE NOCASE, id").fetchall()
2741
+ rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL AND workspace = ? "
2742
+ "ORDER BY name COLLATE NOCASE, id", (workspace_of(self.store),)).fetchall()
2425
2743
  return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
2426
2744
  "groups": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
2427
2745
 
@@ -2679,11 +2997,40 @@ class Packs:
2679
2997
  self.trash = {}
2680
2998
  with store.lock, closing(sqlite3.connect(store.path)) as db, db:
2681
2999
  db.execute("CREATE TABLE IF NOT EXISTS packs ("
2682
- "id TEXT PRIMARY KEY, name TEXT NOT NULL, version TEXT NOT NULL, "
2683
- "manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL)")
3000
+ "id TEXT NOT NULL, name TEXT NOT NULL, version TEXT NOT NULL, "
3001
+ "manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL, "
3002
+ "workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (id, workspace))")
2684
3003
  db.execute("CREATE TABLE IF NOT EXISTS pack_items ("
2685
3004
  "pack TEXT NOT NULL, kind TEXT NOT NULL, key TEXT NOT NULL, item TEXT NOT NULL, "
2686
- "PRIMARY KEY (pack, kind, key))")
3005
+ "workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (pack, workspace, kind, key))")
3006
+ # A pack's id is its manifest's, not globally unique, so two
3007
+ # workspaces can hold the same pack: the workspace is part of both
3008
+ # keys (docs/workspaces.md). A store from before workspaces keyed
3009
+ # packs by id alone; it is rebuilt once, every row to the default.
3010
+ if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(packs)")}:
3011
+ self._rebuild(db, store.default_ws, "packs",
3012
+ "id, name, version, manifest, presets, installed_at",
3013
+ "id TEXT NOT NULL, name TEXT NOT NULL, version TEXT NOT NULL, "
3014
+ "manifest TEXT NOT NULL, presets TEXT NOT NULL, installed_at TEXT NOT NULL, "
3015
+ "workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (id, workspace)")
3016
+ if "workspace" not in {r[1] for r in db.execute("PRAGMA table_info(pack_items)")}:
3017
+ self._rebuild(db, store.default_ws, "pack_items", "pack, kind, key, item",
3018
+ "pack TEXT NOT NULL, kind TEXT NOT NULL, key TEXT NOT NULL, item TEXT NOT NULL, "
3019
+ "workspace TEXT NOT NULL DEFAULT '', PRIMARY KEY (pack, workspace, kind, key)")
3020
+
3021
+ @staticmethod
3022
+ def _rebuild(db, ws, table, cols, schema):
3023
+ """A table keyed without a workspace, rebuilt with one its old rows
3024
+ take: SQLite cannot add to a PRIMARY KEY, so the rows move through a
3025
+ fresh table (the sources `type`/`config` migration's shape)."""
3026
+ rows = db.execute(f"SELECT {cols} FROM {table}").fetchall()
3027
+ db.execute(f"ALTER TABLE {table} RENAME TO {table}_old")
3028
+ db.execute(f"CREATE TABLE {table} ({schema})")
3029
+ names = cols.split(", ")
3030
+ ph = ", ".join("?" * (len(names) + 1))
3031
+ for r in rows:
3032
+ db.execute(f"INSERT INTO {table} ({cols}, workspace) VALUES ({ph})", (*r, ws))
3033
+ db.execute(f"DROP TABLE {table}_old")
2687
3034
 
2688
3035
  def _connect(self):
2689
3036
  db = sqlite3.connect(self.store.path)
@@ -2693,12 +3040,15 @@ class Packs:
2693
3040
  def _items(self, pid):
2694
3041
  with self.store.lock, self._connect() as db:
2695
3042
  return {(r["kind"], r["key"]): r["item"]
2696
- for r in db.execute("SELECT kind, key, item FROM pack_items WHERE pack = ?", (pid,))}
3043
+ for r in db.execute("SELECT kind, key, item FROM pack_items WHERE pack = ? AND workspace = ?",
3044
+ (pid, workspace_of(self.store)))}
2697
3045
 
2698
3046
  def list(self) -> list:
2699
3047
  with self.store.lock, self._connect() as db:
2700
- packs = db.execute("SELECT * FROM packs ORDER BY name COLLATE NOCASE").fetchall()
2701
- items = db.execute("SELECT pack, kind, item FROM pack_items").fetchall()
3048
+ ws = workspace_of(self.store)
3049
+ packs = db.execute("SELECT * FROM packs WHERE workspace = ? ORDER BY name COLLATE NOCASE",
3050
+ (ws,)).fetchall()
3051
+ items = db.execute("SELECT pack, kind, item FROM pack_items WHERE workspace = ?", (ws,)).fetchall()
2702
3052
  workflows = {w.get("id"): w.get("name") for w in self._doc("promptlab.workflows")[1].get("list", [])}
2703
3053
  out = []
2704
3054
  for p in packs:
@@ -2718,7 +3068,8 @@ class Packs:
2718
3068
  def presets(self) -> list:
2719
3069
  """Every installed pack's Setup presets, each marked with its pack."""
2720
3070
  with self.store.lock, self._connect() as db:
2721
- rows = db.execute("SELECT id, presets FROM packs ORDER BY name COLLATE NOCASE").fetchall()
3071
+ rows = db.execute("SELECT id, presets FROM packs WHERE workspace = ? ORDER BY name COLLATE NOCASE",
3072
+ (workspace_of(self.store),)).fetchall()
2722
3073
  return [{**p, "pack": r["id"]} for r in rows for p in json.loads(r["presets"])]
2723
3074
 
2724
3075
  # The page's documents are the server's to write here too, through the
@@ -2757,9 +3108,11 @@ class Packs:
2757
3108
  why = pack_requires_problem(m, set(plugins))
2758
3109
  if why:
2759
3110
  return None, (400, why)
3111
+ ws = workspace_of(self.store)
2760
3112
  with self.lock:
2761
3113
  with self.store.lock, self._connect() as db:
2762
- have = db.execute("SELECT version FROM packs WHERE id = ?", (m["id"],)).fetchone()
3114
+ have = db.execute("SELECT version FROM packs WHERE id = ? AND workspace = ?",
3115
+ (m["id"], ws)).fetchone()
2763
3116
  if have and have["version"] == m["packVersion"]:
2764
3117
  return {"pack": m["id"], "installed": False}, None
2765
3118
  owned = self._items(m["id"])
@@ -2854,22 +3207,23 @@ class Packs:
2854
3207
  self._write_doc("promptlab.versions", keep)
2855
3208
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
2856
3209
  with self.store.lock, self._connect() as db, db:
2857
- db.execute("INSERT INTO packs (id, name, version, manifest, presets, installed_at) "
2858
- "VALUES (?, ?, ?, ?, ?, ?) ON CONFLICT(id) DO UPDATE SET name = excluded.name, "
2859
- "version = excluded.version, manifest = excluded.manifest, "
3210
+ db.execute("INSERT INTO packs (id, name, version, manifest, presets, installed_at, workspace) "
3211
+ "VALUES (?, ?, ?, ?, ?, ?, ?) ON CONFLICT(id, workspace) DO UPDATE SET "
3212
+ "name = excluded.name, version = excluded.version, manifest = excluded.manifest, "
2860
3213
  "presets = excluded.presets, installed_at = excluded.installed_at",
2861
3214
  (m["id"], m["name"].strip(), m["packVersion"], json.dumps(m),
2862
- json.dumps(pack["presets"]), now))
3215
+ json.dumps(pack["presets"]), now, ws))
2863
3216
  for (kind, key), item in made.items():
2864
- db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item) VALUES (?, ?, ?, ?)",
2865
- (m["id"], kind, key, item))
3217
+ db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item, workspace) "
3218
+ "VALUES (?, ?, ?, ?, ?)", (m["id"], kind, key, item, ws))
2866
3219
  return {"pack": m["id"], "installed": True}, None
2867
3220
 
2868
3221
  def remove(self, pid):
2869
3222
  """What the pack made, into the trash, at once. Returns (token, None)."""
3223
+ ws = workspace_of(self.store)
2870
3224
  with self.lock:
2871
3225
  with self.store.lock, self._connect() as db:
2872
- row = db.execute("SELECT * FROM packs WHERE id = ?", (pid,)).fetchone()
3226
+ row = db.execute("SELECT * FROM packs WHERE id = ? AND workspace = ?", (pid, ws)).fetchone()
2873
3227
  if row is None:
2874
3228
  return None, (404, "no such pack")
2875
3229
  items = self._items(pid)
@@ -2894,8 +3248,8 @@ class Packs:
2894
3248
  if gone:
2895
3249
  self._write_doc("promptlab.workflows", drop)
2896
3250
  with self.store.lock, self._connect() as db, db:
2897
- db.execute("DELETE FROM pack_items WHERE pack = ?", (pid,))
2898
- db.execute("DELETE FROM packs WHERE id = ?", (pid,))
3251
+ db.execute("DELETE FROM pack_items WHERE pack = ? AND workspace = ?", (pid, ws))
3252
+ db.execute("DELETE FROM packs WHERE id = ? AND workspace = ?", (pid, ws))
2899
3253
  token = secrets.token_hex(6)
2900
3254
  self.trash[token] = entry
2901
3255
  return token, None
@@ -2915,13 +3269,14 @@ class Packs:
2915
3269
  self._write_doc("promptlab.workflows",
2916
3270
  lambda body: {**body, "list": [*body.get("list", []), *entry["pipelines"]]})
2917
3271
  r = entry["row"]
3272
+ ws = r.get("workspace", self.store.default_ws)
2918
3273
  with self.store.lock, self._connect() as db, db:
2919
- db.execute("INSERT OR REPLACE INTO packs (id, name, version, manifest, presets, installed_at) "
2920
- "VALUES (?, ?, ?, ?, ?, ?)",
2921
- (r["id"], r["name"], r["version"], r["manifest"], r["presets"], r["installed_at"]))
3274
+ db.execute("INSERT OR REPLACE INTO packs (id, name, version, manifest, presets, installed_at, workspace) "
3275
+ "VALUES (?, ?, ?, ?, ?, ?, ?)",
3276
+ (r["id"], r["name"], r["version"], r["manifest"], r["presets"], r["installed_at"], ws))
2922
3277
  for kind, key, item in entry["items"]:
2923
- db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item) VALUES (?, ?, ?, ?)",
2924
- (r["id"], kind, key, item))
3278
+ db.execute("INSERT OR REPLACE INTO pack_items (pack, kind, key, item, workspace) "
3279
+ "VALUES (?, ?, ?, ?, ?)", (r["id"], kind, key, item, ws))
2925
3280
  del self.trash[token]
2926
3281
  return {"pack": r["id"]}, None
2927
3282
 
@@ -2932,7 +3287,8 @@ class Packs:
2932
3287
  removed it and has datasets of its own is left alone. Returns what
2933
3288
  happened, or None where nothing was tried."""
2934
3289
  with self.store.lock, self._connect() as db:
2935
- have = db.execute("SELECT version FROM packs WHERE id = 'demo'").fetchone()
3290
+ have = db.execute("SELECT version FROM packs WHERE id = 'demo' AND workspace = ?",
3291
+ (workspace_of(self.store),)).fetchone()
2936
3292
  if have is None and DATASETS.list():
2937
3293
  return None
2938
3294
  got, err = self.install(demo_pack())
@@ -3701,6 +4057,14 @@ class Queue:
3701
4057
  # `dataset` column, read as its one group.
3702
4058
  if "groups" not in cols:
3703
4059
  db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
4060
+ # The workspace a run belongs to (docs/workspaces.md): History is
4061
+ # per-workspace, so the list and every id-keyed read filter by it.
4062
+ # The worker loop is the one global reader -- it grades every
4063
+ # workspace's runs by id. A store from before workspaces backfills
4064
+ # to the default.
4065
+ if "workspace" not in cols:
4066
+ db.execute("ALTER TABLE queue ADD COLUMN workspace TEXT")
4067
+ db.execute("UPDATE queue SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
3704
4068
 
3705
4069
  # ---- rows -----------------------------------------------------------
3706
4070
 
@@ -3710,7 +4074,7 @@ class Queue:
3710
4074
  # the order ALTER TABLE added them in is no order _row can count on.
3711
4075
  SELECT = ("SELECT q.id, q.status, q.cancel, q.submitted_at, q.started_at, "
3712
4076
  "q.finished_at, q.snapshot, q.results, q.progress, q.totals, q.error, "
3713
- "q.rerun_of, q.verdict, q.verdicts, o.submitted_at FROM queue q "
4077
+ "q.rerun_of, q.verdict, q.verdicts, o.submitted_at, q.workspace FROM queue q "
3714
4078
  "LEFT JOIN queue o ON o.id = q.rerun_of")
3715
4079
 
3716
4080
  @staticmethod
@@ -3726,6 +4090,7 @@ class Queue:
3726
4090
  "rerunOf": r[11], "rerunOfAt": r[14],
3727
4091
  "verdict": r[12],
3728
4092
  "verdicts": json.loads(r[13]) if r[13] else None,
4093
+ "workspace": r[15],
3729
4094
  }
3730
4095
 
3731
4096
  # A row from before run documents has no version, and nothing here can
@@ -3738,14 +4103,20 @@ class Queue:
3738
4103
  def _readable(row):
3739
4104
  return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
3740
4105
 
3741
- def _all(self, db):
3742
- return [row for row in (self._row(r) for r in db.execute(self.SELECT))
4106
+ def _all(self, db, ws=None):
4107
+ """Every readable run; a workspace's when `ws` is given, else the lot --
4108
+ the worker and startup's prompt backfill read across every workspace."""
4109
+ sql = self.SELECT + (" WHERE q.workspace = ?" if ws else "")
4110
+ return [row for row in (self._row(r) for r in db.execute(sql, (ws,) if ws else ()))
3743
4111
  if self._readable(row)]
3744
4112
 
3745
- def get(self, rid):
4113
+ def get(self, rid, ws=None):
4114
+ """One run by id, or None. `ws` scopes the lookup so a workspace cannot
4115
+ read another's run by id (docs/workspaces.md); the worker reads
4116
+ unscoped, by the globally unique id."""
4117
+ sql = self.SELECT + " WHERE q.id = ?" + (" AND q.workspace = ?" if ws else "")
3746
4118
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3747
- row = self._row(db.execute(self.SELECT + " WHERE q.id = ?",
3748
- (rid,)).fetchone())
4119
+ row = self._row(db.execute(sql, (rid, ws) if ws else (rid,)).fetchone())
3749
4120
  return row if self._readable(row) else None
3750
4121
 
3751
4122
  def list(self, limit=RUNS_PAGE, before=None, before_id=None, full=False):
@@ -3757,7 +4128,7 @@ class Queue:
3757
4128
  second (#238). `before` alone stops at the second. Each row is
3758
4129
  brief_row's unless `full` asks for the whole of it."""
3759
4130
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3760
- rows = self._all(db)
4131
+ rows = self._all(db, workspace_of(self.store))
3761
4132
  key = lambda r: (r["submittedAt"], r["id"])
3762
4133
  rows = [r for r in rows if before is None or key(r) < (before, before_id or "")]
3763
4134
  rows.sort(key=key, reverse=True)
@@ -3788,13 +4159,13 @@ class Queue:
3788
4159
  total = len(run_items(run))
3789
4160
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3790
4161
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
3791
- "snapshot, results, progress, totals, dataset, rerun_of, groups) "
3792
- "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?)",
4162
+ "snapshot, results, progress, totals, dataset, rerun_of, groups, workspace) "
4163
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
3793
4164
  (rid, "queued", now, json.dumps(run), "[]",
3794
4165
  json.dumps({"current": None, "n": 0, "total": total}),
3795
4166
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
3796
4167
  None if dataset is None else json.dumps(dataset), rerun_of,
3797
- None if groups is None else json.dumps(groups)))
4168
+ None if groups is None else json.dumps(groups), workspace_of(self.store)))
3798
4169
  # Its prompts' uses, in the same transaction: a run is in the
3799
4170
  # library the moment it is queued, or not queued at all.
3800
4171
  if self.prompts is not None:
@@ -3882,6 +4253,27 @@ class Queue:
3882
4253
  (rundir / "dataset.json").write_text(json.dumps(body))
3883
4254
  return ["--dataset", str(rundir / "dataset.json")], None
3884
4255
 
4256
+ def _grading_args(self, run, rundir):
4257
+ """The worker's eval group bodies: --dataset for a run that grades
4258
+ against one group, which keeps its single-body path and old-row
4259
+ pinning, and --groups for a run that links several (#233) -- every body
4260
+ it kept, by `<id>@<n>`, written beside the run document. Returns
4261
+ (args, None) or (None, why)."""
4262
+ refs = [eval_group_ref(t) for t in run["snapshot"].get("evals", []) or []]
4263
+ keys = {group_key(r) for r in refs if r is not None}
4264
+ if len(keys) <= 1:
4265
+ return self._dataset_args(run, rundir)
4266
+ kept = self.groups(run["id"], raw=True)
4267
+ if kept is None:
4268
+ return None, "the run kept no eval group bodies"
4269
+ missing = sorted({(r.get("name") or r.get("id")) for r in refs
4270
+ if r is not None and group_key(r) not in kept})
4271
+ if missing:
4272
+ return None, f"the run kept no body of the eval group {', '.join(missing)}"
4273
+ rundir.mkdir(parents=True, exist_ok=True)
4274
+ (rundir / "groups.json").write_text(json.dumps(kept))
4275
+ return ["--groups", str(rundir / "groups.json")], None
4276
+
3885
4277
  def _behind(self, rid):
3886
4278
  """How many submissions stand between this one and the worker, by
3887
4279
  submit time -- what a waiting form names when it says what it is
@@ -3939,12 +4331,15 @@ class Queue:
3939
4331
  return self.get(rid), None
3940
4332
 
3941
4333
  def clear(self):
3942
- """Empty the queue: every run, in every browser. The materialised
3943
- item files go too, so a cleared run leaves nothing behind to be
3944
- re-dequeued by a stray worker."""
4334
+ """Empty this workspace's History: its runs, in every browser. The
4335
+ materialised item files go too, so a cleared run leaves nothing behind
4336
+ to be re-dequeued by a stray worker. Another workspace's runs stay."""
4337
+ ws = workspace_of(self.store)
3945
4338
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3946
- n = db.execute("DELETE FROM queue").rowcount
3947
- for d in self.dir.iterdir():
4339
+ gone = [r[0] for r in db.execute("SELECT id FROM queue WHERE workspace = ?", (ws,))]
4340
+ n = db.execute("DELETE FROM queue WHERE workspace = ?", (ws,)).rowcount
4341
+ for rid in gone:
4342
+ d = self.dir / rid
3948
4343
  if d.is_dir():
3949
4344
  (d / "cancel").unlink(missing_ok=True)
3950
4345
  shutil.rmtree(d, ignore_errors=True)
@@ -4020,7 +4415,7 @@ class Queue:
4020
4415
  return None, (403, err)
4021
4416
  if NODE is None:
4022
4417
  return None, (500, "node is not installed, so nothing can run")
4023
- dataset, err = self._dataset_args(run, rundir)
4418
+ dataset, err = self._grading_args(run, rundir)
4024
4419
  if err:
4025
4420
  return None, (409, err)
4026
4421
  plugins, err = self._plugin_args(run)
@@ -4086,7 +4481,7 @@ class Queue:
4086
4481
  return None, (500, "node is not installed, so nothing can run")
4087
4482
  rundir = self.dir / run["id"]
4088
4483
  rundir.mkdir(parents=True, exist_ok=True)
4089
- dataset, err = self._dataset_args(run, rundir)
4484
+ dataset, err = self._grading_args(run, rundir)
4090
4485
  if err:
4091
4486
  return None, (409, err)
4092
4487
  plugins, err = self._plugin_args(run)
@@ -4202,7 +4597,7 @@ class Queue:
4202
4597
  return self._finish(rid, "failed", error=err)
4203
4598
  if NODE is None:
4204
4599
  return self._finish(rid, "failed", error="node is not installed, so nothing can run")
4205
- dataset, err = self._dataset_args(run, rundir)
4600
+ dataset, err = self._grading_args(run, rundir)
4206
4601
  if err:
4207
4602
  return self._finish(rid, "failed", error=err)
4208
4603
  plugins, err = self._plugin_args(run)
@@ -4340,12 +4735,19 @@ class Queue:
4340
4735
  if run is None:
4341
4736
  self._stop.wait(QUEUE_WAIT)
4342
4737
  continue
4738
+ # The worker grades every workspace's runs; while it grades this
4739
+ # one, the thread is bound to the run's workspace, so any dataset
4740
+ # or Source it resolves (a legacy row's pinned body) reads from
4741
+ # there, not the default (docs/workspaces.md).
4742
+ set_workspace(run.get("workspace"))
4343
4743
  try:
4344
4744
  self._execute(run)
4345
4745
  except Exception as e:
4346
4746
  # The type, never the message -- the relay's rule, for the
4347
4747
  # relay's reason: an exception message can carry a key.
4348
4748
  self._finish(run["id"], "failed", error=f"the worker failed: {type(e).__name__}")
4749
+ finally:
4750
+ set_workspace(None)
4349
4751
 
4350
4752
  def _dequeue(self):
4351
4753
  """The oldest queued run this server can read. One it cannot is never
@@ -5034,7 +5436,7 @@ class Handler(BaseHTTPRequestHandler):
5034
5436
  # Runs now execute on the server and nowhere else, so the page
5035
5437
  # says what this one is missing, instead of a request failing
5036
5438
  # oddly mid-run.
5037
- head += carried("labstate", {"docs": STORE.served(),
5439
+ head += carried("labstate", {"docs": STORE.served(self.ws),
5038
5440
  "runs": {"queue": QUEUE is not None, "node": NODE is not None,
5039
5441
  "convert": CONVERT is not None}})
5040
5442
  # The page marks where the carried data goes (web/index.html); the
@@ -5058,9 +5460,39 @@ class Handler(BaseHTTPRequestHandler):
5058
5460
  if self.grouped:
5059
5461
  self.path = "/api/datasets" + self.path[len(self.GROUPS_ROUTE):]
5060
5462
 
5463
+ # The workspace a request is in (docs/workspaces.md), the default when it
5464
+ # names none. Set it to the store's flagged default even with no store, so
5465
+ # the None-store guards below read a harmless value.
5466
+ ws = None
5467
+
5468
+ def _scope(self):
5469
+ """Resolve and bind the request's workspace: the `/w/<slug>` path
5470
+ prefix, else the `X-Workspace` header, else the flagged default. The
5471
+ prefix is stripped from self.path so every route below is addressed the
5472
+ same way inside a workspace or out of it -- `/w/<slug>` becomes `/`,
5473
+ which serves the SPA shell, and `/w/<slug>/api/...` becomes `/api/...`.
5474
+ An unknown slug falls back to the default; the page's Gone state for a
5475
+ vanished workspace is phase 3. The slug is a bookmarkable address, not a
5476
+ secret: isolation here is a filter, not access control."""
5477
+ raw = self.path
5478
+ qpos = raw.find("?")
5479
+ path, query = (raw[:qpos], raw[qpos:]) if qpos >= 0 else (raw, "")
5480
+ slug = None
5481
+ if path == "/w" or path.startswith("/w/"):
5482
+ slug, sep, tail = path[3:].partition("/")
5483
+ self.path = ("/" + tail if sep or tail else "/") + query
5484
+ ws = None
5485
+ if STORE is not None:
5486
+ named = slug or self.headers.get("X-Workspace")
5487
+ ws = STORE.workspace(named) if named else None
5488
+ ws = ws or STORE.default_ws
5489
+ self.ws = ws
5490
+ set_workspace(ws)
5491
+
5061
5492
  def do_GET(self):
5062
5493
  if not self._authorised():
5063
5494
  return
5495
+ self._scope()
5064
5496
  self._alias()
5065
5497
  path = self.path.split("?", 1)[0]
5066
5498
  # The lab is one page: a Connection and an Input make a scenario,
@@ -5096,7 +5528,7 @@ class Handler(BaseHTTPRequestHandler):
5096
5528
  if path == "/api/state":
5097
5529
  if STORE is None:
5098
5530
  return self._send(404, b"not found", "text/plain")
5099
- return self._json(200, {"docs": STORE.served()})
5531
+ return self._json(200, {"docs": STORE.served(self.ws)})
5100
5532
  # The run queue (#530): a run is a row the server owns, and History
5101
5533
  # reads the server. One run, or the list.
5102
5534
  if path == "/api/queue":
@@ -5115,7 +5547,7 @@ class Handler(BaseHTTPRequestHandler):
5115
5547
  # The bodies of the eval groups a run grades with, by `<id>@<n>`
5116
5548
  # (§17); null for a run that kept none, as /dataset answers.
5117
5549
  run_id = path.split("/")[3]
5118
- if QUEUE is None or QUEUE.get(run_id) is None:
5550
+ if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
5119
5551
  return self._send(404, b"not found", "text/plain")
5120
5552
  return self._json(200, QUEUE.groups(run_id))
5121
5553
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
@@ -5129,13 +5561,13 @@ class Handler(BaseHTTPRequestHandler):
5129
5561
  # time such a run was opened, which is the console people are
5130
5562
  # told to watch for real faults. A run that is not there is 404.
5131
5563
  run_id = path.split("/")[3]
5132
- if QUEUE is None or QUEUE.get(run_id) is None:
5564
+ if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
5133
5565
  return self._send(404, b"not found", "text/plain")
5134
5566
  return self._json(200, QUEUE.dataset(run_id))
5135
5567
  if path.startswith("/api/queue/"):
5136
5568
  if QUEUE is None:
5137
5569
  return self._send(404, b"not found", "text/plain")
5138
- run = QUEUE.get(path[len("/api/queue/"):])
5570
+ run = QUEUE.get(path[len("/api/queue/"):], self.ws)
5139
5571
  if run is None:
5140
5572
  return self._send(404, b"not found", "text/plain")
5141
5573
  # How far back in the queue this submission sits, for the form's
@@ -5279,6 +5711,7 @@ class Handler(BaseHTTPRequestHandler):
5279
5711
  def do_DELETE(self):
5280
5712
  if not self._authorised() or not self._from_this_page():
5281
5713
  return
5714
+ self._scope()
5282
5715
  self._alias()
5283
5716
  path = self.path.split("?", 1)[0]
5284
5717
  if path == "/api/connections/google":
@@ -5347,6 +5780,7 @@ class Handler(BaseHTTPRequestHandler):
5347
5780
  def do_PATCH(self):
5348
5781
  if not self._authorised() or not self._from_this_page():
5349
5782
  return
5783
+ self._scope()
5350
5784
  self._alias()
5351
5785
  parts = self.path.split("?", 1)[0].split("/")
5352
5786
  if len(parts) != 4 or parts[1] != "api":
@@ -5368,6 +5802,8 @@ class Handler(BaseHTTPRequestHandler):
5368
5802
  return self._json(200, source)
5369
5803
  if parts[2] == "queue" and QUEUE is not None:
5370
5804
  payload = self._payload() or {}
5805
+ if QUEUE.get(parts[3], self.ws) is None:
5806
+ return self._send(404, b"not found", "text/plain")
5371
5807
  run, err = QUEUE.set_comment(parts[3], payload.get("comment"))
5372
5808
  if err:
5373
5809
  code, message = err
@@ -5378,6 +5814,7 @@ class Handler(BaseHTTPRequestHandler):
5378
5814
  def do_PUT(self):
5379
5815
  if not self._authorised() or not self._from_this_page():
5380
5816
  return
5817
+ self._scope()
5381
5818
  self._alias()
5382
5819
  parts = self.path.split("?", 1)[0].split("/")
5383
5820
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
@@ -5441,6 +5878,7 @@ class Handler(BaseHTTPRequestHandler):
5441
5878
  return self._json(500, {"error": f"the relay failed: {type(e).__name__}"})
5442
5879
 
5443
5880
  def _post(self):
5881
+ self._scope()
5444
5882
  self._alias()
5445
5883
  path = self.path.split("?", 1)[0]
5446
5884
  if path == "/api/state":
@@ -5623,7 +6061,7 @@ class Handler(BaseHTTPRequestHandler):
5623
6061
  if len(json.dumps(d.get("body"))) > MAX_DOC:
5624
6062
  return self._json(413, {"error": f"{name} is too large"})
5625
6063
  versions, stale = STORE.write({n: {"version": d["version"], "body": d.get("body")}
5626
- for n, d in docs.items()})
6064
+ for n, d in docs.items()}, self.ws)
5627
6065
  if stale is not None:
5628
6066
  return self._json(409, {"stale": stale})
5629
6067
  # A pin names a group's version, so the group keeps that version from
@@ -5687,10 +6125,8 @@ class Handler(BaseHTTPRequestHandler):
5687
6125
  return self._json(400, {"error": f"the run cannot be queued: {body}"})
5688
6126
  ref["n"], ref["version"] = n, fingerprint(body)
5689
6127
  groups[group_key(ref)] = body
5690
- # The worker is handed one body until it reads one per group (#233).
5691
- if len(groups) > 1:
5692
- return self._json(400, {"error": "the run cannot be queued: its evals grade against "
5693
- "one version of one Library eval group at a time"})
6128
+ # The worker reads a body per group now (#233), so a run may link
6129
+ # several; each body is kept with the row under `<id>@<n>`.
5694
6130
  return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
5695
6131
 
5696
6132
  # ---- Datasets ----------------------------------------------------------
@@ -5865,6 +6301,10 @@ class Handler(BaseHTTPRequestHandler):
5865
6301
  parts = path.split("/")
5866
6302
  rid = parts[3]
5867
6303
  action = parts[4] if len(parts) > 4 else ""
6304
+ # A run is reachable only from its own workspace: an id from another is
6305
+ # a run this workspace does not have (docs/workspaces.md).
6306
+ if QUEUE.get(rid, self.ws) is None:
6307
+ return self._send(404, b"not found", "text/plain")
5868
6308
  if action == "cancel":
5869
6309
  run, err = QUEUE.cancel(rid)
5870
6310
  elif action == "resume":
@@ -5989,9 +6429,10 @@ class Handler(BaseHTTPRequestHandler):
5989
6429
  if len(parts) == 3:
5990
6430
  payload = self._payload() or {}
5991
6431
  kind = payload.get("type", DEFAULT_SOURCE_TYPE)
5992
- # A type made with a sign-in (a flow, with Microsoft's) cannot be
5993
- # made in a lab without that sign-in.
5994
- sign_in = (SOURCE_TYPES.get(kind) or {}).get("signIn")
6432
+ # A platform made with a sign-in (Power Automate, with Microsoft's)
6433
+ # cannot be made in a lab without that sign-in; an unknown kind or
6434
+ # platform is refused by create() below with its own sentence.
6435
+ sign_in = (source_entry(kind, payload.get("config")) or {}).get("signIn")
5995
6436
  if sign_in and not SIGN_INS.get(sign_in, lambda: False)():
5996
6437
  return self._json(400, {"error": "this lab has no Microsoft app to sign in with (Setup › Connections)"
5997
6438
  if sign_in == "microsoft" else f"this lab has no {sign_in} sign-in"})