evals-lab 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -513,6 +513,13 @@ def take_record(name, data):
513
513
 
514
514
  WORKFLOW_PLATFORMS["power-automate"]["take"] = take_record
515
515
 
516
+ # The lab's own workflow platforms and wizards, kept apart so the mirror can be
517
+ # rebuilt from them as plugins come and go (Plugins.apply), and so a plugin is
518
+ # refused the id of one the lab ships (#303). Wizards are a page concept; the
519
+ # server holds only their ids, to refuse two plugins claiming one.
520
+ BUILTIN_WORKFLOW_PLATFORMS = {k: dict(v) for k, v in WORKFLOW_PLATFORMS.items()}
521
+ BUILTIN_WIZARD_IDS = {"test-workflow"}
522
+
516
523
  # The sign-ins a Source type may be made with, and whether this lab has each:
517
524
  # a type naming one this lab lacks cannot be made.
518
525
  SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
@@ -908,6 +915,360 @@ class Store:
908
915
  self._put(db, n, None if n in DOC_GLOBAL else ws, version, body, at)
909
916
  return {n: have(n)["version"] + 1 for n in docs}, None
910
917
 
918
+ # ---- shared-by-choice connections (docs/workspaces.md) ------------------
919
+ # A Target profile or an Account (kind "profile"/"account", id the
920
+ # profile's or the account's) is usable in a workspace by its share set in
921
+ # `connection_shares`: a row for that workspace, or the SHARE_ALL sentinel.
922
+ # No row at all is the migration and new-connection default -- shared with
923
+ # every workspace, so nothing stops running the moment workspaces exist
924
+ # (open question 2, resolved All); narrowing a connection is adding the
925
+ # rows that say where it may be used. SHARE_NEW is never read here: a
926
+ # workspace created after a SHARE_NEW share was set has it materialised into
927
+ # a concrete row at creation (Workspaces.create), so "New workspaces" is
928
+ # exactly the future ones and not the ones that already existed. The server
929
+ # is the only writer and the only enforcer -- the key a share gates is
930
+ # never served (#258), so neither the page nor the worker could enforce it.
931
+
932
+ def shared(self, kind: str, cid: str, ws) -> bool:
933
+ with self.lock, closing(sqlite3.connect(self.path)) as db:
934
+ rows = {r[0] for r in db.execute(
935
+ "SELECT workspace FROM connection_shares WHERE kind = ? AND id = ?", (kind, cid))}
936
+ return not rows or SHARE_ALL in rows or ws in rows
937
+
938
+ def shares(self) -> dict:
939
+ """Every connection's share set, for the Connections menus: keyed by
940
+ "<kind>:<id>", the workspace values as stored (workspace ids and the
941
+ sentinels). A connection with no row is absent, which the page reads as
942
+ shared with All. No key is anywhere in this (#258)."""
943
+ out: dict = {}
944
+ with self.lock, closing(sqlite3.connect(self.path)) as db:
945
+ for kind, cid, ws in db.execute("SELECT kind, id, workspace FROM connection_shares"):
946
+ out.setdefault(f"{kind}:{cid}", []).append(ws)
947
+ return out
948
+
949
+ def set_shares(self, kind: str, cid: str, workspaces) -> None:
950
+ """Replace a connection's share set. SHARE_ALL means every workspace,
951
+ so it is kept alone; only the sentinels and live workspace ids are let
952
+ in, so a stale id cannot linger in the table."""
953
+ with self.lock, closing(sqlite3.connect(self.path)) as db, db:
954
+ valid = {r[0] for r in db.execute("SELECT id FROM workspaces WHERE trash IS NULL")}
955
+ chosen = [w for w in dict.fromkeys(workspaces or [])
956
+ if w in (SHARE_ALL, SHARE_NEW) or w in valid]
957
+ if SHARE_ALL in chosen:
958
+ chosen = [SHARE_ALL]
959
+ db.execute("DELETE FROM connection_shares WHERE kind = ? AND id = ?", (kind, cid))
960
+ db.executemany("INSERT INTO connection_shares (kind, id, workspace) VALUES (?, ?, ?)",
961
+ [(kind, cid, w) for w in chosen])
962
+
963
+ def share_with(self, kind: str, cid: str, ws, on: bool) -> None:
964
+ """Add or remove one workspace from a connection's share set -- the Run
965
+ bar's Share link and its Undo (docs/workspaces.md). The link is offered
966
+ only where the connection is actually narrowed, so adding a row is what
967
+ grants the blocked workspace its use and Undo takes it back."""
968
+ with self.lock, closing(sqlite3.connect(self.path)) as db, db:
969
+ if on:
970
+ db.execute("INSERT OR IGNORE INTO connection_shares (kind, id, workspace) "
971
+ "VALUES (?, ?, ?)", (kind, cid, ws))
972
+ else:
973
+ db.execute("DELETE FROM connection_shares WHERE kind = ? AND id = ? AND workspace = ?",
974
+ (kind, cid, ws))
975
+
976
+ def workspace_name(self, ws) -> str:
977
+ """A workspace's name, for the refusal sentence the Run bar shows."""
978
+ with self.lock, closing(sqlite3.connect(self.path)) as db:
979
+ row = db.execute("SELECT name FROM workspaces WHERE id = ?", (ws,)).fetchone()
980
+ return row[0] if row else "this workspace"
981
+
982
+ def profile_name(self, pid: str) -> str:
983
+ """A Target profile's display name, by id, from the global profiles
984
+ document -- for the refusal sentence. The id itself when none is held."""
985
+ with self.lock, closing(sqlite3.connect(self.path)) as db:
986
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.profiles' "
987
+ "AND workspace IS NULL").fetchone()
988
+ try:
989
+ for p in (json.loads(row[0]) or {}).get("list", []) if row and row[0] else []:
990
+ if isinstance(p, dict) and str(p.get("id") or "") == pid:
991
+ return p.get("name") or pid
992
+ except (ValueError, TypeError):
993
+ pass
994
+ return pid
995
+
996
+
997
+ # A workspace's name is one line, capped like a dataset's or a prompt's.
998
+ WORKSPACE_NAME_MAX = 80
999
+
1000
+
1001
+ class Workspaces:
1002
+ """
1003
+ The workspace registry (docs/workspaces.md): the lab-wide list the
1004
+ Setup › Workspaces table manages. Phase 2 is the registry and its UI; the
1005
+ active workspace stays the flagged default (switching is phase 3), so a
1006
+ workspace created here is reached by its address only once that lands.
1007
+
1008
+ Create, rename, archive, unarchive, Regenerate (a new slug from the name),
1009
+ and a server trash like the others: delete puts a workspace into the trash
1010
+ with an undo window, restore brings it back, and a lazy sweep purges it
1011
+ after TRASH_SECONDS -- taking its scoped data with it (its pipelines, runs,
1012
+ datasets, prompts, Sources, packs and documents), since that is what a
1013
+ permanent delete means. The child tables co-scope through their parent id.
1014
+
1015
+ The slug is kept through a rename; only Regenerate changes it. The last
1016
+ active workspace cannot be archived, so a flag-holder always exists;
1017
+ archiving or deleting the flagged default moves the flag to the most
1018
+ recently used other active workspace (docs/workspaces.md, decision 7).
1019
+ """
1020
+
1021
+ def __init__(self, store: Store):
1022
+ self.store = store
1023
+ self.dir = store.path.parent
1024
+
1025
+ def _connect(self):
1026
+ return closing(sqlite3.connect(self.store.path))
1027
+
1028
+ @staticmethod
1029
+ def _slugify(raw) -> str:
1030
+ s = re.sub(r"[^a-z0-9]+", "-", str(raw or "").lower()).strip("-")[:SLUG_MAX].strip("-")
1031
+ return s or "workspace"
1032
+
1033
+ @staticmethod
1034
+ def _unique(base: str, taken: set) -> str:
1035
+ if base not in taken:
1036
+ return base
1037
+ n = 2
1038
+ while True:
1039
+ suffix = f"-{n}"
1040
+ cand = (base[:SLUG_MAX - len(suffix)].strip("-") or "workspace") + suffix
1041
+ if cand not in taken:
1042
+ return cand
1043
+ n += 1
1044
+
1045
+ @staticmethod
1046
+ def _row(r, pipes, runs) -> dict:
1047
+ return {"id": r[0], "slug": r[1], "name": r[2], "archived": bool(r[3]),
1048
+ "isDefault": bool(r[4]), "lastUsed": r[5], "createdAt": r[6],
1049
+ "pipelines": pipes.get(r[0], 0), "runs": runs.get(r[0], 0)}
1050
+
1051
+ def list(self) -> list:
1052
+ """Every workspace not in the trash, with how many pipelines and runs
1053
+ it holds -- the two counts the Manage table shows (docs/other-tabs.md).
1054
+ Active first, then archived, each by name."""
1055
+ with self.store.lock, self._connect() as db:
1056
+ rows = db.execute(
1057
+ "SELECT id, slug, name, archived, is_default, last_used, created_at "
1058
+ "FROM workspaces WHERE trash IS NULL "
1059
+ "ORDER BY archived, name COLLATE NOCASE").fetchall()
1060
+ runs = {w: n for w, n in db.execute(
1061
+ "SELECT workspace, COUNT(*) FROM queue GROUP BY workspace")}
1062
+ pipes = {}
1063
+ for w, body in db.execute(
1064
+ "SELECT workspace, body FROM docs WHERE name = 'promptlab.workflows'"):
1065
+ try:
1066
+ pipes[w] = len((json.loads(body) or {}).get("list", [])) if body else 0
1067
+ except (ValueError, TypeError):
1068
+ pipes[w] = 0
1069
+ return [self._row(r, pipes, runs) for r in rows]
1070
+
1071
+ def get(self, wid: str) -> dict:
1072
+ """One workspace by id, with its counts, or None."""
1073
+ return next((w for w in self.list() if w["id"] == wid), None)
1074
+
1075
+ def create(self, name, slug=None, shares=None) -> tuple:
1076
+ """A new, active workspace; its slug is minted from the name (or a slug
1077
+ asked for), unique across every workspace including the trash, since
1078
+ the column is unique. Returns (id, None) or (None, error). The demo
1079
+ pack is the Handler's to install, since that reaches Packs.
1080
+
1081
+ Shared-by-choice connections (docs/workspaces.md): `shares` is the
1082
+ New-workspace dialog's Connections picker -- a list of {kind, id} the
1083
+ workspace may use, each written as a concrete row. With none given
1084
+ (the plain Add workspace, the CLI), every connection shared with New
1085
+ workspaces (SHARE_NEW) is materialised into a concrete row instead, so
1086
+ "New workspaces" resolves for this one though it was created after the
1087
+ share was set."""
1088
+ name = (name or "").strip()
1089
+ if not name:
1090
+ return None, (400, "a workspace needs a name")
1091
+ if len(name) > WORKSPACE_NAME_MAX:
1092
+ return None, (400, f"a name is at most {WORKSPACE_NAME_MAX} characters")
1093
+ wid = secrets.token_hex(6)
1094
+ now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
1095
+ with self.store.lock, self._connect() as db, db:
1096
+ taken = {r[0] for r in db.execute("SELECT slug FROM workspaces")}
1097
+ s = self._unique(self._slugify(slug or name), taken)
1098
+ db.execute("INSERT INTO workspaces (id, slug, name, archived, is_default, last_used, created_at) "
1099
+ "VALUES (?, ?, ?, 0, 0, ?, ?)", (wid, s, name, now, now))
1100
+ grant = ([(c.get("kind"), c.get("id")) for c in shares if isinstance(c, dict)]
1101
+ if isinstance(shares, list)
1102
+ else list(db.execute("SELECT kind, id FROM connection_shares WHERE workspace = ?",
1103
+ (SHARE_NEW,))))
1104
+ db.executemany("INSERT OR IGNORE INTO connection_shares (kind, id, workspace) VALUES (?, ?, ?)",
1105
+ [(k, i, wid) for k, i in grant if k and i])
1106
+ return wid, None
1107
+
1108
+ def rename(self, wid, name) -> tuple:
1109
+ """A new name; the slug is kept, so an old bookmark still resolves
1110
+ (docs/workspaces.md). Returns (id, None) or (None, error)."""
1111
+ name = (name or "").strip()
1112
+ if not name:
1113
+ return None, (400, "a workspace needs a name")
1114
+ if len(name) > WORKSPACE_NAME_MAX:
1115
+ return None, (400, f"a name is at most {WORKSPACE_NAME_MAX} characters")
1116
+ with self.store.lock, self._connect() as db, db:
1117
+ if db.execute("SELECT 1 FROM workspaces WHERE id = ? AND trash IS NULL", (wid,)).fetchone() is None:
1118
+ return None, (404, "no such workspace")
1119
+ db.execute("UPDATE workspaces SET name = ? WHERE id = ?", (name, wid))
1120
+ return wid, None
1121
+
1122
+ def regenerate(self, wid) -> tuple:
1123
+ """A new slug minted from the current name -- the one write that
1124
+ changes an address (docs/workspaces.md). It does not redirect: that is
1125
+ the warning the dialog carries. Returns (id, None) or (None, error)."""
1126
+ with self.store.lock, self._connect() as db, db:
1127
+ row = db.execute("SELECT name FROM workspaces WHERE id = ? AND trash IS NULL", (wid,)).fetchone()
1128
+ if row is None:
1129
+ return None, (404, "no such workspace")
1130
+ taken = {r[0] for r in db.execute("SELECT slug FROM workspaces WHERE id != ?", (wid,))}
1131
+ db.execute("UPDATE workspaces SET slug = ? WHERE id = ?",
1132
+ (self._unique(self._slugify(row[0]), taken), wid))
1133
+ return wid, None
1134
+
1135
+ def archive(self, wid) -> tuple:
1136
+ """Into the archive, at once; Undo unarchives it. The last active
1137
+ workspace cannot be archived; archiving the flagged default moves the
1138
+ flag to the most recently used other active workspace, keeping the
1139
+ server's cached default in step. Returns (id, None) or (None, error)."""
1140
+ with self.store.lock, self._connect() as db, db:
1141
+ row = db.execute("SELECT archived, is_default FROM workspaces WHERE id = ? AND trash IS NULL",
1142
+ (wid,)).fetchone()
1143
+ if row is None:
1144
+ return None, (404, "no such workspace")
1145
+ if row[0]:
1146
+ return wid, None
1147
+ others = db.execute(
1148
+ "SELECT id FROM workspaces WHERE archived = 0 AND trash IS NULL AND id != ? "
1149
+ "ORDER BY last_used DESC, created_at DESC", (wid,)).fetchall()
1150
+ if not others:
1151
+ return None, (409, "the last active workspace cannot be archived")
1152
+ db.execute("UPDATE workspaces SET archived = 1 WHERE id = ?", (wid,))
1153
+ if row[1]:
1154
+ db.execute("UPDATE workspaces SET is_default = 0 WHERE id = ?", (wid,))
1155
+ db.execute("UPDATE workspaces SET is_default = 1 WHERE id = ?", (others[0][0],))
1156
+ self.store.default_ws = others[0][0]
1157
+ return wid, None
1158
+
1159
+ def set_default(self, wid) -> tuple:
1160
+ """Make [wid] the flagged default -- the workspace the CLI and an
1161
+ address with no /w/ prefix resolve to (docs/workspaces.md). Only an
1162
+ active workspace can be the default. Returns (id, None) or
1163
+ (None, error)."""
1164
+ with self.store.lock, self._connect() as db, db:
1165
+ row = db.execute("SELECT archived FROM workspaces WHERE id = ? AND trash IS NULL",
1166
+ (wid,)).fetchone()
1167
+ if row is None:
1168
+ return None, (404, "no such workspace")
1169
+ if row[0]:
1170
+ return None, (409, "an archived workspace cannot be the default")
1171
+ db.execute("UPDATE workspaces SET is_default = 0 WHERE is_default = 1")
1172
+ db.execute("UPDATE workspaces SET is_default = 1 WHERE id = ?", (wid,))
1173
+ self.store.default_ws = wid
1174
+ return wid, None
1175
+
1176
+ def unarchive(self, wid) -> tuple:
1177
+ with self.store.lock, self._connect() as db, db:
1178
+ if db.execute("SELECT 1 FROM workspaces WHERE id = ? AND trash IS NULL", (wid,)).fetchone() is None:
1179
+ return None, (404, "no such workspace")
1180
+ db.execute("UPDATE workspaces SET archived = 0 WHERE id = ?", (wid,))
1181
+ return wid, None
1182
+
1183
+ def remove(self, wid) -> tuple:
1184
+ """Into the trash, at once; Undo restores it. An active workspace is
1185
+ archived first (the Manage menu offers Delete only on an archived one),
1186
+ so a trashed workspace is never the flagged default. Returns
1187
+ (token, None) or (None, error)."""
1188
+ token = secrets.token_hex(6)
1189
+ with self.store.lock, self._connect() as db, db:
1190
+ row = db.execute("SELECT archived FROM workspaces WHERE id = ? AND trash IS NULL",
1191
+ (wid,)).fetchone()
1192
+ if row is None:
1193
+ return None, (404, "no such workspace")
1194
+ if not row[0]:
1195
+ return None, (409, "archive a workspace before deleting it")
1196
+ db.execute("UPDATE workspaces SET trash = ?, trashed_at = ? WHERE id = ?",
1197
+ (token, time.time(), wid))
1198
+ return token, None
1199
+
1200
+ def restore(self, token) -> tuple:
1201
+ """A trashed workspace back, archived as it was. Returns (id, None) or
1202
+ (None, error)."""
1203
+ with self.store.lock, self._connect() as db, db:
1204
+ row = db.execute("SELECT id FROM workspaces WHERE trash = ?", (str(token),)).fetchone()
1205
+ if row is None:
1206
+ return None, (404, "no such trash entry")
1207
+ db.execute("UPDATE workspaces SET trash = NULL, trashed_at = NULL WHERE id = ?", (row[0],))
1208
+ return row[0], None
1209
+
1210
+ def _purge(self, db, ids) -> tuple:
1211
+ """Delete each workspace in `ids` and all its scoped data -- the rows
1212
+ across every scopable table and its documents -- returning the Source
1213
+ and run ids whose on-disk directories the caller then removes. The
1214
+ child tables co-scope through their parent id (docs/workspaces.md)."""
1215
+ sids, rids = [], []
1216
+ for ws in ids:
1217
+ sids += [r[0] for r in db.execute(
1218
+ "SELECT id FROM sources WHERE workspace = ? AND system = 0", (ws,))]
1219
+ rids += [r[0] for r in db.execute("SELECT id FROM queue WHERE workspace = ?", (ws,))]
1220
+ db.execute("DELETE FROM source_files WHERE source IN "
1221
+ "(SELECT id FROM sources WHERE workspace = ? AND system = 0)", (ws,))
1222
+ db.execute("DELETE FROM source_definitions WHERE source IN "
1223
+ "(SELECT id FROM sources WHERE workspace = ? AND system = 0)", (ws,))
1224
+ db.execute("DELETE FROM sources WHERE workspace = ? AND system = 0", (ws,))
1225
+ db.execute("DELETE FROM eval_group_versions WHERE group_id IN "
1226
+ "(SELECT id FROM datasets WHERE workspace = ?)", (ws,))
1227
+ db.execute("DELETE FROM dataset_rules_archive WHERE dataset_id IN "
1228
+ "(SELECT id FROM datasets WHERE workspace = ?)", (ws,))
1229
+ db.execute("DELETE FROM dataset_body_archive WHERE dataset_id IN "
1230
+ "(SELECT id FROM datasets WHERE workspace = ?)", (ws,))
1231
+ db.execute("DELETE FROM datasets WHERE workspace = ?", (ws,))
1232
+ db.execute("DELETE FROM filter_set_versions WHERE set_id IN "
1233
+ "(SELECT id FROM filter_sets WHERE workspace = ?)", (ws,))
1234
+ db.execute("DELETE FROM filter_sets WHERE workspace = ?", (ws,))
1235
+ db.execute("DELETE FROM prompt_uses WHERE prompt_id IN "
1236
+ "(SELECT id FROM prompts WHERE workspace = ?)", (ws,))
1237
+ db.execute("DELETE FROM prompt_versions WHERE prompt_id IN "
1238
+ "(SELECT id FROM prompts WHERE workspace = ?)", (ws,))
1239
+ db.execute("DELETE FROM prompts WHERE workspace = ?", (ws,))
1240
+ db.execute("DELETE FROM pack_items WHERE workspace = ?", (ws,))
1241
+ db.execute("DELETE FROM packs WHERE workspace = ?", (ws,))
1242
+ db.execute("DELETE FROM queue WHERE workspace = ?", (ws,))
1243
+ db.execute("DELETE FROM runs WHERE workspace = ?", (ws,))
1244
+ db.execute("DELETE FROM docs WHERE workspace = ?", (ws,))
1245
+ db.execute("DELETE FROM connection_shares WHERE workspace = ?", (ws,))
1246
+ db.execute("DELETE FROM workspaces WHERE id = ?", (ws,))
1247
+ return sids, rids
1248
+
1249
+ def _rmdirs(self, sids, rids):
1250
+ for sid in sids:
1251
+ shutil.rmtree(self.dir / "sources" / sid, ignore_errors=True)
1252
+ for rid in rids:
1253
+ shutil.rmtree(self.dir / "runs" / rid, ignore_errors=True)
1254
+
1255
+ def lazy_trash(self):
1256
+ """A /api/workspaces request purges what has been trashed longer than
1257
+ TRASH_SECONDS, lazily as the other trashes do."""
1258
+ with self.store.lock, self._connect() as db, db:
1259
+ gone = [r[0] for r in db.execute(
1260
+ "SELECT id FROM workspaces WHERE trash IS NOT NULL AND trashed_at < ?",
1261
+ (time.time() - TRASH_SECONDS,))]
1262
+ sids, rids = self._purge(db, gone)
1263
+ self._rmdirs(sids, rids)
1264
+
1265
+ def empty_trash(self):
1266
+ """The startup sweep: a restart has nothing to undo."""
1267
+ with self.store.lock, self._connect() as db, db:
1268
+ gone = [r[0] for r in db.execute("SELECT id FROM workspaces WHERE trash IS NOT NULL")]
1269
+ sids, rids = self._purge(db, gone)
1270
+ self._rmdirs(sids, rids)
1271
+
911
1272
 
912
1273
  # A filename, the target filesystem's view rather than the caller's: a
913
1274
  # basename is all that survives, restricted to characters that filesystem
@@ -1908,6 +2269,22 @@ def upgrade_body(body):
1908
2269
  return recorded_ids_v7(group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}))
1909
2270
 
1910
2271
 
2272
+ # The filter-set body's version: evals-core.ts's FILTER_SET_VERSION. Version 1
2273
+ # is the first -- the { version, filters } body #334 introduces. A filter set
2274
+ # is a Library row the lab stores and versions like an eval group (#336,
2275
+ # FilterSets below); this and its twin are the body it keeps, held to
2276
+ # evals-core.ts by proxy-check.py.
2277
+ FILTER_SET_BODY_VERSION = 1
2278
+
2279
+
2280
+ def upgrade_filter_set_body(body):
2281
+ """An earlier filter-set body as today's (version 1): evals-core.ts's
2282
+ upgradeFilterSetBody, in Python. There is no earlier version yet, so a
2283
+ version-1 body comes back as it was, and so does anything that is not a
2284
+ body. Pure."""
2285
+ return body
2286
+
2287
+
1911
2288
  def body_prompt(body):
1912
2289
  """The prompt an earlier body held, or "": what the library is given."""
1913
2290
  prompt = body.get("prompt") if isinstance(body, dict) else None
@@ -2007,6 +2384,90 @@ def unique_dataset_name(name: str, taken: set) -> str:
2007
2384
  return f"{name} ({n})"
2008
2385
 
2009
2386
 
2387
+ # ---- Filter sets (the store's, FilterSets below) ---------------------------
2388
+ #
2389
+ # A filter set is a Library row like an eval group: a versioned { version,
2390
+ # filters } body (evals-core.ts FilterSetBody), kept so a pipeline can link it
2391
+ # and a run grade against the body it was submitted with (#336). The fields
2392
+ # are the body's; the per-filter shape is the core's to judge, as a case's
2393
+ # metrics are (dataset_problem). The suffixing is the generic one.
2394
+ FILTER_SET_FIELDS = ("version", "filters")
2395
+ FILTER_SET_NAME_MAX = 80
2396
+ unique_filter_set_name = unique_dataset_name
2397
+
2398
+
2399
+ def filter_set_problem(body) -> str:
2400
+ """Why [body] is not a filter set's body, in one sentence, or "":
2401
+ evals-core.ts's filterSetBodyProblems, to the depth the server checks."""
2402
+ if not isinstance(body, dict):
2403
+ return "a filter set's body is a JSON object"
2404
+ for k in body:
2405
+ if k not in FILTER_SET_FIELDS:
2406
+ return f"a filter set's body has \"{k}\", which is not a filter-set field"
2407
+ for k in FILTER_SET_FIELDS:
2408
+ if k not in body:
2409
+ return f"a filter set's body has no \"{k}\""
2410
+ if body["version"] != FILTER_SET_BODY_VERSION:
2411
+ return f"a filter set's body is version {FILTER_SET_BODY_VERSION}"
2412
+ if not isinstance(body["filters"], list) or not all(isinstance(f, dict) for f in body["filters"]):
2413
+ return "filters has to be a list of filters"
2414
+ return ""
2415
+
2416
+
2417
+ def filter_set_name(raw):
2418
+ """A name, trimmed, or (None, why)."""
2419
+ if not isinstance(raw, str) or not raw.strip():
2420
+ return None, "a filter set needs a name"
2421
+ name = raw.strip()
2422
+ if len(name) > FILTER_SET_NAME_MAX:
2423
+ return None, "that name is too long"
2424
+ return name, None
2425
+
2426
+
2427
+ def filter_set_links(doc) -> list:
2428
+ """Every filter-set link a run or pipeline document carries: a job's
2429
+ Responses steps list them under `filterSets` (evals-core.ts FilterSetLink).
2430
+ A link names a Library set (`set`, followed latest or pinned at `pin`) or
2431
+ holds a private one inline (`own`); only a Library link reads the store.
2432
+ Runs wires the step and migrates the inline rules in #339; this is the one
2433
+ reader of where the links sit, for the store (#336) to resolve and keep."""
2434
+ out = []
2435
+ jobs = doc.get("jobs") if isinstance(doc, dict) else None
2436
+ for job in jobs if isinstance(jobs, list) else []:
2437
+ steps = job.get("steps") if isinstance(job, dict) else None
2438
+ for st in steps if isinstance(steps, list) else []:
2439
+ links = st.get("filterSets") if isinstance(st, dict) else None
2440
+ for link in links if isinstance(links, list) else []:
2441
+ if isinstance(link, dict):
2442
+ out.append(link)
2443
+ return out
2444
+
2445
+
2446
+ def filter_set_ref(link):
2447
+ """The Library filter set a link reads, as the dict itself (so a caller may
2448
+ stamp it in place), or None for a private (`own`) link."""
2449
+ if not isinstance(link, dict):
2450
+ return None
2451
+ s = link.get("set")
2452
+ return s if isinstance(s, dict) and isinstance(s.get("id"), str) else None
2453
+
2454
+
2455
+ def filter_set_pins_in(workflows) -> set:
2456
+ """(filter-set id, version) for every filter-set link a stored pipeline
2457
+ pins: the promptlab.workflows body, each pipeline as its `work`, as
2458
+ pins_in reads eval-group pins. A Library link with an integer `pin`
2459
+ freezes that version from the moment a pipeline pins it (§17)."""
2460
+ out = set()
2461
+ listed = workflows.get("list") if isinstance(workflows, dict) else None
2462
+ for w in listed if isinstance(listed, list) else []:
2463
+ work = w.get("work") if isinstance(w, dict) else None
2464
+ for link in filter_set_links(work):
2465
+ ref = filter_set_ref(link)
2466
+ if ref is not None and type(link.get("pin")) is int:
2467
+ out.add((ref["id"], link["pin"]))
2468
+ return out
2469
+
2470
+
2010
2471
  # ---- The Prompt library -----------------------------------------------------
2011
2472
  #
2012
2473
  # Every prompt the lab has, run or not, each keeping every version of its text,
@@ -2449,8 +2910,11 @@ class Datasets:
2449
2910
  newest one's number, which a pin may name, and 1 before any is kept,
2450
2911
  the row's body being version 1 in waiting."""
2451
2912
  parsed = upgrade_body(json.loads(r["body"]))
2913
+ # The grader rides in the summary: a run carries the profile each
2914
+ # linked group asks, and the page resolves a run from the list.
2452
2915
  out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
2453
- "version": r["version"], "versions": versions, "updated": r["updated_at"]}
2916
+ "version": r["version"], "versions": versions, "updated": r["updated_at"],
2917
+ "grader": parsed.get("grader")}
2454
2918
  if body:
2455
2919
  out["body"] = parsed
2456
2920
  return out
@@ -2803,6 +3267,319 @@ class Datasets:
2803
3267
  return [{k: v for k, v in self.get(did).items() if k != "body"} for did in created], None
2804
3268
 
2805
3269
 
3270
+ class FilterSets:
3271
+ """Filter sets as a Library row, versioned like eval groups (#336,
3272
+ docs/pipeline-model.md §18). The shape mirrors Datasets exactly -- a row
3273
+ per set, every version kept in `filter_set_versions` for a link to pin and
3274
+ a run to name -- without a dataset's prompt, scoring or import/export: a
3275
+ filter set's body is just { version, filters }. A run keeps the body of
3276
+ each set it links (Queue.submit's `filter_sets`), so it grades against what
3277
+ it was submitted with whatever the set holds later."""
3278
+
3279
+ def __init__(self, store: Store):
3280
+ self.store = store
3281
+ with store.lock, closing(sqlite3.connect(store.path)) as db, db:
3282
+ db.execute("CREATE TABLE IF NOT EXISTS filter_sets ("
3283
+ "id TEXT PRIMARY KEY, name TEXT NOT NULL, "
3284
+ "version INTEGER NOT NULL, body TEXT NOT NULL, "
3285
+ "created_at TEXT NOT NULL, updated_at TEXT NOT NULL, "
3286
+ "trash TEXT, trashed_at REAL, workspace TEXT)")
3287
+ # Every version of each set, numbered from 1, as eval_group_versions
3288
+ # keeps a group's (§17): `ran` freezes it for good once a run grades
3289
+ # with it, a pin freezes it while a pipeline holds the pin, and
3290
+ # version 1 is minted from the row's body at its first save, pin or
3291
+ # run. A set from before workspaces backfills to the default.
3292
+ db.execute("CREATE TABLE IF NOT EXISTS filter_set_versions ("
3293
+ "set_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
3294
+ "created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
3295
+ "ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (set_id, n))")
3296
+ db.execute("UPDATE filter_sets SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
3297
+
3298
+ @staticmethod
3299
+ def _now():
3300
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
3301
+
3302
+ @staticmethod
3303
+ def _doc(r, body=True, versions=1):
3304
+ """A row as the API answers it: its summary, and its body with it, read
3305
+ as today's version. `version` is the save counter a write is arbitrated
3306
+ by; `versions` is its newest kept version's number, which a pin may
3307
+ name, 1 before any is kept."""
3308
+ parsed = upgrade_filter_set_body(json.loads(r["body"]))
3309
+ out = {"id": r["id"], "name": r["name"],
3310
+ "filters": len(parsed.get("filters") or []),
3311
+ "version": r["version"], "versions": versions, "updated": r["updated_at"]}
3312
+ if body:
3313
+ out["body"] = parsed
3314
+ return out
3315
+
3316
+ @staticmethod
3317
+ def _counts(db) -> dict:
3318
+ return dict(db.execute("SELECT set_id, MAX(n) FROM filter_set_versions GROUP BY set_id"))
3319
+
3320
+ def _row_doc(self, db, r, body=True):
3321
+ return self._doc(r, body, self._counts(db).get(r["id"], 1))
3322
+
3323
+ def _connect(self):
3324
+ db = sqlite3.connect(self.store.path)
3325
+ db.row_factory = sqlite3.Row
3326
+ return closing(db)
3327
+
3328
+ def _live(self, db, fid):
3329
+ return db.execute("SELECT * FROM filter_sets WHERE id = ? AND trash IS NULL AND workspace = ?",
3330
+ (fid, workspace_of(self.store))).fetchone()
3331
+
3332
+ def _names(self, db, but=None):
3333
+ return {r[0] for r in db.execute(
3334
+ "SELECT name FROM filter_sets WHERE trash IS NULL AND id IS NOT ? AND workspace = ?",
3335
+ (but, workspace_of(self.store)))}
3336
+
3337
+ def list(self) -> list:
3338
+ with self.store.lock, self._connect() as db:
3339
+ counts = self._counts(db)
3340
+ return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
3341
+ "SELECT * FROM filter_sets WHERE trash IS NULL AND workspace = ? ORDER BY name COLLATE NOCASE, id",
3342
+ (workspace_of(self.store),))]
3343
+
3344
+ def get(self, fid):
3345
+ with self.store.lock, self._connect() as db:
3346
+ r = self._live(db, fid)
3347
+ return self._row_doc(db, r) if r else None
3348
+
3349
+ # ---- a set's versions (docs/pipeline-model.md §17, §18) --------------
3350
+
3351
+ @staticmethod
3352
+ def _head(db, fid):
3353
+ return db.execute("SELECT * FROM filter_set_versions WHERE set_id = ? "
3354
+ "ORDER BY n DESC LIMIT 1", (fid,)).fetchone()
3355
+
3356
+ def _mint(self, db, r):
3357
+ """The newest version of row [r], adding version 1 from its body first
3358
+ if it has none, edited when the row last was -- as a group's _mint."""
3359
+ head = self._head(db, r["id"])
3360
+ if head is not None:
3361
+ return head
3362
+ try:
3363
+ edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
3364
+ except ValueError:
3365
+ edited = 0
3366
+ db.execute("INSERT INTO filter_set_versions (set_id, n, body, created_at, edited_at) "
3367
+ "VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
3368
+ return self._head(db, r["id"])
3369
+
3370
+ def _pins(self, db) -> set:
3371
+ """Every (set, version) a stored pipeline pins, read in the caller's
3372
+ transaction from this workspace's workflows document."""
3373
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows' AND workspace = ?",
3374
+ (workspace_of(self.store),)).fetchone()
3375
+ return filter_set_pins_in(json.loads(row[0])) if row and row[0] else set()
3376
+
3377
+ def _cut(self, db, fid, text):
3378
+ n = self._head(db, fid)["n"] + 1
3379
+ db.execute("INSERT INTO filter_set_versions (set_id, n, body, created_at, edited_at) "
3380
+ "VALUES (?, ?, ?, ?, ?)", (fid, n, text, self._now(), time.time()))
3381
+ return n
3382
+
3383
+ def _keep(self, db, r, text):
3384
+ """[text] as row [r]'s newest version: edited in place while no run has
3385
+ graded with it, no link pins it and it was edited in the last
3386
+ GROUP_IDLE_SECONDS, and a new version otherwise -- a group's _keep."""
3387
+ head = self._mint(db, r)
3388
+ if text == head["body"]:
3389
+ return
3390
+ fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
3391
+ if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
3392
+ db.execute("UPDATE filter_set_versions SET body = ?, edited_at = ? WHERE set_id = ? AND n = ?",
3393
+ (text, time.time(), r["id"], head["n"]))
3394
+ else:
3395
+ self._cut(db, r["id"], text)
3396
+
3397
+ def versions(self, fid):
3398
+ """A set's versions, newest first, without their bodies -- version 1
3399
+ alone, read from the row, before any is kept -- or None."""
3400
+ with self.store.lock, self._connect() as db:
3401
+ r = self._live(db, fid)
3402
+ if r is None:
3403
+ return None
3404
+ pins = self._pins(db)
3405
+ rows = db.execute("SELECT * FROM filter_set_versions WHERE set_id = ? ORDER BY n DESC",
3406
+ (fid,)).fetchall()
3407
+ if not rows:
3408
+ return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (fid, 1) in pins,
3409
+ "fingerprint": fingerprint(json.loads(r["body"]))}]
3410
+ return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
3411
+ "pinned": (fid, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
3412
+ for v in rows]
3413
+
3414
+ def version_body(self, fid, n):
3415
+ """Version [n]'s body, read as today's, or None."""
3416
+ with self.store.lock, self._connect() as db:
3417
+ r = self._live(db, fid)
3418
+ if r is None:
3419
+ return None
3420
+ v = db.execute("SELECT body FROM filter_set_versions WHERE set_id = ? AND n = ?",
3421
+ (fid, n)).fetchone()
3422
+ if v is None and n == 1 and self._head(db, fid) is None:
3423
+ v = (r["body"],)
3424
+ return upgrade_filter_set_body(json.loads(v[0])) if v else None
3425
+
3426
+ def restore_version(self, fid, n):
3427
+ """An older version's body as the newest version, and the row's:
3428
+ nothing is rewritten, so a run or a pin naming any version still reads
3429
+ what it named. Returns (doc, None)."""
3430
+ with self.store.lock, self._connect() as db, db:
3431
+ r = self._live(db, fid)
3432
+ if r is None:
3433
+ return None, (404, "no such filter set")
3434
+ head = self._mint(db, r)
3435
+ old = db.execute("SELECT body FROM filter_set_versions WHERE set_id = ? AND n = ?",
3436
+ (fid, n)).fetchone()
3437
+ if old is None:
3438
+ return None, (404, "no such version")
3439
+ if old["body"] != head["body"]:
3440
+ self._cut(db, fid, old["body"])
3441
+ db.execute("UPDATE filter_sets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
3442
+ (old["body"], r["version"] + 1, self._now(), fid))
3443
+ return self._row_doc(db, self._live(db, fid)), None
3444
+
3445
+ def mint_pinned(self, pins):
3446
+ """Version 1 of each pinned set that has none yet: a pin names a
3447
+ version, so the version has to be kept from the moment it does."""
3448
+ with self.store.lock, self._connect() as db, db:
3449
+ for fid, _ in pins:
3450
+ r = self._live(db, fid)
3451
+ if r is not None:
3452
+ self._mint(db, r)
3453
+
3454
+ def resolve(self, fid, pin=None):
3455
+ """The version a run submitted now grades with -- [pin], or the newest
3456
+ -- marked as graded with, in the same transaction, so no save can edit
3457
+ it in place between this and the run keeping its body. Returns
3458
+ (n, body), (None, why) for a pin the set has no version of, or None for
3459
+ a set the lab does not have -- as a group's resolve."""
3460
+ with self.store.lock, self._connect() as db, db:
3461
+ r = self._live(db, fid) if isinstance(fid, str) else None
3462
+ if r is None:
3463
+ return None
3464
+ head = self._mint(db, r)
3465
+ v = head if pin is None else db.execute(
3466
+ "SELECT * FROM filter_set_versions WHERE set_id = ? AND n = ?", (fid, pin)).fetchone()
3467
+ if v is None:
3468
+ return None, f"{r['name']} has no version {pin}"
3469
+ db.execute("UPDATE filter_set_versions SET ran = 1 WHERE set_id = ? AND n = ?", (fid, v["n"]))
3470
+ return v["n"], json.loads(v["body"])
3471
+
3472
+ def snapshot(self, fid):
3473
+ """The body a run submitted now grades against, and its fingerprint;
3474
+ None for a set the lab does not have."""
3475
+ with self.store.lock, self._connect() as db:
3476
+ r = self._live(db, fid) if isinstance(fid, str) else None
3477
+ if r is None:
3478
+ return None
3479
+ body = json.loads(r["body"])
3480
+ return body, fingerprint(body)
3481
+
3482
+ def _insert(self, db, name, body):
3483
+ fid = secrets.token_hex(6)
3484
+ now = self._now()
3485
+ db.execute("INSERT INTO filter_sets (id, name, version, body, created_at, updated_at, workspace) "
3486
+ "VALUES (?, ?, 1, ?, ?, ?, ?)",
3487
+ (fid, name, json.dumps(body), now, now, workspace_of(self.store)))
3488
+ return fid
3489
+
3490
+ def create(self, name, body=None):
3491
+ """A new filter set, blank unless [body] is given -- New, and Duplicate,
3492
+ which sends the source's body. A name already taken gets ` (2)`."""
3493
+ name, why = filter_set_name(name)
3494
+ if why:
3495
+ return None, (400, why)
3496
+ body = {"version": FILTER_SET_BODY_VERSION, "filters": []} if body is None \
3497
+ else upgrade_filter_set_body(body)
3498
+ why = filter_set_problem(body)
3499
+ if why:
3500
+ return None, (400, why)
3501
+ with self.store.lock, self._connect() as db, db:
3502
+ fid = self._insert(db, unique_filter_set_name(name, self._names(db)), body)
3503
+ return self.get(fid), None
3504
+
3505
+ def rename(self, fid, name):
3506
+ """A new label. The body and its version are untouched: a rename is not
3507
+ an edit a browser holding the body has to reload for."""
3508
+ name, why = filter_set_name(name)
3509
+ if why:
3510
+ return None, (400, why)
3511
+ with self.store.lock, self._connect() as db, db:
3512
+ if self._live(db, fid) is None:
3513
+ return None, (404, "no such filter set")
3514
+ if name.lower() in {n.lower() for n in self._names(db, but=fid)}:
3515
+ return None, (409, f"{name!r} is taken by another filter set")
3516
+ db.execute("UPDATE filter_sets SET name = ?, updated_at = ? WHERE id = ?",
3517
+ (name, self._now(), fid))
3518
+ return self.get(fid), None
3519
+
3520
+ def save(self, fid, version, body):
3521
+ """The body, written at [version] -- the one it began from. Returns
3522
+ ({version}, None), or (None, error); a stale version's error carries
3523
+ the current row."""
3524
+ if type(version) is not int:
3525
+ return None, (400, "a save names the version it began from")
3526
+ body = upgrade_filter_set_body(body)
3527
+ why = filter_set_problem(body)
3528
+ if why:
3529
+ return None, (400, why)
3530
+ with self.store.lock, self._connect() as db, db:
3531
+ r = self._live(db, fid)
3532
+ if r is None:
3533
+ return None, (404, "no such filter set")
3534
+ if r["version"] != version:
3535
+ return None, (409, {"current": self._row_doc(db, r)})
3536
+ text = json.dumps(body)
3537
+ self._keep(db, r, text)
3538
+ db.execute("UPDATE filter_sets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
3539
+ (text, version + 1, self._now(), fid))
3540
+ return {"version": version + 1}, None
3541
+
3542
+ def remove(self, fid):
3543
+ """Into the trash, at once; Undo restores it. Returns (token, None)."""
3544
+ token = secrets.token_hex(6)
3545
+ with self.store.lock, self._connect() as db, db:
3546
+ if self._live(db, fid) is None:
3547
+ return None, (404, "no such filter set")
3548
+ db.execute("UPDATE filter_sets SET trash = ?, trashed_at = ? WHERE id = ?",
3549
+ (token, time.time(), fid))
3550
+ return token, None
3551
+
3552
+ def restore(self, token):
3553
+ """A trashed set back, under a new ` (2)` name if its own has been taken
3554
+ since. Returns (doc, None)."""
3555
+ with self.store.lock, self._connect() as db, db:
3556
+ r = db.execute("SELECT * FROM filter_sets WHERE trash = ? AND workspace = ?",
3557
+ (str(token), workspace_of(self.store))).fetchone()
3558
+ if r is None:
3559
+ return None, (404, "no such trash entry")
3560
+ name = unique_filter_set_name(r["name"], self._names(db, but=r["id"]))
3561
+ db.execute("UPDATE filter_sets SET trash = NULL, trashed_at = NULL, name = ? WHERE id = ?",
3562
+ (name, r["id"]))
3563
+ fid = r["id"]
3564
+ return self.get(fid), None
3565
+
3566
+ @staticmethod
3567
+ def _purge(db, where, args):
3568
+ db.execute(f"DELETE FROM filter_set_versions WHERE set_id IN (SELECT id FROM filter_sets WHERE {where})", args)
3569
+ db.execute(f"DELETE FROM filter_sets WHERE {where}", args)
3570
+
3571
+ def lazy_trash(self):
3572
+ """A filter-sets request empties what has been trashed longer than
3573
+ TRASH_SECONDS, as a datasets request does."""
3574
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3575
+ self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
3576
+
3577
+ def empty_trash(self):
3578
+ """The startup sweep: a restart has nothing to undo."""
3579
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3580
+ self._purge(db, "trash IS NOT NULL", ())
3581
+
3582
+
2806
3583
  def make_thumbs(sid, name):
2807
3584
  """JPEGs a browser can render, beside a stored file: a tile-size one
2808
3585
  for every image type, and a full-size render for the formats a browser
@@ -3328,7 +4105,7 @@ PLUGIN_VERSIONS = (1,)
3328
4105
  PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
3329
4106
  PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
3330
4107
  PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
3331
- REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
4108
+ REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes", "workflowPlatforms", "wizards")
3332
4109
  # evalTypes as a manifest written before pipeline version 11 spells it: a
3333
4110
  # plugin's own file, which the lab cannot upgrade, so it is read for good.
3334
4111
  OLD_REGISTRIES = {"testTypes": "evalTypes"}
@@ -3382,7 +4159,7 @@ def read_plugin(data: bytes):
3382
4159
  reg = m.get("registers") or {}
3383
4160
  if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
3384
4161
  return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
3385
- for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
4162
+ for k in ("outputKinds", "modifiers", "evalTypes", "wizards", *OLD_REGISTRIES):
3386
4163
  if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
3387
4164
  return None, f"registers.{k} is a list of ids"
3388
4165
  conns = reg.get("connectionTypes", [])
@@ -3397,6 +4174,25 @@ def read_plugin(data: bytes):
3397
4174
  or c.get("auth", "bearer") not in AUTH_WAYS):
3398
4175
  return None, ("a connection type the plugin registers is { id, settings, chatPath, "
3399
4176
  f"auth }}, auth one of {', '.join(AUTH_WAYS)}")
4177
+ # A workflow platform (#303): the generic `workflow` kind drives it, and the
4178
+ # server reads only the data it enforces -- the files it takes, whether it
4179
+ # keeps a definition, and the sign-in it is made with. The sign-in must be
4180
+ # one this lab already has (or none): a platform needing a new server-held
4181
+ # OAuth grant is server code and a secret store a plugin cannot ship, so it
4182
+ # stays lab-only (docs/workflow-sources.md phase 7).
4183
+ platforms = reg.get("workflowPlatforms", [])
4184
+ if not isinstance(platforms, list):
4185
+ return None, "registers.workflowPlatforms is a list"
4186
+ for p in platforms:
4187
+ uploads = p.get("uploads", {}) if isinstance(p, dict) else None
4188
+ if (not isinstance(p, dict) or not isinstance(p.get("id"), str) or not p["id"]
4189
+ or not isinstance(uploads, dict)
4190
+ or not all(isinstance(e, str) and e.startswith(".") and isinstance(t, str)
4191
+ for e, t in uploads.items())
4192
+ or not isinstance(p.get("keepsDefinition", False), bool)
4193
+ or (p.get("signIn") is not None and p.get("signIn") not in SIGN_INS)):
4194
+ return None, ("a workflow platform the plugin registers is { id, uploads, "
4195
+ "keepsDefinition, signIn }, signIn null or a sign-in the lab has")
3400
4196
  why = pack_requires_problem({"requires": {"lab": (m.get("requires") or {}).get("lab")}}, set())
3401
4197
  if why:
3402
4198
  return None, why.replace("the pack", "the plugin")
@@ -3406,8 +4202,9 @@ def read_plugin(data: bytes):
3406
4202
  def registered_ids(manifest: dict) -> set:
3407
4203
  """(registry, id) for everything a plugin's manifest says it registers."""
3408
4204
  reg = plugin_registers(manifest)
3409
- out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
3410
- return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
4205
+ out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes", "wizards") for x in reg.get(k, [])}
4206
+ out |= {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
4207
+ return out | {("workflowPlatforms", p["id"]) for p in reg.get("workflowPlatforms", [])}
3411
4208
 
3412
4209
 
3413
4210
  # ---- Connections: the lab's grants to outside services (#127) --------------
@@ -3738,20 +4535,35 @@ class Plugins:
3738
4535
  return [dict(r) for r in db.execute("SELECT * FROM plugins ORDER BY id")]
3739
4536
 
3740
4537
  def apply(self):
3741
- """The server's connection-type mirror: the built-in types and every
3742
- installed plugin's, rebuilt in place so every reader sees the same."""
4538
+ """The server's connection-type and workflow-platform mirrors: the
4539
+ built-in entries and every installed plugin's, rebuilt in place so every
4540
+ reader sees the same."""
3743
4541
  types, paths, auth = dict(BUILTIN_CONNECTION_TYPES), dict(BUILTIN_CHAT_PATHS), dict(BUILTIN_AUTH)
3744
4542
  local = set(BUILTIN_LOCAL)
4543
+ platforms = {k: dict(v) for k, v in BUILTIN_WORKFLOW_PLATFORMS.items()}
3745
4544
  for r in self._rows():
3746
- for c in (json.loads(r["manifest"]).get("registers") or {}).get("connectionTypes", []):
4545
+ reg = json.loads(r["manifest"]).get("registers") or {}
4546
+ for c in reg.get("connectionTypes", []):
3747
4547
  types[c["id"]] = tuple(c.get("settings", []))
3748
4548
  paths[c["id"]] = c.get("chatPath", "/chat/completions")
3749
4549
  auth[c["id"]] = c.get("auth", "bearer")
3750
4550
  if c.get("local") is True:
3751
4551
  local.add(c["id"])
4552
+ # A plugin platform's enforcement data, read from the manifest: the
4553
+ # files it takes, checked and redacted by the generic record
4554
+ # redactor like any flow's (take_record), whether it keeps a
4555
+ # definition, and the sign-in (none, or one the lab has) a Source of
4556
+ # it is made with. Its code -- api, stepsOf, evaluate … -- is the
4557
+ # page's and the runner's; this server never runs it.
4558
+ for p in reg.get("workflowPlatforms", []):
4559
+ platforms[p["id"]] = {"label": p.get("label", p["id"]), "take": take_record,
4560
+ "uploads": p.get("uploads") or {},
4561
+ "definition": bool(p.get("keepsDefinition")),
4562
+ "signIn": p.get("signIn")}
3752
4563
  LOCAL_CONNECTIONS.clear()
3753
4564
  LOCAL_CONNECTIONS.update(local)
3754
- for table, value in ((CONNECTION_TYPES, types), (CONNECTION_CHAT_PATHS, paths), (CONNECTION_AUTH, auth)):
4565
+ for table, value in ((CONNECTION_TYPES, types), (CONNECTION_CHAT_PATHS, paths),
4566
+ (CONNECTION_AUTH, auth), (WORKFLOW_PLATFORMS, platforms)):
3755
4567
  table.clear()
3756
4568
  table.update(value)
3757
4569
 
@@ -3801,9 +4613,13 @@ class Plugins:
3801
4613
  if same and same["sha256"] == plugin["sha256"]:
3802
4614
  return {"plugin": m["id"], "installed": False}, None
3803
4615
  mine = registered_ids(m)
4616
+ builtin = {"connectionTypes": (BUILTIN_CONNECTION_TYPES, "connection type"),
4617
+ "workflowPlatforms": (BUILTIN_WORKFLOW_PLATFORMS, "workflow platform"),
4618
+ "wizards": (BUILTIN_WIZARD_IDS, "wizard")}
3804
4619
  for (reg, x) in mine:
3805
- if reg == "connectionTypes" and x in BUILTIN_CONNECTION_TYPES:
3806
- return None, (400, f"the plugin registers the connection type {x}, which the lab has already")
4620
+ have, what = builtin.get(reg, (None, None))
4621
+ if have is not None and x in have:
4622
+ return None, (400, f"the plugin registers the {what} {x}, which the lab has already")
3807
4623
  for r in rows:
3808
4624
  if r["id"] == m["id"]:
3809
4625
  continue
@@ -4057,6 +4873,12 @@ class Queue:
4057
4873
  # `dataset` column, read as its one group.
4058
4874
  if "groups" not in cols:
4059
4875
  db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
4876
+ # The body of each filter set a run links, by `<id>@<n>` (#336,
4877
+ # docs/pipeline-model.md §18): a run cleans and gates its replies
4878
+ # against these, never the set as it reads later. A row from before
4879
+ # them has none, which is a run that links none.
4880
+ if "filter_sets" not in cols:
4881
+ db.execute("ALTER TABLE queue ADD COLUMN filter_sets TEXT")
4060
4882
  # The workspace a run belongs to (docs/workspaces.md): History is
4061
4883
  # per-workspace, so the list and every id-keyed read filter by it.
4062
4884
  # The worker loop is the one global reader -- it grades every
@@ -4142,30 +4964,35 @@ class Queue:
4142
4964
 
4143
4965
  # ---- submit ---------------------------------------------------------
4144
4966
 
4145
- def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
4967
+ def submit(self, run: dict, dataset=None, rerun_of=None, groups=None, filter_sets=None):
4146
4968
  """
4147
4969
  A new queued run. `run` is the run document (docs/pipeline-model.md
4148
4970
  §5): the pipeline, the profiles it resolved to without their keys, its
4149
4971
  content's file list in order and with its repeats, and each eval
4150
4972
  group's version; `groups` is those versions' bodies, by `<id>@<n>`
4151
4973
  (§17), and `dataset` the one body a row from before them kept, which a
4152
- re-run of one carries on; `rerun_of` is the run a re-run was queued
4153
- from. Returns the row. Its items are that list, or the one inline
4154
- text, each through every scenario -- so the total is the list's
4155
- length, repeats and all, the same count the runner and the page make.
4974
+ re-run of one carries on; `filter_sets` is the body of each Library
4975
+ filter set the run links, by `<id>@<n>` (§18), so a run cleans and
4976
+ gates its replies against what it was submitted with; `rerun_of` is the
4977
+ run a re-run was queued from. Returns the row. Its items are that list,
4978
+ or the one inline text, each through every scenario -- so the total is
4979
+ the list's length, repeats and all, the same count the runner and the
4980
+ page make.
4156
4981
  """
4157
4982
  rid = secrets.token_hex(6)
4158
4983
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
4159
4984
  total = len(run_items(run))
4160
4985
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
4161
4986
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
4162
- "snapshot, results, progress, totals, dataset, rerun_of, groups, workspace) "
4163
- "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
4987
+ "snapshot, results, progress, totals, dataset, rerun_of, groups, filter_sets, workspace) "
4988
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
4164
4989
  (rid, "queued", now, json.dumps(run), "[]",
4165
4990
  json.dumps({"current": None, "n": 0, "total": total}),
4166
4991
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
4167
4992
  None if dataset is None else json.dumps(dataset), rerun_of,
4168
- None if groups is None else json.dumps(groups), workspace_of(self.store)))
4993
+ None if groups is None else json.dumps(groups),
4994
+ None if filter_sets is None else json.dumps(filter_sets),
4995
+ workspace_of(self.store)))
4169
4996
  # Its prompts' uses, in the same transaction: a run is in the
4170
4997
  # library the moment it is queued, or not queued at all.
4171
4998
  if self.prompts is not None:
@@ -4214,6 +5041,21 @@ class Queue:
4214
5041
  return None
4215
5042
  return body if raw else upgrade_body(body)
4216
5043
 
5044
+ def filter_sets(self, rid, raw=False):
5045
+ """The bodies of the Library filter sets a readable run links, by
5046
+ `<id>@<n>` (§18), or None for a run that is not there or links none --
5047
+ what its replies were cleaned and gated against, whatever the sets hold
5048
+ now. A body kept at an earlier version reads as one of today's, unless
5049
+ [raw]."""
5050
+ if self.get(rid) is None:
5051
+ return None
5052
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
5053
+ r = db.execute("SELECT filter_sets FROM queue WHERE id = ?", (rid,)).fetchone()
5054
+ kept = json.loads(r[0]) if r and r[0] is not None else None
5055
+ if kept is None:
5056
+ return None
5057
+ return kept if raw else {k: upgrade_filter_set_body(b) for k, b in kept.items()}
5058
+
4217
5059
  def _plugin_args(self, run):
4218
5060
  """The worker's --plugins, when the run recorded any; a run whose
4219
5061
  plugins have changed since is refused, naming them."""
@@ -4274,6 +5116,28 @@ class Queue:
4274
5116
  (rundir / "groups.json").write_text(json.dumps(kept))
4275
5117
  return ["--groups", str(rundir / "groups.json")], None
4276
5118
 
5119
+ def _filter_args(self, run, rundir):
5120
+ """The worker's --filter-sets for a run that links a Library filter set
5121
+ in Responses: the body it kept for each, by `<id>@<n>`, written beside
5122
+ the run document, so it cleans and gates its replies against what it
5123
+ was submitted with (§18). A private (`own`) set carries its body in the
5124
+ document, so it needs nothing here, and a run that links no Library set
5125
+ passes no --filter-sets. Returns (args, None) or (None, why)."""
5126
+ refs = [filter_set_ref(link) for link in filter_set_links(run["snapshot"])]
5127
+ keys = {group_key(r) for r in refs if r is not None}
5128
+ if not keys:
5129
+ return [], None
5130
+ kept = self.filter_sets(run["id"], raw=True)
5131
+ if kept is None:
5132
+ return None, "the run kept no filter-set bodies"
5133
+ missing = sorted({(r.get("name") or r.get("id")) for r in refs
5134
+ if r is not None and group_key(r) not in kept})
5135
+ if missing:
5136
+ return None, f"the run kept no body of the filter set {', '.join(missing)}"
5137
+ rundir.mkdir(parents=True, exist_ok=True)
5138
+ (rundir / "filter-sets.json").write_text(json.dumps(kept))
5139
+ return ["--filter-sets", str(rundir / "filter-sets.json")], None
5140
+
4277
5141
  def _behind(self, rid):
4278
5142
  """How many submissions stand between this one and the worker, by
4279
5143
  submit time -- what a waiting form names when it says what it is
@@ -4366,8 +5230,10 @@ class Queue:
4366
5230
  if err:
4367
5231
  return None, (409, err)
4368
5232
  # The group bodies the original kept: a re-run grades with exactly
4369
- # them, whatever the groups or the pins read now.
5233
+ # them, whatever the groups or the pins read now; its filter-set
5234
+ # bodies travel the same way, so it cleans and gates as the original did.
4370
5235
  kept = self._kept(rid)
5236
+ kept_filters = self.filter_sets(rid, raw=True)
4371
5237
  ref = evals_dataset(snap)
4372
5238
  body = None
4373
5239
  if ref is not None and kept is None:
@@ -4386,7 +5252,7 @@ class Queue:
4386
5252
  _, err = worker_destinations(snap)
4387
5253
  if err:
4388
5254
  return None, (403, err)
4389
- return self.submit(snap, body, rerun_of=rid, groups=kept), None
5255
+ return self.submit(snap, body, rerun_of=rid, groups=kept, filter_sets=kept_filters), None
4390
5256
 
4391
5257
  def rerun_item(self, rid, index):
4392
5258
  """
@@ -4421,7 +5287,10 @@ class Queue:
4421
5287
  plugins, err = self._plugin_args(run)
4422
5288
  if err:
4423
5289
  return None, (409, err)
4424
- dataset = [*dataset, *plugins]
5290
+ filters, err = self._filter_args(run, rundir)
5291
+ if err:
5292
+ return None, (409, err)
5293
+ dataset = [*dataset, *plugins, *filters]
4425
5294
  results = [r for r in run["results"] if r is not None]
4426
5295
  (rundir / "results.json").write_text(json.dumps(results))
4427
5296
  args = [NODE, str(HERE / "run-evals.js"), "--run", str(rundir / "run.json"),
@@ -4603,7 +5472,10 @@ class Queue:
4603
5472
  plugins, err = self._plugin_args(run)
4604
5473
  if err:
4605
5474
  return self._finish(rid, "failed", error=err)
4606
- dataset = [*dataset, *plugins]
5475
+ filters, err = self._filter_args(run, rundir)
5476
+ if err:
5477
+ return self._finish(rid, "failed", error=err)
5478
+ dataset = [*dataset, *plugins, *filters]
4607
5479
  results = [r for r in run["results"] if r is not None]
4608
5480
  (rundir / "results.json").write_text(json.dumps(results))
4609
5481
  progress_path = rundir / "progress.jsonl"
@@ -4883,10 +5755,40 @@ def build_bundle(payload):
4883
5755
  return buf.getvalue(), None
4884
5756
 
4885
5757
 
5758
+ # Shared-by-choice enforcement (docs/workspaces.md): the one sentence the Run
5759
+ # bar turns into a Share link, and the profiles that earn it -- the run's
5760
+ # Target profiles not shared with workspace `ws`. A local profile reaches
5761
+ # nothing, so sharing does not gate it. Keyed by the run's profiles table,
5762
+ # whose keys are the profile ids the share set is written against.
5763
+ def not_shared_sentence(names, wsname: str) -> str:
5764
+ verb = "is" if len(names) == 1 else "are"
5765
+ return f"{', '.join(names)} {verb} not shared with {wsname}."
5766
+
5767
+
5768
+ def unshared_profiles(run: dict, ws):
5769
+ table = run.get("profiles") if isinstance(run, dict) else None
5770
+ if not isinstance(table, dict) or STORE is None:
5771
+ return []
5772
+ out = []
5773
+ for pid, conn in table.items():
5774
+ if isinstance(conn, dict) and conn.get("type") in LOCAL_CONNECTIONS:
5775
+ continue
5776
+ if not STORE.shared("profile", pid, ws):
5777
+ out.append({"id": pid, "name": (isinstance(conn, dict) and conn.get("name")) or pid})
5778
+ return out
5779
+
5780
+
4886
5781
  def worker_destinations(run: dict):
4887
5782
  table = run.get("profiles") if isinstance(run, dict) else None
4888
5783
  if not isinstance(table, dict):
4889
5784
  return None, "a run document carries its profiles as an object of id → connection"
5785
+ # The run grades in the workspace the thread is bound to -- the request's
5786
+ # at submit, the run's at dequeue/re-run (docs/workspaces.md). A profile
5787
+ # not shared with it is refused here, the seam every run path passes.
5788
+ ws = workspace_of(STORE) if STORE is not None else None
5789
+ bad = unshared_profiles(run, ws)
5790
+ if bad:
5791
+ return None, not_shared_sentence([b["name"] for b in bad], STORE.workspace_name(ws))
4890
5792
  stored = []
4891
5793
  if STORE is not None:
4892
5794
  doc = STORE.all().get("promptlab.profiles") or {}
@@ -4934,10 +5836,12 @@ def worker_destinations(run: dict):
4934
5836
  # 12: a Contains metric's Ignore case holds item by item too, kept as written.
4935
5837
  # 13: an eval is a link to an eval group, or a group of the pipeline's own,
4936
5838
  # and the document has an overall pass rule (docs/pipeline-model.md §17).
4937
- PIPELINE_VERSION = 13
5839
+ # 14: a job's Responses stage links filter sets on a `filters` step, and the
5840
+ # inline Drop items / Reject rules migrate into a private one (§18, #339).
5841
+ PIPELINE_VERSION = 14
4938
5842
  # What a stored run may be: the current version, and the ones evals-core.ts's
4939
5843
  # upgradePipeline reads. A new submission is upgraded to the current one.
4940
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
5844
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14)
4941
5845
  TARGET_CAP = 4
4942
5846
  # Target steps whose words the Prompt library does not record as a use: they
4943
5847
  # ask no model (evals-core.ts's STEP_TYPES.echo).
@@ -5298,9 +6202,11 @@ def run_problems(run):
5298
6202
  # there was a store, and it stays usable without one being asked for.
5299
6203
  if DATA_DIR:
5300
6204
  STORE = Store(Path(DATA_DIR) / "lab.db")
6205
+ WORKSPACES = Workspaces(STORE)
5301
6206
  SOURCES = Sources(STORE)
5302
6207
  PROMPTS = Prompts(STORE)
5303
6208
  DATASETS = Datasets(STORE, PROMPTS)
6209
+ FILTER_SETS = FilterSets(STORE)
5304
6210
  PACKS = Packs(STORE)
5305
6211
  PLUGINS = Plugins(STORE)
5306
6212
  CONNECTIONS = Connections(STORE)
@@ -5309,7 +6215,8 @@ if DATA_DIR:
5309
6215
  QUEUE = Queue(STORE)
5310
6216
  QUEUE.prompts = PROMPTS
5311
6217
  else:
5312
- STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
6218
+ STORE = WORKSPACES = SOURCES = PROMPTS = DATASETS = FILTER_SETS = PACKS = PLUGINS = \
6219
+ CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
5313
6220
 
5314
6221
 
5315
6222
  class NoRedirects(urllib.request.HTTPRedirectHandler):
@@ -5564,6 +6471,14 @@ class Handler(BaseHTTPRequestHandler):
5564
6471
  if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
5565
6472
  return self._send(404, b"not found", "text/plain")
5566
6473
  return self._json(200, QUEUE.dataset(run_id))
6474
+ if path.startswith("/api/queue/") and path.endswith("/filter-sets") and path.count("/") == 4:
6475
+ # The bodies of the filter sets a run cleaned and gated its replies
6476
+ # against, by `<id>@<n>` (§18); null for a run that linked none, as
6477
+ # /dataset and /groups answer.
6478
+ run_id = path.split("/")[3]
6479
+ if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
6480
+ return self._send(404, b"not found", "text/plain")
6481
+ return self._json(200, QUEUE.filter_sets(run_id))
5567
6482
  if path.startswith("/api/queue/"):
5568
6483
  if QUEUE is None:
5569
6484
  return self._send(404, b"not found", "text/plain")
@@ -5583,6 +6498,13 @@ class Handler(BaseHTTPRequestHandler):
5583
6498
  if PLUGINS is None:
5584
6499
  return self._send(404, b"not found", "text/plain")
5585
6500
  return self._json(200, {"plugins": PLUGINS.list(), "cap": PLUGIN_CAP})
6501
+ if path == "/api/connections/shares":
6502
+ # Which workspaces may use each Target profile and Account
6503
+ # (docs/workspaces.md): the share sets the Connections menus read.
6504
+ # Keyed by "<kind>:<id>", workspace ids and the sentinels, no key.
6505
+ if STORE is None:
6506
+ return self._send(404, b"not found", "text/plain")
6507
+ return self._json(200, {"shares": STORE.shares()})
5586
6508
  if path == "/api/connections":
5587
6509
  if CONNECTIONS is None:
5588
6510
  return self._send(404, b"not found", "text/plain")
@@ -5607,6 +6529,11 @@ class Handler(BaseHTTPRequestHandler):
5607
6529
  if PACKS is None:
5608
6530
  return self._send(404, b"not found", "text/plain")
5609
6531
  return self._json(200, {"packs": PACKS.list(), "cap": PACK_CAP})
6532
+ if path == "/api/workspaces":
6533
+ if WORKSPACES is None:
6534
+ return self._send(404, b"not found", "text/plain")
6535
+ WORKSPACES.lazy_trash()
6536
+ return self._json(200, {"workspaces": WORKSPACES.list()})
5610
6537
  if path.startswith("/api/samples/thumbs/") or path.startswith("/api/samples/zoom/"):
5611
6538
  # The sample-library tiles, store-independent: baked into the
5612
6539
  # image (or absent on a checkout), and read by the library strip,
@@ -5650,6 +6577,8 @@ class Handler(BaseHTTPRequestHandler):
5650
6577
  })
5651
6578
  if path.startswith("/api/datasets"):
5652
6579
  return self._datasets_get(path)
6580
+ if path.startswith("/api/filter-sets"):
6581
+ return self._filter_sets_get(path)
5653
6582
  if path == "/api/prompts" or path.startswith("/api/prompts/"):
5654
6583
  return self._prompts_get(path)
5655
6584
  if path.startswith("/api/sources"):
@@ -5735,6 +6664,14 @@ class Handler(BaseHTTPRequestHandler):
5735
6664
  if err:
5736
6665
  return self._json(err[0], {"error": err[1]})
5737
6666
  return self._json(200, {"trash": token})
6667
+ if path.startswith("/api/workspaces/"):
6668
+ parts = path.split("/")
6669
+ if WORKSPACES is None or len(parts) != 4:
6670
+ return self._send(404, b"not found", "text/plain")
6671
+ token, err = WORKSPACES.remove(parts[3])
6672
+ if err:
6673
+ return self._json(err[0], {"error": err[1]})
6674
+ return self._json(200, {"trash": token})
5738
6675
  if path.startswith("/api/datasets/"):
5739
6676
  parts = path.split("/")
5740
6677
  if DATASETS is None or len(parts) != 4:
@@ -5743,6 +6680,14 @@ class Handler(BaseHTTPRequestHandler):
5743
6680
  if err:
5744
6681
  return self._json(err[0], {"error": err[1]})
5745
6682
  return self._json(200, {"trash": token})
6683
+ if path.startswith("/api/filter-sets/"):
6684
+ parts = path.split("/")
6685
+ if FILTER_SETS is None or len(parts) != 4:
6686
+ return self._send(404, b"not found", "text/plain")
6687
+ token, err = FILTER_SETS.remove(parts[3])
6688
+ if err:
6689
+ return self._json(err[0], {"error": err[1]})
6690
+ return self._json(200, {"trash": token})
5746
6691
  if path.startswith("/api/prompts/"):
5747
6692
  parts = path.split("/")
5748
6693
  if PROMPTS is None or len(parts) != 4:
@@ -5793,6 +6738,14 @@ class Handler(BaseHTTPRequestHandler):
5793
6738
  if err:
5794
6739
  return self._json(err[0], {"error": err[1]})
5795
6740
  return self._json(200, dataset)
6741
+ if parts[2] == "filter-sets" and FILTER_SETS is not None:
6742
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
6743
+ if err:
6744
+ return err()
6745
+ fs, err = FILTER_SETS.rename(parts[3], payload.get("name") if isinstance(payload, dict) else None)
6746
+ if err:
6747
+ return self._json(err[0], {"error": err[1]})
6748
+ return self._json(200, fs)
5796
6749
  if parts[2] == "sources" and SOURCES is not None:
5797
6750
  payload = self._payload() or {}
5798
6751
  source, err = SOURCES.rename(parts[3], payload.get("name"))
@@ -5809,6 +6762,12 @@ class Handler(BaseHTTPRequestHandler):
5809
6762
  code, message = err
5810
6763
  return self._json(code, {"error": message})
5811
6764
  return self._json(200, run)
6765
+ if parts[2] == "workspaces" and WORKSPACES is not None:
6766
+ payload = self._payload() or {}
6767
+ wid, err = WORKSPACES.rename(parts[3], payload.get("name"))
6768
+ if err:
6769
+ return self._json(err[0], {"error": err[1]})
6770
+ return self._json(200, {"workspace": WORKSPACES.get(wid), "workspaces": WORKSPACES.list()})
5812
6771
  return self._send(404, b"not found", "text/plain")
5813
6772
 
5814
6773
  def do_PUT(self):
@@ -5817,6 +6776,16 @@ class Handler(BaseHTTPRequestHandler):
5817
6776
  self._scope()
5818
6777
  self._alias()
5819
6778
  parts = self.path.split("?", 1)[0].split("/")
6779
+ if parts == ["", "api", "connections", "shares"] and STORE is not None:
6780
+ # A connection's whole share set, replaced (docs/workspaces.md): the
6781
+ # Connections checkbox menu's All / list / New workspaces. The reply
6782
+ # is the whole map, so every menu redraws from one response.
6783
+ payload = self._payload() or {}
6784
+ target = self._share_target(payload)
6785
+ if target is None:
6786
+ return self._json(400, {"error": "a share names a kind (profile or account) and an id"})
6787
+ STORE.set_shares(target[0], target[1], payload.get("workspaces") or [])
6788
+ return self._json(200, {"shares": STORE.shares()})
5820
6789
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
5821
6790
  return self._prompts_put(parts[3])
5822
6791
  if parts == ["", "api", "google"] and GOOGLE is not None:
@@ -5851,6 +6820,17 @@ class Handler(BaseHTTPRequestHandler):
5851
6820
  code, said = err
5852
6821
  return self._json(code, said if isinstance(said, dict) else {"error": said})
5853
6822
  return self._json(200, saved)
6823
+ if len(parts) == 4 and parts[:3] == ["", "api", "filter-sets"] and FILTER_SETS is not None:
6824
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
6825
+ if err:
6826
+ return err()
6827
+ if not isinstance(payload, dict):
6828
+ return self._json(400, {"error": "a save is { version, body }"})
6829
+ saved, err = FILTER_SETS.save(parts[3], payload.get("version"), payload.get("body"))
6830
+ if err:
6831
+ code, said = err
6832
+ return self._json(code, said if isinstance(said, dict) else {"error": said})
6833
+ return self._json(200, saved)
5854
6834
  if len(parts) != 4 or parts[:3] != ["", "api", "datasets"] or DATASETS is None:
5855
6835
  return self._send(404, b"not found", "text/plain")
5856
6836
  payload, err = self._dataset_payload(DATASET_CAP)
@@ -5889,10 +6869,26 @@ class Handler(BaseHTTPRequestHandler):
5889
6869
  return self._queue_action(path)
5890
6870
  if path.startswith("/api/datasets"):
5891
6871
  return self._datasets_post(path)
6872
+ if path.startswith("/api/filter-sets"):
6873
+ return self._filter_sets_post(path)
5892
6874
  if path == "/api/packs" or path.startswith("/api/packs/"):
5893
6875
  return self._packs_post(path)
6876
+ if path == "/api/workspaces" or path.startswith("/api/workspaces/"):
6877
+ return self._workspaces_post(path)
5894
6878
  if path == "/api/plugins" or path.startswith("/api/plugins/"):
5895
6879
  return self._plugins_post(path)
6880
+ if path == "/api/connections/share":
6881
+ # The Run bar's Share link, and its Undo (docs/workspaces.md): the
6882
+ # current workspace into (or, undo, out of) a connection's share
6883
+ # set, at once. The reply is the whole map, as the menu's PUT is.
6884
+ if STORE is None:
6885
+ return self._send(404, b"not found", "text/plain")
6886
+ payload = self._payload() or {}
6887
+ target = self._share_target(payload)
6888
+ if target is None:
6889
+ return self._json(400, {"error": "a share names a kind (profile or account) and an id"})
6890
+ STORE.share_with(target[0], target[1], self.ws, payload.get("undo") is not True)
6891
+ return self._json(200, {"shares": STORE.shares()})
5896
6892
  if path == "/api/connections/google/start":
5897
6893
  return self._google_start()
5898
6894
  if path == "/api/prompts" or path.startswith("/api/prompts/"):
@@ -5914,6 +6910,51 @@ class Handler(BaseHTTPRequestHandler):
5914
6910
  return self._models_ask(payload)
5915
6911
  return self._send(404, b"not found", "text/plain")
5916
6912
 
6913
+ def _workspaces_post(self, path):
6914
+ """Create, archive, unarchive, Regenerate, set the default and restore,
6915
+ as the registry API (docs/other-tabs.md). Each mutation answers with the affected
6916
+ workspace and the whole list, so the Manage table and its counts
6917
+ redraw from one response; delete is a DELETE and answers with a trash
6918
+ token for Undo."""
6919
+ if WORKSPACES is None:
6920
+ return self._send(404, b"not found", "text/plain")
6921
+ parts = path.split("/")
6922
+ said = lambda wid: self._json(200, {"workspace": WORKSPACES.get(wid), "workspaces": WORKSPACES.list()})
6923
+ if len(parts) == 3:
6924
+ payload = self._payload() or {}
6925
+ # The New-workspace dialog's Connections picker (docs/workspaces.md):
6926
+ # the connections the new workspace may use, each written a concrete
6927
+ # share row; with none named, the SHARE_NEW set is materialised.
6928
+ shares = payload.get("connections") if isinstance(payload.get("connections"), list) else None
6929
+ wid, err = WORKSPACES.create(payload.get("name"), payload.get("slug"), shares)
6930
+ if err:
6931
+ return self._json(err[0], {"error": err[1]})
6932
+ # Starting content (docs/workspaces.md): the demo pack lands in the
6933
+ # new workspace, not the one the request is in, so bind the thread
6934
+ # to it for the install and bind it back after.
6935
+ if payload.get("content") == "demo" and PACKS is not None:
6936
+ set_workspace(wid)
6937
+ try:
6938
+ PACKS.install(demo_pack())
6939
+ finally:
6940
+ set_workspace(self.ws)
6941
+ return said(wid)
6942
+ if len(parts) == 6 and parts[3] == "trash" and parts[5] == "restore":
6943
+ self._payload()
6944
+ wid, err = WORKSPACES.restore(parts[4])
6945
+ if err:
6946
+ return self._json(err[0], {"error": err[1]})
6947
+ return said(wid)
6948
+ if len(parts) == 5 and parts[4] in ("archive", "unarchive", "regenerate", "default"):
6949
+ self._payload()
6950
+ act = {"archive": WORKSPACES.archive, "unarchive": WORKSPACES.unarchive,
6951
+ "regenerate": WORKSPACES.regenerate, "default": WORKSPACES.set_default}[parts[4]]
6952
+ wid, err = act(parts[3])
6953
+ if err:
6954
+ return self._json(err[0], {"error": err[1]})
6955
+ return said(wid)
6956
+ return self._send(404, b"not found", "text/plain")
6957
+
5917
6958
  def _packs_post(self, path):
5918
6959
  if PACKS is None:
5919
6960
  return self._send(404, b"not found", "text/plain")
@@ -5975,7 +7016,44 @@ class Handler(BaseHTTPRequestHandler):
5975
7016
  return self._json(err[0], {"error": err[1]})
5976
7017
  return self._json(201 if got["installed"] else 200, got)
5977
7018
 
7019
+ def _share_target(self, payload):
7020
+ """(kind, id) when `payload` names a shareable connection (kind profile
7021
+ or account, a non-empty id), else None -- the caller answers 400. The
7022
+ two kinds are the ones that carry a Workspaces field (docs/workspaces.md)."""
7023
+ kind, cid = payload.get("kind"), payload.get("id")
7024
+ return (kind, cid) if kind in ("profile", "account") and isinstance(cid, str) and cid else None
7025
+
7026
+ def _unshared(self, payload):
7027
+ """A saved Target profile a relayed request names by its held key, not
7028
+ shared with this request's workspace (docs/workspaces.md): (id, name),
7029
+ else None. Only a held key (KEY_HELD<id>) names a stored profile; a key
7030
+ typed into the Setup form is a connection not yet saved, so nothing
7031
+ gates it. This is the relay/Test seam, beside worker_destinations' run
7032
+ seam -- the two places a connection resolves to a key, so the two
7033
+ places sharing is enforced. The caller writes the 403: a helper that
7034
+ answers here would send the body and still fall through to the proxy."""
7035
+ if STORE is None:
7036
+ return None
7037
+ raw = str((payload or {}).get("key") or "")
7038
+ if not held(raw):
7039
+ return None
7040
+ pid = raw[len(KEY_HELD):]
7041
+ if STORE.shared("profile", pid, self.ws):
7042
+ return None
7043
+ return pid, STORE.profile_name(pid)
7044
+
7045
+ def _refuse_unshared(self, payload):
7046
+ """The 403 a relay seam answers when `payload` names an unshared
7047
+ profile, else None (nothing written)."""
7048
+ u = self._unshared(payload)
7049
+ if u is None:
7050
+ return False
7051
+ return self._json(403, {"error": not_shared_sentence([u[1]], STORE.workspace_name(self.ws)),
7052
+ "unshared": [{"kind": "profile", "id": u[0], "name": u[1]}]}) or True
7053
+
5978
7054
  def _models_list(self, payload):
7055
+ if self._refuse_unshared(payload):
7056
+ return
5979
7057
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
5980
7058
  key = relay_key(payload)
5981
7059
  ctype = str(payload.get("type") or "")
@@ -6027,6 +7105,8 @@ class Handler(BaseHTTPRequestHandler):
6027
7105
  # (evals-core.ts's connectionRequest); the server decides where it goes
6028
7106
  # and how the key travels, per the type -- the same mirror the models
6029
7107
  # list uses, so a client cannot point the relay at a path of its own.
7108
+ if self._refuse_unshared(payload):
7109
+ return
6030
7110
  base = api_base(str(payload.get("url") or "")) or api_base(OLLAMA)
6031
7111
  key = relay_key(payload)
6032
7112
  if not header_safe(key):
@@ -6064,10 +7144,14 @@ class Handler(BaseHTTPRequestHandler):
6064
7144
  for n, d in docs.items()}, self.ws)
6065
7145
  if stale is not None:
6066
7146
  return self._json(409, {"stale": stale})
6067
- # A pin names a group's version, so the group keeps that version from
6068
- # the moment a pipeline pins it (§17).
6069
- if "promptlab.workflows" in docs and DATASETS is not None:
6070
- DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
7147
+ # A pin names a group's or filter set's version, so it keeps that
7148
+ # version from the moment a pipeline pins it (§17, §18).
7149
+ if "promptlab.workflows" in docs:
7150
+ body = docs["promptlab.workflows"].get("body")
7151
+ if DATASETS is not None:
7152
+ DATASETS.mint_pinned(pins_in(body))
7153
+ if FILTER_SETS is not None:
7154
+ FILTER_SETS.mint_pinned(filter_set_pins_in(body))
6071
7155
  return self._json(200, {"versions": versions})
6072
7156
 
6073
7157
  # ---- The run queue (#530) ----------------------------------------------
@@ -6100,6 +7184,14 @@ class Handler(BaseHTTPRequestHandler):
6100
7184
  if revs[name] is None:
6101
7185
  return self._json(400, {"error": f"{name!r} is not in that Source"})
6102
7186
  content["revs"] = revs
7187
+ # A profile not shared with this workspace blocks the run with the
7188
+ # sentence the Run bar shows and the connections it names, so its Share
7189
+ # link can grant them at once (docs/workspaces.md).
7190
+ ws = workspace_of(STORE) if STORE is not None else None
7191
+ bad = unshared_profiles(run, ws)
7192
+ if bad:
7193
+ return self._json(403, {"error": not_shared_sentence([b["name"] for b in bad], STORE.workspace_name(ws)),
7194
+ "unshared": [{"kind": "profile", **b} for b in bad]})
6103
7195
  _, why = worker_destinations(run)
6104
7196
  if why:
6105
7197
  return self._json(403, {"error": why})
@@ -6125,9 +7217,29 @@ class Handler(BaseHTTPRequestHandler):
6125
7217
  return self._json(400, {"error": f"the run cannot be queued: {body}"})
6126
7218
  ref["n"], ref["version"] = n, fingerprint(body)
6127
7219
  groups[group_key(ref)] = body
7220
+ # Each Library filter set the run links, resolved the same way (§18): a
7221
+ # pinned link to its pin, anything else to the set's newest version. The
7222
+ # reference records the version and its fingerprint, and the body is
7223
+ # kept with the row under `<id>@<n>` -- the run cleans and gates its
7224
+ # replies against that, and the version is frozen from now on. A private
7225
+ # (`own`) link carries its body in the document, so it reads no store.
7226
+ filter_sets = {}
7227
+ for link in filter_set_links(run):
7228
+ ref = filter_set_ref(link)
7229
+ if ref is None:
7230
+ continue
7231
+ pin = link.get("pin")
7232
+ got = FILTER_SETS.resolve(ref.get("id"), pin if type(pin) is int else None) if FILTER_SETS else None
7233
+ if got is None:
7234
+ return self._json(400, {"error": "the run links no filter set this lab has"})
7235
+ n, body = got
7236
+ if n is None:
7237
+ return self._json(400, {"error": f"the run cannot be queued: {body}"})
7238
+ ref["n"], ref["version"] = n, fingerprint(body)
7239
+ filter_sets[group_key(ref)] = body
6128
7240
  # The worker reads a body per group now (#233), so a run may link
6129
7241
  # several; each body is kept with the row under `<id>@<n>`.
6130
- return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
7242
+ return self._json(201, {"run": QUEUE.submit(run, groups=groups, filter_sets=filter_sets or None)})
6131
7243
 
6132
7244
  # ---- Datasets ----------------------------------------------------------
6133
7245
  # Rows in the store (Datasets above). The page reads one by id; an export
@@ -6294,6 +7406,58 @@ class Handler(BaseHTTPRequestHandler):
6294
7406
  return self._json(201, dataset)
6295
7407
  return self._send(404, b"not found", "text/plain")
6296
7408
 
7409
+ # ---- Filter sets -------------------------------------------------------
7410
+ # Rows in the store (FilterSets above), read and written as datasets are
7411
+ # (docs/pipeline-model.md §18); no export or import, which datasets have for
7412
+ # CI. The page reads one by id, lists them, and keeps its versions.
7413
+
7414
+ def _filter_sets_get(self, path):
7415
+ if FILTER_SETS is None:
7416
+ return self._send(404, b"not found", "text/plain")
7417
+ FILTER_SETS.lazy_trash()
7418
+ parts = path.split("/")
7419
+ if len(parts) == 3:
7420
+ return self._json(200, {"filterSets": FILTER_SETS.list()})
7421
+ if len(parts) == 5 and parts[4] == "versions":
7422
+ got = FILTER_SETS.versions(parts[3])
7423
+ return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
7424
+ if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
7425
+ got = FILTER_SETS.version_body(parts[3], int(parts[5]))
7426
+ return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
7427
+ else self._send(404, b"not found", "text/plain")
7428
+ if len(parts) == 4:
7429
+ d = FILTER_SETS.get(parts[3])
7430
+ return self._json(200, d) if d else self._send(404, b"not found", "text/plain")
7431
+ return self._send(404, b"not found", "text/plain")
7432
+
7433
+ def _filter_sets_post(self, path):
7434
+ if FILTER_SETS is None:
7435
+ return self._send(404, b"not found", "text/plain")
7436
+ parts = path.split("/")
7437
+ if len(parts) == 6 and parts[3] == "trash" and parts[5] == "restore":
7438
+ self._payload()
7439
+ fs, err = FILTER_SETS.restore(parts[4])
7440
+ if err:
7441
+ return self._json(err[0], {"error": err[1]})
7442
+ return self._json(200, fs)
7443
+ if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
7444
+ self._payload()
7445
+ fs, err = FILTER_SETS.restore_version(parts[3], int(parts[5]))
7446
+ if err:
7447
+ return self._json(err[0], {"error": err[1]})
7448
+ return self._json(200, fs)
7449
+ if len(parts) == 3:
7450
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
7451
+ if err:
7452
+ return err()
7453
+ if not isinstance(payload, dict):
7454
+ return self._json(400, {"error": "a new filter set is { name, body? }"})
7455
+ fs, err = FILTER_SETS.create(payload.get("name"), payload.get("body"))
7456
+ if err:
7457
+ return self._json(err[0], {"error": err[1]})
7458
+ return self._json(201, fs)
7459
+ return self._send(404, b"not found", "text/plain")
7460
+
6297
7461
  def _queue_action(self, path):
6298
7462
  # Drained, so a keep-alive connection is not left holding the
6299
7463
  # page's `{}` in front of its next request.
@@ -6582,12 +7746,17 @@ def main():
6582
7746
  missing.append("no ImageMagick")
6583
7747
  runs = "ready" if not missing else "CANNOT RUN: " + ", ".join(missing)
6584
7748
  print(f"runs: {runs}", flush=True)
7749
+ if WORKSPACES is not None:
7750
+ # The workspace trash's undo window is per-session, like the others.
7751
+ WORKSPACES.empty_trash()
6585
7752
  if DATASETS is not None:
6586
7753
  # A new lab starts with none: the Datasets tab offers New and Import.
6587
7754
  DATASETS.empty_trash()
6588
7755
  print(f"datasets: {len(DATASETS.list())} in the store", flush=True)
6589
7756
  else:
6590
7757
  print("datasets: (none -- datasets need DATA_DIR)", flush=True)
7758
+ if FILTER_SETS is not None:
7759
+ FILTER_SETS.empty_trash()
6591
7760
  if PROMPTS is not None:
6592
7761
  PROMPTS.empty_trash()
6593
7762
  # The runs from before the library, read into it once.