evals-lab 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -1229,6 +1229,9 @@ class Workspaces:
1229
1229
  db.execute("DELETE FROM dataset_body_archive WHERE dataset_id IN "
1230
1230
  "(SELECT id FROM datasets WHERE workspace = ?)", (ws,))
1231
1231
  db.execute("DELETE FROM datasets WHERE workspace = ?", (ws,))
1232
+ db.execute("DELETE FROM filter_set_versions WHERE set_id IN "
1233
+ "(SELECT id FROM filter_sets WHERE workspace = ?)", (ws,))
1234
+ db.execute("DELETE FROM filter_sets WHERE workspace = ?", (ws,))
1232
1235
  db.execute("DELETE FROM prompt_uses WHERE prompt_id IN "
1233
1236
  "(SELECT id FROM prompts WHERE workspace = ?)", (ws,))
1234
1237
  db.execute("DELETE FROM prompt_versions WHERE prompt_id IN "
@@ -2266,6 +2269,22 @@ def upgrade_body(body):
2266
2269
  return recorded_ids_v7(group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}))
2267
2270
 
2268
2271
 
2272
+ # The filter-set body's version: evals-core.ts's FILTER_SET_VERSION. Version 1
2273
+ # is the first -- the { version, filters } body #334 introduces. A filter set
2274
+ # is a Library row the lab stores and versions like an eval group (#336,
2275
+ # FilterSets below); this and its twin are the body it keeps, held to
2276
+ # evals-core.ts by proxy-check.py.
2277
+ FILTER_SET_BODY_VERSION = 1
2278
+
2279
+
2280
+ def upgrade_filter_set_body(body):
2281
+ """An earlier filter-set body as today's (version 1): evals-core.ts's
2282
+ upgradeFilterSetBody, in Python. There is no earlier version yet, so a
2283
+ version-1 body comes back as it was, and so does anything that is not a
2284
+ body. Pure."""
2285
+ return body
2286
+
2287
+
2269
2288
  def body_prompt(body):
2270
2289
  """The prompt an earlier body held, or "": what the library is given."""
2271
2290
  prompt = body.get("prompt") if isinstance(body, dict) else None
@@ -2365,6 +2384,90 @@ def unique_dataset_name(name: str, taken: set) -> str:
2365
2384
  return f"{name} ({n})"
2366
2385
 
2367
2386
 
2387
+ # ---- Filter sets (the store's, FilterSets below) ---------------------------
2388
+ #
2389
+ # A filter set is a Library row like an eval group: a versioned { version,
2390
+ # filters } body (evals-core.ts FilterSetBody), kept so a pipeline can link it
2391
+ # and a run grade against the body it was submitted with (#336). The fields
2392
+ # are the body's; the per-filter shape is the core's to judge, as a case's
2393
+ # metrics are (dataset_problem). The suffixing is the generic one.
2394
+ FILTER_SET_FIELDS = ("version", "filters")
2395
+ FILTER_SET_NAME_MAX = 80
2396
+ unique_filter_set_name = unique_dataset_name
2397
+
2398
+
2399
+ def filter_set_problem(body) -> str:
2400
+ """Why [body] is not a filter set's body, in one sentence, or "":
2401
+ evals-core.ts's filterSetBodyProblems, to the depth the server checks."""
2402
+ if not isinstance(body, dict):
2403
+ return "a filter set's body is a JSON object"
2404
+ for k in body:
2405
+ if k not in FILTER_SET_FIELDS:
2406
+ return f"a filter set's body has \"{k}\", which is not a filter-set field"
2407
+ for k in FILTER_SET_FIELDS:
2408
+ if k not in body:
2409
+ return f"a filter set's body has no \"{k}\""
2410
+ if body["version"] != FILTER_SET_BODY_VERSION:
2411
+ return f"a filter set's body is version {FILTER_SET_BODY_VERSION}"
2412
+ if not isinstance(body["filters"], list) or not all(isinstance(f, dict) for f in body["filters"]):
2413
+ return "filters has to be a list of filters"
2414
+ return ""
2415
+
2416
+
2417
+ def filter_set_name(raw):
2418
+ """A name, trimmed, or (None, why)."""
2419
+ if not isinstance(raw, str) or not raw.strip():
2420
+ return None, "a filter set needs a name"
2421
+ name = raw.strip()
2422
+ if len(name) > FILTER_SET_NAME_MAX:
2423
+ return None, "that name is too long"
2424
+ return name, None
2425
+
2426
+
2427
+ def filter_set_links(doc) -> list:
2428
+ """Every filter-set link a run or pipeline document carries: a job's
2429
+ Responses steps list them under `filterSets` (evals-core.ts FilterSetLink).
2430
+ A link names a Library set (`set`, followed latest or pinned at `pin`) or
2431
+ holds a private one inline (`own`); only a Library link reads the store.
2432
+ Runs wires the step and migrates the inline rules in #339; this is the one
2433
+ reader of where the links sit, for the store (#336) to resolve and keep."""
2434
+ out = []
2435
+ jobs = doc.get("jobs") if isinstance(doc, dict) else None
2436
+ for job in jobs if isinstance(jobs, list) else []:
2437
+ steps = job.get("steps") if isinstance(job, dict) else None
2438
+ for st in steps if isinstance(steps, list) else []:
2439
+ links = st.get("filterSets") if isinstance(st, dict) else None
2440
+ for link in links if isinstance(links, list) else []:
2441
+ if isinstance(link, dict):
2442
+ out.append(link)
2443
+ return out
2444
+
2445
+
2446
+ def filter_set_ref(link):
2447
+ """The Library filter set a link reads, as the dict itself (so a caller may
2448
+ stamp it in place), or None for a private (`own`) link."""
2449
+ if not isinstance(link, dict):
2450
+ return None
2451
+ s = link.get("set")
2452
+ return s if isinstance(s, dict) and isinstance(s.get("id"), str) else None
2453
+
2454
+
2455
+ def filter_set_pins_in(workflows) -> set:
2456
+ """(filter-set id, version) for every filter-set link a stored pipeline
2457
+ pins: the promptlab.workflows body, each pipeline as its `work`, as
2458
+ pins_in reads eval-group pins. A Library link with an integer `pin`
2459
+ freezes that version from the moment a pipeline pins it (§17)."""
2460
+ out = set()
2461
+ listed = workflows.get("list") if isinstance(workflows, dict) else None
2462
+ for w in listed if isinstance(listed, list) else []:
2463
+ work = w.get("work") if isinstance(w, dict) else None
2464
+ for link in filter_set_links(work):
2465
+ ref = filter_set_ref(link)
2466
+ if ref is not None and type(link.get("pin")) is int:
2467
+ out.add((ref["id"], link["pin"]))
2468
+ return out
2469
+
2470
+
2368
2471
  # ---- The Prompt library -----------------------------------------------------
2369
2472
  #
2370
2473
  # Every prompt the lab has, run or not, each keeping every version of its text,
@@ -3164,6 +3267,319 @@ class Datasets:
3164
3267
  return [{k: v for k, v in self.get(did).items() if k != "body"} for did in created], None
3165
3268
 
3166
3269
 
3270
+ class FilterSets:
3271
+ """Filter sets as a Library row, versioned like eval groups (#336,
3272
+ docs/pipeline-model.md §18). The shape mirrors Datasets exactly -- a row
3273
+ per set, every version kept in `filter_set_versions` for a link to pin and
3274
+ a run to name -- without a dataset's prompt, scoring or import/export: a
3275
+ filter set's body is just { version, filters }. A run keeps the body of
3276
+ each set it links (Queue.submit's `filter_sets`), so it grades against what
3277
+ it was submitted with whatever the set holds later."""
3278
+
3279
+ def __init__(self, store: Store):
3280
+ self.store = store
3281
+ with store.lock, closing(sqlite3.connect(store.path)) as db, db:
3282
+ db.execute("CREATE TABLE IF NOT EXISTS filter_sets ("
3283
+ "id TEXT PRIMARY KEY, name TEXT NOT NULL, "
3284
+ "version INTEGER NOT NULL, body TEXT NOT NULL, "
3285
+ "created_at TEXT NOT NULL, updated_at TEXT NOT NULL, "
3286
+ "trash TEXT, trashed_at REAL, workspace TEXT)")
3287
+ # Every version of each set, numbered from 1, as eval_group_versions
3288
+ # keeps a group's (§17): `ran` freezes it for good once a run grades
3289
+ # with it, a pin freezes it while a pipeline holds the pin, and
3290
+ # version 1 is minted from the row's body at its first save, pin or
3291
+ # run. A set from before workspaces backfills to the default.
3292
+ db.execute("CREATE TABLE IF NOT EXISTS filter_set_versions ("
3293
+ "set_id TEXT NOT NULL, n INTEGER NOT NULL, body TEXT NOT NULL, "
3294
+ "created_at TEXT NOT NULL, edited_at REAL NOT NULL, "
3295
+ "ran INTEGER NOT NULL DEFAULT 0, PRIMARY KEY (set_id, n))")
3296
+ db.execute("UPDATE filter_sets SET workspace = ? WHERE workspace IS NULL", (store.default_ws,))
3297
+
3298
+ @staticmethod
3299
+ def _now():
3300
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
3301
+
3302
+ @staticmethod
3303
+ def _doc(r, body=True, versions=1):
3304
+ """A row as the API answers it: its summary, and its body with it, read
3305
+ as today's version. `version` is the save counter a write is arbitrated
3306
+ by; `versions` is its newest kept version's number, which a pin may
3307
+ name, 1 before any is kept."""
3308
+ parsed = upgrade_filter_set_body(json.loads(r["body"]))
3309
+ out = {"id": r["id"], "name": r["name"],
3310
+ "filters": len(parsed.get("filters") or []),
3311
+ "version": r["version"], "versions": versions, "updated": r["updated_at"]}
3312
+ if body:
3313
+ out["body"] = parsed
3314
+ return out
3315
+
3316
+ @staticmethod
3317
+ def _counts(db) -> dict:
3318
+ return dict(db.execute("SELECT set_id, MAX(n) FROM filter_set_versions GROUP BY set_id"))
3319
+
3320
+ def _row_doc(self, db, r, body=True):
3321
+ return self._doc(r, body, self._counts(db).get(r["id"], 1))
3322
+
3323
+ def _connect(self):
3324
+ db = sqlite3.connect(self.store.path)
3325
+ db.row_factory = sqlite3.Row
3326
+ return closing(db)
3327
+
3328
+ def _live(self, db, fid):
3329
+ return db.execute("SELECT * FROM filter_sets WHERE id = ? AND trash IS NULL AND workspace = ?",
3330
+ (fid, workspace_of(self.store))).fetchone()
3331
+
3332
+ def _names(self, db, but=None):
3333
+ return {r[0] for r in db.execute(
3334
+ "SELECT name FROM filter_sets WHERE trash IS NULL AND id IS NOT ? AND workspace = ?",
3335
+ (but, workspace_of(self.store)))}
3336
+
3337
+ def list(self) -> list:
3338
+ with self.store.lock, self._connect() as db:
3339
+ counts = self._counts(db)
3340
+ return [self._doc(r, False, counts.get(r["id"], 1)) for r in db.execute(
3341
+ "SELECT * FROM filter_sets WHERE trash IS NULL AND workspace = ? ORDER BY name COLLATE NOCASE, id",
3342
+ (workspace_of(self.store),))]
3343
+
3344
+ def get(self, fid):
3345
+ with self.store.lock, self._connect() as db:
3346
+ r = self._live(db, fid)
3347
+ return self._row_doc(db, r) if r else None
3348
+
3349
+ # ---- a set's versions (docs/pipeline-model.md §17, §18) --------------
3350
+
3351
+ @staticmethod
3352
+ def _head(db, fid):
3353
+ return db.execute("SELECT * FROM filter_set_versions WHERE set_id = ? "
3354
+ "ORDER BY n DESC LIMIT 1", (fid,)).fetchone()
3355
+
3356
+ def _mint(self, db, r):
3357
+ """The newest version of row [r], adding version 1 from its body first
3358
+ if it has none, edited when the row last was -- as a group's _mint."""
3359
+ head = self._head(db, r["id"])
3360
+ if head is not None:
3361
+ return head
3362
+ try:
3363
+ edited = calendar.timegm(time.strptime(r["updated_at"], "%Y-%m-%dT%H:%M:%SZ"))
3364
+ except ValueError:
3365
+ edited = 0
3366
+ db.execute("INSERT INTO filter_set_versions (set_id, n, body, created_at, edited_at) "
3367
+ "VALUES (?, 1, ?, ?, ?)", (r["id"], r["body"], r["updated_at"], edited))
3368
+ return self._head(db, r["id"])
3369
+
3370
+ def _pins(self, db) -> set:
3371
+ """Every (set, version) a stored pipeline pins, read in the caller's
3372
+ transaction from this workspace's workflows document."""
3373
+ row = db.execute("SELECT body FROM docs WHERE name = 'promptlab.workflows' AND workspace = ?",
3374
+ (workspace_of(self.store),)).fetchone()
3375
+ return filter_set_pins_in(json.loads(row[0])) if row and row[0] else set()
3376
+
3377
+ def _cut(self, db, fid, text):
3378
+ n = self._head(db, fid)["n"] + 1
3379
+ db.execute("INSERT INTO filter_set_versions (set_id, n, body, created_at, edited_at) "
3380
+ "VALUES (?, ?, ?, ?, ?)", (fid, n, text, self._now(), time.time()))
3381
+ return n
3382
+
3383
+ def _keep(self, db, r, text):
3384
+ """[text] as row [r]'s newest version: edited in place while no run has
3385
+ graded with it, no link pins it and it was edited in the last
3386
+ GROUP_IDLE_SECONDS, and a new version otherwise -- a group's _keep."""
3387
+ head = self._mint(db, r)
3388
+ if text == head["body"]:
3389
+ return
3390
+ fresh = time.time() - head["edited_at"] < GROUP_IDLE_SECONDS
3391
+ if fresh and not head["ran"] and (r["id"], head["n"]) not in self._pins(db):
3392
+ db.execute("UPDATE filter_set_versions SET body = ?, edited_at = ? WHERE set_id = ? AND n = ?",
3393
+ (text, time.time(), r["id"], head["n"]))
3394
+ else:
3395
+ self._cut(db, r["id"], text)
3396
+
3397
+ def versions(self, fid):
3398
+ """A set's versions, newest first, without their bodies -- version 1
3399
+ alone, read from the row, before any is kept -- or None."""
3400
+ with self.store.lock, self._connect() as db:
3401
+ r = self._live(db, fid)
3402
+ if r is None:
3403
+ return None
3404
+ pins = self._pins(db)
3405
+ rows = db.execute("SELECT * FROM filter_set_versions WHERE set_id = ? ORDER BY n DESC",
3406
+ (fid,)).fetchall()
3407
+ if not rows:
3408
+ return [{"n": 1, "created": r["created_at"], "ran": False, "pinned": (fid, 1) in pins,
3409
+ "fingerprint": fingerprint(json.loads(r["body"]))}]
3410
+ return [{"n": v["n"], "created": v["created_at"], "ran": bool(v["ran"]),
3411
+ "pinned": (fid, v["n"]) in pins, "fingerprint": fingerprint(json.loads(v["body"]))}
3412
+ for v in rows]
3413
+
3414
+ def version_body(self, fid, n):
3415
+ """Version [n]'s body, read as today's, or None."""
3416
+ with self.store.lock, self._connect() as db:
3417
+ r = self._live(db, fid)
3418
+ if r is None:
3419
+ return None
3420
+ v = db.execute("SELECT body FROM filter_set_versions WHERE set_id = ? AND n = ?",
3421
+ (fid, n)).fetchone()
3422
+ if v is None and n == 1 and self._head(db, fid) is None:
3423
+ v = (r["body"],)
3424
+ return upgrade_filter_set_body(json.loads(v[0])) if v else None
3425
+
3426
+ def restore_version(self, fid, n):
3427
+ """An older version's body as the newest version, and the row's:
3428
+ nothing is rewritten, so a run or a pin naming any version still reads
3429
+ what it named. Returns (doc, None)."""
3430
+ with self.store.lock, self._connect() as db, db:
3431
+ r = self._live(db, fid)
3432
+ if r is None:
3433
+ return None, (404, "no such filter set")
3434
+ head = self._mint(db, r)
3435
+ old = db.execute("SELECT body FROM filter_set_versions WHERE set_id = ? AND n = ?",
3436
+ (fid, n)).fetchone()
3437
+ if old is None:
3438
+ return None, (404, "no such version")
3439
+ if old["body"] != head["body"]:
3440
+ self._cut(db, fid, old["body"])
3441
+ db.execute("UPDATE filter_sets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
3442
+ (old["body"], r["version"] + 1, self._now(), fid))
3443
+ return self._row_doc(db, self._live(db, fid)), None
3444
+
3445
+ def mint_pinned(self, pins):
3446
+ """Version 1 of each pinned set that has none yet: a pin names a
3447
+ version, so the version has to be kept from the moment it does."""
3448
+ with self.store.lock, self._connect() as db, db:
3449
+ for fid, _ in pins:
3450
+ r = self._live(db, fid)
3451
+ if r is not None:
3452
+ self._mint(db, r)
3453
+
3454
+ def resolve(self, fid, pin=None):
3455
+ """The version a run submitted now grades with -- [pin], or the newest
3456
+ -- marked as graded with, in the same transaction, so no save can edit
3457
+ it in place between this and the run keeping its body. Returns
3458
+ (n, body), (None, why) for a pin the set has no version of, or None for
3459
+ a set the lab does not have -- as a group's resolve."""
3460
+ with self.store.lock, self._connect() as db, db:
3461
+ r = self._live(db, fid) if isinstance(fid, str) else None
3462
+ if r is None:
3463
+ return None
3464
+ head = self._mint(db, r)
3465
+ v = head if pin is None else db.execute(
3466
+ "SELECT * FROM filter_set_versions WHERE set_id = ? AND n = ?", (fid, pin)).fetchone()
3467
+ if v is None:
3468
+ return None, f"{r['name']} has no version {pin}"
3469
+ db.execute("UPDATE filter_set_versions SET ran = 1 WHERE set_id = ? AND n = ?", (fid, v["n"]))
3470
+ return v["n"], json.loads(v["body"])
3471
+
3472
+ def snapshot(self, fid):
3473
+ """The body a run submitted now grades against, and its fingerprint;
3474
+ None for a set the lab does not have."""
3475
+ with self.store.lock, self._connect() as db:
3476
+ r = self._live(db, fid) if isinstance(fid, str) else None
3477
+ if r is None:
3478
+ return None
3479
+ body = json.loads(r["body"])
3480
+ return body, fingerprint(body)
3481
+
3482
+ def _insert(self, db, name, body):
3483
+ fid = secrets.token_hex(6)
3484
+ now = self._now()
3485
+ db.execute("INSERT INTO filter_sets (id, name, version, body, created_at, updated_at, workspace) "
3486
+ "VALUES (?, ?, 1, ?, ?, ?, ?)",
3487
+ (fid, name, json.dumps(body), now, now, workspace_of(self.store)))
3488
+ return fid
3489
+
3490
+ def create(self, name, body=None):
3491
+ """A new filter set, blank unless [body] is given -- New, and Duplicate,
3492
+ which sends the source's body. A name already taken gets ` (2)`."""
3493
+ name, why = filter_set_name(name)
3494
+ if why:
3495
+ return None, (400, why)
3496
+ body = {"version": FILTER_SET_BODY_VERSION, "filters": []} if body is None \
3497
+ else upgrade_filter_set_body(body)
3498
+ why = filter_set_problem(body)
3499
+ if why:
3500
+ return None, (400, why)
3501
+ with self.store.lock, self._connect() as db, db:
3502
+ fid = self._insert(db, unique_filter_set_name(name, self._names(db)), body)
3503
+ return self.get(fid), None
3504
+
3505
+ def rename(self, fid, name):
3506
+ """A new label. The body and its version are untouched: a rename is not
3507
+ an edit a browser holding the body has to reload for."""
3508
+ name, why = filter_set_name(name)
3509
+ if why:
3510
+ return None, (400, why)
3511
+ with self.store.lock, self._connect() as db, db:
3512
+ if self._live(db, fid) is None:
3513
+ return None, (404, "no such filter set")
3514
+ if name.lower() in {n.lower() for n in self._names(db, but=fid)}:
3515
+ return None, (409, f"{name!r} is taken by another filter set")
3516
+ db.execute("UPDATE filter_sets SET name = ?, updated_at = ? WHERE id = ?",
3517
+ (name, self._now(), fid))
3518
+ return self.get(fid), None
3519
+
3520
+ def save(self, fid, version, body):
3521
+ """The body, written at [version] -- the one it began from. Returns
3522
+ ({version}, None), or (None, error); a stale version's error carries
3523
+ the current row."""
3524
+ if type(version) is not int:
3525
+ return None, (400, "a save names the version it began from")
3526
+ body = upgrade_filter_set_body(body)
3527
+ why = filter_set_problem(body)
3528
+ if why:
3529
+ return None, (400, why)
3530
+ with self.store.lock, self._connect() as db, db:
3531
+ r = self._live(db, fid)
3532
+ if r is None:
3533
+ return None, (404, "no such filter set")
3534
+ if r["version"] != version:
3535
+ return None, (409, {"current": self._row_doc(db, r)})
3536
+ text = json.dumps(body)
3537
+ self._keep(db, r, text)
3538
+ db.execute("UPDATE filter_sets SET body = ?, version = ?, updated_at = ? WHERE id = ?",
3539
+ (text, version + 1, self._now(), fid))
3540
+ return {"version": version + 1}, None
3541
+
3542
+ def remove(self, fid):
3543
+ """Into the trash, at once; Undo restores it. Returns (token, None)."""
3544
+ token = secrets.token_hex(6)
3545
+ with self.store.lock, self._connect() as db, db:
3546
+ if self._live(db, fid) is None:
3547
+ return None, (404, "no such filter set")
3548
+ db.execute("UPDATE filter_sets SET trash = ?, trashed_at = ? WHERE id = ?",
3549
+ (token, time.time(), fid))
3550
+ return token, None
3551
+
3552
+ def restore(self, token):
3553
+ """A trashed set back, under a new ` (2)` name if its own has been taken
3554
+ since. Returns (doc, None)."""
3555
+ with self.store.lock, self._connect() as db, db:
3556
+ r = db.execute("SELECT * FROM filter_sets WHERE trash = ? AND workspace = ?",
3557
+ (str(token), workspace_of(self.store))).fetchone()
3558
+ if r is None:
3559
+ return None, (404, "no such trash entry")
3560
+ name = unique_filter_set_name(r["name"], self._names(db, but=r["id"]))
3561
+ db.execute("UPDATE filter_sets SET trash = NULL, trashed_at = NULL, name = ? WHERE id = ?",
3562
+ (name, r["id"]))
3563
+ fid = r["id"]
3564
+ return self.get(fid), None
3565
+
3566
+ @staticmethod
3567
+ def _purge(db, where, args):
3568
+ db.execute(f"DELETE FROM filter_set_versions WHERE set_id IN (SELECT id FROM filter_sets WHERE {where})", args)
3569
+ db.execute(f"DELETE FROM filter_sets WHERE {where}", args)
3570
+
3571
+ def lazy_trash(self):
3572
+ """A filter-sets request empties what has been trashed longer than
3573
+ TRASH_SECONDS, as a datasets request does."""
3574
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3575
+ self._purge(db, "trash IS NOT NULL AND trashed_at < ?", (time.time() - TRASH_SECONDS,))
3576
+
3577
+ def empty_trash(self):
3578
+ """The startup sweep: a restart has nothing to undo."""
3579
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
3580
+ self._purge(db, "trash IS NOT NULL", ())
3581
+
3582
+
3167
3583
  def make_thumbs(sid, name):
3168
3584
  """JPEGs a browser can render, beside a stored file: a tile-size one
3169
3585
  for every image type, and a full-size render for the formats a browser
@@ -4457,6 +4873,12 @@ class Queue:
4457
4873
  # `dataset` column, read as its one group.
4458
4874
  if "groups" not in cols:
4459
4875
  db.execute("ALTER TABLE queue ADD COLUMN groups TEXT")
4876
+ # The body of each filter set a run links, by `<id>@<n>` (#336,
4877
+ # docs/pipeline-model.md §18): a run cleans and gates its replies
4878
+ # against these, never the set as it reads later. A row from before
4879
+ # them has none, which is a run that links none.
4880
+ if "filter_sets" not in cols:
4881
+ db.execute("ALTER TABLE queue ADD COLUMN filter_sets TEXT")
4460
4882
  # The workspace a run belongs to (docs/workspaces.md): History is
4461
4883
  # per-workspace, so the list and every id-keyed read filter by it.
4462
4884
  # The worker loop is the one global reader -- it grades every
@@ -4542,30 +4964,35 @@ class Queue:
4542
4964
 
4543
4965
  # ---- submit ---------------------------------------------------------
4544
4966
 
4545
- def submit(self, run: dict, dataset=None, rerun_of=None, groups=None):
4967
+ def submit(self, run: dict, dataset=None, rerun_of=None, groups=None, filter_sets=None):
4546
4968
  """
4547
4969
  A new queued run. `run` is the run document (docs/pipeline-model.md
4548
4970
  §5): the pipeline, the profiles it resolved to without their keys, its
4549
4971
  content's file list in order and with its repeats, and each eval
4550
4972
  group's version; `groups` is those versions' bodies, by `<id>@<n>`
4551
4973
  (§17), and `dataset` the one body a row from before them kept, which a
4552
- re-run of one carries on; `rerun_of` is the run a re-run was queued
4553
- from. Returns the row. Its items are that list, or the one inline
4554
- text, each through every scenario -- so the total is the list's
4555
- length, repeats and all, the same count the runner and the page make.
4974
+ re-run of one carries on; `filter_sets` is the body of each Library
4975
+ filter set the run links, by `<id>@<n>` (§18), so a run cleans and
4976
+ gates its replies against what it was submitted with; `rerun_of` is the
4977
+ run a re-run was queued from. Returns the row. Its items are that list,
4978
+ or the one inline text, each through every scenario -- so the total is
4979
+ the list's length, repeats and all, the same count the runner and the
4980
+ page make.
4556
4981
  """
4557
4982
  rid = secrets.token_hex(6)
4558
4983
  now = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
4559
4984
  total = len(run_items(run))
4560
4985
  with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
4561
4986
  db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
4562
- "snapshot, results, progress, totals, dataset, rerun_of, groups, workspace) "
4563
- "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
4987
+ "snapshot, results, progress, totals, dataset, rerun_of, groups, filter_sets, workspace) "
4988
+ "VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
4564
4989
  (rid, "queued", now, json.dumps(run), "[]",
4565
4990
  json.dumps({"current": None, "n": 0, "total": total}),
4566
4991
  json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
4567
4992
  None if dataset is None else json.dumps(dataset), rerun_of,
4568
- None if groups is None else json.dumps(groups), workspace_of(self.store)))
4993
+ None if groups is None else json.dumps(groups),
4994
+ None if filter_sets is None else json.dumps(filter_sets),
4995
+ workspace_of(self.store)))
4569
4996
  # Its prompts' uses, in the same transaction: a run is in the
4570
4997
  # library the moment it is queued, or not queued at all.
4571
4998
  if self.prompts is not None:
@@ -4614,6 +5041,21 @@ class Queue:
4614
5041
  return None
4615
5042
  return body if raw else upgrade_body(body)
4616
5043
 
5044
+ def filter_sets(self, rid, raw=False):
5045
+ """The bodies of the Library filter sets a readable run links, by
5046
+ `<id>@<n>` (§18), or None for a run that is not there or links none --
5047
+ what its replies were cleaned and gated against, whatever the sets hold
5048
+ now. A body kept at an earlier version reads as one of today's, unless
5049
+ [raw]."""
5050
+ if self.get(rid) is None:
5051
+ return None
5052
+ with self.lock, closing(sqlite3.connect(self.store.path)) as db:
5053
+ r = db.execute("SELECT filter_sets FROM queue WHERE id = ?", (rid,)).fetchone()
5054
+ kept = json.loads(r[0]) if r and r[0] is not None else None
5055
+ if kept is None:
5056
+ return None
5057
+ return kept if raw else {k: upgrade_filter_set_body(b) for k, b in kept.items()}
5058
+
4617
5059
  def _plugin_args(self, run):
4618
5060
  """The worker's --plugins, when the run recorded any; a run whose
4619
5061
  plugins have changed since is refused, naming them."""
@@ -4674,6 +5116,28 @@ class Queue:
4674
5116
  (rundir / "groups.json").write_text(json.dumps(kept))
4675
5117
  return ["--groups", str(rundir / "groups.json")], None
4676
5118
 
5119
+ def _filter_args(self, run, rundir):
5120
+ """The worker's --filter-sets for a run that links a Library filter set
5121
+ in Responses: the body it kept for each, by `<id>@<n>`, written beside
5122
+ the run document, so it cleans and gates its replies against what it
5123
+ was submitted with (§18). A private (`own`) set carries its body in the
5124
+ document, so it needs nothing here, and a run that links no Library set
5125
+ passes no --filter-sets. Returns (args, None) or (None, why)."""
5126
+ refs = [filter_set_ref(link) for link in filter_set_links(run["snapshot"])]
5127
+ keys = {group_key(r) for r in refs if r is not None}
5128
+ if not keys:
5129
+ return [], None
5130
+ kept = self.filter_sets(run["id"], raw=True)
5131
+ if kept is None:
5132
+ return None, "the run kept no filter-set bodies"
5133
+ missing = sorted({(r.get("name") or r.get("id")) for r in refs
5134
+ if r is not None and group_key(r) not in kept})
5135
+ if missing:
5136
+ return None, f"the run kept no body of the filter set {', '.join(missing)}"
5137
+ rundir.mkdir(parents=True, exist_ok=True)
5138
+ (rundir / "filter-sets.json").write_text(json.dumps(kept))
5139
+ return ["--filter-sets", str(rundir / "filter-sets.json")], None
5140
+
4677
5141
  def _behind(self, rid):
4678
5142
  """How many submissions stand between this one and the worker, by
4679
5143
  submit time -- what a waiting form names when it says what it is
@@ -4766,8 +5230,10 @@ class Queue:
4766
5230
  if err:
4767
5231
  return None, (409, err)
4768
5232
  # The group bodies the original kept: a re-run grades with exactly
4769
- # them, whatever the groups or the pins read now.
5233
+ # them, whatever the groups or the pins read now; its filter-set
5234
+ # bodies travel the same way, so it cleans and gates as the original did.
4770
5235
  kept = self._kept(rid)
5236
+ kept_filters = self.filter_sets(rid, raw=True)
4771
5237
  ref = evals_dataset(snap)
4772
5238
  body = None
4773
5239
  if ref is not None and kept is None:
@@ -4786,7 +5252,7 @@ class Queue:
4786
5252
  _, err = worker_destinations(snap)
4787
5253
  if err:
4788
5254
  return None, (403, err)
4789
- return self.submit(snap, body, rerun_of=rid, groups=kept), None
5255
+ return self.submit(snap, body, rerun_of=rid, groups=kept, filter_sets=kept_filters), None
4790
5256
 
4791
5257
  def rerun_item(self, rid, index):
4792
5258
  """
@@ -4821,7 +5287,10 @@ class Queue:
4821
5287
  plugins, err = self._plugin_args(run)
4822
5288
  if err:
4823
5289
  return None, (409, err)
4824
- dataset = [*dataset, *plugins]
5290
+ filters, err = self._filter_args(run, rundir)
5291
+ if err:
5292
+ return None, (409, err)
5293
+ dataset = [*dataset, *plugins, *filters]
4825
5294
  results = [r for r in run["results"] if r is not None]
4826
5295
  (rundir / "results.json").write_text(json.dumps(results))
4827
5296
  args = [NODE, str(HERE / "run-evals.js"), "--run", str(rundir / "run.json"),
@@ -5003,7 +5472,10 @@ class Queue:
5003
5472
  plugins, err = self._plugin_args(run)
5004
5473
  if err:
5005
5474
  return self._finish(rid, "failed", error=err)
5006
- dataset = [*dataset, *plugins]
5475
+ filters, err = self._filter_args(run, rundir)
5476
+ if err:
5477
+ return self._finish(rid, "failed", error=err)
5478
+ dataset = [*dataset, *plugins, *filters]
5007
5479
  results = [r for r in run["results"] if r is not None]
5008
5480
  (rundir / "results.json").write_text(json.dumps(results))
5009
5481
  progress_path = rundir / "progress.jsonl"
@@ -5364,10 +5836,12 @@ def worker_destinations(run: dict):
5364
5836
  # 12: a Contains metric's Ignore case holds item by item too, kept as written.
5365
5837
  # 13: an eval is a link to an eval group, or a group of the pipeline's own,
5366
5838
  # and the document has an overall pass rule (docs/pipeline-model.md §17).
5367
- PIPELINE_VERSION = 13
5839
+ # 14: a job's Responses stage links filter sets on a `filters` step, and the
5840
+ # inline Drop items / Reject rules migrate into a private one (§18, #339).
5841
+ PIPELINE_VERSION = 14
5368
5842
  # What a stored run may be: the current version, and the ones evals-core.ts's
5369
5843
  # upgradePipeline reads. A new submission is upgraded to the current one.
5370
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
5844
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14)
5371
5845
  TARGET_CAP = 4
5372
5846
  # Target steps whose words the Prompt library does not record as a use: they
5373
5847
  # ask no model (evals-core.ts's STEP_TYPES.echo).
@@ -5732,6 +6206,7 @@ if DATA_DIR:
5732
6206
  SOURCES = Sources(STORE)
5733
6207
  PROMPTS = Prompts(STORE)
5734
6208
  DATASETS = Datasets(STORE, PROMPTS)
6209
+ FILTER_SETS = FilterSets(STORE)
5735
6210
  PACKS = Packs(STORE)
5736
6211
  PLUGINS = Plugins(STORE)
5737
6212
  CONNECTIONS = Connections(STORE)
@@ -5740,7 +6215,8 @@ if DATA_DIR:
5740
6215
  QUEUE = Queue(STORE)
5741
6216
  QUEUE.prompts = PROMPTS
5742
6217
  else:
5743
- STORE = WORKSPACES = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
6218
+ STORE = WORKSPACES = SOURCES = PROMPTS = DATASETS = FILTER_SETS = PACKS = PLUGINS = \
6219
+ CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
5744
6220
 
5745
6221
 
5746
6222
  class NoRedirects(urllib.request.HTTPRedirectHandler):
@@ -5995,6 +6471,14 @@ class Handler(BaseHTTPRequestHandler):
5995
6471
  if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
5996
6472
  return self._send(404, b"not found", "text/plain")
5997
6473
  return self._json(200, QUEUE.dataset(run_id))
6474
+ if path.startswith("/api/queue/") and path.endswith("/filter-sets") and path.count("/") == 4:
6475
+ # The bodies of the filter sets a run cleaned and gated its replies
6476
+ # against, by `<id>@<n>` (§18); null for a run that linked none, as
6477
+ # /dataset and /groups answer.
6478
+ run_id = path.split("/")[3]
6479
+ if QUEUE is None or QUEUE.get(run_id, self.ws) is None:
6480
+ return self._send(404, b"not found", "text/plain")
6481
+ return self._json(200, QUEUE.filter_sets(run_id))
5998
6482
  if path.startswith("/api/queue/"):
5999
6483
  if QUEUE is None:
6000
6484
  return self._send(404, b"not found", "text/plain")
@@ -6093,6 +6577,8 @@ class Handler(BaseHTTPRequestHandler):
6093
6577
  })
6094
6578
  if path.startswith("/api/datasets"):
6095
6579
  return self._datasets_get(path)
6580
+ if path.startswith("/api/filter-sets"):
6581
+ return self._filter_sets_get(path)
6096
6582
  if path == "/api/prompts" or path.startswith("/api/prompts/"):
6097
6583
  return self._prompts_get(path)
6098
6584
  if path.startswith("/api/sources"):
@@ -6194,6 +6680,14 @@ class Handler(BaseHTTPRequestHandler):
6194
6680
  if err:
6195
6681
  return self._json(err[0], {"error": err[1]})
6196
6682
  return self._json(200, {"trash": token})
6683
+ if path.startswith("/api/filter-sets/"):
6684
+ parts = path.split("/")
6685
+ if FILTER_SETS is None or len(parts) != 4:
6686
+ return self._send(404, b"not found", "text/plain")
6687
+ token, err = FILTER_SETS.remove(parts[3])
6688
+ if err:
6689
+ return self._json(err[0], {"error": err[1]})
6690
+ return self._json(200, {"trash": token})
6197
6691
  if path.startswith("/api/prompts/"):
6198
6692
  parts = path.split("/")
6199
6693
  if PROMPTS is None or len(parts) != 4:
@@ -6244,6 +6738,14 @@ class Handler(BaseHTTPRequestHandler):
6244
6738
  if err:
6245
6739
  return self._json(err[0], {"error": err[1]})
6246
6740
  return self._json(200, dataset)
6741
+ if parts[2] == "filter-sets" and FILTER_SETS is not None:
6742
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
6743
+ if err:
6744
+ return err()
6745
+ fs, err = FILTER_SETS.rename(parts[3], payload.get("name") if isinstance(payload, dict) else None)
6746
+ if err:
6747
+ return self._json(err[0], {"error": err[1]})
6748
+ return self._json(200, fs)
6247
6749
  if parts[2] == "sources" and SOURCES is not None:
6248
6750
  payload = self._payload() or {}
6249
6751
  source, err = SOURCES.rename(parts[3], payload.get("name"))
@@ -6318,6 +6820,17 @@ class Handler(BaseHTTPRequestHandler):
6318
6820
  code, said = err
6319
6821
  return self._json(code, said if isinstance(said, dict) else {"error": said})
6320
6822
  return self._json(200, saved)
6823
+ if len(parts) == 4 and parts[:3] == ["", "api", "filter-sets"] and FILTER_SETS is not None:
6824
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
6825
+ if err:
6826
+ return err()
6827
+ if not isinstance(payload, dict):
6828
+ return self._json(400, {"error": "a save is { version, body }"})
6829
+ saved, err = FILTER_SETS.save(parts[3], payload.get("version"), payload.get("body"))
6830
+ if err:
6831
+ code, said = err
6832
+ return self._json(code, said if isinstance(said, dict) else {"error": said})
6833
+ return self._json(200, saved)
6321
6834
  if len(parts) != 4 or parts[:3] != ["", "api", "datasets"] or DATASETS is None:
6322
6835
  return self._send(404, b"not found", "text/plain")
6323
6836
  payload, err = self._dataset_payload(DATASET_CAP)
@@ -6356,6 +6869,8 @@ class Handler(BaseHTTPRequestHandler):
6356
6869
  return self._queue_action(path)
6357
6870
  if path.startswith("/api/datasets"):
6358
6871
  return self._datasets_post(path)
6872
+ if path.startswith("/api/filter-sets"):
6873
+ return self._filter_sets_post(path)
6359
6874
  if path == "/api/packs" or path.startswith("/api/packs/"):
6360
6875
  return self._packs_post(path)
6361
6876
  if path == "/api/workspaces" or path.startswith("/api/workspaces/"):
@@ -6629,10 +7144,14 @@ class Handler(BaseHTTPRequestHandler):
6629
7144
  for n, d in docs.items()}, self.ws)
6630
7145
  if stale is not None:
6631
7146
  return self._json(409, {"stale": stale})
6632
- # A pin names a group's version, so the group keeps that version from
6633
- # the moment a pipeline pins it (§17).
6634
- if "promptlab.workflows" in docs and DATASETS is not None:
6635
- DATASETS.mint_pinned(pins_in(docs["promptlab.workflows"].get("body")))
7147
+ # A pin names a group's or filter set's version, so it keeps that
7148
+ # version from the moment a pipeline pins it (§17, §18).
7149
+ if "promptlab.workflows" in docs:
7150
+ body = docs["promptlab.workflows"].get("body")
7151
+ if DATASETS is not None:
7152
+ DATASETS.mint_pinned(pins_in(body))
7153
+ if FILTER_SETS is not None:
7154
+ FILTER_SETS.mint_pinned(filter_set_pins_in(body))
6636
7155
  return self._json(200, {"versions": versions})
6637
7156
 
6638
7157
  # ---- The run queue (#530) ----------------------------------------------
@@ -6698,9 +7217,29 @@ class Handler(BaseHTTPRequestHandler):
6698
7217
  return self._json(400, {"error": f"the run cannot be queued: {body}"})
6699
7218
  ref["n"], ref["version"] = n, fingerprint(body)
6700
7219
  groups[group_key(ref)] = body
7220
+ # Each Library filter set the run links, resolved the same way (§18): a
7221
+ # pinned link to its pin, anything else to the set's newest version. The
7222
+ # reference records the version and its fingerprint, and the body is
7223
+ # kept with the row under `<id>@<n>` -- the run cleans and gates its
7224
+ # replies against that, and the version is frozen from now on. A private
7225
+ # (`own`) link carries its body in the document, so it reads no store.
7226
+ filter_sets = {}
7227
+ for link in filter_set_links(run):
7228
+ ref = filter_set_ref(link)
7229
+ if ref is None:
7230
+ continue
7231
+ pin = link.get("pin")
7232
+ got = FILTER_SETS.resolve(ref.get("id"), pin if type(pin) is int else None) if FILTER_SETS else None
7233
+ if got is None:
7234
+ return self._json(400, {"error": "the run links no filter set this lab has"})
7235
+ n, body = got
7236
+ if n is None:
7237
+ return self._json(400, {"error": f"the run cannot be queued: {body}"})
7238
+ ref["n"], ref["version"] = n, fingerprint(body)
7239
+ filter_sets[group_key(ref)] = body
6701
7240
  # The worker reads a body per group now (#233), so a run may link
6702
7241
  # several; each body is kept with the row under `<id>@<n>`.
6703
- return self._json(201, {"run": QUEUE.submit(run, groups=groups)})
7242
+ return self._json(201, {"run": QUEUE.submit(run, groups=groups, filter_sets=filter_sets or None)})
6704
7243
 
6705
7244
  # ---- Datasets ----------------------------------------------------------
6706
7245
  # Rows in the store (Datasets above). The page reads one by id; an export
@@ -6867,6 +7406,58 @@ class Handler(BaseHTTPRequestHandler):
6867
7406
  return self._json(201, dataset)
6868
7407
  return self._send(404, b"not found", "text/plain")
6869
7408
 
7409
+ # ---- Filter sets -------------------------------------------------------
7410
+ # Rows in the store (FilterSets above), read and written as datasets are
7411
+ # (docs/pipeline-model.md §18); no export or import, which datasets have for
7412
+ # CI. The page reads one by id, lists them, and keeps its versions.
7413
+
7414
+ def _filter_sets_get(self, path):
7415
+ if FILTER_SETS is None:
7416
+ return self._send(404, b"not found", "text/plain")
7417
+ FILTER_SETS.lazy_trash()
7418
+ parts = path.split("/")
7419
+ if len(parts) == 3:
7420
+ return self._json(200, {"filterSets": FILTER_SETS.list()})
7421
+ if len(parts) == 5 and parts[4] == "versions":
7422
+ got = FILTER_SETS.versions(parts[3])
7423
+ return self._json(200, {"versions": got}) if got is not None else self._send(404, b"not found", "text/plain")
7424
+ if len(parts) == 6 and parts[4] == "versions" and parts[5].isdigit():
7425
+ got = FILTER_SETS.version_body(parts[3], int(parts[5]))
7426
+ return self._json(200, {"n": int(parts[5]), "body": got}) if got is not None \
7427
+ else self._send(404, b"not found", "text/plain")
7428
+ if len(parts) == 4:
7429
+ d = FILTER_SETS.get(parts[3])
7430
+ return self._json(200, d) if d else self._send(404, b"not found", "text/plain")
7431
+ return self._send(404, b"not found", "text/plain")
7432
+
7433
+ def _filter_sets_post(self, path):
7434
+ if FILTER_SETS is None:
7435
+ return self._send(404, b"not found", "text/plain")
7436
+ parts = path.split("/")
7437
+ if len(parts) == 6 and parts[3] == "trash" and parts[5] == "restore":
7438
+ self._payload()
7439
+ fs, err = FILTER_SETS.restore(parts[4])
7440
+ if err:
7441
+ return self._json(err[0], {"error": err[1]})
7442
+ return self._json(200, fs)
7443
+ if len(parts) == 7 and parts[4] == "versions" and parts[6] == "restore" and parts[5].isdigit():
7444
+ self._payload()
7445
+ fs, err = FILTER_SETS.restore_version(parts[3], int(parts[5]))
7446
+ if err:
7447
+ return self._json(err[0], {"error": err[1]})
7448
+ return self._json(200, fs)
7449
+ if len(parts) == 3:
7450
+ payload, err = self._dataset_payload(DATASET_CAP, "a filter set")
7451
+ if err:
7452
+ return err()
7453
+ if not isinstance(payload, dict):
7454
+ return self._json(400, {"error": "a new filter set is { name, body? }"})
7455
+ fs, err = FILTER_SETS.create(payload.get("name"), payload.get("body"))
7456
+ if err:
7457
+ return self._json(err[0], {"error": err[1]})
7458
+ return self._json(201, fs)
7459
+ return self._send(404, b"not found", "text/plain")
7460
+
6870
7461
  def _queue_action(self, path):
6871
7462
  # Drained, so a keep-alive connection is not left holding the
6872
7463
  # page's `{}` in front of its next request.
@@ -7164,6 +7755,8 @@ def main():
7164
7755
  print(f"datasets: {len(DATASETS.list())} in the store", flush=True)
7165
7756
  else:
7166
7757
  print("datasets: (none -- datasets need DATA_DIR)", flush=True)
7758
+ if FILTER_SETS is not None:
7759
+ FILTER_SETS.empty_trash()
7167
7760
  if PROMPTS is not None:
7168
7761
  PROMPTS.empty_trash()
7169
7762
  # The runs from before the library, read into it once.