evals-lab 0.1.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
24
24
 
25
25
  import base64
26
26
  import calendar
27
+ import gzip
27
28
  import hashlib
28
29
  import hmac
29
30
  import io
@@ -247,6 +248,35 @@ TYPES = {
247
248
  ".png": "image/png",
248
249
  }
249
250
 
251
+ # What is gzipped for a client that asks (#225): text, which compresses five
252
+ # to ten times, and nothing already compressed. A body under GZIP_MIN goes as
253
+ # it is, since the header and the gzip frame would outweigh the saving.
254
+ COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
255
+ GZIP_MIN = 1024
256
+ # The built page's bundles, gzipped once: a new build is new names.
257
+ GZIPPED: dict = {}
258
+
259
+
260
+ def accepts_gzip(header: str) -> bool:
261
+ """Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
262
+ above 0. `gzip;q=0` is a refusal, not a request."""
263
+ star = False
264
+ for part in (header or "").split(","):
265
+ name, _, params = part.partition(";")
266
+ name, q = name.strip().lower(), 1.0
267
+ for p in params.split(";"):
268
+ k, _, v = p.partition("=")
269
+ if k.strip().lower() == "q":
270
+ try:
271
+ q = float(v)
272
+ except ValueError:
273
+ q = 0.0
274
+ if name == "gzip":
275
+ return q > 0
276
+ if name == "*":
277
+ star = q > 0
278
+ return star
279
+
250
280
 
251
281
  # What a Source may accept and ever be served back as -- deliberately not
252
282
  # TYPES: an upload that could come back as text/html is stored XSS, and the
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
469
499
  # A key taken off this list is no longer served or written, and its rows stay
470
500
  # in the store: a document is not migrated or deleted because nothing reads it.
471
501
  # promptlab.cases, .rules and their .base copies went that way when a dataset
472
- # became a row of its own (#86).
502
+ # became a row of its own (#86), and promptlab.mappings once a dataset named
503
+ # the Source it grades (#199).
473
504
  SYNCED = ("promptlab.workflows", "promptlab.profiles",
474
- "promptlab.versions", "promptlab.tokens", "promptlab.mappings")
505
+ "promptlab.versions", "promptlab.tokens")
475
506
  MAX_DOC = 8 * 1024 * 1024
476
507
  RUNS_PAGE = 25
477
508
  # A dataset request's caps, read from Content-Length before the body is, as a
@@ -1306,7 +1337,7 @@ class Sources:
1306
1337
 
1307
1338
  # ---- Datasets ----------------------------------------------------------------
1308
1339
  #
1309
- # A dataset is data a graded test names by id: its cases. (The prompt a new
1340
+ # A dataset is data a graded eval names by id: its cases. (The prompt a new
1310
1341
  # scenario starts from is the Prompt library's Default, below; a dataset from
1311
1342
  # before the library held one, and gave it to the library once.) It is one row in the
1312
1343
  # store's SQLite, its body one JSON document with a version that goes up by one
@@ -1325,44 +1356,140 @@ class Sources:
1325
1356
  # A lab from before this held a one-time import's `meta` row saying it ran;
1326
1357
  # it is left where it is, and nothing reads it.
1327
1358
 
1328
- DATASET_FIELDS = ("cases",)
1359
+ DATASET_FIELDS = ("version", "source", "cases")
1360
+ # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 5 was
1361
+ # told by its `source` alone, and earlier ones by neither.
1362
+ DATASET_BODY_VERSION = 6
1329
1363
  DATASET_NAME_MAX = 80
1330
- # The file forms Export writes and Import reads. Export writes version 4;
1331
- # Import reads it and versions 1 to 3, upgraded, and refuses anything else,
1364
+ # The file forms Export writes and Import reads. Export writes version 6;
1365
+ # Import reads it and versions 1 to 5, upgraded, and refuses anything else,
1332
1366
  # as a pipeline of another version is refused. Versions 1 to 3 carried a
1333
1367
  # prompt, which an import gives to the Prompt library.
1334
1368
  EXPORT_ONE = "evals-lab/dataset"
1335
1369
  EXPORT_ALL = "evals-lab/datasets"
1336
- EXPORT_VERSION = 4
1337
- IMPORT_VERSIONS = (1, 2, 3, 4)
1370
+ EXPORT_VERSION = 6
1371
+ IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
1338
1372
 
1339
1373
 
1340
1374
  def blank_dataset() -> dict:
1341
- return {"cases": []}
1375
+ return {"version": DATASET_BODY_VERSION, "source": None, "cases": []}
1376
+
1377
+
1378
+ # vocab: the names older versions gave a case's fields
1379
+ CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
1380
+ CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
1381
+
1382
+
1383
+ def _words(s) -> list:
1384
+ """evals-core.ts's words: the letters and digits of [s], lowercased."""
1385
+ return re.findall(r"[^\W_]+", str(s or "").lower())
1386
+
1387
+
1388
+ def _term_in(items, term) -> bool:
1389
+ """evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
1390
+ t = _words(term)
1391
+ if not t:
1392
+ return False
1393
+ for item in items:
1394
+ w = _words(item)
1395
+ if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
1396
+ return True
1397
+ return False
1398
+
1399
+
1400
+ def case_metrics(c: dict) -> list:
1401
+ """A version-4 case's expectations as the metrics that say the same:
1402
+ evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
1403
+ through fixtures/dataset-v6.json."""
1404
+ def strs(v):
1405
+ return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
1406
+ if c.get("discarded") is True:
1407
+ return [{"type": "discarded"}]
1408
+ out = []
1409
+ expect, allow = strs(c.get("expect")), strs(c.get("allow"))
1410
+ if expect:
1411
+ out.append({"type": "contains-all", "values": "\n".join(expect)})
1412
+ for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
1413
+ if strs(g):
1414
+ out.append({"type": "contains-any", "values": "\n".join(strs(g))})
1415
+ for t in strs(c.get("forbid")):
1416
+ # An exception excuses only the forbidden term inside it.
1417
+ except_ = [a for a in allow if _term_in([a], t)]
1418
+ out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
1419
+ whole = lambda v: v if type(v) is int else None
1420
+ lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
1421
+ if lo is not None or hi is not None:
1422
+ out.append({"type": "item-count", "min": lo, "max": hi})
1423
+ for t in strs(c.get("watch")):
1424
+ out.append({"type": "contains-any", "values": t, "weight": 0})
1425
+ return out
1426
+
1427
+
1428
+ def case_of_v4(c):
1429
+ """One case of any earlier version as a version-5 one: evals-core.ts's
1430
+ caseOfV4, in Python."""
1431
+ if not isinstance(c, dict):
1432
+ return c
1433
+ was = {}
1434
+ for k, v in c.items():
1435
+ key = CASE_RENAMED.get(k, k)
1436
+ if key not in was or key == k:
1437
+ was[key] = v
1438
+ if isinstance(was.get("item"), str):
1439
+ was.pop("filename", None)
1440
+ out = {}
1441
+ if "id" in was:
1442
+ out["id"] = was["id"]
1443
+ out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
1444
+ out["todo"] = was.get("todo") is True
1445
+ out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
1446
+ out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
1447
+ for k, v in was.items():
1448
+ if k not in out and k not in CASE_V4:
1449
+ out[k] = v
1450
+ return out
1451
+
1452
+
1453
+ # The metrics whose Ignore case version 6 made mean what it says for a reply
1454
+ # read as a list: evals-core.ts's CASE_FOLDING.
1455
+ CASE_FOLDING = ("contains", "contains-all", "contains-any")
1456
+
1457
+
1458
+ def case_of_v5(c):
1459
+ """A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
1460
+ Python. Each Contains metric says Ignore case, as version 5 matched a
1461
+ list's items whatever it said."""
1462
+ if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
1463
+ return c
1464
+ return {**c, "metrics": [{**m, "ignoreCase": True}
1465
+ if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
1466
+ else m for m in c["metrics"]]}
1342
1467
 
1343
1468
 
1344
1469
  def upgrade_body(body):
1345
- """An earlier body as today's: evals-core.ts's upgradeDatasetBody, in
1346
- Python. Version 1's `imageCases` are `cases`, each case's
1347
- `minTags`/`maxTags` its `minCount`/`maxCount`, and its `replays` and
1348
- `conformance` go (fixtures/replays.json holds the parser's tests).
1349
- Version 2's `rules` go -- they clean a job's answer, so they are the
1350
- job's -- and the terms a case watches for are its `watch`. Version 3's
1351
- `prompt` goes: the Prompt library holds prompts now. A caller that needs
1352
- the rules or the prompt takes them first (`body_rules`, `body_prompt`).
1353
- Anything else comes back as it was."""
1354
- if not isinstance(body, dict):
1470
+ """An earlier body as today's (version 6): evals-core.ts's
1471
+ upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
1472
+ its `replays` and `conformance` go (fixtures/replays.json holds the
1473
+ parser's tests). Version 2's `rules` go -- they clean a job's answer, so
1474
+ they are the job's. Version 3's `prompt` goes: the Prompt library holds
1475
+ prompts now. Version 4's case named its item `filename` and said what a
1476
+ good answer is in expectations; each becomes its metric, `why` the
1477
+ `note`, `traits` go, and the body names no Source yet. A caller that
1478
+ needs the rules or the prompt takes them first (`body_rules`,
1479
+ `body_prompt`). A body naming its Source is version 5, whose Contains
1480
+ metrics each come to say Ignore case (`case_of_v5`). A body saying it is
1481
+ version 6 comes back as it was; so does anything that is not a body."""
1482
+ if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
1355
1483
  return body
1356
- if "imageCases" not in body and "rules" not in body:
1357
- if "prompt" not in body:
1358
- return body
1359
- return {k: v for k, v in body.items() if k != "prompt"}
1360
- renamed = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch"} # vocab: older names
1484
+ if "source" in body:
1485
+ up = {"version": DATASET_BODY_VERSION, **body}
1486
+ if isinstance(body.get("cases"), list):
1487
+ up["cases"] = [case_of_v5(c) for c in body["cases"]]
1488
+ return up
1361
1489
  cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
1362
1490
  if not isinstance(cases, list):
1363
1491
  return body
1364
- cases = [{renamed.get(k, k): v for k, v in c.items()} if isinstance(c, dict) else c for c in cases]
1365
- return {"cases": canonical_cases(cases)}
1492
+ return {"version": DATASET_BODY_VERSION, "source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}
1366
1493
 
1367
1494
 
1368
1495
  def body_prompt(body):
@@ -1377,21 +1504,6 @@ def body_rules(body):
1377
1504
  return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
1378
1505
 
1379
1506
 
1380
- def canonical_cases(cases: list) -> list:
1381
- """Every case naming its file as `filename`: evals-core.ts's
1382
- canonicalCases, in Python. `old` is the key a set re-synced from an
1383
- app's own repository arrives with; the key keeps its place, so only its
1384
- spelling changes."""
1385
- old = "photo" # vocab: the older spelling of filename
1386
- out = []
1387
- for c in cases:
1388
- if isinstance(c, dict) and old in c:
1389
- c = {("filename" if k == old else k): v for k, v in c.items()
1390
- if not (k == old and "filename" in c)}
1391
- out.append(c)
1392
- return out
1393
-
1394
-
1395
1507
  def dataset_problem(body) -> str:
1396
1508
  """Why [body] is not a dataset's body, in one sentence, or ""."""
1397
1509
  if not isinstance(body, dict):
@@ -1402,8 +1514,14 @@ def dataset_problem(body) -> str:
1402
1514
  for k in DATASET_FIELDS:
1403
1515
  if k not in body:
1404
1516
  return f"a dataset's body has no \"{k}\""
1517
+ if body["version"] != DATASET_BODY_VERSION:
1518
+ return f"a dataset's body is version {DATASET_BODY_VERSION}"
1405
1519
  if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
1406
1520
  return "cases has to be a list of cases"
1521
+ src = body["source"]
1522
+ if src is not None and not (isinstance(src, dict) and isinstance(src.get("id"), str)
1523
+ and isinstance(src.get("name"), str)):
1524
+ return "a dataset names its Source as { id, name }, or null"
1407
1525
  return ""
1408
1526
 
1409
1527
 
@@ -1476,7 +1594,7 @@ def scenario_ref(sc, i):
1476
1594
  what version 4's upgrade gives one, by position."""
1477
1595
  sid = sc.get("id") if isinstance(sc, dict) else None
1478
1596
  name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
1479
- return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Scenario {i + 1}")
1597
+ return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
1480
1598
 
1481
1599
 
1482
1600
  class Prompts:
@@ -1632,27 +1750,28 @@ class Prompts:
1632
1750
  recorded once: a second call for the same run changes nothing."""
1633
1751
  # The caller's connection: a Row still reads by index, as it does.
1634
1752
  db.row_factory = sqlite3.Row
1635
- for i, sc in enumerate(run.get("scenarios") or []):
1636
- if not isinstance(sc, dict):
1637
- continue
1753
+ for i, sc, k, cell in cells_of(run):
1638
1754
  sid, sname = scenario_ref(sc, i)
1639
- for k, cell in enumerate(sc.get("stages") or []):
1640
- text = cell.get("prompt") if isinstance(cell, dict) else None
1641
- if not isinstance(text, str) or not text.strip():
1642
- continue
1643
- pid = version = None
1644
- ref = cell.get("from")
1645
- if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
1646
- pid = ref["id"]
1647
- hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
1648
- "ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
1649
- version = hit[0] if hit else self._cut(db, pid, text)
1650
- if pid is None:
1651
- hit = self._matching(db, text)
1652
- pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
1653
- db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
1654
- "scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
1655
- (pid, version, rid, sid, sname, k, at))
1755
+ # Echo's words are no wording under test: the item is its reply,
1756
+ # and over Prompt only the words are the reply itself.
1757
+ if isinstance(cell, dict) and cell.get("type") in UNRECORDED_STEPS:
1758
+ continue
1759
+ text = cell.get("prompt") if isinstance(cell, dict) else None
1760
+ if not isinstance(text, str) or not text.strip():
1761
+ continue
1762
+ pid = version = None
1763
+ ref = cell.get("from")
1764
+ if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
1765
+ pid = ref["id"]
1766
+ hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
1767
+ "ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
1768
+ version = hit[0] if hit else self._cut(db, pid, text)
1769
+ if pid is None:
1770
+ hit = self._matching(db, text)
1771
+ pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
1772
+ db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
1773
+ "scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
1774
+ (pid, version, rid, sid, sname, k, at))
1656
1775
 
1657
1776
  def backfill(self, runs):
1658
1777
  """The runs from before the library, read into it once. [runs] is
@@ -1787,6 +1906,11 @@ class Datasets:
1787
1906
  # so nothing is lost if a pipeline was missed.
1788
1907
  db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
1789
1908
  "dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
1909
+ # Each row's body as it was before the conversion below rewrote
1910
+ # it (#199): a version-4 case's expectations became metrics, and
1911
+ # the body it was typed as is kept, as the rules were.
1912
+ db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
1913
+ "dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
1790
1914
  # Rows from an earlier version are converted once, in place: a
1791
1915
  # dataset is typed in by hand and costly to re-enter, so it is
1792
1916
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -1805,6 +1929,8 @@ class Datasets:
1805
1929
  up = upgrade_body(body)
1806
1930
  if up is not body:
1807
1931
  self._archive(db, did, body)
1932
+ db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
1933
+ (did, raw, self._now()))
1808
1934
  if prompts is not None:
1809
1935
  given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
1810
1936
  db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
@@ -1873,7 +1999,6 @@ class Datasets:
1873
1999
  def _insert(self, db, name, body):
1874
2000
  did = secrets.token_hex(6)
1875
2001
  now = self._now()
1876
- body = {**body, "cases": canonical_cases(body["cases"])}
1877
2002
  db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
1878
2003
  "VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
1879
2004
  return did
@@ -1884,7 +2009,7 @@ class Datasets:
1884
2009
  name, why = dataset_name(name)
1885
2010
  if why:
1886
2011
  return None, (400, why)
1887
- body = blank_dataset() if body is None else body
2012
+ body = blank_dataset() if body is None else upgrade_body(body)
1888
2013
  why = dataset_problem(body)
1889
2014
  if why:
1890
2015
  return None, (400, why)
@@ -1913,10 +2038,12 @@ class Datasets:
1913
2038
  the current row."""
1914
2039
  if type(version) is not int:
1915
2040
  return None, (400, "a save names the version it began from")
2041
+ # A body of an earlier version -- from a page loaded before this one --
2042
+ # is read as today's, as an import is.
2043
+ body = upgrade_body(body)
1916
2044
  why = dataset_problem(body)
1917
2045
  if why:
1918
2046
  return None, (400, why)
1919
- body = {**body, "cases": canonical_cases(body["cases"])}
1920
2047
  with self.store.lock, self._connect() as db, db:
1921
2048
  r = self._live(db, did)
1922
2049
  if r is None:
@@ -2327,6 +2454,13 @@ class Packs:
2327
2454
  ds_ids = {}
2328
2455
  cuts = [] # (kind, id, the document as it was), kept before it changes
2329
2456
  for key, name, body, raw in pack["datasets"]:
2457
+ # The Source a dataset grades may be the pack's own, named by
2458
+ # its folder: pointed at the Source the pack made here.
2459
+ ref = body.get("source")
2460
+ if isinstance(ref, dict):
2461
+ sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
2462
+ if sid:
2463
+ body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
2330
2464
  did = owned.get(("dataset", key))
2331
2465
  current = DATASETS.get(did) if did else None
2332
2466
  if current:
@@ -2348,8 +2482,10 @@ class Packs:
2348
2482
  new_work = []
2349
2483
  for key, doc in pack["pipelines"]:
2350
2484
  doc = json.loads(json.dumps(doc))
2351
- tests = doc.get("tests")
2352
- for t in tests if isinstance(tests, list) else [tests]:
2485
+ # A pack written before version 11 spells its evals `tests`;
2486
+ # the page upgrades the pipeline as it reads it.
2487
+ evals = doc.get("evals", doc.get("tests"))
2488
+ for t in evals if isinstance(evals, list) else [evals]:
2353
2489
  ref = t.get("dataset") if isinstance(t, dict) else None
2354
2490
  if isinstance(ref, dict):
2355
2491
  did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
@@ -2486,7 +2622,7 @@ class Packs:
2486
2622
  #
2487
2623
  # A plugin is code, installed like a pack (docs/packs.md): a zip of a
2488
2624
  # manifest and the compiled JavaScript that registers what the lab lacks -- a
2489
- # kind of answer, a modifier, a test type, a connection type. Its code runs in
2625
+ # kind of answer, a modifier, an eval type, a connection type. Its code runs in
2490
2626
  # the page and in the runner, where the registries live; this server never
2491
2627
  # runs it. It reads the manifest's `registers` as data, so it can refuse two
2492
2628
  # plugins registering one id, and so a connection type's settings, chat path
@@ -2504,7 +2640,19 @@ PLUGIN_VERSIONS = (1,)
2504
2640
  PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
2505
2641
  PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
2506
2642
  PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
2507
- REGISTRIES = ("outputKinds", "modifiers", "testTypes", "connectionTypes")
2643
+ REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
2644
+ # evalTypes as a manifest written before pipeline version 11 spells it: a
2645
+ # plugin's own file, which the lab cannot upgrade, so it is read for good.
2646
+ OLD_REGISTRIES = {"testTypes": "evalTypes"}
2647
+
2648
+
2649
+ def plugin_registers(manifest: dict) -> dict:
2650
+ """A manifest's registers under today's names: testTypes read as evalTypes."""
2651
+ reg = dict(manifest.get("registers") or {})
2652
+ for old, now in OLD_REGISTRIES.items():
2653
+ if old in reg:
2654
+ reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
2655
+ return reg
2508
2656
  AUTH_WAYS = ("bearer", "x-api-key", "none")
2509
2657
 
2510
2658
 
@@ -2544,9 +2692,9 @@ def read_plugin(data: bytes):
2544
2692
  if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
2545
2693
  return None, "the plugin's entry names no JavaScript file it holds"
2546
2694
  reg = m.get("registers") or {}
2547
- if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
2695
+ if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
2548
2696
  return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
2549
- for k in ("outputKinds", "modifiers", "testTypes"):
2697
+ for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
2550
2698
  if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
2551
2699
  return None, f"registers.{k} is a list of ids"
2552
2700
  conns = reg.get("connectionTypes", [])
@@ -2569,24 +2717,64 @@ def read_plugin(data: bytes):
2569
2717
 
2570
2718
  def registered_ids(manifest: dict) -> set:
2571
2719
  """(registry, id) for everything a plugin's manifest says it registers."""
2572
- reg = manifest.get("registers") or {}
2573
- out = {(k, x) for k in ("outputKinds", "modifiers", "testTypes") for x in reg.get(k, [])}
2720
+ reg = plugin_registers(manifest)
2721
+ out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
2574
2722
  return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
2575
2723
 
2576
2724
 
2577
2725
  # ---- Connections: the lab's grants to outside services (#127) --------------
2578
2726
  #
2579
2727
  # One Google grant per lab, for Sources that read a Drive folder. The OAuth
2580
- # client is the deployment's (env vars), never the repository's or the
2581
- # store's; the refresh token the grant yields is the store's, in a table of
2582
- # its own, so /api/state -- which serves the synced documents to the page --
2583
- # can never carry it. The page is told only whether there is a grant and
2584
- # whose. The hosts are fixed here, not chosen by anything a request carries,
2585
- # and reached through OPENER: no redirects, no proxy from the environment.
2728
+ # client it signs in with is the deployment's (GOOGLE_* below), then the one
2729
+ # entered in Setup (GoogleApp), then the published app (PUBLIC_GOOGLE_APP),
2730
+ # as Microsoft's is (docs/power-automate.md). The refresh token the grant
2731
+ # yields is the store's, in a table of its own, so /api/state -- which serves
2732
+ # the synced documents to the page -- can never carry it, and nor can the
2733
+ # client's secret. The page is told only whether there is a grant and whose.
2734
+ # The hosts are fixed here, not chosen by anything a request carries, and
2735
+ # reached through OPENER: no redirects, no proxy from the environment.
2586
2736
  GOOGLE_CLIENT_ID = os.environ.get("GOOGLE_CLIENT_ID") or None
2587
2737
  GOOGLE_CLIENT_SECRET = os.environ.get("GOOGLE_CLIENT_SECRET") or None
2588
2738
  GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") or None
2589
- GOOGLE_APP_ID = os.environ.get("GOOGLE_APP_ID") or None
2739
+ # A deployment reached at its own address signs in with a Web application
2740
+ # client, whose redirect it registers; "installed" is Google's Desktop app.
2741
+ GOOGLE_CLIENT_TYPE = os.environ.get("GOOGLE_CLIENT_TYPE") or "web"
2742
+ GOOGLE_CLIENT_TYPES = ("installed", "web")
2743
+ # An OAuth client's id is its project's number, a dash, and Google's own
2744
+ # suffix: the number is the app id the Picker is told, so it is never asked.
2745
+ GOOGLE_CLIENT = re.compile(r"(\d+)-[0-9a-z]+\.apps\.googleusercontent\.com")
2746
+ GOOGLE_KEY = re.compile(r"[A-Za-z0-9_-]{30,60}")
2747
+
2748
+
2749
+ def read_published_google(path):
2750
+ """The published app from the file the npm package's build writes
2751
+ beside server.py: (app, None), (None, None) when there is no file, or
2752
+ (None, why) for a file that is not one."""
2753
+ if not path.is_file():
2754
+ return None, None
2755
+ try:
2756
+ got = json.loads(path.read_text("utf-8"))
2757
+ except (OSError, ValueError):
2758
+ return None, f"{path} is not JSON"
2759
+ if not (isinstance(got, dict) and isinstance(got.get("clientId"), str)
2760
+ and GOOGLE_CLIENT.fullmatch(got["clientId"]) and isinstance(got.get("clientSecret"), str)
2761
+ and got["clientSecret"] and isinstance(got.get("apiKey"), str) and GOOGLE_KEY.fullmatch(got["apiKey"])):
2762
+ return None, f"{path} is not a Google app: {{ clientId, clientSecret, apiKey }}"
2763
+ return {"clientId": got["clientId"], "clientSecret": got["clientSecret"],
2764
+ "apiKey": got["apiKey"], "clientType": "installed"}, None
2765
+
2766
+
2767
+ # The Desktop-app client published for every lab, so a lab installed from
2768
+ # npm signs in with nothing to set up. A Desktop client signs in back to any
2769
+ # port on this machine with no redirect registered, and Google does not hold
2770
+ # its secret to be one: it is in every copy of the package by design. It
2771
+ # serves only a lab reached on this machine -- one reached at its own address
2772
+ # brings a Web application client of its own. Never in the repository: the
2773
+ # package's build writes it beside server.py from the Package workflow's
2774
+ # PUBLIC_GOOGLE_APP secret (docs/google-drive.md § The published app), and
2775
+ # a checkout or the image has none.
2776
+ GOOGLE_APP_FILE = HERE / "google-app.json"
2777
+ PUBLIC_GOOGLE_APP, PUBLIC_GOOGLE_PROBLEM = read_published_google(GOOGLE_APP_FILE)
2590
2778
  # drive.file: only what the user picks in Google's Picker, and non-sensitive,
2591
2779
  # so the app can be published without Google's verification (#127).
2592
2780
  GOOGLE_SCOPE = "https://www.googleapis.com/auth/drive.file"
@@ -2600,6 +2788,94 @@ GOOGLE_CALLBACK = "/api/connections/google/callback"
2600
2788
  GOOGLE_STATE_SECONDS = 600
2601
2789
 
2602
2790
 
2791
+ def google_app():
2792
+ """The Google app this lab signs in with -- secret included, for the
2793
+ server's own use only -- and where it came from; None with none."""
2794
+ if GOOGLE_CLIENT_ID:
2795
+ return {"clientId": GOOGLE_CLIENT_ID, "clientSecret": GOOGLE_CLIENT_SECRET,
2796
+ "apiKey": GOOGLE_API_KEY, "clientType": GOOGLE_CLIENT_TYPE, "from": "env"}
2797
+ kept = GOOGLE.get() if GOOGLE is not None else None
2798
+ if kept:
2799
+ return {**kept, "from": "lab"}
2800
+ if PUBLIC_GOOGLE_APP:
2801
+ return {**PUBLIC_GOOGLE_APP, "from": "default"}
2802
+ return None
2803
+
2804
+
2805
+ def google_public(app):
2806
+ """What the page is told of an app: everything but its secret."""
2807
+ if app is None:
2808
+ return None
2809
+ m = GOOGLE_CLIENT.fullmatch(app.get("clientId") or "")
2810
+ return {"clientId": app.get("clientId"), "clientType": app.get("clientType"),
2811
+ "apiKey": app.get("apiKey"), "appId": m.group(1) if m else None,
2812
+ "hasSecret": bool(app.get("clientSecret")), "from": app["from"]}
2813
+
2814
+
2815
+ def loopback_origin(origin) -> bool:
2816
+ """Whether a page's origin is this machine: where a Desktop client may
2817
+ send the browser back to."""
2818
+ host = urllib.parse.urlsplit(origin).hostname or ""
2819
+ return host in ("localhost", "::1") or host.startswith("127.")
2820
+
2821
+
2822
+ class GoogleApp:
2823
+ """The Google app entered in Setup: a client and its secret, from the
2824
+ client file Google hands out, and the Picker's key. Kept so a lab with no
2825
+ deployment around it needs no environment variable; the secret is
2826
+ written here and never read back out to the page."""
2827
+
2828
+ KEYS = ("google.clientId", "google.clientSecret", "google.apiKey", "google.clientType")
2829
+
2830
+ def __init__(self, store: Store):
2831
+ self.store = store
2832
+ with store.lock, closing(sqlite3.connect(store.path)) as db, db:
2833
+ db.execute("CREATE TABLE IF NOT EXISTS settings (key TEXT PRIMARY KEY, value TEXT NOT NULL)")
2834
+
2835
+ def _read(self, db):
2836
+ rows = dict(db.execute("SELECT key, value FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS))
2837
+ return dict(zip(("clientId", "clientSecret", "apiKey", "clientType"),
2838
+ (rows.get(k) or None for k in self.KEYS)))
2839
+
2840
+ def get(self):
2841
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
2842
+ got = self._read(db)
2843
+ return {**got, "clientType": got["clientType"] or "installed"} if got["clientId"] else None
2844
+
2845
+ def set(self, payload):
2846
+ """Keeps what is sent, or with no client id forgets it all: (kept,
2847
+ error, whether the client changed). A secret not sent is kept while
2848
+ the client is the same one, and forgotten when it is not: a secret
2849
+ belongs to its client."""
2850
+ if not isinstance(payload, dict):
2851
+ return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
2852
+ client = payload.get("clientId")
2853
+ secret, key = payload.get("clientSecret"), payload.get("apiKey")
2854
+ ctype = payload.get("clientType") or "installed"
2855
+ if not isinstance(client, str) or not all(v is None or isinstance(v, str) for v in (secret, key)):
2856
+ return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
2857
+ client, key = client.strip(), (key or "").strip()
2858
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2859
+ had = self._read(db)
2860
+ changed = (had["clientId"] or "") != client
2861
+ if not client:
2862
+ db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
2863
+ return None, None, changed
2864
+ if not GOOGLE_CLIENT.fullmatch(client):
2865
+ return None, (400, "a client ID ends .apps.googleusercontent.com"), False
2866
+ if key and not GOOGLE_KEY.fullmatch(key):
2867
+ return None, (400, "that is not an API key"), False
2868
+ if ctype not in GOOGLE_CLIENT_TYPES:
2869
+ return None, (400, "a client is a Desktop app or a Web application"), False
2870
+ if secret is None:
2871
+ secret = None if changed else had["clientSecret"]
2872
+ secret = (secret or "").strip()
2873
+ db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
2874
+ db.executemany("INSERT INTO settings VALUES (?, ?)",
2875
+ [(k, v) for k, v in zip(self.KEYS, (client, secret, key, ctype)) if v])
2876
+ return self.get(), None, changed
2877
+
2878
+
2603
2879
  class Connections:
2604
2880
  """The lab's grants, held server-side; the page sees their state only."""
2605
2881
 
@@ -2619,7 +2895,8 @@ class Connections:
2619
2895
 
2620
2896
  @staticmethod
2621
2897
  def configured() -> bool:
2622
- return all((GOOGLE_CLIENT_ID, GOOGLE_CLIENT_SECRET, GOOGLE_API_KEY, GOOGLE_APP_ID))
2898
+ app = google_app()
2899
+ return bool(app and app.get("clientId") and app.get("clientSecret"))
2623
2900
 
2624
2901
  def _row(self):
2625
2902
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
@@ -2633,15 +2910,21 @@ class Connections:
2633
2910
 
2634
2911
  def start(self, origin: str):
2635
2912
  """The consent address the page sends the browser to."""
2913
+ app = google_app()
2636
2914
  if not self.configured():
2637
2915
  return None, (409, "Google is not configured for this lab")
2916
+ if app["clientType"] == "installed" and not loopback_origin(origin):
2917
+ return None, (409, "a Desktop app client signs in only on this machine: "
2918
+ "give this lab a Web application client in Configure…")
2638
2919
  state = secrets.token_urlsafe(24)
2639
2920
  now = time.time()
2640
2921
  with self.lock:
2641
2922
  self.states = {s: v for s, v in self.states.items() if v[0] > now}
2642
- self.states[state] = (now + GOOGLE_STATE_SECONDS, origin)
2923
+ # The app is kept with the state: the code Google sends back is
2924
+ # exchanged with the client that asked for it, whatever changes.
2925
+ self.states[state] = (now + GOOGLE_STATE_SECONDS, origin, app)
2643
2926
  return GOOGLE_CONSENT_URL + "?" + urllib.parse.urlencode({
2644
- "client_id": GOOGLE_CLIENT_ID, "redirect_uri": origin + GOOGLE_CALLBACK,
2927
+ "client_id": app["clientId"], "redirect_uri": origin + GOOGLE_CALLBACK,
2645
2928
  "response_type": "code", "scope": GOOGLE_SCOPE, "state": state,
2646
2929
  "access_type": "offline", "prompt": "consent", "include_granted_scopes": "true",
2647
2930
  }), None
@@ -2661,7 +2944,7 @@ class Connections:
2661
2944
  which never include the code or a token."""
2662
2945
  state = (query.get("state") or [""])[0]
2663
2946
  with self.lock:
2664
- lapses, origin = self.states.pop(state, (0, ""))
2947
+ lapses, origin, app = self.states.pop(state, (0, "", None))
2665
2948
  if lapses <= time.time():
2666
2949
  return "that sign-in had lapsed or was not this lab's; sign in again"
2667
2950
  if query.get("error"):
@@ -2671,7 +2954,7 @@ class Connections:
2671
2954
  return "Google sent no code back"
2672
2955
  try:
2673
2956
  got = self._call(GOOGLE_TOKEN_URL, {
2674
- "code": code, "client_id": GOOGLE_CLIENT_ID, "client_secret": GOOGLE_CLIENT_SECRET,
2957
+ "code": code, "client_id": app["clientId"], "client_secret": app["clientSecret"],
2675
2958
  "redirect_uri": origin + GOOGLE_CALLBACK, "grant_type": "authorization_code"})
2676
2959
  except (OSError, ValueError) as e:
2677
2960
  return f"the code could not be exchanged ({type(e).__name__})"
@@ -2689,6 +2972,12 @@ class Connections:
2689
2972
  time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())))
2690
2973
  return None
2691
2974
 
2975
+ def forget(self):
2976
+ """Drops the grant without a word to Google: the client it was made
2977
+ with is no longer this lab's, so the grant is no use to it."""
2978
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2979
+ db.execute("DELETE FROM connections WHERE id = 'google'")
2980
+
2692
2981
  def sign_out(self):
2693
2982
  """Forgets the grant, and asks Google to revoke it; a revoke that
2694
2983
  fails still forgets it here, which is what signing out means."""
@@ -2782,7 +3071,7 @@ class Plugins:
2782
3071
  return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
2783
3072
  "entry": json.loads(r["manifest"])["entry"],
2784
3073
  "description": json.loads(r["manifest"]).get("description") or "",
2785
- "registers": json.loads(r["manifest"]).get("registers") or {},
3074
+ "registers": plugin_registers(json.loads(r["manifest"])),
2786
3075
  "installed": r["installed_at"]} for r in self._rows()]
2787
3076
 
2788
3077
  def stamp(self) -> list:
@@ -2958,6 +3247,55 @@ def worker_refusal(code, stderr, env):
2958
3247
  return text if len(text) <= 2000 else text[:2000] + "…"
2959
3248
 
2960
3249
 
3250
+
3251
+ # A list of runs is read for its figures -- History's table, Home's recent
3252
+ # runs, Runs' progress -- and a run's results are mostly what Results alone
3253
+ # shows: every reply, every stage's request and answer, every item a score
3254
+ # found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
3255
+ # list carries a brief copy of its results, `brief` says so, and one run
3256
+ # (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
3257
+ # whether it ran; a brief score keeps its verdict and counts in place of its
3258
+ # lists. The replies go too: an eval over the whole run reads them, and the
3259
+ # page asks for that run whole rather than every list carrying them.
3260
+ BRIEF_SCORE = ("pass", "score", "points", "skipped")
3261
+ COUNTED = ("found", "missed", "invented")
3262
+
3263
+
3264
+ def brief_score(score):
3265
+ if not isinstance(score, dict):
3266
+ return score
3267
+ out = {k: score[k] for k in BRIEF_SCORE if k in score}
3268
+ for k in COUNTED:
3269
+ if isinstance(score.get(k), list):
3270
+ out[k] = len(score[k])
3271
+ return out
3272
+
3273
+
3274
+ def brief_cell(cell):
3275
+ if not isinstance(cell, dict):
3276
+ return cell
3277
+ out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
3278
+ if isinstance(cell.get("res"), dict):
3279
+ out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
3280
+ if isinstance(cell.get("scores"), dict):
3281
+ out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
3282
+ elif "scores" in cell:
3283
+ out["scores"] = cell["scores"]
3284
+ # A run from before version 6 kept its one test's score as `score`, which
3285
+ # the page reads as `scores.t1`: kept under its own name for that.
3286
+ if "score" in cell:
3287
+ out["score"] = brief_score(cell["score"])
3288
+ return out
3289
+
3290
+
3291
+ def brief_row(row):
3292
+ def item(it):
3293
+ if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
3294
+ return it
3295
+ return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
3296
+ return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
3297
+
3298
+
2961
3299
  class Queue:
2962
3300
  STATUS = ("queued", "running", "done", "incomplete", "cancelled",
2963
3301
  "failed", "interrupted")
@@ -3022,15 +3360,17 @@ class Queue:
3022
3360
  (rid,)).fetchone())
3023
3361
  return row if self._readable(row) else None
3024
3362
 
3025
- def list(self, limit=RUNS_PAGE, before=None):
3363
+ def list(self, limit=RUNS_PAGE, before=None, full=False):
3026
3364
  """Runs, newest first, and whether more follow. `before` is a
3027
3365
  `submittedAt` the page of runs stops at, so History can page through
3028
- them the way it pages the runs store."""
3366
+ them the way it pages the runs store. Each row is brief_row's unless
3367
+ `full` asks for the whole of it."""
3029
3368
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3030
3369
  rows = self._all(db)
3031
3370
  rows = [r for r in rows if before is None or r["submittedAt"] < before]
3032
3371
  rows.sort(key=lambda r: r["submittedAt"], reverse=True)
3033
- return rows[:limit], len(rows) > limit
3372
+ page = rows[:limit]
3373
+ return (page if full else [brief_row(r) for r in page]), len(rows) > limit
3034
3374
 
3035
3375
  def _set(self, rid, **fields):
3036
3376
  sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
@@ -3099,7 +3439,7 @@ class Queue:
3099
3439
  written beside the run document. A row queued before runs kept their
3100
3440
  dataset has none, and is pinned to the dataset as it reads now, once,
3101
3441
  so every later pass over it agrees. Returns (args, None) or (None, why)."""
3102
- ref = tests_dataset(run["snapshot"])
3442
+ ref = evals_dataset(run["snapshot"])
3103
3443
  if ref is None:
3104
3444
  return [], None
3105
3445
  body = self.dataset(run["id"], raw=True)
@@ -3589,24 +3929,24 @@ def worker_destinations(run: dict):
3589
3929
  base = api_base(str(conn.get("url") or "")) or api_base(OLLAMA)
3590
3930
  why = allowed(base, "")
3591
3931
  if why:
3592
- return None, f"Setup profile {name}: {why}"
3932
+ return None, f"Target profile {name}: {why}"
3593
3933
  profile = next((p for p in stored if isinstance(p, dict) and p.get("id") == pid), None)
3594
3934
  if profile is None:
3595
- return None, f"Setup profile {name} not found"
3935
+ return None, f"Target profile {name} not found"
3596
3936
  key = str(profile.get("key") or "").strip()
3597
3937
  if key:
3598
3938
  if not header_safe(key):
3599
- return None, f"Setup profile {name} has a key that cannot go in a header"
3939
+ return None, f"Target profile {name} has a key that cannot go in a header"
3600
3940
  why = allowed(base, key)
3601
3941
  if why:
3602
- return None, f"Setup profile {name}: {why}"
3942
+ return None, f"Target profile {name}: {why}"
3603
3943
  env[key_var(pid)] = key
3604
3944
  return env, None
3605
3945
 
3606
3946
 
3607
3947
  # A run document's fields, checked before it is accepted -- the rules
3608
3948
  # docs/pipeline-model.md §6 gives the server, in Python because the server is
3609
- # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or a test
3949
+ # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
3610
3950
  # means is the runner's to judge, and it fails the run with a sentence if it
3611
3951
  # cannot; what is here is what the server itself depends on: a version it
3612
3952
  # reads, its own cap, content it can count and copy, and connections that
@@ -3614,16 +3954,23 @@ def worker_destinations(run: dict):
3614
3954
  # 5: the pipeline and every chain have an id; version 4's scenarios keep
3615
3955
  # theirs, and an older run's chains are read as they are, by position.
3616
3956
  # 6: tests are an ordered list; a stored run's one test (or null) is read as
3617
- # a list of one (tests_dataset).
3957
+ # a list of one (evals_dataset).
3618
3958
  # 7: chains are jobs: the field is `jobs` and each job's type is "job".
3619
3959
  # 8: a job is its steps; the content is job 1's Attach Content step.
3620
- # 9: a test is Metrics; a Single Test or a Graded set is read converted.
3621
- PIPELINE_VERSION = 9
3960
+ # 9: an eval is Metrics; a Single Test or a Graded set is read converted.
3961
+ # 10: a job's steps are its stages, and each scenario is a target whose own
3962
+ # step in each job is what it sends there (docs/pipeline-model.md §16).
3963
+ # 11: `tests` are `evals`; nothing in an eval changes.
3964
+ # 12: a Contains metric's Ignore case holds item by item too, kept as written.
3965
+ PIPELINE_VERSION = 12
3622
3966
  # What a stored run may be: the current version, and the ones evals-core.ts's
3623
- # upgradePipeline reads. A new submission is always the current one.
3624
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9)
3625
- SCENARIO_CAP = 4
3626
- RUN_FIELDS = ("version", "id", "name", "jobs", "scenarios", "tests", "profiles", "comment", "plugins")
3967
+ # upgradePipeline reads. A new submission is upgraded to the current one.
3968
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
3969
+ TARGET_CAP = 4
3970
+ # Target steps whose words the Prompt library does not record as a use: they
3971
+ # ask no model (evals-core.ts's STEP_TYPES.echo).
3972
+ UNRECORDED_STEPS = {"echo"}
3973
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
3627
3974
 
3628
3975
 
3629
3976
  def content_of(doc):
@@ -3780,14 +4127,14 @@ def connection_problems(pid, conn, at, bad):
3780
4127
  bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
3781
4128
 
3782
4129
 
3783
- def tests_dataset(doc):
3784
- """The dataset reference a document's tests grade against, or None: the
3785
- first test that names one. A run grades against one dataset (the core's
3786
- validatePipeline says so). Reads a stored document of either shape --
3787
- version 5's list, or the one test before it -- since rows keep the
3788
- document they were submitted with."""
3789
- tests = doc.get("tests") if isinstance(doc, dict) else None
3790
- for t in tests if isinstance(tests, list) else [tests]:
4130
+ def evals_dataset(doc):
4131
+ """The dataset reference a document's evals grade against, or None: the
4132
+ first eval that names one. A run grades against one dataset (the core's
4133
+ validatePipeline says so). Reads a stored document of any shape --
4134
+ version 11's `evals`, the `tests` before it, version 5's list or the one
4135
+ test before that -- since rows keep the document they were submitted with."""
4136
+ evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
4137
+ for t in evals if isinstance(evals, list) else [evals]:
3791
4138
  if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
3792
4139
  return t["dataset"]
3793
4140
  return None
@@ -3823,16 +4170,20 @@ def upgrade_run(run):
3823
4170
  return up, None
3824
4171
 
3825
4172
 
3826
- def answers_from_item(run, sc, cell):
3827
- """Whether job 1 of a scenario is answered from its item's own text --
3828
- a local type (Echo) over a Source or Text -- so its prompt may be blank:
3829
- evals-core.ts's rule, mirrored. Over Prompt only the prompt is the reply."""
3830
- content = content_of(run)
3831
- if not isinstance(content, dict) or content.get("type") not in ("source", "text"):
3832
- return False
3833
- ref = cell.get("profile") or sc.get("profile")
3834
- conn = (run.get("profiles") or {}).get(ref.get("id")) if isinstance(ref, dict) else None
3835
- return isinstance(conn, dict) and conn.get("type") in LOCAL_CONNECTIONS
4173
+ def cells_of(run):
4174
+ """Every target's step in every job of a run, as (i, target, k, step):
4175
+ version 10's targets and their steps, or an earlier version's scenarios
4176
+ and their cells -- what the Prompt library records a use from, whichever
4177
+ version a stored run is."""
4178
+ if not isinstance(run, dict):
4179
+ return
4180
+ targets = run.get("targets") if isinstance(run.get("targets"), list) else run.get("scenarios")
4181
+ for i, t in enumerate(targets if isinstance(targets, list) else []):
4182
+ if not isinstance(t, dict):
4183
+ continue
4184
+ steps = t.get("steps") if isinstance(t.get("steps"), list) else t.get("stages")
4185
+ for k, step in enumerate(steps if isinstance(steps, list) else []):
4186
+ yield i, t, k, step
3836
4187
 
3837
4188
 
3838
4189
  def run_problems(run):
@@ -3877,36 +4228,41 @@ def run_problems(run):
3877
4228
  bad.append(f"{at} has to be an object")
3878
4229
  continue
3879
4230
  connection_problems(pid, conn, at, bad)
3880
- scenarios = run.get("scenarios")
3881
- if not isinstance(scenarios, list) or not 1 <= len(scenarios) <= SCENARIO_CAP:
3882
- return bad + [f"a run needs between one and {SCENARIO_CAP} scenarios"]
3883
- ids = [sc.get("id") for sc in scenarios if isinstance(sc, dict)]
3884
- for i, sc in enumerate(scenarios):
3885
- at = f"scenario {i + 1}"
3886
- if not isinstance(sc, dict):
4231
+ targets = run.get("targets")
4232
+ if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
4233
+ return bad + [f"a run needs between one and {TARGET_CAP} targets"]
4234
+ ids = [t.get("id") for t in targets if isinstance(t, dict)]
4235
+ for i, t in enumerate(targets):
4236
+ at = f"target {i + 1}"
4237
+ if not isinstance(t, dict):
3887
4238
  bad.append(f"{at} has to be an object")
3888
4239
  continue
3889
4240
  # The id the Prompt library records a use under (version 4).
3890
- if not isinstance(sc.get("id"), str) or not sc["id"].strip():
4241
+ if not isinstance(t.get("id"), str) or not t["id"].strip():
3891
4242
  bad.append(f"{at} has no id")
3892
- elif ids.count(sc["id"]) > 1:
3893
- bad.append(f"{at} has the id of another scenario")
3894
- stages = sc.get("stages")
3895
- if not isinstance(stages, list) or len(stages) != len(jobs):
3896
- bad.append(f"{at} has to have one prompt per job")
4243
+ elif ids.count(t["id"]) > 1:
4244
+ bad.append(f"{at} has the id of another target")
4245
+ steps = t.get("steps")
4246
+ if not isinstance(steps, list) or len(steps) != len(jobs):
4247
+ bad.append(f"{at} has to have one step per job")
3897
4248
  continue
3898
- refs = [sc.get("profile")] + [c.get("profile") for c in stages
3899
- if isinstance(c, dict) and c.get("profile") is not None]
3900
- for k, cell in enumerate(stages):
3901
- if not isinstance(cell, dict) or not isinstance(cell.get("prompt"), str) or (
3902
- not cell["prompt"].strip() and not (k == 0 and answers_from_item(run, sc, cell))):
4249
+ # A target with no profile is one whose steps ask none (Echo) or each
4250
+ # name their own; which steps need one is the core's to judge.
4251
+ refs = ([t["profile"]] if t.get("profile") is not None else []) + [
4252
+ st.get("profile") for st in steps if isinstance(st, dict) and st.get("profile") is not None]
4253
+ # Whether the words may be blank -- Echo answering from the item --
4254
+ # is the core's to judge, and the worker refuses the run in its words.
4255
+ for k, step in enumerate(steps):
4256
+ if not isinstance(step, dict) or not isinstance(step.get("type"), str):
4257
+ bad.append(f"{at}, job {k + 1} has to name what it sends")
4258
+ elif not isinstance(step.get("prompt"), str):
3903
4259
  bad.append(f"{at}, job {k + 1} has no prompt")
3904
- elif cell.get("from") is not None and not (
3905
- isinstance(cell["from"], dict) and isinstance(cell["from"].get("id"), str)):
4260
+ elif step.get("from") is not None and not (
4261
+ isinstance(step["from"], dict) and isinstance(step["from"].get("id"), str)):
3906
4262
  bad.append(f"{at}, job {k + 1} names the prompt it was picked from without an id")
3907
4263
  for ref in refs:
3908
4264
  if not isinstance(ref, dict) or ref.get("id") not in table:
3909
- bad.append(f"{at} names a Setup profile the run does not carry")
4265
+ bad.append(f"{at} names a Target profile the run does not carry")
3910
4266
  content = content_of(run)
3911
4267
  kind = content.get("type") if isinstance(content, dict) else None
3912
4268
  if kind == "source":
@@ -3923,9 +4279,9 @@ def run_problems(run):
3923
4279
  pass
3924
4280
  else:
3925
4281
  bad.append("a run needs a Source's files, some inline text, or Prompt only")
3926
- tests = run.get("tests")
3927
- if not isinstance(tests, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in tests):
3928
- bad.append("tests has to be a list of tests, each naming its type")
4282
+ evals = run.get("evals")
4283
+ if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
4284
+ bad.append("evals has to be a list of evals, each naming its type")
3929
4285
  if run.get("comment") is not None and not isinstance(run["comment"], str):
3930
4286
  bad.append("comment has to be text")
3931
4287
  return bad
@@ -3946,10 +4302,11 @@ if DATA_DIR:
3946
4302
  PLUGINS = Plugins(STORE)
3947
4303
  CONNECTIONS = Connections(STORE)
3948
4304
  MICROSOFT = MicrosoftApp(STORE)
4305
+ GOOGLE = GoogleApp(STORE)
3949
4306
  QUEUE = Queue(STORE)
3950
4307
  QUEUE.prompts = PROMPTS
3951
4308
  else:
3952
- STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = QUEUE = None
4309
+ STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
3953
4310
 
3954
4311
 
3955
4312
  class NoRedirects(urllib.request.HTTPRedirectHandler):
@@ -3989,11 +4346,25 @@ class Handler(BaseHTTPRequestHandler):
3989
4346
 
3990
4347
  # ---- helpers --------------------------------------------------------
3991
4348
 
3992
- def _send(self, code, body: bytes, ctype="application/json", headers=()):
4349
+ def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
4350
+ """`packed` keys a body that never changes under it -- a hashed
4351
+ bundle -- so it is gzipped once and kept, not on every request."""
3993
4352
  self.send_response(code)
3994
4353
  self.send_header("Content-Type", ctype)
3995
4354
  for k, v in headers:
3996
4355
  self.send_header(k, v)
4356
+ if ctype.startswith(COMPRESSIBLE):
4357
+ # Said whether or not this answer is compressed, so a cache
4358
+ # between here and the browser keys on it either way.
4359
+ self.send_header("Vary", "Accept-Encoding")
4360
+ if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
4361
+ if packed is None:
4362
+ body = gzip.compress(body, 6, mtime=0)
4363
+ else:
4364
+ if packed not in GZIPPED:
4365
+ GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
4366
+ body = GZIPPED[packed]
4367
+ self.send_header("Content-Encoding", "gzip")
3997
4368
  self.send_header("Content-Length", str(len(body)))
3998
4369
  self.end_headers()
3999
4370
  self.wfile.write(body)
@@ -4078,7 +4449,7 @@ class Handler(BaseHTTPRequestHandler):
4078
4449
  return
4079
4450
  path = self.path.split("?", 1)[0]
4080
4451
  # The lab is one page: a Connection and an Input make a scenario,
4081
- # Content and Tests are shared, and one to four scenarios run over the
4452
+ # Content and Evals are shared, and one to four scenarios run over the
4082
4453
  # content. It replaced the A/B page it was prototyped beside once it
4083
4454
  # carried everything that page did -- the graded set, the
4084
4455
  # Configuration group, the history -- rather than being left to rot
@@ -4098,14 +4469,14 @@ class Handler(BaseHTTPRequestHandler):
4098
4469
  if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
4099
4470
  return self._send(404, b"not found", "text/plain")
4100
4471
  return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
4101
- (("Cache-Control", "public, max-age=31536000, immutable"),))
4472
+ (("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
4102
4473
  if path == "/api/config":
4103
4474
  # The lab's own limits, for the page to disable rather than
4104
4475
  # hard-code: how many scenarios a run may have.
4105
4476
  # And the Microsoft app a flow is read with, when there is one.
4106
4477
  # The app may be entered in Setup unless the deployment names
4107
4478
  # its own, and only in a lab with a store to keep it in.
4108
- return self._json(200, {"scenarioCap": SCENARIO_CAP, "microsoft": microsoft_config(),
4479
+ return self._json(200, {"targetCap": TARGET_CAP, "microsoft": microsoft_config(),
4109
4480
  "microsoftEditable": MICROSOFT is not None and not M365_CLIENT_ID})
4110
4481
  if path == "/api/state":
4111
4482
  if STORE is None:
@@ -4122,14 +4493,14 @@ class Handler(BaseHTTPRequestHandler):
4122
4493
  except ValueError:
4123
4494
  return self._json(400, {"error": "limit has to be a number"})
4124
4495
  before = (query.get("before") or [None])[0]
4125
- runs, more = QUEUE.list(limit, before)
4496
+ runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
4126
4497
  return self._json(200, {"runs": runs, "more": more})
4127
4498
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
4128
4499
  # The dataset body a graded run was submitted against, which is
4129
4500
  # what its verdicts were graded by; the list never carries it.
4130
4501
  #
4131
4502
  # A run that is there but kept no copy -- one submitted before the
4132
- # lab kept them, or one with no graded test -- answers null, not
4503
+ # lab kept them, or one with no graded eval -- answers null, not
4133
4504
  # 404. It is not an error: the page reads it as "use the dataset
4134
4505
  # as it is now", and a 404 put a red line in the console every
4135
4506
  # time such a run was opened, which is the console people are
@@ -4161,6 +4532,10 @@ class Handler(BaseHTTPRequestHandler):
4161
4532
  if CONNECTIONS is None:
4162
4533
  return self._send(404, b"not found", "text/plain")
4163
4534
  return self._json(200, {"connections": CONNECTIONS.list()})
4535
+ if path == "/api/google":
4536
+ if GOOGLE is None:
4537
+ return self._send(404, b"not found", "text/plain")
4538
+ return self._json(200, {"google": google_public(google_app()), "editable": not GOOGLE_CLIENT_ID})
4164
4539
  if path == GOOGLE_CALLBACK:
4165
4540
  return self._google_callback()
4166
4541
  if path.startswith("/plugins/"):
@@ -4381,6 +4756,17 @@ class Handler(BaseHTTPRequestHandler):
4381
4756
  parts = self.path.split("?", 1)[0].split("/")
4382
4757
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
4383
4758
  return self._prompts_put(parts[3])
4759
+ if parts == ["", "api", "google"] and GOOGLE is not None:
4760
+ if GOOGLE_CLIENT_ID:
4761
+ return self._json(409, {"error": "this lab's Google app is its deployment's (GOOGLE_CLIENT_ID)"})
4762
+ _, err, changed = GOOGLE.set(self._payload())
4763
+ if err:
4764
+ return self._json(err[0], {"error": err[1]})
4765
+ # A grant is its client's: one made with another client cannot be
4766
+ # refreshed with this one, so it goes rather than failing later.
4767
+ if changed and CONNECTIONS is not None:
4768
+ CONNECTIONS.forget()
4769
+ return self._json(200, {"google": google_public(google_app()), "editable": True})
4384
4770
  if parts == ["", "api", "microsoft"] and MICROSOFT is not None:
4385
4771
  if M365_CLIENT_ID:
4386
4772
  return self._json(409, {"error": "this lab's Microsoft app is its deployment's (M365_CLIENT_ID)"})
@@ -4648,14 +5034,14 @@ class Handler(BaseHTTPRequestHandler):
4648
5034
  # A graded run keeps the body of the dataset it names, as it reads
4649
5035
  # now, and records that body's fingerprint on the reference it
4650
5036
  # belongs to: the worker grades against that and nothing else.
4651
- ref = tests_dataset(run)
5037
+ ref = evals_dataset(run)
4652
5038
  body = None
4653
5039
  if ref is not None:
4654
5040
  snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
4655
5041
  if snap is None:
4656
- return self._json(400, {"error": "the run's graded test names no dataset this lab has"})
5042
+ return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
4657
5043
  body, version = snap
4658
- for t in run["tests"]:
5044
+ for t in run["evals"]:
4659
5045
  if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
4660
5046
  t["dataset"]["version"] = version
4661
5047
  return self._json(201, {"run": QUEUE.submit(run, body)})
@@ -5065,6 +5451,10 @@ def main():
5065
5451
  if not (DEMO / "manifest.json").is_file():
5066
5452
  raise SystemExit(f"no {DEMO / 'manifest.json'} -- the demo pack sits beside server.py, "
5067
5453
  "and the image copies it there")
5454
+ # A published app the build wrote and that cannot be read is a broken
5455
+ # package, not one without the app: said at once, not at Sign in.
5456
+ if PUBLIC_GOOGLE_PROBLEM:
5457
+ raise SystemExit(PUBLIC_GOOGLE_PROBLEM)
5068
5458
  print(f"prompt-lab on {HOST}:{PORT} -> {OLLAMA} by default", flush=True)
5069
5459
  print(f"samples: {SAMPLES}", flush=True)
5070
5460
  print(f"page: {WEB_DIST}", flush=True)