evals-lab 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/server.py CHANGED
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
24
24
 
25
25
  import base64
26
26
  import calendar
27
+ import gzip
27
28
  import hashlib
28
29
  import hmac
29
30
  import io
@@ -247,6 +248,35 @@ TYPES = {
247
248
  ".png": "image/png",
248
249
  }
249
250
 
251
+ # What is gzipped for a client that asks (#225): text, which compresses five
252
+ # to ten times, and nothing already compressed. A body under GZIP_MIN goes as
253
+ # it is, since the header and the gzip frame would outweigh the saving.
254
+ COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
255
+ GZIP_MIN = 1024
256
+ # The built page's bundles, gzipped once: a new build is new names.
257
+ GZIPPED: dict = {}
258
+
259
+
260
+ def accepts_gzip(header: str) -> bool:
261
+ """Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
262
+ above 0. `gzip;q=0` is a refusal, not a request."""
263
+ star = False
264
+ for part in (header or "").split(","):
265
+ name, _, params = part.partition(";")
266
+ name, q = name.strip().lower(), 1.0
267
+ for p in params.split(";"):
268
+ k, _, v = p.partition("=")
269
+ if k.strip().lower() == "q":
270
+ try:
271
+ q = float(v)
272
+ except ValueError:
273
+ q = 0.0
274
+ if name == "gzip":
275
+ return q > 0
276
+ if name == "*":
277
+ star = q > 0
278
+ return star
279
+
250
280
 
251
281
  # What a Source may accept and ever be served back as -- deliberately not
252
282
  # TYPES: an upload that could come back as text/html is stored XSS, and the
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
469
499
  # A key taken off this list is no longer served or written, and its rows stay
470
500
  # in the store: a document is not migrated or deleted because nothing reads it.
471
501
  # promptlab.cases, .rules and their .base copies went that way when a dataset
472
- # became a row of its own (#86).
502
+ # became a row of its own (#86), and promptlab.mappings once a dataset named
503
+ # the Source it grades (#199).
473
504
  SYNCED = ("promptlab.workflows", "promptlab.profiles",
474
- "promptlab.versions", "promptlab.tokens", "promptlab.mappings")
505
+ "promptlab.versions", "promptlab.tokens")
475
506
  MAX_DOC = 8 * 1024 * 1024
476
507
  RUNS_PAGE = 25
477
508
  # A dataset request's caps, read from Content-Length before the body is, as a
@@ -1306,7 +1337,7 @@ class Sources:
1306
1337
 
1307
1338
  # ---- Datasets ----------------------------------------------------------------
1308
1339
  #
1309
- # A dataset is data a graded test names by id: its cases. (The prompt a new
1340
+ # A dataset is data a graded eval names by id: its cases. (The prompt a new
1310
1341
  # scenario starts from is the Prompt library's Default, below; a dataset from
1311
1342
  # before the library held one, and gave it to the library once.) It is one row in the
1312
1343
  # store's SQLite, its body one JSON document with a version that goes up by one
@@ -1325,44 +1356,140 @@ class Sources:
1325
1356
  # A lab from before this held a one-time import's `meta` row saying it ran;
1326
1357
  # it is left where it is, and nothing reads it.
1327
1358
 
1328
- DATASET_FIELDS = ("cases",)
1359
+ DATASET_FIELDS = ("version", "source", "cases")
1360
+ # A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 5 was
1361
+ # told by its `source` alone, and earlier ones by neither.
1362
+ DATASET_BODY_VERSION = 6
1329
1363
  DATASET_NAME_MAX = 80
1330
- # The file forms Export writes and Import reads. Export writes version 4;
1331
- # Import reads it and versions 1 to 3, upgraded, and refuses anything else,
1364
+ # The file forms Export writes and Import reads. Export writes version 6;
1365
+ # Import reads it and versions 1 to 5, upgraded, and refuses anything else,
1332
1366
  # as a pipeline of another version is refused. Versions 1 to 3 carried a
1333
1367
  # prompt, which an import gives to the Prompt library.
1334
1368
  EXPORT_ONE = "evals-lab/dataset"
1335
1369
  EXPORT_ALL = "evals-lab/datasets"
1336
- EXPORT_VERSION = 4
1337
- IMPORT_VERSIONS = (1, 2, 3, 4)
1370
+ EXPORT_VERSION = 6
1371
+ IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
1338
1372
 
1339
1373
 
1340
1374
  def blank_dataset() -> dict:
1341
- return {"cases": []}
1375
+ return {"version": DATASET_BODY_VERSION, "source": None, "cases": []}
1376
+
1377
+
1378
+ # vocab: the names older versions gave a case's fields
1379
+ CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
1380
+ CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
1381
+
1382
+
1383
+ def _words(s) -> list:
1384
+ """evals-core.ts's words: the letters and digits of [s], lowercased."""
1385
+ return re.findall(r"[^\W_]+", str(s or "").lower())
1386
+
1387
+
1388
+ def _term_in(items, term) -> bool:
1389
+ """evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
1390
+ t = _words(term)
1391
+ if not t:
1392
+ return False
1393
+ for item in items:
1394
+ w = _words(item)
1395
+ if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
1396
+ return True
1397
+ return False
1398
+
1399
+
1400
+ def case_metrics(c: dict) -> list:
1401
+ """A version-4 case's expectations as the metrics that say the same:
1402
+ evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
1403
+ through fixtures/dataset-v6.json."""
1404
+ def strs(v):
1405
+ return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
1406
+ if c.get("discarded") is True:
1407
+ return [{"type": "discarded"}]
1408
+ out = []
1409
+ expect, allow = strs(c.get("expect")), strs(c.get("allow"))
1410
+ if expect:
1411
+ out.append({"type": "contains-all", "values": "\n".join(expect)})
1412
+ for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
1413
+ if strs(g):
1414
+ out.append({"type": "contains-any", "values": "\n".join(strs(g))})
1415
+ for t in strs(c.get("forbid")):
1416
+ # An exception excuses only the forbidden term inside it.
1417
+ except_ = [a for a in allow if _term_in([a], t)]
1418
+ out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
1419
+ whole = lambda v: v if type(v) is int else None
1420
+ lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
1421
+ if lo is not None or hi is not None:
1422
+ out.append({"type": "item-count", "min": lo, "max": hi})
1423
+ for t in strs(c.get("watch")):
1424
+ out.append({"type": "contains-any", "values": t, "weight": 0})
1425
+ return out
1426
+
1427
+
1428
+ def case_of_v4(c):
1429
+ """One case of any earlier version as a version-5 one: evals-core.ts's
1430
+ caseOfV4, in Python."""
1431
+ if not isinstance(c, dict):
1432
+ return c
1433
+ was = {}
1434
+ for k, v in c.items():
1435
+ key = CASE_RENAMED.get(k, k)
1436
+ if key not in was or key == k:
1437
+ was[key] = v
1438
+ if isinstance(was.get("item"), str):
1439
+ was.pop("filename", None)
1440
+ out = {}
1441
+ if "id" in was:
1442
+ out["id"] = was["id"]
1443
+ out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
1444
+ out["todo"] = was.get("todo") is True
1445
+ out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
1446
+ out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
1447
+ for k, v in was.items():
1448
+ if k not in out and k not in CASE_V4:
1449
+ out[k] = v
1450
+ return out
1451
+
1452
+
1453
+ # The metrics whose Ignore case version 6 made mean what it says for a reply
1454
+ # read as a list: evals-core.ts's CASE_FOLDING.
1455
+ CASE_FOLDING = ("contains", "contains-all", "contains-any")
1456
+
1457
+
1458
+ def case_of_v5(c):
1459
+ """A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
1460
+ Python. Each Contains metric says Ignore case, as version 5 matched a
1461
+ list's items whatever it said."""
1462
+ if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
1463
+ return c
1464
+ return {**c, "metrics": [{**m, "ignoreCase": True}
1465
+ if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
1466
+ else m for m in c["metrics"]]}
1342
1467
 
1343
1468
 
1344
1469
  def upgrade_body(body):
1345
- """An earlier body as today's: evals-core.ts's upgradeDatasetBody, in
1346
- Python. Version 1's `imageCases` are `cases`, each case's
1347
- `minTags`/`maxTags` its `minCount`/`maxCount`, and its `replays` and
1348
- `conformance` go (fixtures/replays.json holds the parser's tests).
1349
- Version 2's `rules` go -- they clean a job's answer, so they are the
1350
- job's -- and the terms a case watches for are its `watch`. Version 3's
1351
- `prompt` goes: the Prompt library holds prompts now. A caller that needs
1352
- the rules or the prompt takes them first (`body_rules`, `body_prompt`).
1353
- Anything else comes back as it was."""
1354
- if not isinstance(body, dict):
1470
+ """An earlier body as today's (version 6): evals-core.ts's
1471
+ upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
1472
+ its `replays` and `conformance` go (fixtures/replays.json holds the
1473
+ parser's tests). Version 2's `rules` go -- they clean a job's answer, so
1474
+ they are the job's. Version 3's `prompt` goes: the Prompt library holds
1475
+ prompts now. Version 4's case named its item `filename` and said what a
1476
+ good answer is in expectations; each becomes its metric, `why` the
1477
+ `note`, `traits` go, and the body names no Source yet. A caller that
1478
+ needs the rules or the prompt takes them first (`body_rules`,
1479
+ `body_prompt`). A body naming its Source is version 5, whose Contains
1480
+ metrics each come to say Ignore case (`case_of_v5`). A body saying it is
1481
+ version 6 comes back as it was; so does anything that is not a body."""
1482
+ if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
1355
1483
  return body
1356
- if "imageCases" not in body and "rules" not in body:
1357
- if "prompt" not in body:
1358
- return body
1359
- return {k: v for k, v in body.items() if k != "prompt"}
1360
- renamed = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch"} # vocab: older names
1484
+ if "source" in body:
1485
+ up = {"version": DATASET_BODY_VERSION, **body}
1486
+ if isinstance(body.get("cases"), list):
1487
+ up["cases"] = [case_of_v5(c) for c in body["cases"]]
1488
+ return up
1361
1489
  cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
1362
1490
  if not isinstance(cases, list):
1363
1491
  return body
1364
- cases = [{renamed.get(k, k): v for k, v in c.items()} if isinstance(c, dict) else c for c in cases]
1365
- return {"cases": canonical_cases(cases)}
1492
+ return {"version": DATASET_BODY_VERSION, "source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}
1366
1493
 
1367
1494
 
1368
1495
  def body_prompt(body):
@@ -1377,21 +1504,6 @@ def body_rules(body):
1377
1504
  return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
1378
1505
 
1379
1506
 
1380
- def canonical_cases(cases: list) -> list:
1381
- """Every case naming its file as `filename`: evals-core.ts's
1382
- canonicalCases, in Python. `old` is the key a set re-synced from an
1383
- app's own repository arrives with; the key keeps its place, so only its
1384
- spelling changes."""
1385
- old = "photo" # vocab: the older spelling of filename
1386
- out = []
1387
- for c in cases:
1388
- if isinstance(c, dict) and old in c:
1389
- c = {("filename" if k == old else k): v for k, v in c.items()
1390
- if not (k == old and "filename" in c)}
1391
- out.append(c)
1392
- return out
1393
-
1394
-
1395
1507
  def dataset_problem(body) -> str:
1396
1508
  """Why [body] is not a dataset's body, in one sentence, or ""."""
1397
1509
  if not isinstance(body, dict):
@@ -1402,8 +1514,14 @@ def dataset_problem(body) -> str:
1402
1514
  for k in DATASET_FIELDS:
1403
1515
  if k not in body:
1404
1516
  return f"a dataset's body has no \"{k}\""
1517
+ if body["version"] != DATASET_BODY_VERSION:
1518
+ return f"a dataset's body is version {DATASET_BODY_VERSION}"
1405
1519
  if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
1406
1520
  return "cases has to be a list of cases"
1521
+ src = body["source"]
1522
+ if src is not None and not (isinstance(src, dict) and isinstance(src.get("id"), str)
1523
+ and isinstance(src.get("name"), str)):
1524
+ return "a dataset names its Source as { id, name }, or null"
1407
1525
  return ""
1408
1526
 
1409
1527
 
@@ -1788,6 +1906,11 @@ class Datasets:
1788
1906
  # so nothing is lost if a pipeline was missed.
1789
1907
  db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
1790
1908
  "dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
1909
+ # Each row's body as it was before the conversion below rewrote
1910
+ # it (#199): a version-4 case's expectations became metrics, and
1911
+ # the body it was typed as is kept, as the rules were.
1912
+ db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
1913
+ "dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
1791
1914
  # Rows from an earlier version are converted once, in place: a
1792
1915
  # dataset is typed in by hand and costly to re-enter, so it is
1793
1916
  # upgraded rather than hidden (AGENTS.md's one exception). The
@@ -1806,6 +1929,8 @@ class Datasets:
1806
1929
  up = upgrade_body(body)
1807
1930
  if up is not body:
1808
1931
  self._archive(db, did, body)
1932
+ db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
1933
+ (did, raw, self._now()))
1809
1934
  if prompts is not None:
1810
1935
  given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
1811
1936
  db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
@@ -1874,7 +1999,6 @@ class Datasets:
1874
1999
  def _insert(self, db, name, body):
1875
2000
  did = secrets.token_hex(6)
1876
2001
  now = self._now()
1877
- body = {**body, "cases": canonical_cases(body["cases"])}
1878
2002
  db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
1879
2003
  "VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
1880
2004
  return did
@@ -1885,7 +2009,7 @@ class Datasets:
1885
2009
  name, why = dataset_name(name)
1886
2010
  if why:
1887
2011
  return None, (400, why)
1888
- body = blank_dataset() if body is None else body
2012
+ body = blank_dataset() if body is None else upgrade_body(body)
1889
2013
  why = dataset_problem(body)
1890
2014
  if why:
1891
2015
  return None, (400, why)
@@ -1914,10 +2038,12 @@ class Datasets:
1914
2038
  the current row."""
1915
2039
  if type(version) is not int:
1916
2040
  return None, (400, "a save names the version it began from")
2041
+ # A body of an earlier version -- from a page loaded before this one --
2042
+ # is read as today's, as an import is.
2043
+ body = upgrade_body(body)
1917
2044
  why = dataset_problem(body)
1918
2045
  if why:
1919
2046
  return None, (400, why)
1920
- body = {**body, "cases": canonical_cases(body["cases"])}
1921
2047
  with self.store.lock, self._connect() as db, db:
1922
2048
  r = self._live(db, did)
1923
2049
  if r is None:
@@ -2328,6 +2454,13 @@ class Packs:
2328
2454
  ds_ids = {}
2329
2455
  cuts = [] # (kind, id, the document as it was), kept before it changes
2330
2456
  for key, name, body, raw in pack["datasets"]:
2457
+ # The Source a dataset grades may be the pack's own, named by
2458
+ # its folder: pointed at the Source the pack made here.
2459
+ ref = body.get("source")
2460
+ if isinstance(ref, dict):
2461
+ sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
2462
+ if sid:
2463
+ body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
2331
2464
  did = owned.get(("dataset", key))
2332
2465
  current = DATASETS.get(did) if did else None
2333
2466
  if current:
@@ -2349,8 +2482,10 @@ class Packs:
2349
2482
  new_work = []
2350
2483
  for key, doc in pack["pipelines"]:
2351
2484
  doc = json.loads(json.dumps(doc))
2352
- tests = doc.get("tests")
2353
- for t in tests if isinstance(tests, list) else [tests]:
2485
+ # A pack written before version 11 spells its evals `tests`;
2486
+ # the page upgrades the pipeline as it reads it.
2487
+ evals = doc.get("evals", doc.get("tests"))
2488
+ for t in evals if isinstance(evals, list) else [evals]:
2354
2489
  ref = t.get("dataset") if isinstance(t, dict) else None
2355
2490
  if isinstance(ref, dict):
2356
2491
  did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
@@ -2487,7 +2622,7 @@ class Packs:
2487
2622
  #
2488
2623
  # A plugin is code, installed like a pack (docs/packs.md): a zip of a
2489
2624
  # manifest and the compiled JavaScript that registers what the lab lacks -- a
2490
- # kind of answer, a modifier, a test type, a connection type. Its code runs in
2625
+ # kind of answer, a modifier, an eval type, a connection type. Its code runs in
2491
2626
  # the page and in the runner, where the registries live; this server never
2492
2627
  # runs it. It reads the manifest's `registers` as data, so it can refuse two
2493
2628
  # plugins registering one id, and so a connection type's settings, chat path
@@ -2505,7 +2640,19 @@ PLUGIN_VERSIONS = (1,)
2505
2640
  PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
2506
2641
  PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
2507
2642
  PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
2508
- REGISTRIES = ("outputKinds", "modifiers", "testTypes", "connectionTypes")
2643
+ REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
2644
+ # evalTypes as a manifest written before pipeline version 11 spells it: a
2645
+ # plugin's own file, which the lab cannot upgrade, so it is read for good.
2646
+ OLD_REGISTRIES = {"testTypes": "evalTypes"}
2647
+
2648
+
2649
+ def plugin_registers(manifest: dict) -> dict:
2650
+ """A manifest's registers under today's names: testTypes read as evalTypes."""
2651
+ reg = dict(manifest.get("registers") or {})
2652
+ for old, now in OLD_REGISTRIES.items():
2653
+ if old in reg:
2654
+ reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
2655
+ return reg
2509
2656
  AUTH_WAYS = ("bearer", "x-api-key", "none")
2510
2657
 
2511
2658
 
@@ -2545,9 +2692,9 @@ def read_plugin(data: bytes):
2545
2692
  if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
2546
2693
  return None, "the plugin's entry names no JavaScript file it holds"
2547
2694
  reg = m.get("registers") or {}
2548
- if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
2695
+ if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
2549
2696
  return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
2550
- for k in ("outputKinds", "modifiers", "testTypes"):
2697
+ for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
2551
2698
  if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
2552
2699
  return None, f"registers.{k} is a list of ids"
2553
2700
  conns = reg.get("connectionTypes", [])
@@ -2570,8 +2717,8 @@ def read_plugin(data: bytes):
2570
2717
 
2571
2718
  def registered_ids(manifest: dict) -> set:
2572
2719
  """(registry, id) for everything a plugin's manifest says it registers."""
2573
- reg = manifest.get("registers") or {}
2574
- out = {(k, x) for k in ("outputKinds", "modifiers", "testTypes") for x in reg.get(k, [])}
2720
+ reg = plugin_registers(manifest)
2721
+ out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
2575
2722
  return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
2576
2723
 
2577
2724
 
@@ -2924,7 +3071,7 @@ class Plugins:
2924
3071
  return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
2925
3072
  "entry": json.loads(r["manifest"])["entry"],
2926
3073
  "description": json.loads(r["manifest"]).get("description") or "",
2927
- "registers": json.loads(r["manifest"]).get("registers") or {},
3074
+ "registers": plugin_registers(json.loads(r["manifest"])),
2928
3075
  "installed": r["installed_at"]} for r in self._rows()]
2929
3076
 
2930
3077
  def stamp(self) -> list:
@@ -3100,6 +3247,55 @@ def worker_refusal(code, stderr, env):
3100
3247
  return text if len(text) <= 2000 else text[:2000] + "…"
3101
3248
 
3102
3249
 
3250
+
3251
+ # A list of runs is read for its figures -- History's table, Home's recent
3252
+ # runs, Runs' progress -- and a run's results are mostly what Results alone
3253
+ # shows: every reply, every stage's request and answer, every item a score
3254
+ # found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
3255
+ # list carries a brief copy of its results, `brief` says so, and one run
3256
+ # (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
3257
+ # whether it ran; a brief score keeps its verdict and counts in place of its
3258
+ # lists. The replies go too: an eval over the whole run reads them, and the
3259
+ # page asks for that run whole rather than every list carrying them.
3260
+ BRIEF_SCORE = ("pass", "score", "points", "skipped")
3261
+ COUNTED = ("found", "missed", "invented")
3262
+
3263
+
3264
+ def brief_score(score):
3265
+ if not isinstance(score, dict):
3266
+ return score
3267
+ out = {k: score[k] for k in BRIEF_SCORE if k in score}
3268
+ for k in COUNTED:
3269
+ if isinstance(score.get(k), list):
3270
+ out[k] = len(score[k])
3271
+ return out
3272
+
3273
+
3274
+ def brief_cell(cell):
3275
+ if not isinstance(cell, dict):
3276
+ return cell
3277
+ out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
3278
+ if isinstance(cell.get("res"), dict):
3279
+ out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
3280
+ if isinstance(cell.get("scores"), dict):
3281
+ out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
3282
+ elif "scores" in cell:
3283
+ out["scores"] = cell["scores"]
3284
+ # A run from before version 6 kept its one test's score as `score`, which
3285
+ # the page reads as `scores.t1`: kept under its own name for that.
3286
+ if "score" in cell:
3287
+ out["score"] = brief_score(cell["score"])
3288
+ return out
3289
+
3290
+
3291
+ def brief_row(row):
3292
+ def item(it):
3293
+ if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
3294
+ return it
3295
+ return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
3296
+ return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
3297
+
3298
+
3103
3299
  class Queue:
3104
3300
  STATUS = ("queued", "running", "done", "incomplete", "cancelled",
3105
3301
  "failed", "interrupted")
@@ -3164,15 +3360,17 @@ class Queue:
3164
3360
  (rid,)).fetchone())
3165
3361
  return row if self._readable(row) else None
3166
3362
 
3167
- def list(self, limit=RUNS_PAGE, before=None):
3363
+ def list(self, limit=RUNS_PAGE, before=None, full=False):
3168
3364
  """Runs, newest first, and whether more follow. `before` is a
3169
3365
  `submittedAt` the page of runs stops at, so History can page through
3170
- them the way it pages the runs store."""
3366
+ them the way it pages the runs store. Each row is brief_row's unless
3367
+ `full` asks for the whole of it."""
3171
3368
  with self.lock, closing(sqlite3.connect(self.store.path)) as db:
3172
3369
  rows = self._all(db)
3173
3370
  rows = [r for r in rows if before is None or r["submittedAt"] < before]
3174
3371
  rows.sort(key=lambda r: r["submittedAt"], reverse=True)
3175
- return rows[:limit], len(rows) > limit
3372
+ page = rows[:limit]
3373
+ return (page if full else [brief_row(r) for r in page]), len(rows) > limit
3176
3374
 
3177
3375
  def _set(self, rid, **fields):
3178
3376
  sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
@@ -3241,7 +3439,7 @@ class Queue:
3241
3439
  written beside the run document. A row queued before runs kept their
3242
3440
  dataset has none, and is pinned to the dataset as it reads now, once,
3243
3441
  so every later pass over it agrees. Returns (args, None) or (None, why)."""
3244
- ref = tests_dataset(run["snapshot"])
3442
+ ref = evals_dataset(run["snapshot"])
3245
3443
  if ref is None:
3246
3444
  return [], None
3247
3445
  body = self.dataset(run["id"], raw=True)
@@ -3748,7 +3946,7 @@ def worker_destinations(run: dict):
3748
3946
 
3749
3947
  # A run document's fields, checked before it is accepted -- the rules
3750
3948
  # docs/pipeline-model.md §6 gives the server, in Python because the server is
3751
- # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or a test
3949
+ # stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
3752
3950
  # means is the runner's to judge, and it fails the run with a sentence if it
3753
3951
  # cannot; what is here is what the server itself depends on: a version it
3754
3952
  # reads, its own cap, content it can count and copy, and connections that
@@ -3756,21 +3954,23 @@ def worker_destinations(run: dict):
3756
3954
  # 5: the pipeline and every chain have an id; version 4's scenarios keep
3757
3955
  # theirs, and an older run's chains are read as they are, by position.
3758
3956
  # 6: tests are an ordered list; a stored run's one test (or null) is read as
3759
- # a list of one (tests_dataset).
3957
+ # a list of one (evals_dataset).
3760
3958
  # 7: chains are jobs: the field is `jobs` and each job's type is "job".
3761
3959
  # 8: a job is its steps; the content is job 1's Attach Content step.
3762
- # 9: a test is Metrics; a Single Test or a Graded set is read converted.
3960
+ # 9: an eval is Metrics; a Single Test or a Graded set is read converted.
3763
3961
  # 10: a job's steps are its stages, and each scenario is a target whose own
3764
3962
  # step in each job is what it sends there (docs/pipeline-model.md §16).
3765
- PIPELINE_VERSION = 10
3963
+ # 11: `tests` are `evals`; nothing in an eval changes.
3964
+ # 12: a Contains metric's Ignore case holds item by item too, kept as written.
3965
+ PIPELINE_VERSION = 12
3766
3966
  # What a stored run may be: the current version, and the ones evals-core.ts's
3767
3967
  # upgradePipeline reads. A new submission is upgraded to the current one.
3768
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
3968
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
3769
3969
  TARGET_CAP = 4
3770
3970
  # Target steps whose words the Prompt library does not record as a use: they
3771
3971
  # ask no model (evals-core.ts's STEP_TYPES.echo).
3772
3972
  UNRECORDED_STEPS = {"echo"}
3773
- RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "tests", "profiles", "comment", "plugins")
3973
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
3774
3974
 
3775
3975
 
3776
3976
  def content_of(doc):
@@ -3927,14 +4127,14 @@ def connection_problems(pid, conn, at, bad):
3927
4127
  bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
3928
4128
 
3929
4129
 
3930
- def tests_dataset(doc):
3931
- """The dataset reference a document's tests grade against, or None: the
3932
- first test that names one. A run grades against one dataset (the core's
3933
- validatePipeline says so). Reads a stored document of either shape --
3934
- version 5's list, or the one test before it -- since rows keep the
3935
- document they were submitted with."""
3936
- tests = doc.get("tests") if isinstance(doc, dict) else None
3937
- for t in tests if isinstance(tests, list) else [tests]:
4130
+ def evals_dataset(doc):
4131
+ """The dataset reference a document's evals grade against, or None: the
4132
+ first eval that names one. A run grades against one dataset (the core's
4133
+ validatePipeline says so). Reads a stored document of any shape --
4134
+ version 11's `evals`, the `tests` before it, version 5's list or the one
4135
+ test before that -- since rows keep the document they were submitted with."""
4136
+ evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
4137
+ for t in evals if isinstance(evals, list) else [evals]:
3938
4138
  if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
3939
4139
  return t["dataset"]
3940
4140
  return None
@@ -4079,9 +4279,9 @@ def run_problems(run):
4079
4279
  pass
4080
4280
  else:
4081
4281
  bad.append("a run needs a Source's files, some inline text, or Prompt only")
4082
- tests = run.get("tests")
4083
- if not isinstance(tests, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in tests):
4084
- bad.append("tests has to be a list of tests, each naming its type")
4282
+ evals = run.get("evals")
4283
+ if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
4284
+ bad.append("evals has to be a list of evals, each naming its type")
4085
4285
  if run.get("comment") is not None and not isinstance(run["comment"], str):
4086
4286
  bad.append("comment has to be text")
4087
4287
  return bad
@@ -4146,11 +4346,25 @@ class Handler(BaseHTTPRequestHandler):
4146
4346
 
4147
4347
  # ---- helpers --------------------------------------------------------
4148
4348
 
4149
- def _send(self, code, body: bytes, ctype="application/json", headers=()):
4349
+ def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
4350
+ """`packed` keys a body that never changes under it -- a hashed
4351
+ bundle -- so it is gzipped once and kept, not on every request."""
4150
4352
  self.send_response(code)
4151
4353
  self.send_header("Content-Type", ctype)
4152
4354
  for k, v in headers:
4153
4355
  self.send_header(k, v)
4356
+ if ctype.startswith(COMPRESSIBLE):
4357
+ # Said whether or not this answer is compressed, so a cache
4358
+ # between here and the browser keys on it either way.
4359
+ self.send_header("Vary", "Accept-Encoding")
4360
+ if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
4361
+ if packed is None:
4362
+ body = gzip.compress(body, 6, mtime=0)
4363
+ else:
4364
+ if packed not in GZIPPED:
4365
+ GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
4366
+ body = GZIPPED[packed]
4367
+ self.send_header("Content-Encoding", "gzip")
4154
4368
  self.send_header("Content-Length", str(len(body)))
4155
4369
  self.end_headers()
4156
4370
  self.wfile.write(body)
@@ -4235,7 +4449,7 @@ class Handler(BaseHTTPRequestHandler):
4235
4449
  return
4236
4450
  path = self.path.split("?", 1)[0]
4237
4451
  # The lab is one page: a Connection and an Input make a scenario,
4238
- # Content and Tests are shared, and one to four scenarios run over the
4452
+ # Content and Evals are shared, and one to four scenarios run over the
4239
4453
  # content. It replaced the A/B page it was prototyped beside once it
4240
4454
  # carried everything that page did -- the graded set, the
4241
4455
  # Configuration group, the history -- rather than being left to rot
@@ -4255,7 +4469,7 @@ class Handler(BaseHTTPRequestHandler):
4255
4469
  if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
4256
4470
  return self._send(404, b"not found", "text/plain")
4257
4471
  return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
4258
- (("Cache-Control", "public, max-age=31536000, immutable"),))
4472
+ (("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
4259
4473
  if path == "/api/config":
4260
4474
  # The lab's own limits, for the page to disable rather than
4261
4475
  # hard-code: how many scenarios a run may have.
@@ -4279,14 +4493,14 @@ class Handler(BaseHTTPRequestHandler):
4279
4493
  except ValueError:
4280
4494
  return self._json(400, {"error": "limit has to be a number"})
4281
4495
  before = (query.get("before") or [None])[0]
4282
- runs, more = QUEUE.list(limit, before)
4496
+ runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
4283
4497
  return self._json(200, {"runs": runs, "more": more})
4284
4498
  if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
4285
4499
  # The dataset body a graded run was submitted against, which is
4286
4500
  # what its verdicts were graded by; the list never carries it.
4287
4501
  #
4288
4502
  # A run that is there but kept no copy -- one submitted before the
4289
- # lab kept them, or one with no graded test -- answers null, not
4503
+ # lab kept them, or one with no graded eval -- answers null, not
4290
4504
  # 404. It is not an error: the page reads it as "use the dataset
4291
4505
  # as it is now", and a 404 put a red line in the console every
4292
4506
  # time such a run was opened, which is the console people are
@@ -4820,14 +5034,14 @@ class Handler(BaseHTTPRequestHandler):
4820
5034
  # A graded run keeps the body of the dataset it names, as it reads
4821
5035
  # now, and records that body's fingerprint on the reference it
4822
5036
  # belongs to: the worker grades against that and nothing else.
4823
- ref = tests_dataset(run)
5037
+ ref = evals_dataset(run)
4824
5038
  body = None
4825
5039
  if ref is not None:
4826
5040
  snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
4827
5041
  if snap is None:
4828
- return self._json(400, {"error": "the run's graded test names no dataset this lab has"})
5042
+ return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
4829
5043
  body, version = snap
4830
- for t in run["tests"]:
5044
+ for t in run["evals"]:
4831
5045
  if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
4832
5046
  t["dataset"]["version"] = version
4833
5047
  return self._json(201, {"run": QUEUE.submit(run, body)})
@@ -1 +1 @@
1
- .gallery-index,.gallery-one{max-width:720px;padding:var(--s4);min-width:0;margin:0 auto}.gallery-index h1{font-size:var(--t-title);font-weight:600}.gallery-index ul{gap:var(--s2);margin:0;padding:0;list-style:none;display:grid}.gallery-index li{gap:var(--s2);flex-wrap:wrap;align-items:baseline;min-height:44px;display:flex}.gallery-index a{color:var(--amber)}.gallery-index code{font:var(--t-note) var(--mono);color:var(--faint)}.gallery-nav{gap:var(--s2);margin-bottom:var(--s3);font-size:var(--t-note);color:var(--dim);align-items:center;min-height:44px;display:flex}.gallery-nav a{color:var(--amber)}.note{color:var(--dim);font-size:var(--t-label);margin:0}
1
+ .gallery-index,.gallery-one{max-width:720px;padding:var(--s4);min-width:0;margin:0 auto}.gallery-index h1{font-size:var(--t-title);font-weight:600}.gallery-index ul{gap:var(--s2);margin:0;padding:0;list-style:none;display:grid}.gallery-index li{gap:var(--s2);flex-wrap:wrap;align-items:baseline;min-height:44px;display:flex}.gallery-index a{color:var(--amber)}.gallery-index code{font:var(--t-note) var(--mono);color:var(--faint)}.gallery-nav{gap:var(--s2);margin-bottom:var(--s3);font-size:var(--t-note);color:var(--dim);align-items:center;min-height:44px;display:flex}.gallery-nav a{color:var(--amber)}.note{color:var(--dim);font-size:var(--t-label);margin:0}.gallery-row{gap:var(--s2);flex-wrap:wrap;display:flex}