evals-lab 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +17 -9
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +3 -7
- package/lab/demo/pipelines/demo-2.json +3 -7
- package/lab/evals-core.mjs +429 -434
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +35 -33
- package/lab/server.py +296 -82
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DsetJSXv.js +0 -3
- package/lab/web/dist/assets/main-B-VtDGxC.css +0 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +0 -19
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +0 -51
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +0 -1
package/lab/server.py
CHANGED
|
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
|
|
|
24
24
|
|
|
25
25
|
import base64
|
|
26
26
|
import calendar
|
|
27
|
+
import gzip
|
|
27
28
|
import hashlib
|
|
28
29
|
import hmac
|
|
29
30
|
import io
|
|
@@ -247,6 +248,35 @@ TYPES = {
|
|
|
247
248
|
".png": "image/png",
|
|
248
249
|
}
|
|
249
250
|
|
|
251
|
+
# What is gzipped for a client that asks (#225): text, which compresses five
|
|
252
|
+
# to ten times, and nothing already compressed. A body under GZIP_MIN goes as
|
|
253
|
+
# it is, since the header and the gzip frame would outweigh the saving.
|
|
254
|
+
COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
|
|
255
|
+
GZIP_MIN = 1024
|
|
256
|
+
# The built page's bundles, gzipped once: a new build is new names.
|
|
257
|
+
GZIPPED: dict = {}
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def accepts_gzip(header: str) -> bool:
|
|
261
|
+
"""Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
|
|
262
|
+
above 0. `gzip;q=0` is a refusal, not a request."""
|
|
263
|
+
star = False
|
|
264
|
+
for part in (header or "").split(","):
|
|
265
|
+
name, _, params = part.partition(";")
|
|
266
|
+
name, q = name.strip().lower(), 1.0
|
|
267
|
+
for p in params.split(";"):
|
|
268
|
+
k, _, v = p.partition("=")
|
|
269
|
+
if k.strip().lower() == "q":
|
|
270
|
+
try:
|
|
271
|
+
q = float(v)
|
|
272
|
+
except ValueError:
|
|
273
|
+
q = 0.0
|
|
274
|
+
if name == "gzip":
|
|
275
|
+
return q > 0
|
|
276
|
+
if name == "*":
|
|
277
|
+
star = q > 0
|
|
278
|
+
return star
|
|
279
|
+
|
|
250
280
|
|
|
251
281
|
# What a Source may accept and ever be served back as -- deliberately not
|
|
252
282
|
# TYPES: an upload that could come back as text/html is stored XSS, and the
|
|
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
|
|
|
469
499
|
# A key taken off this list is no longer served or written, and its rows stay
|
|
470
500
|
# in the store: a document is not migrated or deleted because nothing reads it.
|
|
471
501
|
# promptlab.cases, .rules and their .base copies went that way when a dataset
|
|
472
|
-
# became a row of its own (#86).
|
|
502
|
+
# became a row of its own (#86), and promptlab.mappings once a dataset named
|
|
503
|
+
# the Source it grades (#199).
|
|
473
504
|
SYNCED = ("promptlab.workflows", "promptlab.profiles",
|
|
474
|
-
"promptlab.versions", "promptlab.tokens"
|
|
505
|
+
"promptlab.versions", "promptlab.tokens")
|
|
475
506
|
MAX_DOC = 8 * 1024 * 1024
|
|
476
507
|
RUNS_PAGE = 25
|
|
477
508
|
# A dataset request's caps, read from Content-Length before the body is, as a
|
|
@@ -1306,7 +1337,7 @@ class Sources:
|
|
|
1306
1337
|
|
|
1307
1338
|
# ---- Datasets ----------------------------------------------------------------
|
|
1308
1339
|
#
|
|
1309
|
-
# A dataset is data a graded
|
|
1340
|
+
# A dataset is data a graded eval names by id: its cases. (The prompt a new
|
|
1310
1341
|
# scenario starts from is the Prompt library's Default, below; a dataset from
|
|
1311
1342
|
# before the library held one, and gave it to the library once.) It is one row in the
|
|
1312
1343
|
# store's SQLite, its body one JSON document with a version that goes up by one
|
|
@@ -1325,44 +1356,140 @@ class Sources:
|
|
|
1325
1356
|
# A lab from before this held a one-time import's `meta` row saying it ran;
|
|
1326
1357
|
# it is left where it is, and nothing reads it.
|
|
1327
1358
|
|
|
1328
|
-
DATASET_FIELDS = ("cases"
|
|
1359
|
+
DATASET_FIELDS = ("version", "source", "cases")
|
|
1360
|
+
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 5 was
|
|
1361
|
+
# told by its `source` alone, and earlier ones by neither.
|
|
1362
|
+
DATASET_BODY_VERSION = 6
|
|
1329
1363
|
DATASET_NAME_MAX = 80
|
|
1330
|
-
# The file forms Export writes and Import reads. Export writes version
|
|
1331
|
-
# Import reads it and versions 1 to
|
|
1364
|
+
# The file forms Export writes and Import reads. Export writes version 6;
|
|
1365
|
+
# Import reads it and versions 1 to 5, upgraded, and refuses anything else,
|
|
1332
1366
|
# as a pipeline of another version is refused. Versions 1 to 3 carried a
|
|
1333
1367
|
# prompt, which an import gives to the Prompt library.
|
|
1334
1368
|
EXPORT_ONE = "evals-lab/dataset"
|
|
1335
1369
|
EXPORT_ALL = "evals-lab/datasets"
|
|
1336
|
-
EXPORT_VERSION =
|
|
1337
|
-
IMPORT_VERSIONS = (1, 2, 3, 4)
|
|
1370
|
+
EXPORT_VERSION = 6
|
|
1371
|
+
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
|
|
1338
1372
|
|
|
1339
1373
|
|
|
1340
1374
|
def blank_dataset() -> dict:
|
|
1341
|
-
return {"cases": []}
|
|
1375
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "cases": []}
|
|
1376
|
+
|
|
1377
|
+
|
|
1378
|
+
# vocab: the names older versions gave a case's fields
|
|
1379
|
+
CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
|
|
1380
|
+
CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
|
|
1381
|
+
|
|
1382
|
+
|
|
1383
|
+
def _words(s) -> list:
|
|
1384
|
+
"""evals-core.ts's words: the letters and digits of [s], lowercased."""
|
|
1385
|
+
return re.findall(r"[^\W_]+", str(s or "").lower())
|
|
1386
|
+
|
|
1387
|
+
|
|
1388
|
+
def _term_in(items, term) -> bool:
|
|
1389
|
+
"""evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
|
|
1390
|
+
t = _words(term)
|
|
1391
|
+
if not t:
|
|
1392
|
+
return False
|
|
1393
|
+
for item in items:
|
|
1394
|
+
w = _words(item)
|
|
1395
|
+
if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
|
|
1396
|
+
return True
|
|
1397
|
+
return False
|
|
1398
|
+
|
|
1399
|
+
|
|
1400
|
+
def case_metrics(c: dict) -> list:
|
|
1401
|
+
"""A version-4 case's expectations as the metrics that say the same:
|
|
1402
|
+
evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
|
|
1403
|
+
through fixtures/dataset-v6.json."""
|
|
1404
|
+
def strs(v):
|
|
1405
|
+
return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
|
|
1406
|
+
if c.get("discarded") is True:
|
|
1407
|
+
return [{"type": "discarded"}]
|
|
1408
|
+
out = []
|
|
1409
|
+
expect, allow = strs(c.get("expect")), strs(c.get("allow"))
|
|
1410
|
+
if expect:
|
|
1411
|
+
out.append({"type": "contains-all", "values": "\n".join(expect)})
|
|
1412
|
+
for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
|
|
1413
|
+
if strs(g):
|
|
1414
|
+
out.append({"type": "contains-any", "values": "\n".join(strs(g))})
|
|
1415
|
+
for t in strs(c.get("forbid")):
|
|
1416
|
+
# An exception excuses only the forbidden term inside it.
|
|
1417
|
+
except_ = [a for a in allow if _term_in([a], t)]
|
|
1418
|
+
out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
|
|
1419
|
+
whole = lambda v: v if type(v) is int else None
|
|
1420
|
+
lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
|
|
1421
|
+
if lo is not None or hi is not None:
|
|
1422
|
+
out.append({"type": "item-count", "min": lo, "max": hi})
|
|
1423
|
+
for t in strs(c.get("watch")):
|
|
1424
|
+
out.append({"type": "contains-any", "values": t, "weight": 0})
|
|
1425
|
+
return out
|
|
1426
|
+
|
|
1427
|
+
|
|
1428
|
+
def case_of_v4(c):
|
|
1429
|
+
"""One case of any earlier version as a version-5 one: evals-core.ts's
|
|
1430
|
+
caseOfV4, in Python."""
|
|
1431
|
+
if not isinstance(c, dict):
|
|
1432
|
+
return c
|
|
1433
|
+
was = {}
|
|
1434
|
+
for k, v in c.items():
|
|
1435
|
+
key = CASE_RENAMED.get(k, k)
|
|
1436
|
+
if key not in was or key == k:
|
|
1437
|
+
was[key] = v
|
|
1438
|
+
if isinstance(was.get("item"), str):
|
|
1439
|
+
was.pop("filename", None)
|
|
1440
|
+
out = {}
|
|
1441
|
+
if "id" in was:
|
|
1442
|
+
out["id"] = was["id"]
|
|
1443
|
+
out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
|
|
1444
|
+
out["todo"] = was.get("todo") is True
|
|
1445
|
+
out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
|
|
1446
|
+
out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
|
|
1447
|
+
for k, v in was.items():
|
|
1448
|
+
if k not in out and k not in CASE_V4:
|
|
1449
|
+
out[k] = v
|
|
1450
|
+
return out
|
|
1451
|
+
|
|
1452
|
+
|
|
1453
|
+
# The metrics whose Ignore case version 6 made mean what it says for a reply
|
|
1454
|
+
# read as a list: evals-core.ts's CASE_FOLDING.
|
|
1455
|
+
CASE_FOLDING = ("contains", "contains-all", "contains-any")
|
|
1456
|
+
|
|
1457
|
+
|
|
1458
|
+
def case_of_v5(c):
|
|
1459
|
+
"""A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
|
|
1460
|
+
Python. Each Contains metric says Ignore case, as version 5 matched a
|
|
1461
|
+
list's items whatever it said."""
|
|
1462
|
+
if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
|
|
1463
|
+
return c
|
|
1464
|
+
return {**c, "metrics": [{**m, "ignoreCase": True}
|
|
1465
|
+
if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
|
|
1466
|
+
else m for m in c["metrics"]]}
|
|
1342
1467
|
|
|
1343
1468
|
|
|
1344
1469
|
def upgrade_body(body):
|
|
1345
|
-
"""An earlier body as today's: evals-core.ts's
|
|
1346
|
-
Python. Version 1's `imageCases` are `cases`,
|
|
1347
|
-
|
|
1348
|
-
`
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1470
|
+
"""An earlier body as today's (version 6): evals-core.ts's
|
|
1471
|
+
upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
|
|
1472
|
+
its `replays` and `conformance` go (fixtures/replays.json holds the
|
|
1473
|
+
parser's tests). Version 2's `rules` go -- they clean a job's answer, so
|
|
1474
|
+
they are the job's. Version 3's `prompt` goes: the Prompt library holds
|
|
1475
|
+
prompts now. Version 4's case named its item `filename` and said what a
|
|
1476
|
+
good answer is in expectations; each becomes its metric, `why` the
|
|
1477
|
+
`note`, `traits` go, and the body names no Source yet. A caller that
|
|
1478
|
+
needs the rules or the prompt takes them first (`body_rules`,
|
|
1479
|
+
`body_prompt`). A body naming its Source is version 5, whose Contains
|
|
1480
|
+
metrics each come to say Ignore case (`case_of_v5`). A body saying it is
|
|
1481
|
+
version 6 comes back as it was; so does anything that is not a body."""
|
|
1482
|
+
if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
|
|
1355
1483
|
return body
|
|
1356
|
-
if "
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1484
|
+
if "source" in body:
|
|
1485
|
+
up = {"version": DATASET_BODY_VERSION, **body}
|
|
1486
|
+
if isinstance(body.get("cases"), list):
|
|
1487
|
+
up["cases"] = [case_of_v5(c) for c in body["cases"]]
|
|
1488
|
+
return up
|
|
1361
1489
|
cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
|
|
1362
1490
|
if not isinstance(cases, list):
|
|
1363
1491
|
return body
|
|
1364
|
-
|
|
1365
|
-
return {"cases": canonical_cases(cases)}
|
|
1492
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}
|
|
1366
1493
|
|
|
1367
1494
|
|
|
1368
1495
|
def body_prompt(body):
|
|
@@ -1377,21 +1504,6 @@ def body_rules(body):
|
|
|
1377
1504
|
return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
|
|
1378
1505
|
|
|
1379
1506
|
|
|
1380
|
-
def canonical_cases(cases: list) -> list:
|
|
1381
|
-
"""Every case naming its file as `filename`: evals-core.ts's
|
|
1382
|
-
canonicalCases, in Python. `old` is the key a set re-synced from an
|
|
1383
|
-
app's own repository arrives with; the key keeps its place, so only its
|
|
1384
|
-
spelling changes."""
|
|
1385
|
-
old = "photo" # vocab: the older spelling of filename
|
|
1386
|
-
out = []
|
|
1387
|
-
for c in cases:
|
|
1388
|
-
if isinstance(c, dict) and old in c:
|
|
1389
|
-
c = {("filename" if k == old else k): v for k, v in c.items()
|
|
1390
|
-
if not (k == old and "filename" in c)}
|
|
1391
|
-
out.append(c)
|
|
1392
|
-
return out
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
1507
|
def dataset_problem(body) -> str:
|
|
1396
1508
|
"""Why [body] is not a dataset's body, in one sentence, or ""."""
|
|
1397
1509
|
if not isinstance(body, dict):
|
|
@@ -1402,8 +1514,14 @@ def dataset_problem(body) -> str:
|
|
|
1402
1514
|
for k in DATASET_FIELDS:
|
|
1403
1515
|
if k not in body:
|
|
1404
1516
|
return f"a dataset's body has no \"{k}\""
|
|
1517
|
+
if body["version"] != DATASET_BODY_VERSION:
|
|
1518
|
+
return f"a dataset's body is version {DATASET_BODY_VERSION}"
|
|
1405
1519
|
if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
|
|
1406
1520
|
return "cases has to be a list of cases"
|
|
1521
|
+
src = body["source"]
|
|
1522
|
+
if src is not None and not (isinstance(src, dict) and isinstance(src.get("id"), str)
|
|
1523
|
+
and isinstance(src.get("name"), str)):
|
|
1524
|
+
return "a dataset names its Source as { id, name }, or null"
|
|
1407
1525
|
return ""
|
|
1408
1526
|
|
|
1409
1527
|
|
|
@@ -1788,6 +1906,11 @@ class Datasets:
|
|
|
1788
1906
|
# so nothing is lost if a pipeline was missed.
|
|
1789
1907
|
db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
|
|
1790
1908
|
"dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1909
|
+
# Each row's body as it was before the conversion below rewrote
|
|
1910
|
+
# it (#199): a version-4 case's expectations became metrics, and
|
|
1911
|
+
# the body it was typed as is kept, as the rules were.
|
|
1912
|
+
db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
|
|
1913
|
+
"dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1791
1914
|
# Rows from an earlier version are converted once, in place: a
|
|
1792
1915
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
1793
1916
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -1806,6 +1929,8 @@ class Datasets:
|
|
|
1806
1929
|
up = upgrade_body(body)
|
|
1807
1930
|
if up is not body:
|
|
1808
1931
|
self._archive(db, did, body)
|
|
1932
|
+
db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
|
|
1933
|
+
(did, raw, self._now()))
|
|
1809
1934
|
if prompts is not None:
|
|
1810
1935
|
given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
|
|
1811
1936
|
db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
|
|
@@ -1874,7 +1999,6 @@ class Datasets:
|
|
|
1874
1999
|
def _insert(self, db, name, body):
|
|
1875
2000
|
did = secrets.token_hex(6)
|
|
1876
2001
|
now = self._now()
|
|
1877
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1878
2002
|
db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
|
|
1879
2003
|
"VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
|
|
1880
2004
|
return did
|
|
@@ -1885,7 +2009,7 @@ class Datasets:
|
|
|
1885
2009
|
name, why = dataset_name(name)
|
|
1886
2010
|
if why:
|
|
1887
2011
|
return None, (400, why)
|
|
1888
|
-
body = blank_dataset() if body is None else body
|
|
2012
|
+
body = blank_dataset() if body is None else upgrade_body(body)
|
|
1889
2013
|
why = dataset_problem(body)
|
|
1890
2014
|
if why:
|
|
1891
2015
|
return None, (400, why)
|
|
@@ -1914,10 +2038,12 @@ class Datasets:
|
|
|
1914
2038
|
the current row."""
|
|
1915
2039
|
if type(version) is not int:
|
|
1916
2040
|
return None, (400, "a save names the version it began from")
|
|
2041
|
+
# A body of an earlier version -- from a page loaded before this one --
|
|
2042
|
+
# is read as today's, as an import is.
|
|
2043
|
+
body = upgrade_body(body)
|
|
1917
2044
|
why = dataset_problem(body)
|
|
1918
2045
|
if why:
|
|
1919
2046
|
return None, (400, why)
|
|
1920
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1921
2047
|
with self.store.lock, self._connect() as db, db:
|
|
1922
2048
|
r = self._live(db, did)
|
|
1923
2049
|
if r is None:
|
|
@@ -2328,6 +2454,13 @@ class Packs:
|
|
|
2328
2454
|
ds_ids = {}
|
|
2329
2455
|
cuts = [] # (kind, id, the document as it was), kept before it changes
|
|
2330
2456
|
for key, name, body, raw in pack["datasets"]:
|
|
2457
|
+
# The Source a dataset grades may be the pack's own, named by
|
|
2458
|
+
# its folder: pointed at the Source the pack made here.
|
|
2459
|
+
ref = body.get("source")
|
|
2460
|
+
if isinstance(ref, dict):
|
|
2461
|
+
sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
|
|
2462
|
+
if sid:
|
|
2463
|
+
body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
|
|
2331
2464
|
did = owned.get(("dataset", key))
|
|
2332
2465
|
current = DATASETS.get(did) if did else None
|
|
2333
2466
|
if current:
|
|
@@ -2349,8 +2482,10 @@ class Packs:
|
|
|
2349
2482
|
new_work = []
|
|
2350
2483
|
for key, doc in pack["pipelines"]:
|
|
2351
2484
|
doc = json.loads(json.dumps(doc))
|
|
2352
|
-
|
|
2353
|
-
|
|
2485
|
+
# A pack written before version 11 spells its evals `tests`;
|
|
2486
|
+
# the page upgrades the pipeline as it reads it.
|
|
2487
|
+
evals = doc.get("evals", doc.get("tests"))
|
|
2488
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
2354
2489
|
ref = t.get("dataset") if isinstance(t, dict) else None
|
|
2355
2490
|
if isinstance(ref, dict):
|
|
2356
2491
|
did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
|
|
@@ -2487,7 +2622,7 @@ class Packs:
|
|
|
2487
2622
|
#
|
|
2488
2623
|
# A plugin is code, installed like a pack (docs/packs.md): a zip of a
|
|
2489
2624
|
# manifest and the compiled JavaScript that registers what the lab lacks -- a
|
|
2490
|
-
# kind of answer, a modifier,
|
|
2625
|
+
# kind of answer, a modifier, an eval type, a connection type. Its code runs in
|
|
2491
2626
|
# the page and in the runner, where the registries live; this server never
|
|
2492
2627
|
# runs it. It reads the manifest's `registers` as data, so it can refuse two
|
|
2493
2628
|
# plugins registering one id, and so a connection type's settings, chat path
|
|
@@ -2505,7 +2640,19 @@ PLUGIN_VERSIONS = (1,)
|
|
|
2505
2640
|
PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
|
|
2506
2641
|
PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
|
|
2507
2642
|
PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
|
|
2508
|
-
REGISTRIES = ("outputKinds", "modifiers", "
|
|
2643
|
+
REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
|
|
2644
|
+
# evalTypes as a manifest written before pipeline version 11 spells it: a
|
|
2645
|
+
# plugin's own file, which the lab cannot upgrade, so it is read for good.
|
|
2646
|
+
OLD_REGISTRIES = {"testTypes": "evalTypes"}
|
|
2647
|
+
|
|
2648
|
+
|
|
2649
|
+
def plugin_registers(manifest: dict) -> dict:
|
|
2650
|
+
"""A manifest's registers under today's names: testTypes read as evalTypes."""
|
|
2651
|
+
reg = dict(manifest.get("registers") or {})
|
|
2652
|
+
for old, now in OLD_REGISTRIES.items():
|
|
2653
|
+
if old in reg:
|
|
2654
|
+
reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
|
|
2655
|
+
return reg
|
|
2509
2656
|
AUTH_WAYS = ("bearer", "x-api-key", "none")
|
|
2510
2657
|
|
|
2511
2658
|
|
|
@@ -2545,9 +2692,9 @@ def read_plugin(data: bytes):
|
|
|
2545
2692
|
if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
|
|
2546
2693
|
return None, "the plugin's entry names no JavaScript file it holds"
|
|
2547
2694
|
reg = m.get("registers") or {}
|
|
2548
|
-
if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
|
|
2695
|
+
if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
|
|
2549
2696
|
return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
|
|
2550
|
-
for k in ("outputKinds", "modifiers", "
|
|
2697
|
+
for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
|
|
2551
2698
|
if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
|
|
2552
2699
|
return None, f"registers.{k} is a list of ids"
|
|
2553
2700
|
conns = reg.get("connectionTypes", [])
|
|
@@ -2570,8 +2717,8 @@ def read_plugin(data: bytes):
|
|
|
2570
2717
|
|
|
2571
2718
|
def registered_ids(manifest: dict) -> set:
|
|
2572
2719
|
"""(registry, id) for everything a plugin's manifest says it registers."""
|
|
2573
|
-
reg = manifest
|
|
2574
|
-
out = {(k, x) for k in ("outputKinds", "modifiers", "
|
|
2720
|
+
reg = plugin_registers(manifest)
|
|
2721
|
+
out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
|
|
2575
2722
|
return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
|
|
2576
2723
|
|
|
2577
2724
|
|
|
@@ -2924,7 +3071,7 @@ class Plugins:
|
|
|
2924
3071
|
return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
|
|
2925
3072
|
"entry": json.loads(r["manifest"])["entry"],
|
|
2926
3073
|
"description": json.loads(r["manifest"]).get("description") or "",
|
|
2927
|
-
"registers": json.loads(r["manifest"])
|
|
3074
|
+
"registers": plugin_registers(json.loads(r["manifest"])),
|
|
2928
3075
|
"installed": r["installed_at"]} for r in self._rows()]
|
|
2929
3076
|
|
|
2930
3077
|
def stamp(self) -> list:
|
|
@@ -3100,6 +3247,55 @@ def worker_refusal(code, stderr, env):
|
|
|
3100
3247
|
return text if len(text) <= 2000 else text[:2000] + "…"
|
|
3101
3248
|
|
|
3102
3249
|
|
|
3250
|
+
|
|
3251
|
+
# A list of runs is read for its figures -- History's table, Home's recent
|
|
3252
|
+
# runs, Runs' progress -- and a run's results are mostly what Results alone
|
|
3253
|
+
# shows: every reply, every stage's request and answer, every item a score
|
|
3254
|
+
# found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
|
|
3255
|
+
# list carries a brief copy of its results, `brief` says so, and one run
|
|
3256
|
+
# (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
|
|
3257
|
+
# whether it ran; a brief score keeps its verdict and counts in place of its
|
|
3258
|
+
# lists. The replies go too: an eval over the whole run reads them, and the
|
|
3259
|
+
# page asks for that run whole rather than every list carrying them.
|
|
3260
|
+
BRIEF_SCORE = ("pass", "score", "points", "skipped")
|
|
3261
|
+
COUNTED = ("found", "missed", "invented")
|
|
3262
|
+
|
|
3263
|
+
|
|
3264
|
+
def brief_score(score):
|
|
3265
|
+
if not isinstance(score, dict):
|
|
3266
|
+
return score
|
|
3267
|
+
out = {k: score[k] for k in BRIEF_SCORE if k in score}
|
|
3268
|
+
for k in COUNTED:
|
|
3269
|
+
if isinstance(score.get(k), list):
|
|
3270
|
+
out[k] = len(score[k])
|
|
3271
|
+
return out
|
|
3272
|
+
|
|
3273
|
+
|
|
3274
|
+
def brief_cell(cell):
|
|
3275
|
+
if not isinstance(cell, dict):
|
|
3276
|
+
return cell
|
|
3277
|
+
out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
|
|
3278
|
+
if isinstance(cell.get("res"), dict):
|
|
3279
|
+
out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
|
|
3280
|
+
if isinstance(cell.get("scores"), dict):
|
|
3281
|
+
out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
|
|
3282
|
+
elif "scores" in cell:
|
|
3283
|
+
out["scores"] = cell["scores"]
|
|
3284
|
+
# A run from before version 6 kept its one test's score as `score`, which
|
|
3285
|
+
# the page reads as `scores.t1`: kept under its own name for that.
|
|
3286
|
+
if "score" in cell:
|
|
3287
|
+
out["score"] = brief_score(cell["score"])
|
|
3288
|
+
return out
|
|
3289
|
+
|
|
3290
|
+
|
|
3291
|
+
def brief_row(row):
|
|
3292
|
+
def item(it):
|
|
3293
|
+
if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
|
|
3294
|
+
return it
|
|
3295
|
+
return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
|
|
3296
|
+
return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
|
|
3297
|
+
|
|
3298
|
+
|
|
3103
3299
|
class Queue:
|
|
3104
3300
|
STATUS = ("queued", "running", "done", "incomplete", "cancelled",
|
|
3105
3301
|
"failed", "interrupted")
|
|
@@ -3164,15 +3360,17 @@ class Queue:
|
|
|
3164
3360
|
(rid,)).fetchone())
|
|
3165
3361
|
return row if self._readable(row) else None
|
|
3166
3362
|
|
|
3167
|
-
def list(self, limit=RUNS_PAGE, before=None):
|
|
3363
|
+
def list(self, limit=RUNS_PAGE, before=None, full=False):
|
|
3168
3364
|
"""Runs, newest first, and whether more follow. `before` is a
|
|
3169
3365
|
`submittedAt` the page of runs stops at, so History can page through
|
|
3170
|
-
them the way it pages the runs store.
|
|
3366
|
+
them the way it pages the runs store. Each row is brief_row's unless
|
|
3367
|
+
`full` asks for the whole of it."""
|
|
3171
3368
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3172
3369
|
rows = self._all(db)
|
|
3173
3370
|
rows = [r for r in rows if before is None or r["submittedAt"] < before]
|
|
3174
3371
|
rows.sort(key=lambda r: r["submittedAt"], reverse=True)
|
|
3175
|
-
|
|
3372
|
+
page = rows[:limit]
|
|
3373
|
+
return (page if full else [brief_row(r) for r in page]), len(rows) > limit
|
|
3176
3374
|
|
|
3177
3375
|
def _set(self, rid, **fields):
|
|
3178
3376
|
sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
|
|
@@ -3241,7 +3439,7 @@ class Queue:
|
|
|
3241
3439
|
written beside the run document. A row queued before runs kept their
|
|
3242
3440
|
dataset has none, and is pinned to the dataset as it reads now, once,
|
|
3243
3441
|
so every later pass over it agrees. Returns (args, None) or (None, why)."""
|
|
3244
|
-
ref =
|
|
3442
|
+
ref = evals_dataset(run["snapshot"])
|
|
3245
3443
|
if ref is None:
|
|
3246
3444
|
return [], None
|
|
3247
3445
|
body = self.dataset(run["id"], raw=True)
|
|
@@ -3748,7 +3946,7 @@ def worker_destinations(run: dict):
|
|
|
3748
3946
|
|
|
3749
3947
|
# A run document's fields, checked before it is accepted -- the rules
|
|
3750
3948
|
# docs/pipeline-model.md §6 gives the server, in Python because the server is
|
|
3751
|
-
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or
|
|
3949
|
+
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
|
|
3752
3950
|
# means is the runner's to judge, and it fails the run with a sentence if it
|
|
3753
3951
|
# cannot; what is here is what the server itself depends on: a version it
|
|
3754
3952
|
# reads, its own cap, content it can count and copy, and connections that
|
|
@@ -3756,21 +3954,23 @@ def worker_destinations(run: dict):
|
|
|
3756
3954
|
# 5: the pipeline and every chain have an id; version 4's scenarios keep
|
|
3757
3955
|
# theirs, and an older run's chains are read as they are, by position.
|
|
3758
3956
|
# 6: tests are an ordered list; a stored run's one test (or null) is read as
|
|
3759
|
-
# a list of one (
|
|
3957
|
+
# a list of one (evals_dataset).
|
|
3760
3958
|
# 7: chains are jobs: the field is `jobs` and each job's type is "job".
|
|
3761
3959
|
# 8: a job is its steps; the content is job 1's Attach Content step.
|
|
3762
|
-
# 9:
|
|
3960
|
+
# 9: an eval is Metrics; a Single Test or a Graded set is read converted.
|
|
3763
3961
|
# 10: a job's steps are its stages, and each scenario is a target whose own
|
|
3764
3962
|
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
3765
|
-
|
|
3963
|
+
# 11: `tests` are `evals`; nothing in an eval changes.
|
|
3964
|
+
# 12: a Contains metric's Ignore case holds item by item too, kept as written.
|
|
3965
|
+
PIPELINE_VERSION = 12
|
|
3766
3966
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
3767
3967
|
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
3768
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
|
|
3968
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
|
3769
3969
|
TARGET_CAP = 4
|
|
3770
3970
|
# Target steps whose words the Prompt library does not record as a use: they
|
|
3771
3971
|
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
3772
3972
|
UNRECORDED_STEPS = {"echo"}
|
|
3773
|
-
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "
|
|
3973
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
|
|
3774
3974
|
|
|
3775
3975
|
|
|
3776
3976
|
def content_of(doc):
|
|
@@ -3927,14 +4127,14 @@ def connection_problems(pid, conn, at, bad):
|
|
|
3927
4127
|
bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
|
|
3928
4128
|
|
|
3929
4129
|
|
|
3930
|
-
def
|
|
3931
|
-
"""The dataset reference a document's
|
|
3932
|
-
first
|
|
3933
|
-
validatePipeline says so). Reads a stored document of
|
|
3934
|
-
version
|
|
3935
|
-
document they were submitted with."""
|
|
3936
|
-
|
|
3937
|
-
for t in
|
|
4130
|
+
def evals_dataset(doc):
|
|
4131
|
+
"""The dataset reference a document's evals grade against, or None: the
|
|
4132
|
+
first eval that names one. A run grades against one dataset (the core's
|
|
4133
|
+
validatePipeline says so). Reads a stored document of any shape --
|
|
4134
|
+
version 11's `evals`, the `tests` before it, version 5's list or the one
|
|
4135
|
+
test before that -- since rows keep the document they were submitted with."""
|
|
4136
|
+
evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
|
|
4137
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
3938
4138
|
if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
|
|
3939
4139
|
return t["dataset"]
|
|
3940
4140
|
return None
|
|
@@ -4079,9 +4279,9 @@ def run_problems(run):
|
|
|
4079
4279
|
pass
|
|
4080
4280
|
else:
|
|
4081
4281
|
bad.append("a run needs a Source's files, some inline text, or Prompt only")
|
|
4082
|
-
|
|
4083
|
-
if not isinstance(
|
|
4084
|
-
bad.append("
|
|
4282
|
+
evals = run.get("evals")
|
|
4283
|
+
if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
|
|
4284
|
+
bad.append("evals has to be a list of evals, each naming its type")
|
|
4085
4285
|
if run.get("comment") is not None and not isinstance(run["comment"], str):
|
|
4086
4286
|
bad.append("comment has to be text")
|
|
4087
4287
|
return bad
|
|
@@ -4146,11 +4346,25 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4146
4346
|
|
|
4147
4347
|
# ---- helpers --------------------------------------------------------
|
|
4148
4348
|
|
|
4149
|
-
def _send(self, code, body: bytes, ctype="application/json", headers=()):
|
|
4349
|
+
def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
|
|
4350
|
+
"""`packed` keys a body that never changes under it -- a hashed
|
|
4351
|
+
bundle -- so it is gzipped once and kept, not on every request."""
|
|
4150
4352
|
self.send_response(code)
|
|
4151
4353
|
self.send_header("Content-Type", ctype)
|
|
4152
4354
|
for k, v in headers:
|
|
4153
4355
|
self.send_header(k, v)
|
|
4356
|
+
if ctype.startswith(COMPRESSIBLE):
|
|
4357
|
+
# Said whether or not this answer is compressed, so a cache
|
|
4358
|
+
# between here and the browser keys on it either way.
|
|
4359
|
+
self.send_header("Vary", "Accept-Encoding")
|
|
4360
|
+
if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
|
|
4361
|
+
if packed is None:
|
|
4362
|
+
body = gzip.compress(body, 6, mtime=0)
|
|
4363
|
+
else:
|
|
4364
|
+
if packed not in GZIPPED:
|
|
4365
|
+
GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
|
|
4366
|
+
body = GZIPPED[packed]
|
|
4367
|
+
self.send_header("Content-Encoding", "gzip")
|
|
4154
4368
|
self.send_header("Content-Length", str(len(body)))
|
|
4155
4369
|
self.end_headers()
|
|
4156
4370
|
self.wfile.write(body)
|
|
@@ -4235,7 +4449,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4235
4449
|
return
|
|
4236
4450
|
path = self.path.split("?", 1)[0]
|
|
4237
4451
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
4238
|
-
# Content and
|
|
4452
|
+
# Content and Evals are shared, and one to four scenarios run over the
|
|
4239
4453
|
# content. It replaced the A/B page it was prototyped beside once it
|
|
4240
4454
|
# carried everything that page did -- the graded set, the
|
|
4241
4455
|
# Configuration group, the history -- rather than being left to rot
|
|
@@ -4255,7 +4469,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4255
4469
|
if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
|
|
4256
4470
|
return self._send(404, b"not found", "text/plain")
|
|
4257
4471
|
return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
|
|
4258
|
-
(("Cache-Control", "public, max-age=31536000, immutable"),))
|
|
4472
|
+
(("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
|
|
4259
4473
|
if path == "/api/config":
|
|
4260
4474
|
# The lab's own limits, for the page to disable rather than
|
|
4261
4475
|
# hard-code: how many scenarios a run may have.
|
|
@@ -4279,14 +4493,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4279
4493
|
except ValueError:
|
|
4280
4494
|
return self._json(400, {"error": "limit has to be a number"})
|
|
4281
4495
|
before = (query.get("before") or [None])[0]
|
|
4282
|
-
runs, more = QUEUE.list(limit, before)
|
|
4496
|
+
runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
|
|
4283
4497
|
return self._json(200, {"runs": runs, "more": more})
|
|
4284
4498
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
4285
4499
|
# The dataset body a graded run was submitted against, which is
|
|
4286
4500
|
# what its verdicts were graded by; the list never carries it.
|
|
4287
4501
|
#
|
|
4288
4502
|
# A run that is there but kept no copy -- one submitted before the
|
|
4289
|
-
# lab kept them, or one with no graded
|
|
4503
|
+
# lab kept them, or one with no graded eval -- answers null, not
|
|
4290
4504
|
# 404. It is not an error: the page reads it as "use the dataset
|
|
4291
4505
|
# as it is now", and a 404 put a red line in the console every
|
|
4292
4506
|
# time such a run was opened, which is the console people are
|
|
@@ -4820,14 +5034,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4820
5034
|
# A graded run keeps the body of the dataset it names, as it reads
|
|
4821
5035
|
# now, and records that body's fingerprint on the reference it
|
|
4822
5036
|
# belongs to: the worker grades against that and nothing else.
|
|
4823
|
-
ref =
|
|
5037
|
+
ref = evals_dataset(run)
|
|
4824
5038
|
body = None
|
|
4825
5039
|
if ref is not None:
|
|
4826
5040
|
snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
|
|
4827
5041
|
if snap is None:
|
|
4828
|
-
return self._json(400, {"error": "the run's graded
|
|
5042
|
+
return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
|
|
4829
5043
|
body, version = snap
|
|
4830
|
-
for t in run["
|
|
5044
|
+
for t in run["evals"]:
|
|
4831
5045
|
if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
|
|
4832
5046
|
t["dataset"]["version"] = version
|
|
4833
5047
|
return self._json(201, {"run": QUEUE.submit(run, body)})
|
|
@@ -1 +1 @@
|
|
|
1
|
-
.gallery-index,.gallery-one{max-width:720px;padding:var(--s4);min-width:0;margin:0 auto}.gallery-index h1{font-size:var(--t-title);font-weight:600}.gallery-index ul{gap:var(--s2);margin:0;padding:0;list-style:none;display:grid}.gallery-index li{gap:var(--s2);flex-wrap:wrap;align-items:baseline;min-height:44px;display:flex}.gallery-index a{color:var(--amber)}.gallery-index code{font:var(--t-note) var(--mono);color:var(--faint)}.gallery-nav{gap:var(--s2);margin-bottom:var(--s3);font-size:var(--t-note);color:var(--dim);align-items:center;min-height:44px;display:flex}.gallery-nav a{color:var(--amber)}.note{color:var(--dim);font-size:var(--t-label);margin:0}
|
|
1
|
+
.gallery-index,.gallery-one{max-width:720px;padding:var(--s4);min-width:0;margin:0 auto}.gallery-index h1{font-size:var(--t-title);font-weight:600}.gallery-index ul{gap:var(--s2);margin:0;padding:0;list-style:none;display:grid}.gallery-index li{gap:var(--s2);flex-wrap:wrap;align-items:baseline;min-height:44px;display:flex}.gallery-index a{color:var(--amber)}.gallery-index code{font:var(--t-note) var(--mono);color:var(--faint)}.gallery-nav{gap:var(--s2);margin-bottom:var(--s3);font-size:var(--t-note);color:var(--dim);align-items:center;min-height:44px;display:flex}.gallery-nav a{color:var(--amber)}.note{color:var(--dim);font-size:var(--t-label);margin:0}.gallery-row{gap:var(--s2);flex-wrap:wrap;display:flex}
|