evals-lab 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +95 -0
- package/README.md +17 -9
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +3 -7
- package/lab/demo/pipelines/demo-2.json +3 -7
- package/lab/evals-core.mjs +510 -444
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +35 -33
- package/lab/server.py +442 -115
- package/lab/web/dist/assets/gallery-B-7oyY37.js +3 -0
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/main-C_b7QoTv.css +1 -0
- package/lab/web/dist/assets/main-DyDG-V9N.js +21 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DsetJSXv.js +0 -3
- package/lab/web/dist/assets/main-B-VtDGxC.css +0 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +0 -19
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +0 -51
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +0 -1
package/lab/server.py
CHANGED
|
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
|
|
|
24
24
|
|
|
25
25
|
import base64
|
|
26
26
|
import calendar
|
|
27
|
+
import gzip
|
|
27
28
|
import hashlib
|
|
28
29
|
import hmac
|
|
29
30
|
import io
|
|
@@ -247,6 +248,35 @@ TYPES = {
|
|
|
247
248
|
".png": "image/png",
|
|
248
249
|
}
|
|
249
250
|
|
|
251
|
+
# What is gzipped for a client that asks (#225): text, which compresses five
|
|
252
|
+
# to ten times, and nothing already compressed. A body under GZIP_MIN goes as
|
|
253
|
+
# it is, since the header and the gzip frame would outweigh the saving.
|
|
254
|
+
COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
|
|
255
|
+
GZIP_MIN = 1024
|
|
256
|
+
# The built page's bundles, gzipped once: a new build is new names.
|
|
257
|
+
GZIPPED: dict = {}
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def accepts_gzip(header: str) -> bool:
|
|
261
|
+
"""Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
|
|
262
|
+
above 0. `gzip;q=0` is a refusal, not a request."""
|
|
263
|
+
star = False
|
|
264
|
+
for part in (header or "").split(","):
|
|
265
|
+
name, _, params = part.partition(";")
|
|
266
|
+
name, q = name.strip().lower(), 1.0
|
|
267
|
+
for p in params.split(";"):
|
|
268
|
+
k, _, v = p.partition("=")
|
|
269
|
+
if k.strip().lower() == "q":
|
|
270
|
+
try:
|
|
271
|
+
q = float(v)
|
|
272
|
+
except ValueError:
|
|
273
|
+
q = 0.0
|
|
274
|
+
if name == "gzip":
|
|
275
|
+
return q > 0
|
|
276
|
+
if name == "*":
|
|
277
|
+
star = q > 0
|
|
278
|
+
return star
|
|
279
|
+
|
|
250
280
|
|
|
251
281
|
# What a Source may accept and ever be served back as -- deliberately not
|
|
252
282
|
# TYPES: an upload that could come back as text/html is stored XSS, and the
|
|
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
|
|
|
469
499
|
# A key taken off this list is no longer served or written, and its rows stay
|
|
470
500
|
# in the store: a document is not migrated or deleted because nothing reads it.
|
|
471
501
|
# promptlab.cases, .rules and their .base copies went that way when a dataset
|
|
472
|
-
# became a row of its own (#86).
|
|
502
|
+
# became a row of its own (#86), and promptlab.mappings once a dataset named
|
|
503
|
+
# the Source it grades (#199).
|
|
473
504
|
SYNCED = ("promptlab.workflows", "promptlab.profiles",
|
|
474
|
-
"promptlab.versions", "promptlab.tokens"
|
|
505
|
+
"promptlab.versions", "promptlab.tokens")
|
|
475
506
|
MAX_DOC = 8 * 1024 * 1024
|
|
476
507
|
RUNS_PAGE = 25
|
|
477
508
|
# A dataset request's caps, read from Content-Length before the body is, as a
|
|
@@ -1306,7 +1337,7 @@ class Sources:
|
|
|
1306
1337
|
|
|
1307
1338
|
# ---- Datasets ----------------------------------------------------------------
|
|
1308
1339
|
#
|
|
1309
|
-
# A dataset is data a graded
|
|
1340
|
+
# A dataset is data a graded eval names by id: its cases. (The prompt a new
|
|
1310
1341
|
# scenario starts from is the Prompt library's Default, below; a dataset from
|
|
1311
1342
|
# before the library held one, and gave it to the library once.) It is one row in the
|
|
1312
1343
|
# store's SQLite, its body one JSON document with a version that goes up by one
|
|
@@ -1325,44 +1356,158 @@ class Sources:
|
|
|
1325
1356
|
# A lab from before this held a one-time import's `meta` row saying it ran;
|
|
1326
1357
|
# it is left where it is, and nothing reads it.
|
|
1327
1358
|
|
|
1328
|
-
DATASET_FIELDS = ("cases"
|
|
1359
|
+
DATASET_FIELDS = ("version", "source", "scoring", "grader", "every", "run", "cases")
|
|
1360
|
+
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 7 is an
|
|
1361
|
+
# eval group (docs/pipeline-model.md §17), version 6 its cases alone; version
|
|
1362
|
+
# 5 was told by its `source` alone, and earlier ones by neither.
|
|
1363
|
+
DATASET_BODY_VERSION = 7
|
|
1329
1364
|
DATASET_NAME_MAX = 80
|
|
1330
|
-
# The file forms Export writes and Import reads. Export writes version
|
|
1331
|
-
# Import reads it and versions 1 to
|
|
1365
|
+
# The file forms Export writes and Import reads. Export writes version 7;
|
|
1366
|
+
# Import reads it and versions 1 to 6, upgraded, and refuses anything else,
|
|
1332
1367
|
# as a pipeline of another version is refused. Versions 1 to 3 carried a
|
|
1333
1368
|
# prompt, which an import gives to the Prompt library.
|
|
1334
1369
|
EXPORT_ONE = "evals-lab/dataset"
|
|
1335
1370
|
EXPORT_ALL = "evals-lab/datasets"
|
|
1336
|
-
EXPORT_VERSION =
|
|
1337
|
-
IMPORT_VERSIONS = (1, 2, 3, 4)
|
|
1371
|
+
EXPORT_VERSION = 7
|
|
1372
|
+
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6, 7)
|
|
1373
|
+
SCORING_MODES = ("all", "weighted")
|
|
1338
1374
|
|
|
1339
1375
|
|
|
1340
1376
|
def blank_dataset() -> dict:
|
|
1341
|
-
return {"cases": []}
|
|
1377
|
+
return group_of_v6({"source": None, "cases": []})
|
|
1378
|
+
|
|
1379
|
+
|
|
1380
|
+
# vocab: the names older versions gave a case's fields
|
|
1381
|
+
CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
|
|
1382
|
+
CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
def _words(s) -> list:
|
|
1386
|
+
"""evals-core.ts's words: the letters and digits of [s], lowercased."""
|
|
1387
|
+
return re.findall(r"[^\W_]+", str(s or "").lower())
|
|
1388
|
+
|
|
1389
|
+
|
|
1390
|
+
def _term_in(items, term) -> bool:
|
|
1391
|
+
"""evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
|
|
1392
|
+
t = _words(term)
|
|
1393
|
+
if not t:
|
|
1394
|
+
return False
|
|
1395
|
+
for item in items:
|
|
1396
|
+
w = _words(item)
|
|
1397
|
+
if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
|
|
1398
|
+
return True
|
|
1399
|
+
return False
|
|
1400
|
+
|
|
1401
|
+
|
|
1402
|
+
def case_metrics(c: dict) -> list:
|
|
1403
|
+
"""A version-4 case's expectations as the metrics that say the same:
|
|
1404
|
+
evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
|
|
1405
|
+
through fixtures/dataset-v7.json."""
|
|
1406
|
+
def strs(v):
|
|
1407
|
+
return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
|
|
1408
|
+
if c.get("discarded") is True:
|
|
1409
|
+
return [{"type": "discarded"}]
|
|
1410
|
+
out = []
|
|
1411
|
+
expect, allow = strs(c.get("expect")), strs(c.get("allow"))
|
|
1412
|
+
if expect:
|
|
1413
|
+
out.append({"type": "contains-all", "values": "\n".join(expect)})
|
|
1414
|
+
for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
|
|
1415
|
+
if strs(g):
|
|
1416
|
+
out.append({"type": "contains-any", "values": "\n".join(strs(g))})
|
|
1417
|
+
for t in strs(c.get("forbid")):
|
|
1418
|
+
# An exception excuses only the forbidden term inside it.
|
|
1419
|
+
except_ = [a for a in allow if _term_in([a], t)]
|
|
1420
|
+
out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
|
|
1421
|
+
whole = lambda v: v if type(v) is int else None
|
|
1422
|
+
lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
|
|
1423
|
+
if lo is not None or hi is not None:
|
|
1424
|
+
out.append({"type": "item-count", "min": lo, "max": hi})
|
|
1425
|
+
for t in strs(c.get("watch")):
|
|
1426
|
+
out.append({"type": "contains-any", "values": t, "weight": 0})
|
|
1427
|
+
return out
|
|
1428
|
+
|
|
1429
|
+
|
|
1430
|
+
def case_of_v4(c):
|
|
1431
|
+
"""One case of any earlier version as a version-5 one: evals-core.ts's
|
|
1432
|
+
caseOfV4, in Python."""
|
|
1433
|
+
if not isinstance(c, dict):
|
|
1434
|
+
return c
|
|
1435
|
+
was = {}
|
|
1436
|
+
for k, v in c.items():
|
|
1437
|
+
key = CASE_RENAMED.get(k, k)
|
|
1438
|
+
if key not in was or key == k:
|
|
1439
|
+
was[key] = v
|
|
1440
|
+
if isinstance(was.get("item"), str):
|
|
1441
|
+
was.pop("filename", None)
|
|
1442
|
+
out = {}
|
|
1443
|
+
if "id" in was:
|
|
1444
|
+
out["id"] = was["id"]
|
|
1445
|
+
out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
|
|
1446
|
+
out["todo"] = was.get("todo") is True
|
|
1447
|
+
out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
|
|
1448
|
+
out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
|
|
1449
|
+
for k, v in was.items():
|
|
1450
|
+
if k not in out and k not in CASE_V4:
|
|
1451
|
+
out[k] = v
|
|
1452
|
+
return out
|
|
1453
|
+
|
|
1454
|
+
|
|
1455
|
+
# The metrics whose Ignore case version 6 made mean what it says for a reply
|
|
1456
|
+
# read as a list: evals-core.ts's CASE_FOLDING.
|
|
1457
|
+
CASE_FOLDING = ("contains", "contains-all", "contains-any")
|
|
1458
|
+
|
|
1459
|
+
|
|
1460
|
+
def case_of_v5(c):
|
|
1461
|
+
"""A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
|
|
1462
|
+
Python. Each Contains metric says Ignore case, as version 5 matched a
|
|
1463
|
+
list's items whatever it said."""
|
|
1464
|
+
if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
|
|
1465
|
+
return c
|
|
1466
|
+
return {**c, "metrics": [{**m, "ignoreCase": True}
|
|
1467
|
+
if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
|
|
1468
|
+
else m for m in c["metrics"]]}
|
|
1469
|
+
|
|
1470
|
+
|
|
1471
|
+
def group_of_v6(body: dict) -> dict:
|
|
1472
|
+
"""A version-6 body as a version-7 eval group: evals-core.ts's
|
|
1473
|
+
groupOfV6. Scored All, the lab's grader, and no metrics of its own for
|
|
1474
|
+
every item or the whole run -- what a Metrics eval naming the dataset with
|
|
1475
|
+
none of its own graded."""
|
|
1476
|
+
rest = {k: v for k, v in body.items() if k != "version"}
|
|
1477
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "scoring": {"mode": "all", "threshold": None},
|
|
1478
|
+
"grader": None, "every": [], "run": [], **rest}
|
|
1342
1479
|
|
|
1343
1480
|
|
|
1344
1481
|
def upgrade_body(body):
|
|
1345
|
-
"""An earlier body as today's: evals-core.ts's
|
|
1346
|
-
Python. Version 1's `imageCases` are `cases`,
|
|
1347
|
-
|
|
1348
|
-
`
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1482
|
+
"""An earlier body as today's (version 7): evals-core.ts's
|
|
1483
|
+
upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
|
|
1484
|
+
its `replays` and `conformance` go (fixtures/replays.json holds the
|
|
1485
|
+
parser's tests). Version 2's `rules` go -- they clean a job's answer, so
|
|
1486
|
+
they are the job's. Version 3's `prompt` goes: the Prompt library holds
|
|
1487
|
+
prompts now. Version 4's case named its item `filename` and said what a
|
|
1488
|
+
good answer is in expectations; each becomes its metric, `why` the
|
|
1489
|
+
`note`, `traits` go, and the body names no Source yet. A caller that
|
|
1490
|
+
needs the rules or the prompt takes them first (`body_rules`,
|
|
1491
|
+
`body_prompt`). A body naming its Source is version 5, whose Contains
|
|
1492
|
+
metrics each come to say Ignore case (`case_of_v5`). Version 6 gains a
|
|
1493
|
+
group's scoring, grader, Every item and Whole run (`group_of_v6`). A body
|
|
1494
|
+
saying it is version 7 comes back as it was; so does anything that is
|
|
1495
|
+
not a body."""
|
|
1496
|
+
if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
|
|
1355
1497
|
return body
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
return
|
|
1360
|
-
|
|
1498
|
+
# A body saying any other version is one this lab does not read, and is
|
|
1499
|
+
# left for dataset_problem to refuse.
|
|
1500
|
+
if "version" in body:
|
|
1501
|
+
return group_of_v6(body) if body["version"] == 6 else body
|
|
1502
|
+
if "source" in body:
|
|
1503
|
+
up = dict(body)
|
|
1504
|
+
if isinstance(body.get("cases"), list):
|
|
1505
|
+
up["cases"] = [case_of_v5(c) for c in body["cases"]]
|
|
1506
|
+
return group_of_v6(up)
|
|
1361
1507
|
cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
|
|
1362
1508
|
if not isinstance(cases, list):
|
|
1363
1509
|
return body
|
|
1364
|
-
|
|
1365
|
-
return {"cases": canonical_cases(cases)}
|
|
1510
|
+
return group_of_v6({"source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]})
|
|
1366
1511
|
|
|
1367
1512
|
|
|
1368
1513
|
def body_prompt(body):
|
|
@@ -1377,21 +1522,6 @@ def body_rules(body):
|
|
|
1377
1522
|
return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
|
|
1378
1523
|
|
|
1379
1524
|
|
|
1380
|
-
def canonical_cases(cases: list) -> list:
|
|
1381
|
-
"""Every case naming its file as `filename`: evals-core.ts's
|
|
1382
|
-
canonicalCases, in Python. `old` is the key a set re-synced from an
|
|
1383
|
-
app's own repository arrives with; the key keeps its place, so only its
|
|
1384
|
-
spelling changes."""
|
|
1385
|
-
old = "photo" # vocab: the older spelling of filename
|
|
1386
|
-
out = []
|
|
1387
|
-
for c in cases:
|
|
1388
|
-
if isinstance(c, dict) and old in c:
|
|
1389
|
-
c = {("filename" if k == old else k): v for k, v in c.items()
|
|
1390
|
-
if not (k == old and "filename" in c)}
|
|
1391
|
-
out.append(c)
|
|
1392
|
-
return out
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
1525
|
def dataset_problem(body) -> str:
|
|
1396
1526
|
"""Why [body] is not a dataset's body, in one sentence, or ""."""
|
|
1397
1527
|
if not isinstance(body, dict):
|
|
@@ -1402,8 +1532,29 @@ def dataset_problem(body) -> str:
|
|
|
1402
1532
|
for k in DATASET_FIELDS:
|
|
1403
1533
|
if k not in body:
|
|
1404
1534
|
return f"a dataset's body has no \"{k}\""
|
|
1535
|
+
if body["version"] != DATASET_BODY_VERSION:
|
|
1536
|
+
return f"a dataset's body is version {DATASET_BODY_VERSION}"
|
|
1405
1537
|
if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
|
|
1406
1538
|
return "cases has to be a list of cases"
|
|
1539
|
+
def is_ref(v):
|
|
1540
|
+
return isinstance(v, dict) and isinstance(v.get("id"), str) and isinstance(v.get("name"), str)
|
|
1541
|
+
if body["source"] is not None and not is_ref(body["source"]):
|
|
1542
|
+
return "a dataset names its Source as { id, name }, or null"
|
|
1543
|
+
# A group's own sections: their metrics are evals-core.ts's validateEvals'
|
|
1544
|
+
# to judge, as a case's are; the shape is the server's.
|
|
1545
|
+
sc = body["scoring"]
|
|
1546
|
+
if not isinstance(sc, dict) or sc.get("mode") not in SCORING_MODES:
|
|
1547
|
+
return "a group is scored all or weighted"
|
|
1548
|
+
number = lambda v: type(v) in (int, float)
|
|
1549
|
+
if sc["mode"] == "weighted" and not number(sc.get("threshold")):
|
|
1550
|
+
return "a group scored in points needs Pass at: the points an item has to reach"
|
|
1551
|
+
if sc.get("threshold") is not None and not number(sc.get("threshold")):
|
|
1552
|
+
return "a group's Pass at has to be a number"
|
|
1553
|
+
if body["grader"] is not None and not is_ref(body["grader"]):
|
|
1554
|
+
return "a group names its grader as { id, name }, or null"
|
|
1555
|
+
for k, label in (("every", "Every item"), ("run", "Whole run")):
|
|
1556
|
+
if not isinstance(body[k], list) or not all(isinstance(m, dict) for m in body[k]):
|
|
1557
|
+
return f"{label} has to be a list of metrics"
|
|
1407
1558
|
return ""
|
|
1408
1559
|
|
|
1409
1560
|
|
|
@@ -1788,6 +1939,11 @@ class Datasets:
|
|
|
1788
1939
|
# so nothing is lost if a pipeline was missed.
|
|
1789
1940
|
db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
|
|
1790
1941
|
"dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1942
|
+
# Each row's body as it was before the conversion below rewrote
|
|
1943
|
+
# it (#199): a version-4 case's expectations became metrics, and
|
|
1944
|
+
# the body it was typed as is kept, as the rules were.
|
|
1945
|
+
db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
|
|
1946
|
+
"dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1791
1947
|
# Rows from an earlier version are converted once, in place: a
|
|
1792
1948
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
1793
1949
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -1803,9 +1959,15 @@ class Datasets:
|
|
|
1803
1959
|
given = False
|
|
1804
1960
|
for did, name, version, raw in rows:
|
|
1805
1961
|
body = json.loads(raw)
|
|
1962
|
+
# A version-6 body is read as version 7 (`_doc`) and saved as
|
|
1963
|
+
# one at its next edit, never rewritten here (§17).
|
|
1964
|
+
if isinstance(body, dict) and body.get("version") == 6:
|
|
1965
|
+
continue
|
|
1806
1966
|
up = upgrade_body(body)
|
|
1807
1967
|
if up is not body:
|
|
1808
1968
|
self._archive(db, did, body)
|
|
1969
|
+
db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
|
|
1970
|
+
(did, raw, self._now()))
|
|
1809
1971
|
if prompts is not None:
|
|
1810
1972
|
given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
|
|
1811
1973
|
db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
|
|
@@ -1831,8 +1993,9 @@ class Datasets:
|
|
|
1831
1993
|
|
|
1832
1994
|
@staticmethod
|
|
1833
1995
|
def _doc(r, body=True):
|
|
1834
|
-
"""A row as the API answers it: a DatasetSummary, and its body with it
|
|
1835
|
-
|
|
1996
|
+
"""A row as the API answers it: a DatasetSummary, and its body with it,
|
|
1997
|
+
read as today's version."""
|
|
1998
|
+
parsed = upgrade_body(json.loads(r["body"]))
|
|
1836
1999
|
out = {"id": r["id"], "name": r["name"], "cases": len(parsed.get("cases") or []),
|
|
1837
2000
|
"version": r["version"], "updated": r["updated_at"]}
|
|
1838
2001
|
if body:
|
|
@@ -1874,7 +2037,6 @@ class Datasets:
|
|
|
1874
2037
|
def _insert(self, db, name, body):
|
|
1875
2038
|
did = secrets.token_hex(6)
|
|
1876
2039
|
now = self._now()
|
|
1877
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1878
2040
|
db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
|
|
1879
2041
|
"VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
|
|
1880
2042
|
return did
|
|
@@ -1885,7 +2047,7 @@ class Datasets:
|
|
|
1885
2047
|
name, why = dataset_name(name)
|
|
1886
2048
|
if why:
|
|
1887
2049
|
return None, (400, why)
|
|
1888
|
-
body = blank_dataset() if body is None else body
|
|
2050
|
+
body = blank_dataset() if body is None else upgrade_body(body)
|
|
1889
2051
|
why = dataset_problem(body)
|
|
1890
2052
|
if why:
|
|
1891
2053
|
return None, (400, why)
|
|
@@ -1914,10 +2076,12 @@ class Datasets:
|
|
|
1914
2076
|
the current row."""
|
|
1915
2077
|
if type(version) is not int:
|
|
1916
2078
|
return None, (400, "a save names the version it began from")
|
|
2079
|
+
# A body of an earlier version -- from a page loaded before this one --
|
|
2080
|
+
# is read as today's, as an import is.
|
|
2081
|
+
body = upgrade_body(body)
|
|
1917
2082
|
why = dataset_problem(body)
|
|
1918
2083
|
if why:
|
|
1919
2084
|
return None, (400, why)
|
|
1920
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1921
2085
|
with self.store.lock, self._connect() as db, db:
|
|
1922
2086
|
r = self._live(db, did)
|
|
1923
2087
|
if r is None:
|
|
@@ -1975,7 +2139,7 @@ class Datasets:
|
|
|
1975
2139
|
rows = db.execute("SELECT * FROM datasets WHERE trash IS NULL "
|
|
1976
2140
|
"ORDER BY name COLLATE NOCASE, id").fetchall()
|
|
1977
2141
|
return {"format": EXPORT_ALL, "version": EXPORT_VERSION,
|
|
1978
|
-
"datasets": [{"name": r["name"], "body": json.loads(r["body"])} for r in rows]}
|
|
2142
|
+
"datasets": [{"name": r["name"], "body": upgrade_body(json.loads(r["body"]))} for r in rows]}
|
|
1979
2143
|
|
|
1980
2144
|
def import_file(self, doc):
|
|
1981
2145
|
"""Either export's file, as new datasets: import always creates, ids
|
|
@@ -2328,6 +2492,13 @@ class Packs:
|
|
|
2328
2492
|
ds_ids = {}
|
|
2329
2493
|
cuts = [] # (kind, id, the document as it was), kept before it changes
|
|
2330
2494
|
for key, name, body, raw in pack["datasets"]:
|
|
2495
|
+
# The Source a dataset grades may be the pack's own, named by
|
|
2496
|
+
# its folder: pointed at the Source the pack made here.
|
|
2497
|
+
ref = body.get("source")
|
|
2498
|
+
if isinstance(ref, dict):
|
|
2499
|
+
sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
|
|
2500
|
+
if sid:
|
|
2501
|
+
body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
|
|
2331
2502
|
did = owned.get(("dataset", key))
|
|
2332
2503
|
current = DATASETS.get(did) if did else None
|
|
2333
2504
|
if current:
|
|
@@ -2349,8 +2520,10 @@ class Packs:
|
|
|
2349
2520
|
new_work = []
|
|
2350
2521
|
for key, doc in pack["pipelines"]:
|
|
2351
2522
|
doc = json.loads(json.dumps(doc))
|
|
2352
|
-
|
|
2353
|
-
|
|
2523
|
+
# A pack written before version 11 spells its evals `tests`;
|
|
2524
|
+
# the page upgrades the pipeline as it reads it.
|
|
2525
|
+
evals = doc.get("evals", doc.get("tests"))
|
|
2526
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
2354
2527
|
ref = t.get("dataset") if isinstance(t, dict) else None
|
|
2355
2528
|
if isinstance(ref, dict):
|
|
2356
2529
|
did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
|
|
@@ -2487,7 +2660,7 @@ class Packs:
|
|
|
2487
2660
|
#
|
|
2488
2661
|
# A plugin is code, installed like a pack (docs/packs.md): a zip of a
|
|
2489
2662
|
# manifest and the compiled JavaScript that registers what the lab lacks -- a
|
|
2490
|
-
# kind of answer, a modifier,
|
|
2663
|
+
# kind of answer, a modifier, an eval type, a connection type. Its code runs in
|
|
2491
2664
|
# the page and in the runner, where the registries live; this server never
|
|
2492
2665
|
# runs it. It reads the manifest's `registers` as data, so it can refuse two
|
|
2493
2666
|
# plugins registering one id, and so a connection type's settings, chat path
|
|
@@ -2505,7 +2678,19 @@ PLUGIN_VERSIONS = (1,)
|
|
|
2505
2678
|
PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
|
|
2506
2679
|
PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
|
|
2507
2680
|
PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
|
|
2508
|
-
REGISTRIES = ("outputKinds", "modifiers", "
|
|
2681
|
+
REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
|
|
2682
|
+
# evalTypes as a manifest written before pipeline version 11 spells it: a
|
|
2683
|
+
# plugin's own file, which the lab cannot upgrade, so it is read for good.
|
|
2684
|
+
OLD_REGISTRIES = {"testTypes": "evalTypes"}
|
|
2685
|
+
|
|
2686
|
+
|
|
2687
|
+
def plugin_registers(manifest: dict) -> dict:
|
|
2688
|
+
"""A manifest's registers under today's names: testTypes read as evalTypes."""
|
|
2689
|
+
reg = dict(manifest.get("registers") or {})
|
|
2690
|
+
for old, now in OLD_REGISTRIES.items():
|
|
2691
|
+
if old in reg:
|
|
2692
|
+
reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
|
|
2693
|
+
return reg
|
|
2509
2694
|
AUTH_WAYS = ("bearer", "x-api-key", "none")
|
|
2510
2695
|
|
|
2511
2696
|
|
|
@@ -2545,9 +2730,9 @@ def read_plugin(data: bytes):
|
|
|
2545
2730
|
if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
|
|
2546
2731
|
return None, "the plugin's entry names no JavaScript file it holds"
|
|
2547
2732
|
reg = m.get("registers") or {}
|
|
2548
|
-
if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
|
|
2733
|
+
if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
|
|
2549
2734
|
return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
|
|
2550
|
-
for k in ("outputKinds", "modifiers", "
|
|
2735
|
+
for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
|
|
2551
2736
|
if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
|
|
2552
2737
|
return None, f"registers.{k} is a list of ids"
|
|
2553
2738
|
conns = reg.get("connectionTypes", [])
|
|
@@ -2570,8 +2755,8 @@ def read_plugin(data: bytes):
|
|
|
2570
2755
|
|
|
2571
2756
|
def registered_ids(manifest: dict) -> set:
|
|
2572
2757
|
"""(registry, id) for everything a plugin's manifest says it registers."""
|
|
2573
|
-
reg = manifest
|
|
2574
|
-
out = {(k, x) for k in ("outputKinds", "modifiers", "
|
|
2758
|
+
reg = plugin_registers(manifest)
|
|
2759
|
+
out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
|
|
2575
2760
|
return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
|
|
2576
2761
|
|
|
2577
2762
|
|
|
@@ -2924,7 +3109,7 @@ class Plugins:
|
|
|
2924
3109
|
return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
|
|
2925
3110
|
"entry": json.loads(r["manifest"])["entry"],
|
|
2926
3111
|
"description": json.loads(r["manifest"]).get("description") or "",
|
|
2927
|
-
"registers": json.loads(r["manifest"])
|
|
3112
|
+
"registers": plugin_registers(json.loads(r["manifest"])),
|
|
2928
3113
|
"installed": r["installed_at"]} for r in self._rows()]
|
|
2929
3114
|
|
|
2930
3115
|
def stamp(self) -> list:
|
|
@@ -3100,6 +3285,55 @@ def worker_refusal(code, stderr, env):
|
|
|
3100
3285
|
return text if len(text) <= 2000 else text[:2000] + "…"
|
|
3101
3286
|
|
|
3102
3287
|
|
|
3288
|
+
|
|
3289
|
+
# A list of runs is read for its figures -- History's table, Home's recent
|
|
3290
|
+
# runs, Runs' progress -- and a run's results are mostly what Results alone
|
|
3291
|
+
# shows: every reply, every stage's request and answer, every item a score
|
|
3292
|
+
# found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
|
|
3293
|
+
# list carries a brief copy of its results, `brief` says so, and one run
|
|
3294
|
+
# (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
|
|
3295
|
+
# whether it ran; a brief score keeps its verdict and counts in place of its
|
|
3296
|
+
# lists. The replies go too: an eval over the whole run reads them, and the
|
|
3297
|
+
# page asks for that run whole rather than every list carrying them.
|
|
3298
|
+
BRIEF_SCORE = ("pass", "score", "points", "skipped")
|
|
3299
|
+
COUNTED = ("found", "missed", "invented")
|
|
3300
|
+
|
|
3301
|
+
|
|
3302
|
+
def brief_score(score):
|
|
3303
|
+
if not isinstance(score, dict):
|
|
3304
|
+
return score
|
|
3305
|
+
out = {k: score[k] for k in BRIEF_SCORE if k in score}
|
|
3306
|
+
for k in COUNTED:
|
|
3307
|
+
if isinstance(score.get(k), list):
|
|
3308
|
+
out[k] = len(score[k])
|
|
3309
|
+
return out
|
|
3310
|
+
|
|
3311
|
+
|
|
3312
|
+
def brief_cell(cell):
|
|
3313
|
+
if not isinstance(cell, dict):
|
|
3314
|
+
return cell
|
|
3315
|
+
out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
|
|
3316
|
+
if isinstance(cell.get("res"), dict):
|
|
3317
|
+
out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
|
|
3318
|
+
if isinstance(cell.get("scores"), dict):
|
|
3319
|
+
out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
|
|
3320
|
+
elif "scores" in cell:
|
|
3321
|
+
out["scores"] = cell["scores"]
|
|
3322
|
+
# A run from before version 6 kept its one test's score as `score`, which
|
|
3323
|
+
# the page reads as `scores.t1`: kept under its own name for that.
|
|
3324
|
+
if "score" in cell:
|
|
3325
|
+
out["score"] = brief_score(cell["score"])
|
|
3326
|
+
return out
|
|
3327
|
+
|
|
3328
|
+
|
|
3329
|
+
def brief_row(row):
|
|
3330
|
+
def item(it):
|
|
3331
|
+
if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
|
|
3332
|
+
return it
|
|
3333
|
+
return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
|
|
3334
|
+
return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
|
|
3335
|
+
|
|
3336
|
+
|
|
3103
3337
|
class Queue:
|
|
3104
3338
|
STATUS = ("queued", "running", "done", "incomplete", "cancelled",
|
|
3105
3339
|
"failed", "interrupted")
|
|
@@ -3127,11 +3361,22 @@ class Queue:
|
|
|
3127
3361
|
"dataset TEXT)")
|
|
3128
3362
|
# A store from before runs kept their dataset gains the column;
|
|
3129
3363
|
# its rows have none, and are pinned when first they need one.
|
|
3130
|
-
|
|
3364
|
+
cols = [c[1] for c in db.execute("PRAGMA table_info(queue)")]
|
|
3365
|
+
if "dataset" not in cols:
|
|
3131
3366
|
db.execute("ALTER TABLE queue ADD COLUMN dataset TEXT")
|
|
3367
|
+
# The run a re-run was queued from (#245); every earlier row is
|
|
3368
|
+
# one of its own.
|
|
3369
|
+
if "rerun_of" not in cols:
|
|
3370
|
+
db.execute("ALTER TABLE queue ADD COLUMN rerun_of TEXT")
|
|
3132
3371
|
|
|
3133
3372
|
# ---- rows -----------------------------------------------------------
|
|
3134
3373
|
|
|
3374
|
+
# A row read with the submit time of the run it re-runs, if any: the
|
|
3375
|
+
# page names a run by that time (History's Run ID), so a "Re-run of"
|
|
3376
|
+
# note reads without fetching the original.
|
|
3377
|
+
SELECT = ("SELECT q.*, o.submitted_at FROM queue q "
|
|
3378
|
+
"LEFT JOIN queue o ON o.id = q.rerun_of")
|
|
3379
|
+
|
|
3135
3380
|
@staticmethod
|
|
3136
3381
|
def _row(r):
|
|
3137
3382
|
if r is None:
|
|
@@ -3142,6 +3387,8 @@ class Queue:
|
|
|
3142
3387
|
"snapshot": json.loads(r[6]), "results": json.loads(r[7]),
|
|
3143
3388
|
"progress": json.loads(r[8]), "totals": json.loads(r[9]),
|
|
3144
3389
|
"error": r[10],
|
|
3390
|
+
"rerunOf": r[12] if len(r) > 12 else None,
|
|
3391
|
+
"rerunOfAt": r[13] if len(r) > 13 else None,
|
|
3145
3392
|
}
|
|
3146
3393
|
|
|
3147
3394
|
# A row from before run documents has no version, and nothing here can
|
|
@@ -3155,24 +3402,26 @@ class Queue:
|
|
|
3155
3402
|
return row is not None and (row["snapshot"] or {}).get("version") in READABLE_VERSIONS
|
|
3156
3403
|
|
|
3157
3404
|
def _all(self, db):
|
|
3158
|
-
return [row for row in (self._row(r) for r in db.execute(
|
|
3405
|
+
return [row for row in (self._row(r) for r in db.execute(self.SELECT))
|
|
3159
3406
|
if self._readable(row)]
|
|
3160
3407
|
|
|
3161
3408
|
def get(self, rid):
|
|
3162
3409
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3163
|
-
row = self._row(db.execute(
|
|
3410
|
+
row = self._row(db.execute(self.SELECT + " WHERE q.id = ?",
|
|
3164
3411
|
(rid,)).fetchone())
|
|
3165
3412
|
return row if self._readable(row) else None
|
|
3166
3413
|
|
|
3167
|
-
def list(self, limit=RUNS_PAGE, before=None):
|
|
3414
|
+
def list(self, limit=RUNS_PAGE, before=None, full=False):
|
|
3168
3415
|
"""Runs, newest first, and whether more follow. `before` is a
|
|
3169
3416
|
`submittedAt` the page of runs stops at, so History can page through
|
|
3170
|
-
them the way it pages the runs store.
|
|
3417
|
+
them the way it pages the runs store. Each row is brief_row's unless
|
|
3418
|
+
`full` asks for the whole of it."""
|
|
3171
3419
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3172
3420
|
rows = self._all(db)
|
|
3173
3421
|
rows = [r for r in rows if before is None or r["submittedAt"] < before]
|
|
3174
3422
|
rows.sort(key=lambda r: r["submittedAt"], reverse=True)
|
|
3175
|
-
|
|
3423
|
+
page = rows[:limit]
|
|
3424
|
+
return (page if full else [brief_row(r) for r in page]), len(rows) > limit
|
|
3176
3425
|
|
|
3177
3426
|
def _set(self, rid, **fields):
|
|
3178
3427
|
sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
|
|
@@ -3181,13 +3430,13 @@ class Queue:
|
|
|
3181
3430
|
|
|
3182
3431
|
# ---- submit ---------------------------------------------------------
|
|
3183
3432
|
|
|
3184
|
-
def submit(self, run: dict, dataset=None):
|
|
3433
|
+
def submit(self, run: dict, dataset=None, rerun_of=None):
|
|
3185
3434
|
"""
|
|
3186
3435
|
A new queued run. `run` is the run document (docs/pipeline-model.md
|
|
3187
3436
|
§5): the pipeline, the profiles it resolved to without their keys, its
|
|
3188
3437
|
content's file list in order and with its repeats, and the dataset's
|
|
3189
|
-
version; `dataset` is that version's body, for a graded run
|
|
3190
|
-
the row. Its items are that list, or the one inline text, each through
|
|
3438
|
+
version; `dataset` is that version's body, for a graded run;
|
|
3439
|
+
`rerun_of` is the run a re-run was queued from. Returns the row. Its items are that list, or the one inline text, each through
|
|
3191
3440
|
every scenario -- so the total is the list's length, repeats and all,
|
|
3192
3441
|
the same count the runner and the page make.
|
|
3193
3442
|
"""
|
|
@@ -3196,11 +3445,12 @@ class Queue:
|
|
|
3196
3445
|
total = len(run_items(run))
|
|
3197
3446
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
3198
3447
|
db.execute("INSERT INTO queue (id, status, cancel, submitted_at, "
|
|
3199
|
-
"snapshot, results, progress, totals, dataset
|
|
3448
|
+
"snapshot, results, progress, totals, dataset, rerun_of) "
|
|
3449
|
+
"VALUES (?, ?, 0, ?, ?, ?, ?, ?, ?, ?)",
|
|
3200
3450
|
(rid, "queued", now, json.dumps(run), "[]",
|
|
3201
3451
|
json.dumps({"current": None, "n": 0, "total": total}),
|
|
3202
3452
|
json.dumps({"ran": 0, "passed": 0, "found": 0, "of": 0}),
|
|
3203
|
-
None if dataset is None else json.dumps(dataset)))
|
|
3453
|
+
None if dataset is None else json.dumps(dataset), rerun_of))
|
|
3204
3454
|
# Its prompts' uses, in the same transaction: a run is in the
|
|
3205
3455
|
# library the moment it is queued, or not queued at all.
|
|
3206
3456
|
if self.prompts is not None:
|
|
@@ -3208,7 +3458,7 @@ class Queue:
|
|
|
3208
3458
|
# Read back under the lock the worker dequeues under, so the
|
|
3209
3459
|
# answer is the run as it was queued: an idle worker can take it
|
|
3210
3460
|
# the moment the lock is let go.
|
|
3211
|
-
return self._row(db.execute(
|
|
3461
|
+
return self._row(db.execute(self.SELECT + " WHERE q.id = ?", (rid,)).fetchone())
|
|
3212
3462
|
|
|
3213
3463
|
def dataset(self, rid, raw=False):
|
|
3214
3464
|
"""The dataset body a readable run was submitted against, or None --
|
|
@@ -3241,7 +3491,7 @@ class Queue:
|
|
|
3241
3491
|
written beside the run document. A row queued before runs kept their
|
|
3242
3492
|
dataset has none, and is pinned to the dataset as it reads now, once,
|
|
3243
3493
|
so every later pass over it agrees. Returns (args, None) or (None, why)."""
|
|
3244
|
-
ref =
|
|
3494
|
+
ref = evals_dataset(run["snapshot"])
|
|
3245
3495
|
if ref is None:
|
|
3246
3496
|
return [], None
|
|
3247
3497
|
body = self.dataset(run["id"], raw=True)
|
|
@@ -3327,6 +3577,46 @@ class Queue:
|
|
|
3327
3577
|
shutil.rmtree(d, ignore_errors=True)
|
|
3328
3578
|
return n
|
|
3329
3579
|
|
|
3580
|
+
def rerun(self, rid):
|
|
3581
|
+
"""
|
|
3582
|
+
A new run from a finished one's document (#245): a new id and submit
|
|
3583
|
+
time, the same document -- the file revisions and the plugins it
|
|
3584
|
+
pinned -- and the dataset body it was graded by, so a re-run measures
|
|
3585
|
+
the model again rather than changed data. The original is left as it
|
|
3586
|
+
was; the new row names it in `rerunOf`. Refused, naming what is
|
|
3587
|
+
missing, when a pinned file or dataset version is no longer kept. A
|
|
3588
|
+
Target's key is read from its profile now, at dequeue, as for any
|
|
3589
|
+
run: none is ever stored in one.
|
|
3590
|
+
"""
|
|
3591
|
+
run = self.get(rid)
|
|
3592
|
+
if run is None:
|
|
3593
|
+
return None, (404, "no such run")
|
|
3594
|
+
if run["status"] in ("queued", "running"):
|
|
3595
|
+
return None, (409, "a run still in progress cannot be re-run")
|
|
3596
|
+
snap = run["snapshot"]
|
|
3597
|
+
_, err = self._pinned_files(snap)
|
|
3598
|
+
if err:
|
|
3599
|
+
return None, (409, err)
|
|
3600
|
+
ref = evals_dataset(snap)
|
|
3601
|
+
body = None
|
|
3602
|
+
if ref is not None:
|
|
3603
|
+
body = self.dataset(rid, raw=True)
|
|
3604
|
+
if body is None:
|
|
3605
|
+
# A run that never started kept no body: the dataset's, if it
|
|
3606
|
+
# still reads as the version the run was submitted against.
|
|
3607
|
+
now = DATASETS.snapshot(ref.get("id")) if DATASETS is not None else None
|
|
3608
|
+
if now is None or not ref.get("version") or now[1] != ref.get("version"):
|
|
3609
|
+
return None, (409, f"the version of the dataset {ref.get('name') or ref.get('id')!r} "
|
|
3610
|
+
"this run was submitted against is no longer kept")
|
|
3611
|
+
body = now[0]
|
|
3612
|
+
_, err = self._plugin_args(run)
|
|
3613
|
+
if err:
|
|
3614
|
+
return None, (409, err)
|
|
3615
|
+
_, err = worker_destinations(snap)
|
|
3616
|
+
if err:
|
|
3617
|
+
return None, (403, err)
|
|
3618
|
+
return self.submit(snap, body, rerun_of=rid), None
|
|
3619
|
+
|
|
3330
3620
|
def rerun_item(self, rid, index):
|
|
3331
3621
|
"""
|
|
3332
3622
|
One item against the run's snapshot, by its index, updating the row in
|
|
@@ -3473,34 +3763,48 @@ class Queue:
|
|
|
3473
3763
|
has gone fails the run with the reason stated. Returns the run dir,
|
|
3474
3764
|
or (None, error)."""
|
|
3475
3765
|
snap = run["snapshot"]
|
|
3476
|
-
|
|
3766
|
+
files, err = self._pinned_files(snap)
|
|
3767
|
+
if err:
|
|
3768
|
+
return None, err
|
|
3477
3769
|
rundir = self.dir / run["id"]
|
|
3478
3770
|
files_dir = rundir / "files"
|
|
3479
3771
|
shutil.rmtree(rundir, ignore_errors=True)
|
|
3480
3772
|
files_dir.mkdir(parents=True)
|
|
3481
|
-
|
|
3482
|
-
|
|
3483
|
-
label = ref.get("name") or ref.get("id")
|
|
3484
|
-
if SOURCES.get(ref.get("id")) is None:
|
|
3485
|
-
return None, f"the source {label!r} is gone"
|
|
3486
|
-
revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
|
|
3487
|
-
changed = []
|
|
3488
|
-
for name in dict.fromkeys(content.get("files") or []):
|
|
3489
|
-
path = SOURCES.file_path(ref["id"], name)
|
|
3490
|
-
if path is None or not path.is_file():
|
|
3491
|
-
return None, f"{name!r} is gone from the source {label!r}"
|
|
3492
|
-
if revs is not None:
|
|
3493
|
-
path = SOURCES.rev_path(ref["id"], name, revs.get(name))
|
|
3494
|
-
if path is None:
|
|
3495
|
-
changed.append(repr(name))
|
|
3496
|
-
continue
|
|
3497
|
-
shutil.copy2(path, files_dir / name)
|
|
3498
|
-
if changed:
|
|
3499
|
-
return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
|
|
3500
|
-
f"changed in the source {label!r} since this run was queued")
|
|
3773
|
+
for name, path in files:
|
|
3774
|
+
shutil.copy2(path, files_dir / name)
|
|
3501
3775
|
(rundir / "run.json").write_text(json.dumps(snap))
|
|
3502
3776
|
return rundir, None
|
|
3503
3777
|
|
|
3778
|
+
@staticmethod
|
|
3779
|
+
def _pinned_files(snap):
|
|
3780
|
+
"""Each file a run document reads, by name, and the path holding the
|
|
3781
|
+
bytes it pinned at submit -- or (None, why), naming the Source that
|
|
3782
|
+
has gone, or the files gone from it or changed since. A document
|
|
3783
|
+
with no Source reads no files."""
|
|
3784
|
+
content = content_of(snap) or {}
|
|
3785
|
+
if content.get("type") != "source":
|
|
3786
|
+
return [], None
|
|
3787
|
+
ref = content.get("ref") or {}
|
|
3788
|
+
label = ref.get("name") or ref.get("id")
|
|
3789
|
+
if SOURCES.get(ref.get("id")) is None:
|
|
3790
|
+
return None, f"the source {label!r} is gone"
|
|
3791
|
+
revs = content.get("revs") if isinstance(content.get("revs"), dict) else None
|
|
3792
|
+
files, changed = [], []
|
|
3793
|
+
for name in dict.fromkeys(content.get("files") or []):
|
|
3794
|
+
path = SOURCES.file_path(ref["id"], name)
|
|
3795
|
+
if path is None or not path.is_file():
|
|
3796
|
+
return None, f"{name!r} is gone from the source {label!r}"
|
|
3797
|
+
if revs is not None:
|
|
3798
|
+
path = SOURCES.rev_path(ref["id"], name, revs.get(name))
|
|
3799
|
+
if path is None:
|
|
3800
|
+
changed.append(repr(name))
|
|
3801
|
+
continue
|
|
3802
|
+
files.append((name, path))
|
|
3803
|
+
if changed:
|
|
3804
|
+
return None, (f"{', '.join(changed)} {'has' if len(changed) == 1 else 'have'} "
|
|
3805
|
+
f"changed in the source {label!r} since this run was queued")
|
|
3806
|
+
return files, None
|
|
3807
|
+
|
|
3504
3808
|
def _execute(self, run):
|
|
3505
3809
|
"""One run, FIFO. Never more than one of these at a time: the worker
|
|
3506
3810
|
is a single thread, so the queue is single-flight by construction."""
|
|
@@ -3748,7 +4052,7 @@ def worker_destinations(run: dict):
|
|
|
3748
4052
|
|
|
3749
4053
|
# A run document's fields, checked before it is accepted -- the rules
|
|
3750
4054
|
# docs/pipeline-model.md §6 gives the server, in Python because the server is
|
|
3751
|
-
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or
|
|
4055
|
+
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
|
|
3752
4056
|
# means is the runner's to judge, and it fails the run with a sentence if it
|
|
3753
4057
|
# cannot; what is here is what the server itself depends on: a version it
|
|
3754
4058
|
# reads, its own cap, content it can count and copy, and connections that
|
|
@@ -3756,21 +4060,23 @@ def worker_destinations(run: dict):
|
|
|
3756
4060
|
# 5: the pipeline and every chain have an id; version 4's scenarios keep
|
|
3757
4061
|
# theirs, and an older run's chains are read as they are, by position.
|
|
3758
4062
|
# 6: tests are an ordered list; a stored run's one test (or null) is read as
|
|
3759
|
-
# a list of one (
|
|
4063
|
+
# a list of one (evals_dataset).
|
|
3760
4064
|
# 7: chains are jobs: the field is `jobs` and each job's type is "job".
|
|
3761
4065
|
# 8: a job is its steps; the content is job 1's Attach Content step.
|
|
3762
|
-
# 9:
|
|
4066
|
+
# 9: an eval is Metrics; a Single Test or a Graded set is read converted.
|
|
3763
4067
|
# 10: a job's steps are its stages, and each scenario is a target whose own
|
|
3764
4068
|
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
3765
|
-
|
|
4069
|
+
# 11: `tests` are `evals`; nothing in an eval changes.
|
|
4070
|
+
# 12: a Contains metric's Ignore case holds item by item too, kept as written.
|
|
4071
|
+
PIPELINE_VERSION = 12
|
|
3766
4072
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
3767
4073
|
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
3768
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
|
|
4074
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
|
3769
4075
|
TARGET_CAP = 4
|
|
3770
4076
|
# Target steps whose words the Prompt library does not record as a use: they
|
|
3771
4077
|
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
3772
4078
|
UNRECORDED_STEPS = {"echo"}
|
|
3773
|
-
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "
|
|
4079
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
|
|
3774
4080
|
|
|
3775
4081
|
|
|
3776
4082
|
def content_of(doc):
|
|
@@ -3927,14 +4233,14 @@ def connection_problems(pid, conn, at, bad):
|
|
|
3927
4233
|
bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
|
|
3928
4234
|
|
|
3929
4235
|
|
|
3930
|
-
def
|
|
3931
|
-
"""The dataset reference a document's
|
|
3932
|
-
first
|
|
3933
|
-
validatePipeline says so). Reads a stored document of
|
|
3934
|
-
version
|
|
3935
|
-
document they were submitted with."""
|
|
3936
|
-
|
|
3937
|
-
for t in
|
|
4236
|
+
def evals_dataset(doc):
|
|
4237
|
+
"""The dataset reference a document's evals grade against, or None: the
|
|
4238
|
+
first eval that names one. A run grades against one dataset (the core's
|
|
4239
|
+
validatePipeline says so). Reads a stored document of any shape --
|
|
4240
|
+
version 11's `evals`, the `tests` before it, version 5's list or the one
|
|
4241
|
+
test before that -- since rows keep the document they were submitted with."""
|
|
4242
|
+
evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
|
|
4243
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
3938
4244
|
if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
|
|
3939
4245
|
return t["dataset"]
|
|
3940
4246
|
return None
|
|
@@ -4079,9 +4385,9 @@ def run_problems(run):
|
|
|
4079
4385
|
pass
|
|
4080
4386
|
else:
|
|
4081
4387
|
bad.append("a run needs a Source's files, some inline text, or Prompt only")
|
|
4082
|
-
|
|
4083
|
-
if not isinstance(
|
|
4084
|
-
bad.append("
|
|
4388
|
+
evals = run.get("evals")
|
|
4389
|
+
if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
|
|
4390
|
+
bad.append("evals has to be a list of evals, each naming its type")
|
|
4085
4391
|
if run.get("comment") is not None and not isinstance(run["comment"], str):
|
|
4086
4392
|
bad.append("comment has to be text")
|
|
4087
4393
|
return bad
|
|
@@ -4146,11 +4452,25 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4146
4452
|
|
|
4147
4453
|
# ---- helpers --------------------------------------------------------
|
|
4148
4454
|
|
|
4149
|
-
def _send(self, code, body: bytes, ctype="application/json", headers=()):
|
|
4455
|
+
def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
|
|
4456
|
+
"""`packed` keys a body that never changes under it -- a hashed
|
|
4457
|
+
bundle -- so it is gzipped once and kept, not on every request."""
|
|
4150
4458
|
self.send_response(code)
|
|
4151
4459
|
self.send_header("Content-Type", ctype)
|
|
4152
4460
|
for k, v in headers:
|
|
4153
4461
|
self.send_header(k, v)
|
|
4462
|
+
if ctype.startswith(COMPRESSIBLE):
|
|
4463
|
+
# Said whether or not this answer is compressed, so a cache
|
|
4464
|
+
# between here and the browser keys on it either way.
|
|
4465
|
+
self.send_header("Vary", "Accept-Encoding")
|
|
4466
|
+
if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
|
|
4467
|
+
if packed is None:
|
|
4468
|
+
body = gzip.compress(body, 6, mtime=0)
|
|
4469
|
+
else:
|
|
4470
|
+
if packed not in GZIPPED:
|
|
4471
|
+
GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
|
|
4472
|
+
body = GZIPPED[packed]
|
|
4473
|
+
self.send_header("Content-Encoding", "gzip")
|
|
4154
4474
|
self.send_header("Content-Length", str(len(body)))
|
|
4155
4475
|
self.end_headers()
|
|
4156
4476
|
self.wfile.write(body)
|
|
@@ -4235,7 +4555,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4235
4555
|
return
|
|
4236
4556
|
path = self.path.split("?", 1)[0]
|
|
4237
4557
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
4238
|
-
# Content and
|
|
4558
|
+
# Content and Evals are shared, and one to four scenarios run over the
|
|
4239
4559
|
# content. It replaced the A/B page it was prototyped beside once it
|
|
4240
4560
|
# carried everything that page did -- the graded set, the
|
|
4241
4561
|
# Configuration group, the history -- rather than being left to rot
|
|
@@ -4255,7 +4575,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4255
4575
|
if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
|
|
4256
4576
|
return self._send(404, b"not found", "text/plain")
|
|
4257
4577
|
return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
|
|
4258
|
-
(("Cache-Control", "public, max-age=31536000, immutable"),))
|
|
4578
|
+
(("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
|
|
4259
4579
|
if path == "/api/config":
|
|
4260
4580
|
# The lab's own limits, for the page to disable rather than
|
|
4261
4581
|
# hard-code: how many scenarios a run may have.
|
|
@@ -4279,14 +4599,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4279
4599
|
except ValueError:
|
|
4280
4600
|
return self._json(400, {"error": "limit has to be a number"})
|
|
4281
4601
|
before = (query.get("before") or [None])[0]
|
|
4282
|
-
runs, more = QUEUE.list(limit, before)
|
|
4602
|
+
runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
|
|
4283
4603
|
return self._json(200, {"runs": runs, "more": more})
|
|
4284
4604
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
4285
4605
|
# The dataset body a graded run was submitted against, which is
|
|
4286
4606
|
# what its verdicts were graded by; the list never carries it.
|
|
4287
4607
|
#
|
|
4288
4608
|
# A run that is there but kept no copy -- one submitted before the
|
|
4289
|
-
# lab kept them, or one with no graded
|
|
4609
|
+
# lab kept them, or one with no graded eval -- answers null, not
|
|
4290
4610
|
# 404. It is not an error: the page reads it as "use the dataset
|
|
4291
4611
|
# as it is now", and a 404 put a red line in the console every
|
|
4292
4612
|
# time such a run was opened, which is the console people are
|
|
@@ -4820,14 +5140,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4820
5140
|
# A graded run keeps the body of the dataset it names, as it reads
|
|
4821
5141
|
# now, and records that body's fingerprint on the reference it
|
|
4822
5142
|
# belongs to: the worker grades against that and nothing else.
|
|
4823
|
-
ref =
|
|
5143
|
+
ref = evals_dataset(run)
|
|
4824
5144
|
body = None
|
|
4825
5145
|
if ref is not None:
|
|
4826
5146
|
snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
|
|
4827
5147
|
if snap is None:
|
|
4828
|
-
return self._json(400, {"error": "the run's graded
|
|
5148
|
+
return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
|
|
4829
5149
|
body, version = snap
|
|
4830
|
-
for t in run["
|
|
5150
|
+
for t in run["evals"]:
|
|
4831
5151
|
if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
|
|
4832
5152
|
t["dataset"]["version"] = version
|
|
4833
5153
|
return self._json(201, {"run": QUEUE.submit(run, body)})
|
|
@@ -4985,6 +5305,9 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4985
5305
|
return self._send(404, b"not found", "text/plain")
|
|
4986
5306
|
|
|
4987
5307
|
def _queue_action(self, path):
|
|
5308
|
+
# Drained, so a keep-alive connection is not left holding the
|
|
5309
|
+
# page's `{}` in front of its next request.
|
|
5310
|
+
self._payload()
|
|
4988
5311
|
parts = path.split("/")
|
|
4989
5312
|
rid = parts[3]
|
|
4990
5313
|
action = parts[4] if len(parts) > 4 else ""
|
|
@@ -4992,6 +5315,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4992
5315
|
run, err = QUEUE.cancel(rid)
|
|
4993
5316
|
elif action == "resume":
|
|
4994
5317
|
run, err = QUEUE.resume(rid)
|
|
5318
|
+
elif action == "rerun" and len(parts) == 5:
|
|
5319
|
+
run, err = QUEUE.rerun(rid)
|
|
5320
|
+
if not err:
|
|
5321
|
+
return self._json(201, {"run": run})
|
|
4995
5322
|
elif action == "items" and len(parts) == 6:
|
|
4996
5323
|
run, err = QUEUE.rerun_item(rid, urllib.parse.unquote(parts[5]))
|
|
4997
5324
|
elif action == "rescore" and len(parts) == 6:
|