evals-lab 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +43 -0
- package/lab/VERSION +1 -1
- package/lab/demo/datasets/demo-1.json +2187 -1644
- package/lab/demo/datasets/demo-2.json +1870 -1485
- package/lab/demo/manifest.json +2 -2
- package/lab/demo/pipelines/demo-1.json +36 -30
- package/lab/demo/pipelines/demo-2.json +36 -30
- package/lab/evals-core.mjs +1122 -710
- package/lab/kinds/list.mjs +1 -1
- package/lab/metrics/builtin.mjs +96 -18
- package/lab/run-evals.js +77 -49
- package/lab/server.py +544 -154
- package/lab/web/dist/assets/{gallery-o7c4lfpn.css → gallery-DFeJkfUw.css} +1 -1
- package/lab/web/dist/assets/gallery-DRBlZ8mP.js +3 -0
- package/lab/web/dist/assets/main-DMpQmW8l.js +21 -0
- package/lab/web/dist/assets/main-wfqC6HcM.css +1 -0
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +1 -0
- package/lab/web/dist/assets/tokens-iGpvbD5R.js +55 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +3 -2
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-61NS6C4m.js +0 -18
- package/lab/web/dist/assets/main-DjQQums6.css +0 -1
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/server.py
CHANGED
|
@@ -24,6 +24,7 @@ given, so both follow their user between browsers; see Store and Sources.
|
|
|
24
24
|
|
|
25
25
|
import base64
|
|
26
26
|
import calendar
|
|
27
|
+
import gzip
|
|
27
28
|
import hashlib
|
|
28
29
|
import hmac
|
|
29
30
|
import io
|
|
@@ -247,6 +248,35 @@ TYPES = {
|
|
|
247
248
|
".png": "image/png",
|
|
248
249
|
}
|
|
249
250
|
|
|
251
|
+
# What is gzipped for a client that asks (#225): text, which compresses five
|
|
252
|
+
# to ten times, and nothing already compressed. A body under GZIP_MIN goes as
|
|
253
|
+
# it is, since the header and the gzip frame would outweigh the saving.
|
|
254
|
+
COMPRESSIBLE = ("text/", "application/json", "image/svg+xml")
|
|
255
|
+
GZIP_MIN = 1024
|
|
256
|
+
# The built page's bundles, gzipped once: a new build is new names.
|
|
257
|
+
GZIPPED: dict = {}
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def accepts_gzip(header: str) -> bool:
|
|
261
|
+
"""Whether an Accept-Encoding header takes gzip: named, or `*`, with a q
|
|
262
|
+
above 0. `gzip;q=0` is a refusal, not a request."""
|
|
263
|
+
star = False
|
|
264
|
+
for part in (header or "").split(","):
|
|
265
|
+
name, _, params = part.partition(";")
|
|
266
|
+
name, q = name.strip().lower(), 1.0
|
|
267
|
+
for p in params.split(";"):
|
|
268
|
+
k, _, v = p.partition("=")
|
|
269
|
+
if k.strip().lower() == "q":
|
|
270
|
+
try:
|
|
271
|
+
q = float(v)
|
|
272
|
+
except ValueError:
|
|
273
|
+
q = 0.0
|
|
274
|
+
if name == "gzip":
|
|
275
|
+
return q > 0
|
|
276
|
+
if name == "*":
|
|
277
|
+
star = q > 0
|
|
278
|
+
return star
|
|
279
|
+
|
|
250
280
|
|
|
251
281
|
# What a Source may accept and ever be served back as -- deliberately not
|
|
252
282
|
# TYPES: an upload that could come back as text/html is stored XSS, and the
|
|
@@ -469,9 +499,10 @@ SIGN_INS = {"microsoft": lambda: microsoft_config() is not None}
|
|
|
469
499
|
# A key taken off this list is no longer served or written, and its rows stay
|
|
470
500
|
# in the store: a document is not migrated or deleted because nothing reads it.
|
|
471
501
|
# promptlab.cases, .rules and their .base copies went that way when a dataset
|
|
472
|
-
# became a row of its own (#86).
|
|
502
|
+
# became a row of its own (#86), and promptlab.mappings once a dataset named
|
|
503
|
+
# the Source it grades (#199).
|
|
473
504
|
SYNCED = ("promptlab.workflows", "promptlab.profiles",
|
|
474
|
-
"promptlab.versions", "promptlab.tokens"
|
|
505
|
+
"promptlab.versions", "promptlab.tokens")
|
|
475
506
|
MAX_DOC = 8 * 1024 * 1024
|
|
476
507
|
RUNS_PAGE = 25
|
|
477
508
|
# A dataset request's caps, read from Content-Length before the body is, as a
|
|
@@ -1306,7 +1337,7 @@ class Sources:
|
|
|
1306
1337
|
|
|
1307
1338
|
# ---- Datasets ----------------------------------------------------------------
|
|
1308
1339
|
#
|
|
1309
|
-
# A dataset is data a graded
|
|
1340
|
+
# A dataset is data a graded eval names by id: its cases. (The prompt a new
|
|
1310
1341
|
# scenario starts from is the Prompt library's Default, below; a dataset from
|
|
1311
1342
|
# before the library held one, and gave it to the library once.) It is one row in the
|
|
1312
1343
|
# store's SQLite, its body one JSON document with a version that goes up by one
|
|
@@ -1325,44 +1356,140 @@ class Sources:
|
|
|
1325
1356
|
# A lab from before this held a one-time import's `meta` row saying it ran;
|
|
1326
1357
|
# it is left where it is, and nothing reads it.
|
|
1327
1358
|
|
|
1328
|
-
DATASET_FIELDS = ("cases"
|
|
1359
|
+
DATASET_FIELDS = ("version", "source", "cases")
|
|
1360
|
+
# A body's own version: evals-core.ts's DATASET_BODY_VERSION. Version 5 was
|
|
1361
|
+
# told by its `source` alone, and earlier ones by neither.
|
|
1362
|
+
DATASET_BODY_VERSION = 6
|
|
1329
1363
|
DATASET_NAME_MAX = 80
|
|
1330
|
-
# The file forms Export writes and Import reads. Export writes version
|
|
1331
|
-
# Import reads it and versions 1 to
|
|
1364
|
+
# The file forms Export writes and Import reads. Export writes version 6;
|
|
1365
|
+
# Import reads it and versions 1 to 5, upgraded, and refuses anything else,
|
|
1332
1366
|
# as a pipeline of another version is refused. Versions 1 to 3 carried a
|
|
1333
1367
|
# prompt, which an import gives to the Prompt library.
|
|
1334
1368
|
EXPORT_ONE = "evals-lab/dataset"
|
|
1335
1369
|
EXPORT_ALL = "evals-lab/datasets"
|
|
1336
|
-
EXPORT_VERSION =
|
|
1337
|
-
IMPORT_VERSIONS = (1, 2, 3, 4)
|
|
1370
|
+
EXPORT_VERSION = 6
|
|
1371
|
+
IMPORT_VERSIONS = (1, 2, 3, 4, 5, 6)
|
|
1338
1372
|
|
|
1339
1373
|
|
|
1340
1374
|
def blank_dataset() -> dict:
|
|
1341
|
-
return {"cases": []}
|
|
1375
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "cases": []}
|
|
1376
|
+
|
|
1377
|
+
|
|
1378
|
+
# vocab: the names older versions gave a case's fields
|
|
1379
|
+
CASE_RENAMED = {"minTags": "minCount", "maxTags": "maxCount", "textInImage": "watch", "photo": "filename"} # vocab: as above
|
|
1380
|
+
CASE_V4 = ("filename", "why", "traits", "expect", "anyOf", "forbid", "allow", "minCount", "maxCount", "watch", "discarded")
|
|
1381
|
+
|
|
1382
|
+
|
|
1383
|
+
def _words(s) -> list:
|
|
1384
|
+
"""evals-core.ts's words: the letters and digits of [s], lowercased."""
|
|
1385
|
+
return re.findall(r"[^\W_]+", str(s or "").lower())
|
|
1386
|
+
|
|
1387
|
+
|
|
1388
|
+
def _term_in(items, term) -> bool:
|
|
1389
|
+
"""evals-core.ts's termIn: [term]'s words in some item, in order and adjacent."""
|
|
1390
|
+
t = _words(term)
|
|
1391
|
+
if not t:
|
|
1392
|
+
return False
|
|
1393
|
+
for item in items:
|
|
1394
|
+
w = _words(item)
|
|
1395
|
+
if any(w[i:i + len(t)] == t for i in range(len(w) - len(t) + 1)):
|
|
1396
|
+
return True
|
|
1397
|
+
return False
|
|
1398
|
+
|
|
1399
|
+
|
|
1400
|
+
def case_metrics(c: dict) -> list:
|
|
1401
|
+
"""A version-4 case's expectations as the metrics that say the same:
|
|
1402
|
+
evals-core.ts's caseMetrics, in Python, and held to it by proxy-check.py
|
|
1403
|
+
through fixtures/dataset-v6.json."""
|
|
1404
|
+
def strs(v):
|
|
1405
|
+
return [x for x in v if isinstance(x, str)] if isinstance(v, list) else []
|
|
1406
|
+
if c.get("discarded") is True:
|
|
1407
|
+
return [{"type": "discarded"}]
|
|
1408
|
+
out = []
|
|
1409
|
+
expect, allow = strs(c.get("expect")), strs(c.get("allow"))
|
|
1410
|
+
if expect:
|
|
1411
|
+
out.append({"type": "contains-all", "values": "\n".join(expect)})
|
|
1412
|
+
for g in c.get("anyOf") if isinstance(c.get("anyOf"), list) else []:
|
|
1413
|
+
if strs(g):
|
|
1414
|
+
out.append({"type": "contains-any", "values": "\n".join(strs(g))})
|
|
1415
|
+
for t in strs(c.get("forbid")):
|
|
1416
|
+
# An exception excuses only the forbidden term inside it.
|
|
1417
|
+
except_ = [a for a in allow if _term_in([a], t)]
|
|
1418
|
+
out.append({"type": "contains", "value": t, "not": True, **({"except": "\n".join(except_)} if except_ else {})})
|
|
1419
|
+
whole = lambda v: v if type(v) is int else None
|
|
1420
|
+
lo, hi = whole(c.get("minCount")), whole(c.get("maxCount"))
|
|
1421
|
+
if lo is not None or hi is not None:
|
|
1422
|
+
out.append({"type": "item-count", "min": lo, "max": hi})
|
|
1423
|
+
for t in strs(c.get("watch")):
|
|
1424
|
+
out.append({"type": "contains-any", "values": t, "weight": 0})
|
|
1425
|
+
return out
|
|
1426
|
+
|
|
1427
|
+
|
|
1428
|
+
def case_of_v4(c):
|
|
1429
|
+
"""One case of any earlier version as a version-5 one: evals-core.ts's
|
|
1430
|
+
caseOfV4, in Python."""
|
|
1431
|
+
if not isinstance(c, dict):
|
|
1432
|
+
return c
|
|
1433
|
+
was = {}
|
|
1434
|
+
for k, v in c.items():
|
|
1435
|
+
key = CASE_RENAMED.get(k, k)
|
|
1436
|
+
if key not in was or key == k:
|
|
1437
|
+
was[key] = v
|
|
1438
|
+
if isinstance(was.get("item"), str):
|
|
1439
|
+
was.pop("filename", None)
|
|
1440
|
+
out = {}
|
|
1441
|
+
if "id" in was:
|
|
1442
|
+
out["id"] = was["id"]
|
|
1443
|
+
out["item"] = was["item"] if isinstance(was.get("item"), str) else was["filename"] if isinstance(was.get("filename"), str) else ""
|
|
1444
|
+
out["todo"] = was.get("todo") is True
|
|
1445
|
+
out["note"] = was["note"] if isinstance(was.get("note"), str) else was["why"] if isinstance(was.get("why"), str) else ""
|
|
1446
|
+
out["metrics"] = case_metrics(was) + (was["metrics"] if isinstance(was.get("metrics"), list) else [])
|
|
1447
|
+
for k, v in was.items():
|
|
1448
|
+
if k not in out and k not in CASE_V4:
|
|
1449
|
+
out[k] = v
|
|
1450
|
+
return out
|
|
1451
|
+
|
|
1452
|
+
|
|
1453
|
+
# The metrics whose Ignore case version 6 made mean what it says for a reply
|
|
1454
|
+
# read as a list: evals-core.ts's CASE_FOLDING.
|
|
1455
|
+
CASE_FOLDING = ("contains", "contains-all", "contains-any")
|
|
1456
|
+
|
|
1457
|
+
|
|
1458
|
+
def case_of_v5(c):
|
|
1459
|
+
"""A version-5 case as a version-6 one: evals-core.ts's caseOfV5, in
|
|
1460
|
+
Python. Each Contains metric says Ignore case, as version 5 matched a
|
|
1461
|
+
list's items whatever it said."""
|
|
1462
|
+
if not isinstance(c, dict) or not isinstance(c.get("metrics"), list):
|
|
1463
|
+
return c
|
|
1464
|
+
return {**c, "metrics": [{**m, "ignoreCase": True}
|
|
1465
|
+
if isinstance(m, dict) and m.get("type") in CASE_FOLDING and m.get("ignoreCase") is not True
|
|
1466
|
+
else m for m in c["metrics"]]}
|
|
1342
1467
|
|
|
1343
1468
|
|
|
1344
1469
|
def upgrade_body(body):
|
|
1345
|
-
"""An earlier body as today's: evals-core.ts's
|
|
1346
|
-
Python. Version 1's `imageCases` are `cases`,
|
|
1347
|
-
|
|
1348
|
-
`
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1470
|
+
"""An earlier body as today's (version 6): evals-core.ts's
|
|
1471
|
+
upgradeDatasetBody, in Python. Version 1's `imageCases` are `cases`, and
|
|
1472
|
+
its `replays` and `conformance` go (fixtures/replays.json holds the
|
|
1473
|
+
parser's tests). Version 2's `rules` go -- they clean a job's answer, so
|
|
1474
|
+
they are the job's. Version 3's `prompt` goes: the Prompt library holds
|
|
1475
|
+
prompts now. Version 4's case named its item `filename` and said what a
|
|
1476
|
+
good answer is in expectations; each becomes its metric, `why` the
|
|
1477
|
+
`note`, `traits` go, and the body names no Source yet. A caller that
|
|
1478
|
+
needs the rules or the prompt takes them first (`body_rules`,
|
|
1479
|
+
`body_prompt`). A body naming its Source is version 5, whose Contains
|
|
1480
|
+
metrics each come to say Ignore case (`case_of_v5`). A body saying it is
|
|
1481
|
+
version 6 comes back as it was; so does anything that is not a body."""
|
|
1482
|
+
if not isinstance(body, dict) or body.get("version") == DATASET_BODY_VERSION:
|
|
1355
1483
|
return body
|
|
1356
|
-
if "
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1484
|
+
if "source" in body:
|
|
1485
|
+
up = {"version": DATASET_BODY_VERSION, **body}
|
|
1486
|
+
if isinstance(body.get("cases"), list):
|
|
1487
|
+
up["cases"] = [case_of_v5(c) for c in body["cases"]]
|
|
1488
|
+
return up
|
|
1361
1489
|
cases = body.get("cases") if isinstance(body.get("cases"), list) else body.get("imageCases")
|
|
1362
1490
|
if not isinstance(cases, list):
|
|
1363
1491
|
return body
|
|
1364
|
-
|
|
1365
|
-
return {"cases": canonical_cases(cases)}
|
|
1492
|
+
return {"version": DATASET_BODY_VERSION, "source": None, "cases": [case_of_v5(case_of_v4(c)) for c in cases]}
|
|
1366
1493
|
|
|
1367
1494
|
|
|
1368
1495
|
def body_prompt(body):
|
|
@@ -1377,21 +1504,6 @@ def body_rules(body):
|
|
|
1377
1504
|
return rules if isinstance(rules, dict) and isinstance(rules.get("rules"), list) else None
|
|
1378
1505
|
|
|
1379
1506
|
|
|
1380
|
-
def canonical_cases(cases: list) -> list:
|
|
1381
|
-
"""Every case naming its file as `filename`: evals-core.ts's
|
|
1382
|
-
canonicalCases, in Python. `old` is the key a set re-synced from an
|
|
1383
|
-
app's own repository arrives with; the key keeps its place, so only its
|
|
1384
|
-
spelling changes."""
|
|
1385
|
-
old = "photo" # vocab: the older spelling of filename
|
|
1386
|
-
out = []
|
|
1387
|
-
for c in cases:
|
|
1388
|
-
if isinstance(c, dict) and old in c:
|
|
1389
|
-
c = {("filename" if k == old else k): v for k, v in c.items()
|
|
1390
|
-
if not (k == old and "filename" in c)}
|
|
1391
|
-
out.append(c)
|
|
1392
|
-
return out
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
1507
|
def dataset_problem(body) -> str:
|
|
1396
1508
|
"""Why [body] is not a dataset's body, in one sentence, or ""."""
|
|
1397
1509
|
if not isinstance(body, dict):
|
|
@@ -1402,8 +1514,14 @@ def dataset_problem(body) -> str:
|
|
|
1402
1514
|
for k in DATASET_FIELDS:
|
|
1403
1515
|
if k not in body:
|
|
1404
1516
|
return f"a dataset's body has no \"{k}\""
|
|
1517
|
+
if body["version"] != DATASET_BODY_VERSION:
|
|
1518
|
+
return f"a dataset's body is version {DATASET_BODY_VERSION}"
|
|
1405
1519
|
if not isinstance(body["cases"], list) or not all(isinstance(c, dict) for c in body["cases"]):
|
|
1406
1520
|
return "cases has to be a list of cases"
|
|
1521
|
+
src = body["source"]
|
|
1522
|
+
if src is not None and not (isinstance(src, dict) and isinstance(src.get("id"), str)
|
|
1523
|
+
and isinstance(src.get("name"), str)):
|
|
1524
|
+
return "a dataset names its Source as { id, name }, or null"
|
|
1407
1525
|
return ""
|
|
1408
1526
|
|
|
1409
1527
|
|
|
@@ -1476,7 +1594,7 @@ def scenario_ref(sc, i):
|
|
|
1476
1594
|
what version 4's upgrade gives one, by position."""
|
|
1477
1595
|
sid = sc.get("id") if isinstance(sc, dict) else None
|
|
1478
1596
|
name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
|
|
1479
|
-
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"
|
|
1597
|
+
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
|
|
1480
1598
|
|
|
1481
1599
|
|
|
1482
1600
|
class Prompts:
|
|
@@ -1632,27 +1750,28 @@ class Prompts:
|
|
|
1632
1750
|
recorded once: a second call for the same run changes nothing."""
|
|
1633
1751
|
# The caller's connection: a Row still reads by index, as it does.
|
|
1634
1752
|
db.row_factory = sqlite3.Row
|
|
1635
|
-
for i, sc in
|
|
1636
|
-
if not isinstance(sc, dict):
|
|
1637
|
-
continue
|
|
1753
|
+
for i, sc, k, cell in cells_of(run):
|
|
1638
1754
|
sid, sname = scenario_ref(sc, i)
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1755
|
+
# Echo's words are no wording under test: the item is its reply,
|
|
1756
|
+
# and over Prompt only the words are the reply itself.
|
|
1757
|
+
if isinstance(cell, dict) and cell.get("type") in UNRECORDED_STEPS:
|
|
1758
|
+
continue
|
|
1759
|
+
text = cell.get("prompt") if isinstance(cell, dict) else None
|
|
1760
|
+
if not isinstance(text, str) or not text.strip():
|
|
1761
|
+
continue
|
|
1762
|
+
pid = version = None
|
|
1763
|
+
ref = cell.get("from")
|
|
1764
|
+
if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
|
|
1765
|
+
pid = ref["id"]
|
|
1766
|
+
hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
|
|
1767
|
+
"ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
|
|
1768
|
+
version = hit[0] if hit else self._cut(db, pid, text)
|
|
1769
|
+
if pid is None:
|
|
1770
|
+
hit = self._matching(db, text)
|
|
1771
|
+
pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
|
|
1772
|
+
db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
|
|
1773
|
+
"scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
1774
|
+
(pid, version, rid, sid, sname, k, at))
|
|
1656
1775
|
|
|
1657
1776
|
def backfill(self, runs):
|
|
1658
1777
|
"""The runs from before the library, read into it once. [runs] is
|
|
@@ -1787,6 +1906,11 @@ class Datasets:
|
|
|
1787
1906
|
# so nothing is lost if a pipeline was missed.
|
|
1788
1907
|
db.execute("CREATE TABLE IF NOT EXISTS dataset_rules_archive ("
|
|
1789
1908
|
"dataset_id TEXT NOT NULL, rules TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1909
|
+
# Each row's body as it was before the conversion below rewrote
|
|
1910
|
+
# it (#199): a version-4 case's expectations became metrics, and
|
|
1911
|
+
# the body it was typed as is kept, as the rules were.
|
|
1912
|
+
db.execute("CREATE TABLE IF NOT EXISTS dataset_body_archive ("
|
|
1913
|
+
"dataset_id TEXT NOT NULL, body TEXT NOT NULL, archived_at TEXT NOT NULL)")
|
|
1790
1914
|
# Rows from an earlier version are converted once, in place: a
|
|
1791
1915
|
# dataset is typed in by hand and costly to re-enter, so it is
|
|
1792
1916
|
# upgraded rather than hidden (AGENTS.md's one exception). The
|
|
@@ -1805,6 +1929,8 @@ class Datasets:
|
|
|
1805
1929
|
up = upgrade_body(body)
|
|
1806
1930
|
if up is not body:
|
|
1807
1931
|
self._archive(db, did, body)
|
|
1932
|
+
db.execute("INSERT INTO dataset_body_archive (dataset_id, body, archived_at) VALUES (?, ?, ?)",
|
|
1933
|
+
(did, raw, self._now()))
|
|
1808
1934
|
if prompts is not None:
|
|
1809
1935
|
given = prompts.adopt(db, body_prompt(body), name, default=not given) is not None or given
|
|
1810
1936
|
db.execute("UPDATE datasets SET body = ?, version = ? WHERE id = ?",
|
|
@@ -1873,7 +1999,6 @@ class Datasets:
|
|
|
1873
1999
|
def _insert(self, db, name, body):
|
|
1874
2000
|
did = secrets.token_hex(6)
|
|
1875
2001
|
now = self._now()
|
|
1876
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1877
2002
|
db.execute("INSERT INTO datasets (id, name, version, body, created_at, updated_at) "
|
|
1878
2003
|
"VALUES (?, ?, 1, ?, ?, ?)", (did, name, json.dumps(body), now, now))
|
|
1879
2004
|
return did
|
|
@@ -1884,7 +2009,7 @@ class Datasets:
|
|
|
1884
2009
|
name, why = dataset_name(name)
|
|
1885
2010
|
if why:
|
|
1886
2011
|
return None, (400, why)
|
|
1887
|
-
body = blank_dataset() if body is None else body
|
|
2012
|
+
body = blank_dataset() if body is None else upgrade_body(body)
|
|
1888
2013
|
why = dataset_problem(body)
|
|
1889
2014
|
if why:
|
|
1890
2015
|
return None, (400, why)
|
|
@@ -1913,10 +2038,12 @@ class Datasets:
|
|
|
1913
2038
|
the current row."""
|
|
1914
2039
|
if type(version) is not int:
|
|
1915
2040
|
return None, (400, "a save names the version it began from")
|
|
2041
|
+
# A body of an earlier version -- from a page loaded before this one --
|
|
2042
|
+
# is read as today's, as an import is.
|
|
2043
|
+
body = upgrade_body(body)
|
|
1916
2044
|
why = dataset_problem(body)
|
|
1917
2045
|
if why:
|
|
1918
2046
|
return None, (400, why)
|
|
1919
|
-
body = {**body, "cases": canonical_cases(body["cases"])}
|
|
1920
2047
|
with self.store.lock, self._connect() as db, db:
|
|
1921
2048
|
r = self._live(db, did)
|
|
1922
2049
|
if r is None:
|
|
@@ -2327,6 +2454,13 @@ class Packs:
|
|
|
2327
2454
|
ds_ids = {}
|
|
2328
2455
|
cuts = [] # (kind, id, the document as it was), kept before it changes
|
|
2329
2456
|
for key, name, body, raw in pack["datasets"]:
|
|
2457
|
+
# The Source a dataset grades may be the pack's own, named by
|
|
2458
|
+
# its folder: pointed at the Source the pack made here.
|
|
2459
|
+
ref = body.get("source")
|
|
2460
|
+
if isinstance(ref, dict):
|
|
2461
|
+
sid = src_ids.get(ref.get("id")) or src_ids.get(ref.get("name"))
|
|
2462
|
+
if sid:
|
|
2463
|
+
body = {**body, "source": {"id": sid, "name": SOURCES.get(sid)["name"]}}
|
|
2330
2464
|
did = owned.get(("dataset", key))
|
|
2331
2465
|
current = DATASETS.get(did) if did else None
|
|
2332
2466
|
if current:
|
|
@@ -2348,8 +2482,10 @@ class Packs:
|
|
|
2348
2482
|
new_work = []
|
|
2349
2483
|
for key, doc in pack["pipelines"]:
|
|
2350
2484
|
doc = json.loads(json.dumps(doc))
|
|
2351
|
-
|
|
2352
|
-
|
|
2485
|
+
# A pack written before version 11 spells its evals `tests`;
|
|
2486
|
+
# the page upgrades the pipeline as it reads it.
|
|
2487
|
+
evals = doc.get("evals", doc.get("tests"))
|
|
2488
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
2353
2489
|
ref = t.get("dataset") if isinstance(t, dict) else None
|
|
2354
2490
|
if isinstance(ref, dict):
|
|
2355
2491
|
did = ds_ids.get(ref.get("id")) or ds_ids.get(ref.get("name"))
|
|
@@ -2486,7 +2622,7 @@ class Packs:
|
|
|
2486
2622
|
#
|
|
2487
2623
|
# A plugin is code, installed like a pack (docs/packs.md): a zip of a
|
|
2488
2624
|
# manifest and the compiled JavaScript that registers what the lab lacks -- a
|
|
2489
|
-
# kind of answer, a modifier,
|
|
2625
|
+
# kind of answer, a modifier, an eval type, a connection type. Its code runs in
|
|
2490
2626
|
# the page and in the runner, where the registries live; this server never
|
|
2491
2627
|
# runs it. It reads the manifest's `registers` as data, so it can refuse two
|
|
2492
2628
|
# plugins registering one id, and so a connection type's settings, chat path
|
|
@@ -2504,7 +2640,19 @@ PLUGIN_VERSIONS = (1,)
|
|
|
2504
2640
|
PLUGIN_CAP = int(os.environ.get("PLUGIN_CAP", str(16 * 1024 ** 2)))
|
|
2505
2641
|
PLUGIN_VERSION_TEXT = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}")
|
|
2506
2642
|
PLUGIN_FILE = re.compile(r"(?:[A-Za-z0-9_-][A-Za-z0-9._-]*/)*[A-Za-z0-9_-][A-Za-z0-9._-]*\.(?:js|mjs|json|map)")
|
|
2507
|
-
REGISTRIES = ("outputKinds", "modifiers", "
|
|
2643
|
+
REGISTRIES = ("outputKinds", "modifiers", "evalTypes", "connectionTypes")
|
|
2644
|
+
# evalTypes as a manifest written before pipeline version 11 spells it: a
|
|
2645
|
+
# plugin's own file, which the lab cannot upgrade, so it is read for good.
|
|
2646
|
+
OLD_REGISTRIES = {"testTypes": "evalTypes"}
|
|
2647
|
+
|
|
2648
|
+
|
|
2649
|
+
def plugin_registers(manifest: dict) -> dict:
|
|
2650
|
+
"""A manifest's registers under today's names: testTypes read as evalTypes."""
|
|
2651
|
+
reg = dict(manifest.get("registers") or {})
|
|
2652
|
+
for old, now in OLD_REGISTRIES.items():
|
|
2653
|
+
if old in reg:
|
|
2654
|
+
reg[now] = [*reg.get(now, []), *[x for x in reg.pop(old) if x not in reg.get(now, [])]]
|
|
2655
|
+
return reg
|
|
2508
2656
|
AUTH_WAYS = ("bearer", "x-api-key", "none")
|
|
2509
2657
|
|
|
2510
2658
|
|
|
@@ -2544,9 +2692,9 @@ def read_plugin(data: bytes):
|
|
|
2544
2692
|
if not isinstance(entry, str) or entry not in files or not entry.endswith((".js", ".mjs")):
|
|
2545
2693
|
return None, "the plugin's entry names no JavaScript file it holds"
|
|
2546
2694
|
reg = m.get("registers") or {}
|
|
2547
|
-
if not isinstance(reg, dict) or any(k not in REGISTRIES for k in reg):
|
|
2695
|
+
if not isinstance(reg, dict) or any(k not in REGISTRIES and k not in OLD_REGISTRIES for k in reg):
|
|
2548
2696
|
return None, f"a plugin's registers are {', '.join(REGISTRIES)}"
|
|
2549
|
-
for k in ("outputKinds", "modifiers", "
|
|
2697
|
+
for k in ("outputKinds", "modifiers", "evalTypes", *OLD_REGISTRIES):
|
|
2550
2698
|
if not isinstance(reg.get(k, []), list) or not all(isinstance(x, str) and x for x in reg.get(k, [])):
|
|
2551
2699
|
return None, f"registers.{k} is a list of ids"
|
|
2552
2700
|
conns = reg.get("connectionTypes", [])
|
|
@@ -2569,24 +2717,64 @@ def read_plugin(data: bytes):
|
|
|
2569
2717
|
|
|
2570
2718
|
def registered_ids(manifest: dict) -> set:
|
|
2571
2719
|
"""(registry, id) for everything a plugin's manifest says it registers."""
|
|
2572
|
-
reg = manifest
|
|
2573
|
-
out = {(k, x) for k in ("outputKinds", "modifiers", "
|
|
2720
|
+
reg = plugin_registers(manifest)
|
|
2721
|
+
out = {(k, x) for k in ("outputKinds", "modifiers", "evalTypes") for x in reg.get(k, [])}
|
|
2574
2722
|
return out | {("connectionTypes", c["id"]) for c in reg.get("connectionTypes", [])}
|
|
2575
2723
|
|
|
2576
2724
|
|
|
2577
2725
|
# ---- Connections: the lab's grants to outside services (#127) --------------
|
|
2578
2726
|
#
|
|
2579
2727
|
# One Google grant per lab, for Sources that read a Drive folder. The OAuth
|
|
2580
|
-
# client is the deployment's (
|
|
2581
|
-
#
|
|
2582
|
-
#
|
|
2583
|
-
#
|
|
2584
|
-
#
|
|
2585
|
-
#
|
|
2728
|
+
# client it signs in with is the deployment's (GOOGLE_* below), then the one
|
|
2729
|
+
# entered in Setup (GoogleApp), then the published app (PUBLIC_GOOGLE_APP),
|
|
2730
|
+
# as Microsoft's is (docs/power-automate.md). The refresh token the grant
|
|
2731
|
+
# yields is the store's, in a table of its own, so /api/state -- which serves
|
|
2732
|
+
# the synced documents to the page -- can never carry it, and nor can the
|
|
2733
|
+
# client's secret. The page is told only whether there is a grant and whose.
|
|
2734
|
+
# The hosts are fixed here, not chosen by anything a request carries, and
|
|
2735
|
+
# reached through OPENER: no redirects, no proxy from the environment.
|
|
2586
2736
|
GOOGLE_CLIENT_ID = os.environ.get("GOOGLE_CLIENT_ID") or None
|
|
2587
2737
|
GOOGLE_CLIENT_SECRET = os.environ.get("GOOGLE_CLIENT_SECRET") or None
|
|
2588
2738
|
GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") or None
|
|
2589
|
-
|
|
2739
|
+
# A deployment reached at its own address signs in with a Web application
|
|
2740
|
+
# client, whose redirect it registers; "installed" is Google's Desktop app.
|
|
2741
|
+
GOOGLE_CLIENT_TYPE = os.environ.get("GOOGLE_CLIENT_TYPE") or "web"
|
|
2742
|
+
GOOGLE_CLIENT_TYPES = ("installed", "web")
|
|
2743
|
+
# An OAuth client's id is its project's number, a dash, and Google's own
|
|
2744
|
+
# suffix: the number is the app id the Picker is told, so it is never asked.
|
|
2745
|
+
GOOGLE_CLIENT = re.compile(r"(\d+)-[0-9a-z]+\.apps\.googleusercontent\.com")
|
|
2746
|
+
GOOGLE_KEY = re.compile(r"[A-Za-z0-9_-]{30,60}")
|
|
2747
|
+
|
|
2748
|
+
|
|
2749
|
+
def read_published_google(path):
|
|
2750
|
+
"""The published app from the file the npm package's build writes
|
|
2751
|
+
beside server.py: (app, None), (None, None) when there is no file, or
|
|
2752
|
+
(None, why) for a file that is not one."""
|
|
2753
|
+
if not path.is_file():
|
|
2754
|
+
return None, None
|
|
2755
|
+
try:
|
|
2756
|
+
got = json.loads(path.read_text("utf-8"))
|
|
2757
|
+
except (OSError, ValueError):
|
|
2758
|
+
return None, f"{path} is not JSON"
|
|
2759
|
+
if not (isinstance(got, dict) and isinstance(got.get("clientId"), str)
|
|
2760
|
+
and GOOGLE_CLIENT.fullmatch(got["clientId"]) and isinstance(got.get("clientSecret"), str)
|
|
2761
|
+
and got["clientSecret"] and isinstance(got.get("apiKey"), str) and GOOGLE_KEY.fullmatch(got["apiKey"])):
|
|
2762
|
+
return None, f"{path} is not a Google app: {{ clientId, clientSecret, apiKey }}"
|
|
2763
|
+
return {"clientId": got["clientId"], "clientSecret": got["clientSecret"],
|
|
2764
|
+
"apiKey": got["apiKey"], "clientType": "installed"}, None
|
|
2765
|
+
|
|
2766
|
+
|
|
2767
|
+
# The Desktop-app client published for every lab, so a lab installed from
|
|
2768
|
+
# npm signs in with nothing to set up. A Desktop client signs in back to any
|
|
2769
|
+
# port on this machine with no redirect registered, and Google does not hold
|
|
2770
|
+
# its secret to be one: it is in every copy of the package by design. It
|
|
2771
|
+
# serves only a lab reached on this machine -- one reached at its own address
|
|
2772
|
+
# brings a Web application client of its own. Never in the repository: the
|
|
2773
|
+
# package's build writes it beside server.py from the Package workflow's
|
|
2774
|
+
# PUBLIC_GOOGLE_APP secret (docs/google-drive.md § The published app), and
|
|
2775
|
+
# a checkout or the image has none.
|
|
2776
|
+
GOOGLE_APP_FILE = HERE / "google-app.json"
|
|
2777
|
+
PUBLIC_GOOGLE_APP, PUBLIC_GOOGLE_PROBLEM = read_published_google(GOOGLE_APP_FILE)
|
|
2590
2778
|
# drive.file: only what the user picks in Google's Picker, and non-sensitive,
|
|
2591
2779
|
# so the app can be published without Google's verification (#127).
|
|
2592
2780
|
GOOGLE_SCOPE = "https://www.googleapis.com/auth/drive.file"
|
|
@@ -2600,6 +2788,94 @@ GOOGLE_CALLBACK = "/api/connections/google/callback"
|
|
|
2600
2788
|
GOOGLE_STATE_SECONDS = 600
|
|
2601
2789
|
|
|
2602
2790
|
|
|
2791
|
+
def google_app():
|
|
2792
|
+
"""The Google app this lab signs in with -- secret included, for the
|
|
2793
|
+
server's own use only -- and where it came from; None with none."""
|
|
2794
|
+
if GOOGLE_CLIENT_ID:
|
|
2795
|
+
return {"clientId": GOOGLE_CLIENT_ID, "clientSecret": GOOGLE_CLIENT_SECRET,
|
|
2796
|
+
"apiKey": GOOGLE_API_KEY, "clientType": GOOGLE_CLIENT_TYPE, "from": "env"}
|
|
2797
|
+
kept = GOOGLE.get() if GOOGLE is not None else None
|
|
2798
|
+
if kept:
|
|
2799
|
+
return {**kept, "from": "lab"}
|
|
2800
|
+
if PUBLIC_GOOGLE_APP:
|
|
2801
|
+
return {**PUBLIC_GOOGLE_APP, "from": "default"}
|
|
2802
|
+
return None
|
|
2803
|
+
|
|
2804
|
+
|
|
2805
|
+
def google_public(app):
|
|
2806
|
+
"""What the page is told of an app: everything but its secret."""
|
|
2807
|
+
if app is None:
|
|
2808
|
+
return None
|
|
2809
|
+
m = GOOGLE_CLIENT.fullmatch(app.get("clientId") or "")
|
|
2810
|
+
return {"clientId": app.get("clientId"), "clientType": app.get("clientType"),
|
|
2811
|
+
"apiKey": app.get("apiKey"), "appId": m.group(1) if m else None,
|
|
2812
|
+
"hasSecret": bool(app.get("clientSecret")), "from": app["from"]}
|
|
2813
|
+
|
|
2814
|
+
|
|
2815
|
+
def loopback_origin(origin) -> bool:
|
|
2816
|
+
"""Whether a page's origin is this machine: where a Desktop client may
|
|
2817
|
+
send the browser back to."""
|
|
2818
|
+
host = urllib.parse.urlsplit(origin).hostname or ""
|
|
2819
|
+
return host in ("localhost", "::1") or host.startswith("127.")
|
|
2820
|
+
|
|
2821
|
+
|
|
2822
|
+
class GoogleApp:
|
|
2823
|
+
"""The Google app entered in Setup: a client and its secret, from the
|
|
2824
|
+
client file Google hands out, and the Picker's key. Kept so a lab with no
|
|
2825
|
+
deployment around it needs no environment variable; the secret is
|
|
2826
|
+
written here and never read back out to the page."""
|
|
2827
|
+
|
|
2828
|
+
KEYS = ("google.clientId", "google.clientSecret", "google.apiKey", "google.clientType")
|
|
2829
|
+
|
|
2830
|
+
def __init__(self, store: Store):
|
|
2831
|
+
self.store = store
|
|
2832
|
+
with store.lock, closing(sqlite3.connect(store.path)) as db, db:
|
|
2833
|
+
db.execute("CREATE TABLE IF NOT EXISTS settings (key TEXT PRIMARY KEY, value TEXT NOT NULL)")
|
|
2834
|
+
|
|
2835
|
+
def _read(self, db):
|
|
2836
|
+
rows = dict(db.execute("SELECT key, value FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS))
|
|
2837
|
+
return dict(zip(("clientId", "clientSecret", "apiKey", "clientType"),
|
|
2838
|
+
(rows.get(k) or None for k in self.KEYS)))
|
|
2839
|
+
|
|
2840
|
+
def get(self):
|
|
2841
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
2842
|
+
got = self._read(db)
|
|
2843
|
+
return {**got, "clientType": got["clientType"] or "installed"} if got["clientId"] else None
|
|
2844
|
+
|
|
2845
|
+
def set(self, payload):
|
|
2846
|
+
"""Keeps what is sent, or with no client id forgets it all: (kept,
|
|
2847
|
+
error, whether the client changed). A secret not sent is kept while
|
|
2848
|
+
the client is the same one, and forgotten when it is not: a secret
|
|
2849
|
+
belongs to its client."""
|
|
2850
|
+
if not isinstance(payload, dict):
|
|
2851
|
+
return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
|
|
2852
|
+
client = payload.get("clientId")
|
|
2853
|
+
secret, key = payload.get("clientSecret"), payload.get("apiKey")
|
|
2854
|
+
ctype = payload.get("clientType") or "installed"
|
|
2855
|
+
if not isinstance(client, str) or not all(v is None or isinstance(v, str) for v in (secret, key)):
|
|
2856
|
+
return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
|
|
2857
|
+
client, key = client.strip(), (key or "").strip()
|
|
2858
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2859
|
+
had = self._read(db)
|
|
2860
|
+
changed = (had["clientId"] or "") != client
|
|
2861
|
+
if not client:
|
|
2862
|
+
db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
|
|
2863
|
+
return None, None, changed
|
|
2864
|
+
if not GOOGLE_CLIENT.fullmatch(client):
|
|
2865
|
+
return None, (400, "a client ID ends .apps.googleusercontent.com"), False
|
|
2866
|
+
if key and not GOOGLE_KEY.fullmatch(key):
|
|
2867
|
+
return None, (400, "that is not an API key"), False
|
|
2868
|
+
if ctype not in GOOGLE_CLIENT_TYPES:
|
|
2869
|
+
return None, (400, "a client is a Desktop app or a Web application"), False
|
|
2870
|
+
if secret is None:
|
|
2871
|
+
secret = None if changed else had["clientSecret"]
|
|
2872
|
+
secret = (secret or "").strip()
|
|
2873
|
+
db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
|
|
2874
|
+
db.executemany("INSERT INTO settings VALUES (?, ?)",
|
|
2875
|
+
[(k, v) for k, v in zip(self.KEYS, (client, secret, key, ctype)) if v])
|
|
2876
|
+
return self.get(), None, changed
|
|
2877
|
+
|
|
2878
|
+
|
|
2603
2879
|
class Connections:
|
|
2604
2880
|
"""The lab's grants, held server-side; the page sees their state only."""
|
|
2605
2881
|
|
|
@@ -2619,7 +2895,8 @@ class Connections:
|
|
|
2619
2895
|
|
|
2620
2896
|
@staticmethod
|
|
2621
2897
|
def configured() -> bool:
|
|
2622
|
-
|
|
2898
|
+
app = google_app()
|
|
2899
|
+
return bool(app and app.get("clientId") and app.get("clientSecret"))
|
|
2623
2900
|
|
|
2624
2901
|
def _row(self):
|
|
2625
2902
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
@@ -2633,15 +2910,21 @@ class Connections:
|
|
|
2633
2910
|
|
|
2634
2911
|
def start(self, origin: str):
|
|
2635
2912
|
"""The consent address the page sends the browser to."""
|
|
2913
|
+
app = google_app()
|
|
2636
2914
|
if not self.configured():
|
|
2637
2915
|
return None, (409, "Google is not configured for this lab")
|
|
2916
|
+
if app["clientType"] == "installed" and not loopback_origin(origin):
|
|
2917
|
+
return None, (409, "a Desktop app client signs in only on this machine: "
|
|
2918
|
+
"give this lab a Web application client in Configure…")
|
|
2638
2919
|
state = secrets.token_urlsafe(24)
|
|
2639
2920
|
now = time.time()
|
|
2640
2921
|
with self.lock:
|
|
2641
2922
|
self.states = {s: v for s, v in self.states.items() if v[0] > now}
|
|
2642
|
-
|
|
2923
|
+
# The app is kept with the state: the code Google sends back is
|
|
2924
|
+
# exchanged with the client that asked for it, whatever changes.
|
|
2925
|
+
self.states[state] = (now + GOOGLE_STATE_SECONDS, origin, app)
|
|
2643
2926
|
return GOOGLE_CONSENT_URL + "?" + urllib.parse.urlencode({
|
|
2644
|
-
"client_id":
|
|
2927
|
+
"client_id": app["clientId"], "redirect_uri": origin + GOOGLE_CALLBACK,
|
|
2645
2928
|
"response_type": "code", "scope": GOOGLE_SCOPE, "state": state,
|
|
2646
2929
|
"access_type": "offline", "prompt": "consent", "include_granted_scopes": "true",
|
|
2647
2930
|
}), None
|
|
@@ -2661,7 +2944,7 @@ class Connections:
|
|
|
2661
2944
|
which never include the code or a token."""
|
|
2662
2945
|
state = (query.get("state") or [""])[0]
|
|
2663
2946
|
with self.lock:
|
|
2664
|
-
lapses, origin = self.states.pop(state, (0, ""))
|
|
2947
|
+
lapses, origin, app = self.states.pop(state, (0, "", None))
|
|
2665
2948
|
if lapses <= time.time():
|
|
2666
2949
|
return "that sign-in had lapsed or was not this lab's; sign in again"
|
|
2667
2950
|
if query.get("error"):
|
|
@@ -2671,7 +2954,7 @@ class Connections:
|
|
|
2671
2954
|
return "Google sent no code back"
|
|
2672
2955
|
try:
|
|
2673
2956
|
got = self._call(GOOGLE_TOKEN_URL, {
|
|
2674
|
-
"code": code, "client_id":
|
|
2957
|
+
"code": code, "client_id": app["clientId"], "client_secret": app["clientSecret"],
|
|
2675
2958
|
"redirect_uri": origin + GOOGLE_CALLBACK, "grant_type": "authorization_code"})
|
|
2676
2959
|
except (OSError, ValueError) as e:
|
|
2677
2960
|
return f"the code could not be exchanged ({type(e).__name__})"
|
|
@@ -2689,6 +2972,12 @@ class Connections:
|
|
|
2689
2972
|
time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())))
|
|
2690
2973
|
return None
|
|
2691
2974
|
|
|
2975
|
+
def forget(self):
|
|
2976
|
+
"""Drops the grant without a word to Google: the client it was made
|
|
2977
|
+
with is no longer this lab's, so the grant is no use to it."""
|
|
2978
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2979
|
+
db.execute("DELETE FROM connections WHERE id = 'google'")
|
|
2980
|
+
|
|
2692
2981
|
def sign_out(self):
|
|
2693
2982
|
"""Forgets the grant, and asks Google to revoke it; a revoke that
|
|
2694
2983
|
fails still forgets it here, which is what signing out means."""
|
|
@@ -2782,7 +3071,7 @@ class Plugins:
|
|
|
2782
3071
|
return [{"id": r["id"], "version": r["version"], "sha256": r["sha256"],
|
|
2783
3072
|
"entry": json.loads(r["manifest"])["entry"],
|
|
2784
3073
|
"description": json.loads(r["manifest"]).get("description") or "",
|
|
2785
|
-
"registers": json.loads(r["manifest"])
|
|
3074
|
+
"registers": plugin_registers(json.loads(r["manifest"])),
|
|
2786
3075
|
"installed": r["installed_at"]} for r in self._rows()]
|
|
2787
3076
|
|
|
2788
3077
|
def stamp(self) -> list:
|
|
@@ -2958,6 +3247,55 @@ def worker_refusal(code, stderr, env):
|
|
|
2958
3247
|
return text if len(text) <= 2000 else text[:2000] + "…"
|
|
2959
3248
|
|
|
2960
3249
|
|
|
3250
|
+
|
|
3251
|
+
# A list of runs is read for its figures -- History's table, Home's recent
|
|
3252
|
+
# runs, Runs' progress -- and a run's results are mostly what Results alone
|
|
3253
|
+
# shows: every reply, every stage's request and answer, every item a score
|
|
3254
|
+
# found or missed. A list of 25 runs was 1.9 MB of that (#225). So a row in a
|
|
3255
|
+
# list carries a brief copy of its results, `brief` says so, and one run
|
|
3256
|
+
# (GET /api/queue/<id>) is always whole. A brief cell keeps its time and
|
|
3257
|
+
# whether it ran; a brief score keeps its verdict and counts in place of its
|
|
3258
|
+
# lists. The replies go too: an eval over the whole run reads them, and the
|
|
3259
|
+
# page asks for that run whole rather than every list carrying them.
|
|
3260
|
+
BRIEF_SCORE = ("pass", "score", "points", "skipped")
|
|
3261
|
+
COUNTED = ("found", "missed", "invented")
|
|
3262
|
+
|
|
3263
|
+
|
|
3264
|
+
def brief_score(score):
|
|
3265
|
+
if not isinstance(score, dict):
|
|
3266
|
+
return score
|
|
3267
|
+
out = {k: score[k] for k in BRIEF_SCORE if k in score}
|
|
3268
|
+
for k in COUNTED:
|
|
3269
|
+
if isinstance(score.get(k), list):
|
|
3270
|
+
out[k] = len(score[k])
|
|
3271
|
+
return out
|
|
3272
|
+
|
|
3273
|
+
|
|
3274
|
+
def brief_cell(cell):
|
|
3275
|
+
if not isinstance(cell, dict):
|
|
3276
|
+
return cell
|
|
3277
|
+
out = {k: v for k, v in cell.items() if k not in ("res", "scores", "score")}
|
|
3278
|
+
if isinstance(cell.get("res"), dict):
|
|
3279
|
+
out["res"] = {k: cell["res"][k] for k in ("ms", "error") if k in cell["res"]}
|
|
3280
|
+
if isinstance(cell.get("scores"), dict):
|
|
3281
|
+
out["scores"] = {k: brief_score(v) for k, v in cell["scores"].items()}
|
|
3282
|
+
elif "scores" in cell:
|
|
3283
|
+
out["scores"] = cell["scores"]
|
|
3284
|
+
# A run from before version 6 kept its one test's score as `score`, which
|
|
3285
|
+
# the page reads as `scores.t1`: kept under its own name for that.
|
|
3286
|
+
if "score" in cell:
|
|
3287
|
+
out["score"] = brief_score(cell["score"])
|
|
3288
|
+
return out
|
|
3289
|
+
|
|
3290
|
+
|
|
3291
|
+
def brief_row(row):
|
|
3292
|
+
def item(it):
|
|
3293
|
+
if not isinstance(it, dict) or not isinstance(it.get("scenarios"), list):
|
|
3294
|
+
return it
|
|
3295
|
+
return {**it, "scenarios": [brief_cell(c) for c in it["scenarios"]]}
|
|
3296
|
+
return {**row, "results": [item(it) for it in row.get("results") or []], "brief": True}
|
|
3297
|
+
|
|
3298
|
+
|
|
2961
3299
|
class Queue:
|
|
2962
3300
|
STATUS = ("queued", "running", "done", "incomplete", "cancelled",
|
|
2963
3301
|
"failed", "interrupted")
|
|
@@ -3022,15 +3360,17 @@ class Queue:
|
|
|
3022
3360
|
(rid,)).fetchone())
|
|
3023
3361
|
return row if self._readable(row) else None
|
|
3024
3362
|
|
|
3025
|
-
def list(self, limit=RUNS_PAGE, before=None):
|
|
3363
|
+
def list(self, limit=RUNS_PAGE, before=None, full=False):
|
|
3026
3364
|
"""Runs, newest first, and whether more follow. `before` is a
|
|
3027
3365
|
`submittedAt` the page of runs stops at, so History can page through
|
|
3028
|
-
them the way it pages the runs store.
|
|
3366
|
+
them the way it pages the runs store. Each row is brief_row's unless
|
|
3367
|
+
`full` asks for the whole of it."""
|
|
3029
3368
|
with self.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
3030
3369
|
rows = self._all(db)
|
|
3031
3370
|
rows = [r for r in rows if before is None or r["submittedAt"] < before]
|
|
3032
3371
|
rows.sort(key=lambda r: r["submittedAt"], reverse=True)
|
|
3033
|
-
|
|
3372
|
+
page = rows[:limit]
|
|
3373
|
+
return (page if full else [brief_row(r) for r in page]), len(rows) > limit
|
|
3034
3374
|
|
|
3035
3375
|
def _set(self, rid, **fields):
|
|
3036
3376
|
sets, vals = ", ".join(f"{k} = ?" for k in fields), list(fields.values())
|
|
@@ -3099,7 +3439,7 @@ class Queue:
|
|
|
3099
3439
|
written beside the run document. A row queued before runs kept their
|
|
3100
3440
|
dataset has none, and is pinned to the dataset as it reads now, once,
|
|
3101
3441
|
so every later pass over it agrees. Returns (args, None) or (None, why)."""
|
|
3102
|
-
ref =
|
|
3442
|
+
ref = evals_dataset(run["snapshot"])
|
|
3103
3443
|
if ref is None:
|
|
3104
3444
|
return [], None
|
|
3105
3445
|
body = self.dataset(run["id"], raw=True)
|
|
@@ -3589,24 +3929,24 @@ def worker_destinations(run: dict):
|
|
|
3589
3929
|
base = api_base(str(conn.get("url") or "")) or api_base(OLLAMA)
|
|
3590
3930
|
why = allowed(base, "")
|
|
3591
3931
|
if why:
|
|
3592
|
-
return None, f"
|
|
3932
|
+
return None, f"Target profile {name}: {why}"
|
|
3593
3933
|
profile = next((p for p in stored if isinstance(p, dict) and p.get("id") == pid), None)
|
|
3594
3934
|
if profile is None:
|
|
3595
|
-
return None, f"
|
|
3935
|
+
return None, f"Target profile {name} not found"
|
|
3596
3936
|
key = str(profile.get("key") or "").strip()
|
|
3597
3937
|
if key:
|
|
3598
3938
|
if not header_safe(key):
|
|
3599
|
-
return None, f"
|
|
3939
|
+
return None, f"Target profile {name} has a key that cannot go in a header"
|
|
3600
3940
|
why = allowed(base, key)
|
|
3601
3941
|
if why:
|
|
3602
|
-
return None, f"
|
|
3942
|
+
return None, f"Target profile {name}: {why}"
|
|
3603
3943
|
env[key_var(pid)] = key
|
|
3604
3944
|
return env, None
|
|
3605
3945
|
|
|
3606
3946
|
|
|
3607
3947
|
# A run document's fields, checked before it is accepted -- the rules
|
|
3608
3948
|
# docs/pipeline-model.md §6 gives the server, in Python because the server is
|
|
3609
|
-
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or
|
|
3949
|
+
# stdlib-only and cannot load evals-core.ts. What a kind, a modifier or an eval
|
|
3610
3950
|
# means is the runner's to judge, and it fails the run with a sentence if it
|
|
3611
3951
|
# cannot; what is here is what the server itself depends on: a version it
|
|
3612
3952
|
# reads, its own cap, content it can count and copy, and connections that
|
|
@@ -3614,16 +3954,23 @@ def worker_destinations(run: dict):
|
|
|
3614
3954
|
# 5: the pipeline and every chain have an id; version 4's scenarios keep
|
|
3615
3955
|
# theirs, and an older run's chains are read as they are, by position.
|
|
3616
3956
|
# 6: tests are an ordered list; a stored run's one test (or null) is read as
|
|
3617
|
-
# a list of one (
|
|
3957
|
+
# a list of one (evals_dataset).
|
|
3618
3958
|
# 7: chains are jobs: the field is `jobs` and each job's type is "job".
|
|
3619
3959
|
# 8: a job is its steps; the content is job 1's Attach Content step.
|
|
3620
|
-
# 9:
|
|
3621
|
-
|
|
3960
|
+
# 9: an eval is Metrics; a Single Test or a Graded set is read converted.
|
|
3961
|
+
# 10: a job's steps are its stages, and each scenario is a target whose own
|
|
3962
|
+
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
3963
|
+
# 11: `tests` are `evals`; nothing in an eval changes.
|
|
3964
|
+
# 12: a Contains metric's Ignore case holds item by item too, kept as written.
|
|
3965
|
+
PIPELINE_VERSION = 12
|
|
3622
3966
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
3623
|
-
# upgradePipeline reads. A new submission is
|
|
3624
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9)
|
|
3625
|
-
|
|
3626
|
-
|
|
3967
|
+
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
3968
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
|
3969
|
+
TARGET_CAP = 4
|
|
3970
|
+
# Target steps whose words the Prompt library does not record as a use: they
|
|
3971
|
+
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
3972
|
+
UNRECORDED_STEPS = {"echo"}
|
|
3973
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "evals", "profiles", "comment", "plugins")
|
|
3627
3974
|
|
|
3628
3975
|
|
|
3629
3976
|
def content_of(doc):
|
|
@@ -3780,14 +4127,14 @@ def connection_problems(pid, conn, at, bad):
|
|
|
3780
4127
|
bad.append(f"{at}: llama.cpp requires its llama-server, not a hosted model")
|
|
3781
4128
|
|
|
3782
4129
|
|
|
3783
|
-
def
|
|
3784
|
-
"""The dataset reference a document's
|
|
3785
|
-
first
|
|
3786
|
-
validatePipeline says so). Reads a stored document of
|
|
3787
|
-
version
|
|
3788
|
-
document they were submitted with."""
|
|
3789
|
-
|
|
3790
|
-
for t in
|
|
4130
|
+
def evals_dataset(doc):
|
|
4131
|
+
"""The dataset reference a document's evals grade against, or None: the
|
|
4132
|
+
first eval that names one. A run grades against one dataset (the core's
|
|
4133
|
+
validatePipeline says so). Reads a stored document of any shape --
|
|
4134
|
+
version 11's `evals`, the `tests` before it, version 5's list or the one
|
|
4135
|
+
test before that -- since rows keep the document they were submitted with."""
|
|
4136
|
+
evals = doc.get("evals", doc.get("tests")) if isinstance(doc, dict) else None
|
|
4137
|
+
for t in evals if isinstance(evals, list) else [evals]:
|
|
3791
4138
|
if isinstance(t, dict) and isinstance(t.get("dataset"), dict):
|
|
3792
4139
|
return t["dataset"]
|
|
3793
4140
|
return None
|
|
@@ -3823,16 +4170,20 @@ def upgrade_run(run):
|
|
|
3823
4170
|
return up, None
|
|
3824
4171
|
|
|
3825
4172
|
|
|
3826
|
-
def
|
|
3827
|
-
"""
|
|
3828
|
-
|
|
3829
|
-
|
|
3830
|
-
|
|
3831
|
-
if not isinstance(
|
|
3832
|
-
return
|
|
3833
|
-
|
|
3834
|
-
|
|
3835
|
-
|
|
4173
|
+
def cells_of(run):
|
|
4174
|
+
"""Every target's step in every job of a run, as (i, target, k, step):
|
|
4175
|
+
version 10's targets and their steps, or an earlier version's scenarios
|
|
4176
|
+
and their cells -- what the Prompt library records a use from, whichever
|
|
4177
|
+
version a stored run is."""
|
|
4178
|
+
if not isinstance(run, dict):
|
|
4179
|
+
return
|
|
4180
|
+
targets = run.get("targets") if isinstance(run.get("targets"), list) else run.get("scenarios")
|
|
4181
|
+
for i, t in enumerate(targets if isinstance(targets, list) else []):
|
|
4182
|
+
if not isinstance(t, dict):
|
|
4183
|
+
continue
|
|
4184
|
+
steps = t.get("steps") if isinstance(t.get("steps"), list) else t.get("stages")
|
|
4185
|
+
for k, step in enumerate(steps if isinstance(steps, list) else []):
|
|
4186
|
+
yield i, t, k, step
|
|
3836
4187
|
|
|
3837
4188
|
|
|
3838
4189
|
def run_problems(run):
|
|
@@ -3877,36 +4228,41 @@ def run_problems(run):
|
|
|
3877
4228
|
bad.append(f"{at} has to be an object")
|
|
3878
4229
|
continue
|
|
3879
4230
|
connection_problems(pid, conn, at, bad)
|
|
3880
|
-
|
|
3881
|
-
if not isinstance(
|
|
3882
|
-
return bad + [f"a run needs between one and {
|
|
3883
|
-
ids = [
|
|
3884
|
-
for i,
|
|
3885
|
-
at = f"
|
|
3886
|
-
if not isinstance(
|
|
4231
|
+
targets = run.get("targets")
|
|
4232
|
+
if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
|
|
4233
|
+
return bad + [f"a run needs between one and {TARGET_CAP} targets"]
|
|
4234
|
+
ids = [t.get("id") for t in targets if isinstance(t, dict)]
|
|
4235
|
+
for i, t in enumerate(targets):
|
|
4236
|
+
at = f"target {i + 1}"
|
|
4237
|
+
if not isinstance(t, dict):
|
|
3887
4238
|
bad.append(f"{at} has to be an object")
|
|
3888
4239
|
continue
|
|
3889
4240
|
# The id the Prompt library records a use under (version 4).
|
|
3890
|
-
if not isinstance(
|
|
4241
|
+
if not isinstance(t.get("id"), str) or not t["id"].strip():
|
|
3891
4242
|
bad.append(f"{at} has no id")
|
|
3892
|
-
elif ids.count(
|
|
3893
|
-
bad.append(f"{at} has the id of another
|
|
3894
|
-
|
|
3895
|
-
if not isinstance(
|
|
3896
|
-
bad.append(f"{at} has to have one
|
|
4243
|
+
elif ids.count(t["id"]) > 1:
|
|
4244
|
+
bad.append(f"{at} has the id of another target")
|
|
4245
|
+
steps = t.get("steps")
|
|
4246
|
+
if not isinstance(steps, list) or len(steps) != len(jobs):
|
|
4247
|
+
bad.append(f"{at} has to have one step per job")
|
|
3897
4248
|
continue
|
|
3898
|
-
|
|
3899
|
-
|
|
3900
|
-
|
|
3901
|
-
if
|
|
3902
|
-
|
|
4249
|
+
# A target with no profile is one whose steps ask none (Echo) or each
|
|
4250
|
+
# name their own; which steps need one is the core's to judge.
|
|
4251
|
+
refs = ([t["profile"]] if t.get("profile") is not None else []) + [
|
|
4252
|
+
st.get("profile") for st in steps if isinstance(st, dict) and st.get("profile") is not None]
|
|
4253
|
+
# Whether the words may be blank -- Echo answering from the item --
|
|
4254
|
+
# is the core's to judge, and the worker refuses the run in its words.
|
|
4255
|
+
for k, step in enumerate(steps):
|
|
4256
|
+
if not isinstance(step, dict) or not isinstance(step.get("type"), str):
|
|
4257
|
+
bad.append(f"{at}, job {k + 1} has to name what it sends")
|
|
4258
|
+
elif not isinstance(step.get("prompt"), str):
|
|
3903
4259
|
bad.append(f"{at}, job {k + 1} has no prompt")
|
|
3904
|
-
elif
|
|
3905
|
-
isinstance(
|
|
4260
|
+
elif step.get("from") is not None and not (
|
|
4261
|
+
isinstance(step["from"], dict) and isinstance(step["from"].get("id"), str)):
|
|
3906
4262
|
bad.append(f"{at}, job {k + 1} names the prompt it was picked from without an id")
|
|
3907
4263
|
for ref in refs:
|
|
3908
4264
|
if not isinstance(ref, dict) or ref.get("id") not in table:
|
|
3909
|
-
bad.append(f"{at} names a
|
|
4265
|
+
bad.append(f"{at} names a Target profile the run does not carry")
|
|
3910
4266
|
content = content_of(run)
|
|
3911
4267
|
kind = content.get("type") if isinstance(content, dict) else None
|
|
3912
4268
|
if kind == "source":
|
|
@@ -3923,9 +4279,9 @@ def run_problems(run):
|
|
|
3923
4279
|
pass
|
|
3924
4280
|
else:
|
|
3925
4281
|
bad.append("a run needs a Source's files, some inline text, or Prompt only")
|
|
3926
|
-
|
|
3927
|
-
if not isinstance(
|
|
3928
|
-
bad.append("
|
|
4282
|
+
evals = run.get("evals")
|
|
4283
|
+
if not isinstance(evals, list) or not all(isinstance(t, dict) and isinstance(t.get("type"), str) for t in evals):
|
|
4284
|
+
bad.append("evals has to be a list of evals, each naming its type")
|
|
3929
4285
|
if run.get("comment") is not None and not isinstance(run["comment"], str):
|
|
3930
4286
|
bad.append("comment has to be text")
|
|
3931
4287
|
return bad
|
|
@@ -3946,10 +4302,11 @@ if DATA_DIR:
|
|
|
3946
4302
|
PLUGINS = Plugins(STORE)
|
|
3947
4303
|
CONNECTIONS = Connections(STORE)
|
|
3948
4304
|
MICROSOFT = MicrosoftApp(STORE)
|
|
4305
|
+
GOOGLE = GoogleApp(STORE)
|
|
3949
4306
|
QUEUE = Queue(STORE)
|
|
3950
4307
|
QUEUE.prompts = PROMPTS
|
|
3951
4308
|
else:
|
|
3952
|
-
STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = QUEUE = None
|
|
4309
|
+
STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
|
|
3953
4310
|
|
|
3954
4311
|
|
|
3955
4312
|
class NoRedirects(urllib.request.HTTPRedirectHandler):
|
|
@@ -3989,11 +4346,25 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
3989
4346
|
|
|
3990
4347
|
# ---- helpers --------------------------------------------------------
|
|
3991
4348
|
|
|
3992
|
-
def _send(self, code, body: bytes, ctype="application/json", headers=()):
|
|
4349
|
+
def _send(self, code, body: bytes, ctype="application/json", headers=(), packed=None):
|
|
4350
|
+
"""`packed` keys a body that never changes under it -- a hashed
|
|
4351
|
+
bundle -- so it is gzipped once and kept, not on every request."""
|
|
3993
4352
|
self.send_response(code)
|
|
3994
4353
|
self.send_header("Content-Type", ctype)
|
|
3995
4354
|
for k, v in headers:
|
|
3996
4355
|
self.send_header(k, v)
|
|
4356
|
+
if ctype.startswith(COMPRESSIBLE):
|
|
4357
|
+
# Said whether or not this answer is compressed, so a cache
|
|
4358
|
+
# between here and the browser keys on it either way.
|
|
4359
|
+
self.send_header("Vary", "Accept-Encoding")
|
|
4360
|
+
if len(body) >= GZIP_MIN and accepts_gzip(self.headers.get("Accept-Encoding", "")):
|
|
4361
|
+
if packed is None:
|
|
4362
|
+
body = gzip.compress(body, 6, mtime=0)
|
|
4363
|
+
else:
|
|
4364
|
+
if packed not in GZIPPED:
|
|
4365
|
+
GZIPPED[packed] = gzip.compress(body, 9, mtime=0)
|
|
4366
|
+
body = GZIPPED[packed]
|
|
4367
|
+
self.send_header("Content-Encoding", "gzip")
|
|
3997
4368
|
self.send_header("Content-Length", str(len(body)))
|
|
3998
4369
|
self.end_headers()
|
|
3999
4370
|
self.wfile.write(body)
|
|
@@ -4078,7 +4449,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4078
4449
|
return
|
|
4079
4450
|
path = self.path.split("?", 1)[0]
|
|
4080
4451
|
# The lab is one page: a Connection and an Input make a scenario,
|
|
4081
|
-
# Content and
|
|
4452
|
+
# Content and Evals are shared, and one to four scenarios run over the
|
|
4082
4453
|
# content. It replaced the A/B page it was prototyped beside once it
|
|
4083
4454
|
# carried everything that page did -- the graded set, the
|
|
4084
4455
|
# Configuration group, the history -- rather than being left to rot
|
|
@@ -4098,14 +4469,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4098
4469
|
if not name or TYPES.get(p.suffix.lower()) is None or not p.is_file():
|
|
4099
4470
|
return self._send(404, b"not found", "text/plain")
|
|
4100
4471
|
return self._send(200, p.read_bytes(), TYPES[p.suffix.lower()],
|
|
4101
|
-
(("Cache-Control", "public, max-age=31536000, immutable"),))
|
|
4472
|
+
(("Cache-Control", "public, max-age=31536000, immutable"),), packed=name)
|
|
4102
4473
|
if path == "/api/config":
|
|
4103
4474
|
# The lab's own limits, for the page to disable rather than
|
|
4104
4475
|
# hard-code: how many scenarios a run may have.
|
|
4105
4476
|
# And the Microsoft app a flow is read with, when there is one.
|
|
4106
4477
|
# The app may be entered in Setup unless the deployment names
|
|
4107
4478
|
# its own, and only in a lab with a store to keep it in.
|
|
4108
|
-
return self._json(200, {"
|
|
4479
|
+
return self._json(200, {"targetCap": TARGET_CAP, "microsoft": microsoft_config(),
|
|
4109
4480
|
"microsoftEditable": MICROSOFT is not None and not M365_CLIENT_ID})
|
|
4110
4481
|
if path == "/api/state":
|
|
4111
4482
|
if STORE is None:
|
|
@@ -4122,14 +4493,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4122
4493
|
except ValueError:
|
|
4123
4494
|
return self._json(400, {"error": "limit has to be a number"})
|
|
4124
4495
|
before = (query.get("before") or [None])[0]
|
|
4125
|
-
runs, more = QUEUE.list(limit, before)
|
|
4496
|
+
runs, more = QUEUE.list(limit, before, full=(query.get("full") or [""])[0] == "1")
|
|
4126
4497
|
return self._json(200, {"runs": runs, "more": more})
|
|
4127
4498
|
if path.startswith("/api/queue/") and path.endswith("/dataset") and path.count("/") == 4:
|
|
4128
4499
|
# The dataset body a graded run was submitted against, which is
|
|
4129
4500
|
# what its verdicts were graded by; the list never carries it.
|
|
4130
4501
|
#
|
|
4131
4502
|
# A run that is there but kept no copy -- one submitted before the
|
|
4132
|
-
# lab kept them, or one with no graded
|
|
4503
|
+
# lab kept them, or one with no graded eval -- answers null, not
|
|
4133
4504
|
# 404. It is not an error: the page reads it as "use the dataset
|
|
4134
4505
|
# as it is now", and a 404 put a red line in the console every
|
|
4135
4506
|
# time such a run was opened, which is the console people are
|
|
@@ -4161,6 +4532,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4161
4532
|
if CONNECTIONS is None:
|
|
4162
4533
|
return self._send(404, b"not found", "text/plain")
|
|
4163
4534
|
return self._json(200, {"connections": CONNECTIONS.list()})
|
|
4535
|
+
if path == "/api/google":
|
|
4536
|
+
if GOOGLE is None:
|
|
4537
|
+
return self._send(404, b"not found", "text/plain")
|
|
4538
|
+
return self._json(200, {"google": google_public(google_app()), "editable": not GOOGLE_CLIENT_ID})
|
|
4164
4539
|
if path == GOOGLE_CALLBACK:
|
|
4165
4540
|
return self._google_callback()
|
|
4166
4541
|
if path.startswith("/plugins/"):
|
|
@@ -4381,6 +4756,17 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4381
4756
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4382
4757
|
if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
|
|
4383
4758
|
return self._prompts_put(parts[3])
|
|
4759
|
+
if parts == ["", "api", "google"] and GOOGLE is not None:
|
|
4760
|
+
if GOOGLE_CLIENT_ID:
|
|
4761
|
+
return self._json(409, {"error": "this lab's Google app is its deployment's (GOOGLE_CLIENT_ID)"})
|
|
4762
|
+
_, err, changed = GOOGLE.set(self._payload())
|
|
4763
|
+
if err:
|
|
4764
|
+
return self._json(err[0], {"error": err[1]})
|
|
4765
|
+
# A grant is its client's: one made with another client cannot be
|
|
4766
|
+
# refreshed with this one, so it goes rather than failing later.
|
|
4767
|
+
if changed and CONNECTIONS is not None:
|
|
4768
|
+
CONNECTIONS.forget()
|
|
4769
|
+
return self._json(200, {"google": google_public(google_app()), "editable": True})
|
|
4384
4770
|
if parts == ["", "api", "microsoft"] and MICROSOFT is not None:
|
|
4385
4771
|
if M365_CLIENT_ID:
|
|
4386
4772
|
return self._json(409, {"error": "this lab's Microsoft app is its deployment's (M365_CLIENT_ID)"})
|
|
@@ -4648,14 +5034,14 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4648
5034
|
# A graded run keeps the body of the dataset it names, as it reads
|
|
4649
5035
|
# now, and records that body's fingerprint on the reference it
|
|
4650
5036
|
# belongs to: the worker grades against that and nothing else.
|
|
4651
|
-
ref =
|
|
5037
|
+
ref = evals_dataset(run)
|
|
4652
5038
|
body = None
|
|
4653
5039
|
if ref is not None:
|
|
4654
5040
|
snap = DATASETS.snapshot(ref.get("id")) if DATASETS else None
|
|
4655
5041
|
if snap is None:
|
|
4656
|
-
return self._json(400, {"error": "the run's graded
|
|
5042
|
+
return self._json(400, {"error": "the run's graded eval names no dataset this lab has"})
|
|
4657
5043
|
body, version = snap
|
|
4658
|
-
for t in run["
|
|
5044
|
+
for t in run["evals"]:
|
|
4659
5045
|
if isinstance(t.get("dataset"), dict) and t["dataset"].get("id") == ref.get("id"):
|
|
4660
5046
|
t["dataset"]["version"] = version
|
|
4661
5047
|
return self._json(201, {"run": QUEUE.submit(run, body)})
|
|
@@ -5065,6 +5451,10 @@ def main():
|
|
|
5065
5451
|
if not (DEMO / "manifest.json").is_file():
|
|
5066
5452
|
raise SystemExit(f"no {DEMO / 'manifest.json'} -- the demo pack sits beside server.py, "
|
|
5067
5453
|
"and the image copies it there")
|
|
5454
|
+
# A published app the build wrote and that cannot be read is a broken
|
|
5455
|
+
# package, not one without the app: said at once, not at Sign in.
|
|
5456
|
+
if PUBLIC_GOOGLE_PROBLEM:
|
|
5457
|
+
raise SystemExit(PUBLIC_GOOGLE_PROBLEM)
|
|
5068
5458
|
print(f"prompt-lab on {HOST}:{PORT} -> {OLLAMA} by default", flush=True)
|
|
5069
5459
|
print(f"samples: {SAMPLES}", flush=True)
|
|
5070
5460
|
print(f"page: {WEB_DIST}", flush=True)
|