evals-lab 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +34 -24
- package/lab/demo/pipelines/demo-2.json +34 -24
- package/lab/evals-core.mjs +701 -284
- package/lab/run-evals.js +42 -16
- package/lab/server.py +251 -75
- package/lab/web/dist/assets/gallery-DsetJSXv.js +3 -0
- package/lab/web/dist/assets/{main-DjQQums6.css → main-B-VtDGxC.css} +1 -1
- package/lab/web/dist/assets/main-ERoD2Kll.js +19 -0
- package/lab/web/dist/assets/tokens-ClRQ7Mui.js +51 -0
- package/lab/web/dist/assets/tokens-D3C8O2Ib.css +1 -0
- package/lab/web/dist/gallery.html +3 -3
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-DipkRvqJ.js +0 -3
- package/lab/web/dist/assets/main-61NS6C4m.js +0 -18
- package/lab/web/dist/assets/tokens-B9intIuT.js +0 -51
- package/lab/web/dist/assets/tokens-s6I-RMVq.css +0 -1
package/lab/run-evals.js
CHANGED
|
@@ -453,21 +453,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
|
|
|
453
453
|
// or, over Prompt only, the prompt -- and the stage reads it as it would a
|
|
454
454
|
// model's.
|
|
455
455
|
//
|
|
456
|
-
// A
|
|
457
|
-
// from
|
|
458
|
-
// only its words: the transport is where the three meet.
|
|
456
|
+
// A target step that asks for a whole request (an HTTP Request, #flows)
|
|
457
|
+
// builds it from the step, the job's flow step and the item's record, and is
|
|
458
|
+
// handed only its words: the transport is where the three meet.
|
|
459
|
+
//
|
|
460
|
+
// Any other step's reply, in a job that reads the flow's step through a Read
|
|
461
|
+
// as, is read the same way: put in the body the flow's API would have sent
|
|
462
|
+
// it in, then read as the flow reads it -- so a model asked in words and
|
|
463
|
+
// production are compared alike (docs/pipeline-model.md §16 › Targets).
|
|
459
464
|
function callsFor(connections, links, text, plan, record) {
|
|
460
465
|
return connections.map((c, k) => {
|
|
461
466
|
const step = plan?.calls?.[k];
|
|
462
467
|
if (core.STEP_TYPES[step?.type]?.asks === "request") {
|
|
463
468
|
return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
|
|
464
469
|
}
|
|
465
|
-
|
|
470
|
+
const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
|
|
466
471
|
? async sent => core.localAnswer(c, text, sent)
|
|
467
472
|
: (sent, url) => ask(links[k], c, sent, url);
|
|
473
|
+
if (!step?.readAs || !step?.step) return call;
|
|
474
|
+
return async (sent, url) => {
|
|
475
|
+
const r = await call(sent, url);
|
|
476
|
+
if (r.error || r.raw == null) return r;
|
|
477
|
+
try {
|
|
478
|
+
return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
|
|
479
|
+
} catch (e) {
|
|
480
|
+
return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
|
|
481
|
+
}
|
|
482
|
+
};
|
|
468
483
|
});
|
|
469
484
|
}
|
|
470
485
|
|
|
486
|
+
// What an expression token reads: the item's record, or a text item read as
|
|
487
|
+
// one, and the loop job 1's flow step sits in.
|
|
488
|
+
const tokenRecord = (run, record, text) => {
|
|
489
|
+
const r = record ?? (text != null ? textRecord(text) : null);
|
|
490
|
+
return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
|
|
491
|
+
};
|
|
492
|
+
|
|
493
|
+
// A prompt as a report restates it: an expression token reads an item's
|
|
494
|
+
// record, and a report has none to name, so its words stay as written.
|
|
495
|
+
const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
|
|
496
|
+
|
|
471
497
|
/** A Setup profile the run carries, asked as a grader: its reply's text, or
|
|
472
498
|
the failure as an error the metric reports. One transport per profile. */
|
|
473
499
|
const graders = new WeakMap();
|
|
@@ -590,7 +616,7 @@ function readUtf8(file) {
|
|
|
590
616
|
// the page reads it: a whole-run test's verdict, settled once every item is
|
|
591
617
|
// in, and a per-item test's counts. [settled] is false for a run that
|
|
592
618
|
// stopped short, whose whole-run tests have not settled.
|
|
593
|
-
const verdictsOf = (run, items, settled) => run.
|
|
619
|
+
const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
|
|
594
620
|
Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
|
|
595
621
|
name: o.label, skipped: o.skipped,
|
|
596
622
|
...(o.whole
|
|
@@ -600,7 +626,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
|
|
|
600
626
|
|
|
601
627
|
// The run as the report restates it: each scenario's stages with the
|
|
602
628
|
// connection each one asked.
|
|
603
|
-
const scenariosOf = run => run.
|
|
629
|
+
const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
|
|
604
630
|
const { stages, connections } = core.stagesFor(run, i);
|
|
605
631
|
return {
|
|
606
632
|
n: i + 1, name: sc.name || null,
|
|
@@ -649,7 +675,7 @@ async function runSnapshot(o) {
|
|
|
649
675
|
// Each scenario once: its stages, the jobs' token sets, and a transport
|
|
650
676
|
// per stage to the profile that stage resolves to, keyed by that
|
|
651
677
|
// profile's id.
|
|
652
|
-
const plans = run.
|
|
678
|
+
const plans = core.targetsOf(run).map((_, i) => {
|
|
653
679
|
const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
|
|
654
680
|
const links = connections.map(c => reach(c, core.keyVar(c.id)));
|
|
655
681
|
return { stages, tokens, connections, links, calls, cells };
|
|
@@ -704,7 +730,7 @@ async function runSnapshot(o) {
|
|
|
704
730
|
const textOf = item.kind === "text" && !item.bare
|
|
705
731
|
? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
|
|
706
732
|
const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
|
|
707
|
-
{ tokens: plan.tokens, text: textOf });
|
|
733
|
+
{ tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
|
|
708
734
|
// A case is found by its file's name, whatever kind of file it is: a
|
|
709
735
|
// text item a dataset grades is graded like an image.
|
|
710
736
|
const kase = item.name ? graded.get(item.name) ?? null : null;
|
|
@@ -822,14 +848,14 @@ function cliRun(o, dataset) {
|
|
|
822
848
|
doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
|
|
823
849
|
if (o.tokens) {
|
|
824
850
|
try {
|
|
825
|
-
doc.jobs[0] = core.
|
|
851
|
+
doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
|
|
826
852
|
} catch (e) {
|
|
827
853
|
broken(`${o.tokens}: ${e.message}`);
|
|
828
854
|
}
|
|
829
855
|
}
|
|
830
856
|
doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
|
|
831
|
-
doc.
|
|
832
|
-
|
|
857
|
+
doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
|
|
858
|
+
steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
|
|
833
859
|
doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
|
|
834
860
|
mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
|
|
835
861
|
doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
|
|
@@ -881,7 +907,7 @@ async function main() {
|
|
|
881
907
|
}
|
|
882
908
|
if (o.run) {
|
|
883
909
|
if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
|
|
884
|
-
broken("--run names everything a run takes --
|
|
910
|
+
broken("--run names everything a run takes -- targets, files and text -- so it "
|
|
885
911
|
+ "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
|
|
886
912
|
}
|
|
887
913
|
const n = Number(o.timeout ?? REQUEST_CAP);
|
|
@@ -920,8 +946,8 @@ async function main() {
|
|
|
920
946
|
}
|
|
921
947
|
const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
|
|
922
948
|
if (o.pipeline) {
|
|
923
|
-
if (run.
|
|
924
|
-
broken(`${o.pipeline} has ${run.
|
|
949
|
+
if (core.targetsOf(run).length !== 1) {
|
|
950
|
+
broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
|
|
925
951
|
+ "against the set -- --run runs them all.");
|
|
926
952
|
}
|
|
927
953
|
if (!core.testsDataset(run)) {
|
|
@@ -934,7 +960,7 @@ async function main() {
|
|
|
934
960
|
// Case by case: each item against its case, as the case metric of the
|
|
935
961
|
// test that names the dataset reads it (scoreCase).
|
|
936
962
|
const { stages, tokens, connections } = core.stagesFor(run, 0);
|
|
937
|
-
const prompt =
|
|
963
|
+
const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
|
|
938
964
|
|
|
939
965
|
// The graded half, through the function the tab builds its own list with.
|
|
940
966
|
const set = core.gradedSetFrom(dataset);
|
|
@@ -1170,7 +1196,7 @@ async function main() {
|
|
|
1170
1196
|
stages: stages.map((s, i) => {
|
|
1171
1197
|
const { id, ...connection } = connections[i];
|
|
1172
1198
|
return { n: i + 1, kind: s.kind, withImage: s.withImage,
|
|
1173
|
-
prompt:
|
|
1199
|
+
prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
|
|
1174
1200
|
}),
|
|
1175
1201
|
} } : {}),
|
|
1176
1202
|
// Which variable a key came from, never the key.
|
package/lab/server.py
CHANGED
|
@@ -1476,7 +1476,7 @@ def scenario_ref(sc, i):
|
|
|
1476
1476
|
what version 4's upgrade gives one, by position."""
|
|
1477
1477
|
sid = sc.get("id") if isinstance(sc, dict) else None
|
|
1478
1478
|
name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
|
|
1479
|
-
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"
|
|
1479
|
+
return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
|
|
1480
1480
|
|
|
1481
1481
|
|
|
1482
1482
|
class Prompts:
|
|
@@ -1632,27 +1632,28 @@ class Prompts:
|
|
|
1632
1632
|
recorded once: a second call for the same run changes nothing."""
|
|
1633
1633
|
# The caller's connection: a Row still reads by index, as it does.
|
|
1634
1634
|
db.row_factory = sqlite3.Row
|
|
1635
|
-
for i, sc in
|
|
1636
|
-
if not isinstance(sc, dict):
|
|
1637
|
-
continue
|
|
1635
|
+
for i, sc, k, cell in cells_of(run):
|
|
1638
1636
|
sid, sname = scenario_ref(sc, i)
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1637
|
+
# Echo's words are no wording under test: the item is its reply,
|
|
1638
|
+
# and over Prompt only the words are the reply itself.
|
|
1639
|
+
if isinstance(cell, dict) and cell.get("type") in UNRECORDED_STEPS:
|
|
1640
|
+
continue
|
|
1641
|
+
text = cell.get("prompt") if isinstance(cell, dict) else None
|
|
1642
|
+
if not isinstance(text, str) or not text.strip():
|
|
1643
|
+
continue
|
|
1644
|
+
pid = version = None
|
|
1645
|
+
ref = cell.get("from")
|
|
1646
|
+
if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
|
|
1647
|
+
pid = ref["id"]
|
|
1648
|
+
hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
|
|
1649
|
+
"ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
|
|
1650
|
+
version = hit[0] if hit else self._cut(db, pid, text)
|
|
1651
|
+
if pid is None:
|
|
1652
|
+
hit = self._matching(db, text)
|
|
1653
|
+
pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
|
|
1654
|
+
db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
|
|
1655
|
+
"scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
1656
|
+
(pid, version, rid, sid, sname, k, at))
|
|
1656
1657
|
|
|
1657
1658
|
def backfill(self, runs):
|
|
1658
1659
|
"""The runs from before the library, read into it once. [runs] is
|
|
@@ -2577,16 +2578,56 @@ def registered_ids(manifest: dict) -> set:
|
|
|
2577
2578
|
# ---- Connections: the lab's grants to outside services (#127) --------------
|
|
2578
2579
|
#
|
|
2579
2580
|
# One Google grant per lab, for Sources that read a Drive folder. The OAuth
|
|
2580
|
-
# client is the deployment's (
|
|
2581
|
-
#
|
|
2582
|
-
#
|
|
2583
|
-
#
|
|
2584
|
-
#
|
|
2585
|
-
#
|
|
2581
|
+
# client it signs in with is the deployment's (GOOGLE_* below), then the one
|
|
2582
|
+
# entered in Setup (GoogleApp), then the published app (PUBLIC_GOOGLE_APP),
|
|
2583
|
+
# as Microsoft's is (docs/power-automate.md). The refresh token the grant
|
|
2584
|
+
# yields is the store's, in a table of its own, so /api/state -- which serves
|
|
2585
|
+
# the synced documents to the page -- can never carry it, and nor can the
|
|
2586
|
+
# client's secret. The page is told only whether there is a grant and whose.
|
|
2587
|
+
# The hosts are fixed here, not chosen by anything a request carries, and
|
|
2588
|
+
# reached through OPENER: no redirects, no proxy from the environment.
|
|
2586
2589
|
GOOGLE_CLIENT_ID = os.environ.get("GOOGLE_CLIENT_ID") or None
|
|
2587
2590
|
GOOGLE_CLIENT_SECRET = os.environ.get("GOOGLE_CLIENT_SECRET") or None
|
|
2588
2591
|
GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") or None
|
|
2589
|
-
|
|
2592
|
+
# A deployment reached at its own address signs in with a Web application
|
|
2593
|
+
# client, whose redirect it registers; "installed" is Google's Desktop app.
|
|
2594
|
+
GOOGLE_CLIENT_TYPE = os.environ.get("GOOGLE_CLIENT_TYPE") or "web"
|
|
2595
|
+
GOOGLE_CLIENT_TYPES = ("installed", "web")
|
|
2596
|
+
# An OAuth client's id is its project's number, a dash, and Google's own
|
|
2597
|
+
# suffix: the number is the app id the Picker is told, so it is never asked.
|
|
2598
|
+
GOOGLE_CLIENT = re.compile(r"(\d+)-[0-9a-z]+\.apps\.googleusercontent\.com")
|
|
2599
|
+
GOOGLE_KEY = re.compile(r"[A-Za-z0-9_-]{30,60}")
|
|
2600
|
+
|
|
2601
|
+
|
|
2602
|
+
def read_published_google(path):
|
|
2603
|
+
"""The published app from the file the npm package's build writes
|
|
2604
|
+
beside server.py: (app, None), (None, None) when there is no file, or
|
|
2605
|
+
(None, why) for a file that is not one."""
|
|
2606
|
+
if not path.is_file():
|
|
2607
|
+
return None, None
|
|
2608
|
+
try:
|
|
2609
|
+
got = json.loads(path.read_text("utf-8"))
|
|
2610
|
+
except (OSError, ValueError):
|
|
2611
|
+
return None, f"{path} is not JSON"
|
|
2612
|
+
if not (isinstance(got, dict) and isinstance(got.get("clientId"), str)
|
|
2613
|
+
and GOOGLE_CLIENT.fullmatch(got["clientId"]) and isinstance(got.get("clientSecret"), str)
|
|
2614
|
+
and got["clientSecret"] and isinstance(got.get("apiKey"), str) and GOOGLE_KEY.fullmatch(got["apiKey"])):
|
|
2615
|
+
return None, f"{path} is not a Google app: {{ clientId, clientSecret, apiKey }}"
|
|
2616
|
+
return {"clientId": got["clientId"], "clientSecret": got["clientSecret"],
|
|
2617
|
+
"apiKey": got["apiKey"], "clientType": "installed"}, None
|
|
2618
|
+
|
|
2619
|
+
|
|
2620
|
+
# The Desktop-app client published for every lab, so a lab installed from
|
|
2621
|
+
# npm signs in with nothing to set up. A Desktop client signs in back to any
|
|
2622
|
+
# port on this machine with no redirect registered, and Google does not hold
|
|
2623
|
+
# its secret to be one: it is in every copy of the package by design. It
|
|
2624
|
+
# serves only a lab reached on this machine -- one reached at its own address
|
|
2625
|
+
# brings a Web application client of its own. Never in the repository: the
|
|
2626
|
+
# package's build writes it beside server.py from the Package workflow's
|
|
2627
|
+
# PUBLIC_GOOGLE_APP secret (docs/google-drive.md § The published app), and
|
|
2628
|
+
# a checkout or the image has none.
|
|
2629
|
+
GOOGLE_APP_FILE = HERE / "google-app.json"
|
|
2630
|
+
PUBLIC_GOOGLE_APP, PUBLIC_GOOGLE_PROBLEM = read_published_google(GOOGLE_APP_FILE)
|
|
2590
2631
|
# drive.file: only what the user picks in Google's Picker, and non-sensitive,
|
|
2591
2632
|
# so the app can be published without Google's verification (#127).
|
|
2592
2633
|
GOOGLE_SCOPE = "https://www.googleapis.com/auth/drive.file"
|
|
@@ -2600,6 +2641,94 @@ GOOGLE_CALLBACK = "/api/connections/google/callback"
|
|
|
2600
2641
|
GOOGLE_STATE_SECONDS = 600
|
|
2601
2642
|
|
|
2602
2643
|
|
|
2644
|
+
def google_app():
|
|
2645
|
+
"""The Google app this lab signs in with -- secret included, for the
|
|
2646
|
+
server's own use only -- and where it came from; None with none."""
|
|
2647
|
+
if GOOGLE_CLIENT_ID:
|
|
2648
|
+
return {"clientId": GOOGLE_CLIENT_ID, "clientSecret": GOOGLE_CLIENT_SECRET,
|
|
2649
|
+
"apiKey": GOOGLE_API_KEY, "clientType": GOOGLE_CLIENT_TYPE, "from": "env"}
|
|
2650
|
+
kept = GOOGLE.get() if GOOGLE is not None else None
|
|
2651
|
+
if kept:
|
|
2652
|
+
return {**kept, "from": "lab"}
|
|
2653
|
+
if PUBLIC_GOOGLE_APP:
|
|
2654
|
+
return {**PUBLIC_GOOGLE_APP, "from": "default"}
|
|
2655
|
+
return None
|
|
2656
|
+
|
|
2657
|
+
|
|
2658
|
+
def google_public(app):
|
|
2659
|
+
"""What the page is told of an app: everything but its secret."""
|
|
2660
|
+
if app is None:
|
|
2661
|
+
return None
|
|
2662
|
+
m = GOOGLE_CLIENT.fullmatch(app.get("clientId") or "")
|
|
2663
|
+
return {"clientId": app.get("clientId"), "clientType": app.get("clientType"),
|
|
2664
|
+
"apiKey": app.get("apiKey"), "appId": m.group(1) if m else None,
|
|
2665
|
+
"hasSecret": bool(app.get("clientSecret")), "from": app["from"]}
|
|
2666
|
+
|
|
2667
|
+
|
|
2668
|
+
def loopback_origin(origin) -> bool:
|
|
2669
|
+
"""Whether a page's origin is this machine: where a Desktop client may
|
|
2670
|
+
send the browser back to."""
|
|
2671
|
+
host = urllib.parse.urlsplit(origin).hostname or ""
|
|
2672
|
+
return host in ("localhost", "::1") or host.startswith("127.")
|
|
2673
|
+
|
|
2674
|
+
|
|
2675
|
+
class GoogleApp:
|
|
2676
|
+
"""The Google app entered in Setup: a client and its secret, from the
|
|
2677
|
+
client file Google hands out, and the Picker's key. Kept so a lab with no
|
|
2678
|
+
deployment around it needs no environment variable; the secret is
|
|
2679
|
+
written here and never read back out to the page."""
|
|
2680
|
+
|
|
2681
|
+
KEYS = ("google.clientId", "google.clientSecret", "google.apiKey", "google.clientType")
|
|
2682
|
+
|
|
2683
|
+
def __init__(self, store: Store):
|
|
2684
|
+
self.store = store
|
|
2685
|
+
with store.lock, closing(sqlite3.connect(store.path)) as db, db:
|
|
2686
|
+
db.execute("CREATE TABLE IF NOT EXISTS settings (key TEXT PRIMARY KEY, value TEXT NOT NULL)")
|
|
2687
|
+
|
|
2688
|
+
def _read(self, db):
|
|
2689
|
+
rows = dict(db.execute("SELECT key, value FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS))
|
|
2690
|
+
return dict(zip(("clientId", "clientSecret", "apiKey", "clientType"),
|
|
2691
|
+
(rows.get(k) or None for k in self.KEYS)))
|
|
2692
|
+
|
|
2693
|
+
def get(self):
|
|
2694
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
2695
|
+
got = self._read(db)
|
|
2696
|
+
return {**got, "clientType": got["clientType"] or "installed"} if got["clientId"] else None
|
|
2697
|
+
|
|
2698
|
+
def set(self, payload):
|
|
2699
|
+
"""Keeps what is sent, or with no client id forgets it all: (kept,
|
|
2700
|
+
error, whether the client changed). A secret not sent is kept while
|
|
2701
|
+
the client is the same one, and forgotten when it is not: a secret
|
|
2702
|
+
belongs to its client."""
|
|
2703
|
+
if not isinstance(payload, dict):
|
|
2704
|
+
return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
|
|
2705
|
+
client = payload.get("clientId")
|
|
2706
|
+
secret, key = payload.get("clientSecret"), payload.get("apiKey")
|
|
2707
|
+
ctype = payload.get("clientType") or "installed"
|
|
2708
|
+
if not isinstance(client, str) or not all(v is None or isinstance(v, str) for v in (secret, key)):
|
|
2709
|
+
return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
|
|
2710
|
+
client, key = client.strip(), (key or "").strip()
|
|
2711
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2712
|
+
had = self._read(db)
|
|
2713
|
+
changed = (had["clientId"] or "") != client
|
|
2714
|
+
if not client:
|
|
2715
|
+
db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
|
|
2716
|
+
return None, None, changed
|
|
2717
|
+
if not GOOGLE_CLIENT.fullmatch(client):
|
|
2718
|
+
return None, (400, "a client ID ends .apps.googleusercontent.com"), False
|
|
2719
|
+
if key and not GOOGLE_KEY.fullmatch(key):
|
|
2720
|
+
return None, (400, "that is not an API key"), False
|
|
2721
|
+
if ctype not in GOOGLE_CLIENT_TYPES:
|
|
2722
|
+
return None, (400, "a client is a Desktop app or a Web application"), False
|
|
2723
|
+
if secret is None:
|
|
2724
|
+
secret = None if changed else had["clientSecret"]
|
|
2725
|
+
secret = (secret or "").strip()
|
|
2726
|
+
db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
|
|
2727
|
+
db.executemany("INSERT INTO settings VALUES (?, ?)",
|
|
2728
|
+
[(k, v) for k, v in zip(self.KEYS, (client, secret, key, ctype)) if v])
|
|
2729
|
+
return self.get(), None, changed
|
|
2730
|
+
|
|
2731
|
+
|
|
2603
2732
|
class Connections:
|
|
2604
2733
|
"""The lab's grants, held server-side; the page sees their state only."""
|
|
2605
2734
|
|
|
@@ -2619,7 +2748,8 @@ class Connections:
|
|
|
2619
2748
|
|
|
2620
2749
|
@staticmethod
|
|
2621
2750
|
def configured() -> bool:
|
|
2622
|
-
|
|
2751
|
+
app = google_app()
|
|
2752
|
+
return bool(app and app.get("clientId") and app.get("clientSecret"))
|
|
2623
2753
|
|
|
2624
2754
|
def _row(self):
|
|
2625
2755
|
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
|
|
@@ -2633,15 +2763,21 @@ class Connections:
|
|
|
2633
2763
|
|
|
2634
2764
|
def start(self, origin: str):
|
|
2635
2765
|
"""The consent address the page sends the browser to."""
|
|
2766
|
+
app = google_app()
|
|
2636
2767
|
if not self.configured():
|
|
2637
2768
|
return None, (409, "Google is not configured for this lab")
|
|
2769
|
+
if app["clientType"] == "installed" and not loopback_origin(origin):
|
|
2770
|
+
return None, (409, "a Desktop app client signs in only on this machine: "
|
|
2771
|
+
"give this lab a Web application client in Configure…")
|
|
2638
2772
|
state = secrets.token_urlsafe(24)
|
|
2639
2773
|
now = time.time()
|
|
2640
2774
|
with self.lock:
|
|
2641
2775
|
self.states = {s: v for s, v in self.states.items() if v[0] > now}
|
|
2642
|
-
|
|
2776
|
+
# The app is kept with the state: the code Google sends back is
|
|
2777
|
+
# exchanged with the client that asked for it, whatever changes.
|
|
2778
|
+
self.states[state] = (now + GOOGLE_STATE_SECONDS, origin, app)
|
|
2643
2779
|
return GOOGLE_CONSENT_URL + "?" + urllib.parse.urlencode({
|
|
2644
|
-
"client_id":
|
|
2780
|
+
"client_id": app["clientId"], "redirect_uri": origin + GOOGLE_CALLBACK,
|
|
2645
2781
|
"response_type": "code", "scope": GOOGLE_SCOPE, "state": state,
|
|
2646
2782
|
"access_type": "offline", "prompt": "consent", "include_granted_scopes": "true",
|
|
2647
2783
|
}), None
|
|
@@ -2661,7 +2797,7 @@ class Connections:
|
|
|
2661
2797
|
which never include the code or a token."""
|
|
2662
2798
|
state = (query.get("state") or [""])[0]
|
|
2663
2799
|
with self.lock:
|
|
2664
|
-
lapses, origin = self.states.pop(state, (0, ""))
|
|
2800
|
+
lapses, origin, app = self.states.pop(state, (0, "", None))
|
|
2665
2801
|
if lapses <= time.time():
|
|
2666
2802
|
return "that sign-in had lapsed or was not this lab's; sign in again"
|
|
2667
2803
|
if query.get("error"):
|
|
@@ -2671,7 +2807,7 @@ class Connections:
|
|
|
2671
2807
|
return "Google sent no code back"
|
|
2672
2808
|
try:
|
|
2673
2809
|
got = self._call(GOOGLE_TOKEN_URL, {
|
|
2674
|
-
"code": code, "client_id":
|
|
2810
|
+
"code": code, "client_id": app["clientId"], "client_secret": app["clientSecret"],
|
|
2675
2811
|
"redirect_uri": origin + GOOGLE_CALLBACK, "grant_type": "authorization_code"})
|
|
2676
2812
|
except (OSError, ValueError) as e:
|
|
2677
2813
|
return f"the code could not be exchanged ({type(e).__name__})"
|
|
@@ -2689,6 +2825,12 @@ class Connections:
|
|
|
2689
2825
|
time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())))
|
|
2690
2826
|
return None
|
|
2691
2827
|
|
|
2828
|
+
def forget(self):
|
|
2829
|
+
"""Drops the grant without a word to Google: the client it was made
|
|
2830
|
+
with is no longer this lab's, so the grant is no use to it."""
|
|
2831
|
+
with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
|
|
2832
|
+
db.execute("DELETE FROM connections WHERE id = 'google'")
|
|
2833
|
+
|
|
2692
2834
|
def sign_out(self):
|
|
2693
2835
|
"""Forgets the grant, and asks Google to revoke it; a revoke that
|
|
2694
2836
|
fails still forgets it here, which is what signing out means."""
|
|
@@ -3589,17 +3731,17 @@ def worker_destinations(run: dict):
|
|
|
3589
3731
|
base = api_base(str(conn.get("url") or "")) or api_base(OLLAMA)
|
|
3590
3732
|
why = allowed(base, "")
|
|
3591
3733
|
if why:
|
|
3592
|
-
return None, f"
|
|
3734
|
+
return None, f"Target profile {name}: {why}"
|
|
3593
3735
|
profile = next((p for p in stored if isinstance(p, dict) and p.get("id") == pid), None)
|
|
3594
3736
|
if profile is None:
|
|
3595
|
-
return None, f"
|
|
3737
|
+
return None, f"Target profile {name} not found"
|
|
3596
3738
|
key = str(profile.get("key") or "").strip()
|
|
3597
3739
|
if key:
|
|
3598
3740
|
if not header_safe(key):
|
|
3599
|
-
return None, f"
|
|
3741
|
+
return None, f"Target profile {name} has a key that cannot go in a header"
|
|
3600
3742
|
why = allowed(base, key)
|
|
3601
3743
|
if why:
|
|
3602
|
-
return None, f"
|
|
3744
|
+
return None, f"Target profile {name}: {why}"
|
|
3603
3745
|
env[key_var(pid)] = key
|
|
3604
3746
|
return env, None
|
|
3605
3747
|
|
|
@@ -3618,12 +3760,17 @@ def worker_destinations(run: dict):
|
|
|
3618
3760
|
# 7: chains are jobs: the field is `jobs` and each job's type is "job".
|
|
3619
3761
|
# 8: a job is its steps; the content is job 1's Attach Content step.
|
|
3620
3762
|
# 9: a test is Metrics; a Single Test or a Graded set is read converted.
|
|
3621
|
-
|
|
3763
|
+
# 10: a job's steps are its stages, and each scenario is a target whose own
|
|
3764
|
+
# step in each job is what it sends there (docs/pipeline-model.md §16).
|
|
3765
|
+
PIPELINE_VERSION = 10
|
|
3622
3766
|
# What a stored run may be: the current version, and the ones evals-core.ts's
|
|
3623
|
-
# upgradePipeline reads. A new submission is
|
|
3624
|
-
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9)
|
|
3625
|
-
|
|
3626
|
-
|
|
3767
|
+
# upgradePipeline reads. A new submission is upgraded to the current one.
|
|
3768
|
+
READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
|
|
3769
|
+
TARGET_CAP = 4
|
|
3770
|
+
# Target steps whose words the Prompt library does not record as a use: they
|
|
3771
|
+
# ask no model (evals-core.ts's STEP_TYPES.echo).
|
|
3772
|
+
UNRECORDED_STEPS = {"echo"}
|
|
3773
|
+
RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "tests", "profiles", "comment", "plugins")
|
|
3627
3774
|
|
|
3628
3775
|
|
|
3629
3776
|
def content_of(doc):
|
|
@@ -3823,16 +3970,20 @@ def upgrade_run(run):
|
|
|
3823
3970
|
return up, None
|
|
3824
3971
|
|
|
3825
3972
|
|
|
3826
|
-
def
|
|
3827
|
-
"""
|
|
3828
|
-
|
|
3829
|
-
|
|
3830
|
-
|
|
3831
|
-
if not isinstance(
|
|
3832
|
-
return
|
|
3833
|
-
|
|
3834
|
-
|
|
3835
|
-
|
|
3973
|
+
def cells_of(run):
|
|
3974
|
+
"""Every target's step in every job of a run, as (i, target, k, step):
|
|
3975
|
+
version 10's targets and their steps, or an earlier version's scenarios
|
|
3976
|
+
and their cells -- what the Prompt library records a use from, whichever
|
|
3977
|
+
version a stored run is."""
|
|
3978
|
+
if not isinstance(run, dict):
|
|
3979
|
+
return
|
|
3980
|
+
targets = run.get("targets") if isinstance(run.get("targets"), list) else run.get("scenarios")
|
|
3981
|
+
for i, t in enumerate(targets if isinstance(targets, list) else []):
|
|
3982
|
+
if not isinstance(t, dict):
|
|
3983
|
+
continue
|
|
3984
|
+
steps = t.get("steps") if isinstance(t.get("steps"), list) else t.get("stages")
|
|
3985
|
+
for k, step in enumerate(steps if isinstance(steps, list) else []):
|
|
3986
|
+
yield i, t, k, step
|
|
3836
3987
|
|
|
3837
3988
|
|
|
3838
3989
|
def run_problems(run):
|
|
@@ -3877,36 +4028,41 @@ def run_problems(run):
|
|
|
3877
4028
|
bad.append(f"{at} has to be an object")
|
|
3878
4029
|
continue
|
|
3879
4030
|
connection_problems(pid, conn, at, bad)
|
|
3880
|
-
|
|
3881
|
-
if not isinstance(
|
|
3882
|
-
return bad + [f"a run needs between one and {
|
|
3883
|
-
ids = [
|
|
3884
|
-
for i,
|
|
3885
|
-
at = f"
|
|
3886
|
-
if not isinstance(
|
|
4031
|
+
targets = run.get("targets")
|
|
4032
|
+
if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
|
|
4033
|
+
return bad + [f"a run needs between one and {TARGET_CAP} targets"]
|
|
4034
|
+
ids = [t.get("id") for t in targets if isinstance(t, dict)]
|
|
4035
|
+
for i, t in enumerate(targets):
|
|
4036
|
+
at = f"target {i + 1}"
|
|
4037
|
+
if not isinstance(t, dict):
|
|
3887
4038
|
bad.append(f"{at} has to be an object")
|
|
3888
4039
|
continue
|
|
3889
4040
|
# The id the Prompt library records a use under (version 4).
|
|
3890
|
-
if not isinstance(
|
|
4041
|
+
if not isinstance(t.get("id"), str) or not t["id"].strip():
|
|
3891
4042
|
bad.append(f"{at} has no id")
|
|
3892
|
-
elif ids.count(
|
|
3893
|
-
bad.append(f"{at} has the id of another
|
|
3894
|
-
|
|
3895
|
-
if not isinstance(
|
|
3896
|
-
bad.append(f"{at} has to have one
|
|
4043
|
+
elif ids.count(t["id"]) > 1:
|
|
4044
|
+
bad.append(f"{at} has the id of another target")
|
|
4045
|
+
steps = t.get("steps")
|
|
4046
|
+
if not isinstance(steps, list) or len(steps) != len(jobs):
|
|
4047
|
+
bad.append(f"{at} has to have one step per job")
|
|
3897
4048
|
continue
|
|
3898
|
-
|
|
3899
|
-
|
|
3900
|
-
|
|
3901
|
-
if
|
|
3902
|
-
|
|
4049
|
+
# A target with no profile is one whose steps ask none (Echo) or each
|
|
4050
|
+
# name their own; which steps need one is the core's to judge.
|
|
4051
|
+
refs = ([t["profile"]] if t.get("profile") is not None else []) + [
|
|
4052
|
+
st.get("profile") for st in steps if isinstance(st, dict) and st.get("profile") is not None]
|
|
4053
|
+
# Whether the words may be blank -- Echo answering from the item --
|
|
4054
|
+
# is the core's to judge, and the worker refuses the run in its words.
|
|
4055
|
+
for k, step in enumerate(steps):
|
|
4056
|
+
if not isinstance(step, dict) or not isinstance(step.get("type"), str):
|
|
4057
|
+
bad.append(f"{at}, job {k + 1} has to name what it sends")
|
|
4058
|
+
elif not isinstance(step.get("prompt"), str):
|
|
3903
4059
|
bad.append(f"{at}, job {k + 1} has no prompt")
|
|
3904
|
-
elif
|
|
3905
|
-
isinstance(
|
|
4060
|
+
elif step.get("from") is not None and not (
|
|
4061
|
+
isinstance(step["from"], dict) and isinstance(step["from"].get("id"), str)):
|
|
3906
4062
|
bad.append(f"{at}, job {k + 1} names the prompt it was picked from without an id")
|
|
3907
4063
|
for ref in refs:
|
|
3908
4064
|
if not isinstance(ref, dict) or ref.get("id") not in table:
|
|
3909
|
-
bad.append(f"{at} names a
|
|
4065
|
+
bad.append(f"{at} names a Target profile the run does not carry")
|
|
3910
4066
|
content = content_of(run)
|
|
3911
4067
|
kind = content.get("type") if isinstance(content, dict) else None
|
|
3912
4068
|
if kind == "source":
|
|
@@ -3946,10 +4102,11 @@ if DATA_DIR:
|
|
|
3946
4102
|
PLUGINS = Plugins(STORE)
|
|
3947
4103
|
CONNECTIONS = Connections(STORE)
|
|
3948
4104
|
MICROSOFT = MicrosoftApp(STORE)
|
|
4105
|
+
GOOGLE = GoogleApp(STORE)
|
|
3949
4106
|
QUEUE = Queue(STORE)
|
|
3950
4107
|
QUEUE.prompts = PROMPTS
|
|
3951
4108
|
else:
|
|
3952
|
-
STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = QUEUE = None
|
|
4109
|
+
STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
|
|
3953
4110
|
|
|
3954
4111
|
|
|
3955
4112
|
class NoRedirects(urllib.request.HTTPRedirectHandler):
|
|
@@ -4105,7 +4262,7 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4105
4262
|
# And the Microsoft app a flow is read with, when there is one.
|
|
4106
4263
|
# The app may be entered in Setup unless the deployment names
|
|
4107
4264
|
# its own, and only in a lab with a store to keep it in.
|
|
4108
|
-
return self._json(200, {"
|
|
4265
|
+
return self._json(200, {"targetCap": TARGET_CAP, "microsoft": microsoft_config(),
|
|
4109
4266
|
"microsoftEditable": MICROSOFT is not None and not M365_CLIENT_ID})
|
|
4110
4267
|
if path == "/api/state":
|
|
4111
4268
|
if STORE is None:
|
|
@@ -4161,6 +4318,10 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4161
4318
|
if CONNECTIONS is None:
|
|
4162
4319
|
return self._send(404, b"not found", "text/plain")
|
|
4163
4320
|
return self._json(200, {"connections": CONNECTIONS.list()})
|
|
4321
|
+
if path == "/api/google":
|
|
4322
|
+
if GOOGLE is None:
|
|
4323
|
+
return self._send(404, b"not found", "text/plain")
|
|
4324
|
+
return self._json(200, {"google": google_public(google_app()), "editable": not GOOGLE_CLIENT_ID})
|
|
4164
4325
|
if path == GOOGLE_CALLBACK:
|
|
4165
4326
|
return self._google_callback()
|
|
4166
4327
|
if path.startswith("/plugins/"):
|
|
@@ -4381,6 +4542,17 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
4381
4542
|
parts = self.path.split("?", 1)[0].split("/")
|
|
4382
4543
|
if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
|
|
4383
4544
|
return self._prompts_put(parts[3])
|
|
4545
|
+
if parts == ["", "api", "google"] and GOOGLE is not None:
|
|
4546
|
+
if GOOGLE_CLIENT_ID:
|
|
4547
|
+
return self._json(409, {"error": "this lab's Google app is its deployment's (GOOGLE_CLIENT_ID)"})
|
|
4548
|
+
_, err, changed = GOOGLE.set(self._payload())
|
|
4549
|
+
if err:
|
|
4550
|
+
return self._json(err[0], {"error": err[1]})
|
|
4551
|
+
# A grant is its client's: one made with another client cannot be
|
|
4552
|
+
# refreshed with this one, so it goes rather than failing later.
|
|
4553
|
+
if changed and CONNECTIONS is not None:
|
|
4554
|
+
CONNECTIONS.forget()
|
|
4555
|
+
return self._json(200, {"google": google_public(google_app()), "editable": True})
|
|
4384
4556
|
if parts == ["", "api", "microsoft"] and MICROSOFT is not None:
|
|
4385
4557
|
if M365_CLIENT_ID:
|
|
4386
4558
|
return self._json(409, {"error": "this lab's Microsoft app is its deployment's (M365_CLIENT_ID)"})
|
|
@@ -5065,6 +5237,10 @@ def main():
|
|
|
5065
5237
|
if not (DEMO / "manifest.json").is_file():
|
|
5066
5238
|
raise SystemExit(f"no {DEMO / 'manifest.json'} -- the demo pack sits beside server.py, "
|
|
5067
5239
|
"and the image copies it there")
|
|
5240
|
+
# A published app the build wrote and that cannot be read is a broken
|
|
5241
|
+
# package, not one without the app: said at once, not at Sign in.
|
|
5242
|
+
if PUBLIC_GOOGLE_PROBLEM:
|
|
5243
|
+
raise SystemExit(PUBLIC_GOOGLE_PROBLEM)
|
|
5068
5244
|
print(f"prompt-lab on {HOST}:{PORT} -> {OLLAMA} by default", flush=True)
|
|
5069
5245
|
print(f"samples: {SAMPLES}", flush=True)
|
|
5070
5246
|
print(f"page: {WEB_DIST}", flush=True)
|