evals-lab 0.1.4 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lab/run-evals.js CHANGED
@@ -453,21 +453,47 @@ let seconds = REQUEST_CAP; // the run's per-request cap, set in main()
453
453
  // or, over Prompt only, the prompt -- and the stage reads it as it would a
454
454
  // model's.
455
455
  //
456
- // A call that asks for a whole request (an HTTP Request, #flows) builds it
457
- // from its step, the scenario's cell and the item's record, and is handed
458
- // only its words: the transport is where the three meet.
456
+ // A target step that asks for a whole request (an HTTP Request, #flows)
457
+ // builds it from the step, the job's flow step and the item's record, and is
458
+ // handed only its words: the transport is where the three meet.
459
+ //
460
+ // Any other step's reply, in a job that reads the flow's step through a Read
461
+ // as, is read the same way: put in the body the flow's API would have sent
462
+ // it in, then read as the flow reads it -- so a model asked in words and
463
+ // production are compared alike (docs/pipeline-model.md §16 › Targets).
459
464
  function callsFor(connections, links, text, plan, record) {
460
465
  return connections.map((c, k) => {
461
466
  const step = plan?.calls?.[k];
462
467
  if (core.STEP_TYPES[step?.type]?.asks === "request") {
463
468
  return () => send(links[k], c, step, plan.cells[k], record ?? textRecord(text));
464
469
  }
465
- return core.CONNECTION_TYPES[core.typeOf(c)]?.local
470
+ const call = core.CONNECTION_TYPES[core.typeOf(c)]?.local
466
471
  ? async sent => core.localAnswer(c, text, sent)
467
472
  : (sent, url) => ask(links[k], c, sent, url);
473
+ if (!step?.readAs || !step?.step) return call;
474
+ return async (sent, url) => {
475
+ const r = await call(sent, url);
476
+ if (r.error || r.raw == null) return r;
477
+ try {
478
+ return { ...r, ...core.readFlowReply(step, r.raw, record ?? textRecord(text)) };
479
+ } catch (e) {
480
+ return { ...r, error: `the reply could not be read as the flow reads it -- ${e.message}` };
481
+ }
482
+ };
468
483
  });
469
484
  }
470
485
 
486
+ // What an expression token reads: the item's record, or a text item read as
487
+ // one, and the loop job 1's flow step sits in.
488
+ const tokenRecord = (run, record, text) => {
489
+ const r = record ?? (text != null ? textRecord(text) : null);
490
+ return r && { scope: r.scope || {}, at: r.at ?? null, loop: core.flowStepOf(run.jobs[0])?.loop ?? null };
491
+ };
492
+
493
+ // A prompt as a report restates it: an expression token reads an item's
494
+ // record, and a report has none to name, so its words stay as written.
495
+ const shown = (text, tokens) => { try { return core.resolvePrompt(text, tokens); } catch { return text; } };
496
+
471
497
  /** A Setup profile the run carries, asked as a grader: its reply's text, or
472
498
  the failure as an error the metric reports. One transport per profile. */
473
499
  const graders = new WeakMap();
@@ -590,7 +616,7 @@ function readUtf8(file) {
590
616
  // the page reads it: a whole-run test's verdict, settled once every item is
591
617
  // in, and a per-item test's counts. [settled] is false for a run that
592
618
  // stopped short, whose whole-run tests have not settled.
593
- const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
619
+ const verdictsOf = (run, items, settled) => core.targetsOf(run).map((_, i) =>
594
620
  Object.fromEntries(core.scenarioTests(run, i, items, settled).map(o => [o.id, {
595
621
  name: o.label, skipped: o.skipped,
596
622
  ...(o.whole
@@ -600,7 +626,7 @@ const verdictsOf = (run, items, settled) => run.scenarios.map((_, i) =>
600
626
 
601
627
  // The run as the report restates it: each scenario's stages with the
602
628
  // connection each one asked.
603
- const scenariosOf = run => run.scenarios.map((sc, i) => {
629
+ const scenariosOf = run => core.targetsOf(run).map((sc, i) => {
604
630
  const { stages, connections } = core.stagesFor(run, i);
605
631
  return {
606
632
  n: i + 1, name: sc.name || null,
@@ -649,7 +675,7 @@ async function runSnapshot(o) {
649
675
  // Each scenario once: its stages, the jobs' token sets, and a transport
650
676
  // per stage to the profile that stage resolves to, keyed by that
651
677
  // profile's id.
652
- const plans = run.scenarios.map((_, i) => {
678
+ const plans = core.targetsOf(run).map((_, i) => {
653
679
  const { stages, tokens, connections, calls, cells } = core.stagesFor(run, i);
654
680
  const links = connections.map(c => reach(c, core.keyVar(c.id)));
655
681
  return { stages, tokens, connections, links, calls, cells };
@@ -704,7 +730,7 @@ async function runSnapshot(o) {
704
730
  const textOf = item.kind === "text" && !item.bare
705
731
  ? (item.text != null ? item.text : readUtf8(path.join(o.source, item.name))) : null;
706
732
  const res = await core.runPipeline(plan.stages, dataUrl, callsFor(plan.connections, plan.links, textOf, plan, record),
707
- { tokens: plan.tokens, text: textOf });
733
+ { tokens: plan.tokens, text: textOf, record: tokenRecord(run, record, textOf) });
708
734
  // A case is found by its file's name, whatever kind of file it is: a
709
735
  // text item a dataset grades is graded like an image.
710
736
  const kase = item.name ? graded.get(item.name) ?? null : null;
@@ -822,14 +848,14 @@ function cliRun(o, dataset) {
822
848
  doc.jobs[0] = core.withOut(doc.jobs[0], core.legacyTagsOut(rulesOf(dataset), "LOWER"));
823
849
  if (o.tokens) {
824
850
  try {
825
- doc.jobs[0] = core.withCall(doc.jobs[0], { tokenMappings: JSON.parse(fs.readFileSync(o.tokens, "utf8")) });
851
+ doc.jobs[0] = core.withTokens(doc.jobs[0], JSON.parse(fs.readFileSync(o.tokens, "utf8")));
826
852
  } catch (e) {
827
853
  broken(`${o.tokens}: ${e.message}`);
828
854
  }
829
855
  }
830
856
  doc = core.withContent(doc, { type: "source", ref: { id: "cli", name: "" }, first: null, loops: 1, files: [] });
831
- doc.scenarios = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
832
- stages: [{ prompt: o.prompt ?? "" }] }];
857
+ doc.targets = [{ id: "cli", name: "", profile: { id: "cli", name: "" },
858
+ steps: [{ type: "prompt", prompt: o.prompt ?? "" }] }];
833
859
  doc.tests = [{ id: "cli", type: "metrics", name: "", continueOnFailure: true, dataset: { id: "cli", name: "" },
834
860
  mode: "all", threshold: null, grader: null, over: "item", metrics: [{ type: "case" }] }];
835
861
  doc.profiles = { cli: { name: null, url: o.url, model: o.model, type: "openai-compatible" } };
@@ -881,7 +907,7 @@ async function main() {
881
907
  }
882
908
  if (o.run) {
883
909
  if (o.pipeline || o.prompt != null || o.tokens || o.model || o.url || o.replies || o.samples || o.files) {
884
- broken("--run names everything a run takes -- scenarios, files and text -- so it "
910
+ broken("--run names everything a run takes -- targets, files and text -- so it "
885
911
  + "cannot be given with --pipeline, --prompt, --tokens, --model, --url, --replies, --samples or --files.");
886
912
  }
887
913
  const n = Number(o.timeout ?? REQUEST_CAP);
@@ -920,8 +946,8 @@ async function main() {
920
946
  }
921
947
  const run = o.pipeline ? readRun(o.pipeline, dataset) : cliRun(o, dataset);
922
948
  if (o.pipeline) {
923
- if (run.scenarios.length !== 1) {
924
- broken(`${o.pipeline} has ${run.scenarios.length} scenarios, and --pipeline grades one `
949
+ if (core.targetsOf(run).length !== 1) {
950
+ broken(`${o.pipeline} has ${core.targetsOf(run).length} targets, and --pipeline grades one `
925
951
  + "against the set -- --run runs them all.");
926
952
  }
927
953
  if (!core.testsDataset(run)) {
@@ -934,7 +960,7 @@ async function main() {
934
960
  // Case by case: each item against its case, as the case metric of the
935
961
  // test that names the dataset reads it (scoreCase).
936
962
  const { stages, tokens, connections } = core.stagesFor(run, 0);
937
- const prompt = core.resolvePrompt(stages[0].text, core.tokenSet(tokens, 0));
963
+ const prompt = shown(stages[0].text, core.tokenSet(tokens, 0));
938
964
 
939
965
  // The graded half, through the function the tab builds its own list with.
940
966
  const set = core.gradedSetFrom(dataset);
@@ -1170,7 +1196,7 @@ async function main() {
1170
1196
  stages: stages.map((s, i) => {
1171
1197
  const { id, ...connection } = connections[i];
1172
1198
  return { n: i + 1, kind: s.kind, withImage: s.withImage,
1173
- prompt: core.resolvePrompt(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1199
+ prompt: shown(s.text, core.tokenSet(tokens, i)), profile: id, connection };
1174
1200
  }),
1175
1201
  } } : {}),
1176
1202
  // Which variable a key came from, never the key.
package/lab/server.py CHANGED
@@ -1476,7 +1476,7 @@ def scenario_ref(sc, i):
1476
1476
  what version 4's upgrade gives one, by position."""
1477
1477
  sid = sc.get("id") if isinstance(sc, dict) else None
1478
1478
  name = (sc.get("name") or "").strip() if isinstance(sc, dict) and isinstance(sc.get("name"), str) else ""
1479
- return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Scenario {i + 1}")
1479
+ return (sid if isinstance(sid, str) and sid else f"s{i + 1}"), (name or f"Target {i + 1}")
1480
1480
 
1481
1481
 
1482
1482
  class Prompts:
@@ -1632,27 +1632,28 @@ class Prompts:
1632
1632
  recorded once: a second call for the same run changes nothing."""
1633
1633
  # The caller's connection: a Row still reads by index, as it does.
1634
1634
  db.row_factory = sqlite3.Row
1635
- for i, sc in enumerate(run.get("scenarios") or []):
1636
- if not isinstance(sc, dict):
1637
- continue
1635
+ for i, sc, k, cell in cells_of(run):
1638
1636
  sid, sname = scenario_ref(sc, i)
1639
- for k, cell in enumerate(sc.get("stages") or []):
1640
- text = cell.get("prompt") if isinstance(cell, dict) else None
1641
- if not isinstance(text, str) or not text.strip():
1642
- continue
1643
- pid = version = None
1644
- ref = cell.get("from")
1645
- if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
1646
- pid = ref["id"]
1647
- hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
1648
- "ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
1649
- version = hit[0] if hit else self._cut(db, pid, text)
1650
- if pid is None:
1651
- hit = self._matching(db, text)
1652
- pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
1653
- db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
1654
- "scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
1655
- (pid, version, rid, sid, sname, k, at))
1637
+ # Echo's words are no wording under test: the item is its reply,
1638
+ # and over Prompt only the words are the reply itself.
1639
+ if isinstance(cell, dict) and cell.get("type") in UNRECORDED_STEPS:
1640
+ continue
1641
+ text = cell.get("prompt") if isinstance(cell, dict) else None
1642
+ if not isinstance(text, str) or not text.strip():
1643
+ continue
1644
+ pid = version = None
1645
+ ref = cell.get("from")
1646
+ if isinstance(ref, dict) and isinstance(ref.get("id"), str) and self._live(db, ref["id"]):
1647
+ pid = ref["id"]
1648
+ hit = db.execute("SELECT version FROM prompt_versions WHERE prompt_id = ? AND text = ? "
1649
+ "ORDER BY version DESC LIMIT 1", (pid, text)).fetchone()
1650
+ version = hit[0] if hit else self._cut(db, pid, text)
1651
+ if pid is None:
1652
+ hit = self._matching(db, text)
1653
+ pid, version = (hit[0], hit[1]) if hit else (self._insert(db, "", text), 1)
1654
+ db.execute("INSERT OR IGNORE INTO prompt_uses (prompt_id, version, run_id, scenario_id, "
1655
+ "scenario_name, job, at) VALUES (?, ?, ?, ?, ?, ?, ?)",
1656
+ (pid, version, rid, sid, sname, k, at))
1656
1657
 
1657
1658
  def backfill(self, runs):
1658
1659
  """The runs from before the library, read into it once. [runs] is
@@ -2577,16 +2578,56 @@ def registered_ids(manifest: dict) -> set:
2577
2578
  # ---- Connections: the lab's grants to outside services (#127) --------------
2578
2579
  #
2579
2580
  # One Google grant per lab, for Sources that read a Drive folder. The OAuth
2580
- # client is the deployment's (env vars), never the repository's or the
2581
- # store's; the refresh token the grant yields is the store's, in a table of
2582
- # its own, so /api/state -- which serves the synced documents to the page --
2583
- # can never carry it. The page is told only whether there is a grant and
2584
- # whose. The hosts are fixed here, not chosen by anything a request carries,
2585
- # and reached through OPENER: no redirects, no proxy from the environment.
2581
+ # client it signs in with is the deployment's (GOOGLE_* below), then the one
2582
+ # entered in Setup (GoogleApp), then the published app (PUBLIC_GOOGLE_APP),
2583
+ # as Microsoft's is (docs/power-automate.md). The refresh token the grant
2584
+ # yields is the store's, in a table of its own, so /api/state -- which serves
2585
+ # the synced documents to the page -- can never carry it, and nor can the
2586
+ # client's secret. The page is told only whether there is a grant and whose.
2587
+ # The hosts are fixed here, not chosen by anything a request carries, and
2588
+ # reached through OPENER: no redirects, no proxy from the environment.
2586
2589
  GOOGLE_CLIENT_ID = os.environ.get("GOOGLE_CLIENT_ID") or None
2587
2590
  GOOGLE_CLIENT_SECRET = os.environ.get("GOOGLE_CLIENT_SECRET") or None
2588
2591
  GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") or None
2589
- GOOGLE_APP_ID = os.environ.get("GOOGLE_APP_ID") or None
2592
+ # A deployment reached at its own address signs in with a Web application
2593
+ # client, whose redirect it registers; "installed" is Google's Desktop app.
2594
+ GOOGLE_CLIENT_TYPE = os.environ.get("GOOGLE_CLIENT_TYPE") or "web"
2595
+ GOOGLE_CLIENT_TYPES = ("installed", "web")
2596
+ # An OAuth client's id is its project's number, a dash, and Google's own
2597
+ # suffix: the number is the app id the Picker is told, so it is never asked.
2598
+ GOOGLE_CLIENT = re.compile(r"(\d+)-[0-9a-z]+\.apps\.googleusercontent\.com")
2599
+ GOOGLE_KEY = re.compile(r"[A-Za-z0-9_-]{30,60}")
2600
+
2601
+
2602
+ def read_published_google(path):
2603
+ """The published app from the file the npm package's build writes
2604
+ beside server.py: (app, None), (None, None) when there is no file, or
2605
+ (None, why) for a file that is not one."""
2606
+ if not path.is_file():
2607
+ return None, None
2608
+ try:
2609
+ got = json.loads(path.read_text("utf-8"))
2610
+ except (OSError, ValueError):
2611
+ return None, f"{path} is not JSON"
2612
+ if not (isinstance(got, dict) and isinstance(got.get("clientId"), str)
2613
+ and GOOGLE_CLIENT.fullmatch(got["clientId"]) and isinstance(got.get("clientSecret"), str)
2614
+ and got["clientSecret"] and isinstance(got.get("apiKey"), str) and GOOGLE_KEY.fullmatch(got["apiKey"])):
2615
+ return None, f"{path} is not a Google app: {{ clientId, clientSecret, apiKey }}"
2616
+ return {"clientId": got["clientId"], "clientSecret": got["clientSecret"],
2617
+ "apiKey": got["apiKey"], "clientType": "installed"}, None
2618
+
2619
+
2620
+ # The Desktop-app client published for every lab, so a lab installed from
2621
+ # npm signs in with nothing to set up. A Desktop client signs in back to any
2622
+ # port on this machine with no redirect registered, and Google does not hold
2623
+ # its secret to be one: it is in every copy of the package by design. It
2624
+ # serves only a lab reached on this machine -- one reached at its own address
2625
+ # brings a Web application client of its own. Never in the repository: the
2626
+ # package's build writes it beside server.py from the Package workflow's
2627
+ # PUBLIC_GOOGLE_APP secret (docs/google-drive.md § The published app), and
2628
+ # a checkout or the image has none.
2629
+ GOOGLE_APP_FILE = HERE / "google-app.json"
2630
+ PUBLIC_GOOGLE_APP, PUBLIC_GOOGLE_PROBLEM = read_published_google(GOOGLE_APP_FILE)
2590
2631
  # drive.file: only what the user picks in Google's Picker, and non-sensitive,
2591
2632
  # so the app can be published without Google's verification (#127).
2592
2633
  GOOGLE_SCOPE = "https://www.googleapis.com/auth/drive.file"
@@ -2600,6 +2641,94 @@ GOOGLE_CALLBACK = "/api/connections/google/callback"
2600
2641
  GOOGLE_STATE_SECONDS = 600
2601
2642
 
2602
2643
 
2644
+ def google_app():
2645
+ """The Google app this lab signs in with -- secret included, for the
2646
+ server's own use only -- and where it came from; None with none."""
2647
+ if GOOGLE_CLIENT_ID:
2648
+ return {"clientId": GOOGLE_CLIENT_ID, "clientSecret": GOOGLE_CLIENT_SECRET,
2649
+ "apiKey": GOOGLE_API_KEY, "clientType": GOOGLE_CLIENT_TYPE, "from": "env"}
2650
+ kept = GOOGLE.get() if GOOGLE is not None else None
2651
+ if kept:
2652
+ return {**kept, "from": "lab"}
2653
+ if PUBLIC_GOOGLE_APP:
2654
+ return {**PUBLIC_GOOGLE_APP, "from": "default"}
2655
+ return None
2656
+
2657
+
2658
+ def google_public(app):
2659
+ """What the page is told of an app: everything but its secret."""
2660
+ if app is None:
2661
+ return None
2662
+ m = GOOGLE_CLIENT.fullmatch(app.get("clientId") or "")
2663
+ return {"clientId": app.get("clientId"), "clientType": app.get("clientType"),
2664
+ "apiKey": app.get("apiKey"), "appId": m.group(1) if m else None,
2665
+ "hasSecret": bool(app.get("clientSecret")), "from": app["from"]}
2666
+
2667
+
2668
+ def loopback_origin(origin) -> bool:
2669
+ """Whether a page's origin is this machine: where a Desktop client may
2670
+ send the browser back to."""
2671
+ host = urllib.parse.urlsplit(origin).hostname or ""
2672
+ return host in ("localhost", "::1") or host.startswith("127.")
2673
+
2674
+
2675
+ class GoogleApp:
2676
+ """The Google app entered in Setup: a client and its secret, from the
2677
+ client file Google hands out, and the Picker's key. Kept so a lab with no
2678
+ deployment around it needs no environment variable; the secret is
2679
+ written here and never read back out to the page."""
2680
+
2681
+ KEYS = ("google.clientId", "google.clientSecret", "google.apiKey", "google.clientType")
2682
+
2683
+ def __init__(self, store: Store):
2684
+ self.store = store
2685
+ with store.lock, closing(sqlite3.connect(store.path)) as db, db:
2686
+ db.execute("CREATE TABLE IF NOT EXISTS settings (key TEXT PRIMARY KEY, value TEXT NOT NULL)")
2687
+
2688
+ def _read(self, db):
2689
+ rows = dict(db.execute("SELECT key, value FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS))
2690
+ return dict(zip(("clientId", "clientSecret", "apiKey", "clientType"),
2691
+ (rows.get(k) or None for k in self.KEYS)))
2692
+
2693
+ def get(self):
2694
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
2695
+ got = self._read(db)
2696
+ return {**got, "clientType": got["clientType"] or "installed"} if got["clientId"] else None
2697
+
2698
+ def set(self, payload):
2699
+ """Keeps what is sent, or with no client id forgets it all: (kept,
2700
+ error, whether the client changed). A secret not sent is kept while
2701
+ the client is the same one, and forgotten when it is not: a secret
2702
+ belongs to its client."""
2703
+ if not isinstance(payload, dict):
2704
+ return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
2705
+ client = payload.get("clientId")
2706
+ secret, key = payload.get("clientSecret"), payload.get("apiKey")
2707
+ ctype = payload.get("clientType") or "installed"
2708
+ if not isinstance(client, str) or not all(v is None or isinstance(v, str) for v in (secret, key)):
2709
+ return None, (400, "a Google app is { clientId, clientSecret, apiKey, clientType }"), False
2710
+ client, key = client.strip(), (key or "").strip()
2711
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2712
+ had = self._read(db)
2713
+ changed = (had["clientId"] or "") != client
2714
+ if not client:
2715
+ db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
2716
+ return None, None, changed
2717
+ if not GOOGLE_CLIENT.fullmatch(client):
2718
+ return None, (400, "a client ID ends .apps.googleusercontent.com"), False
2719
+ if key and not GOOGLE_KEY.fullmatch(key):
2720
+ return None, (400, "that is not an API key"), False
2721
+ if ctype not in GOOGLE_CLIENT_TYPES:
2722
+ return None, (400, "a client is a Desktop app or a Web application"), False
2723
+ if secret is None:
2724
+ secret = None if changed else had["clientSecret"]
2725
+ secret = (secret or "").strip()
2726
+ db.execute("DELETE FROM settings WHERE key IN (?, ?, ?, ?)", self.KEYS)
2727
+ db.executemany("INSERT INTO settings VALUES (?, ?)",
2728
+ [(k, v) for k, v in zip(self.KEYS, (client, secret, key, ctype)) if v])
2729
+ return self.get(), None, changed
2730
+
2731
+
2603
2732
  class Connections:
2604
2733
  """The lab's grants, held server-side; the page sees their state only."""
2605
2734
 
@@ -2619,7 +2748,8 @@ class Connections:
2619
2748
 
2620
2749
  @staticmethod
2621
2750
  def configured() -> bool:
2622
- return all((GOOGLE_CLIENT_ID, GOOGLE_CLIENT_SECRET, GOOGLE_API_KEY, GOOGLE_APP_ID))
2751
+ app = google_app()
2752
+ return bool(app and app.get("clientId") and app.get("clientSecret"))
2623
2753
 
2624
2754
  def _row(self):
2625
2755
  with self.store.lock, closing(sqlite3.connect(self.store.path)) as db:
@@ -2633,15 +2763,21 @@ class Connections:
2633
2763
 
2634
2764
  def start(self, origin: str):
2635
2765
  """The consent address the page sends the browser to."""
2766
+ app = google_app()
2636
2767
  if not self.configured():
2637
2768
  return None, (409, "Google is not configured for this lab")
2769
+ if app["clientType"] == "installed" and not loopback_origin(origin):
2770
+ return None, (409, "a Desktop app client signs in only on this machine: "
2771
+ "give this lab a Web application client in Configure…")
2638
2772
  state = secrets.token_urlsafe(24)
2639
2773
  now = time.time()
2640
2774
  with self.lock:
2641
2775
  self.states = {s: v for s, v in self.states.items() if v[0] > now}
2642
- self.states[state] = (now + GOOGLE_STATE_SECONDS, origin)
2776
+ # The app is kept with the state: the code Google sends back is
2777
+ # exchanged with the client that asked for it, whatever changes.
2778
+ self.states[state] = (now + GOOGLE_STATE_SECONDS, origin, app)
2643
2779
  return GOOGLE_CONSENT_URL + "?" + urllib.parse.urlencode({
2644
- "client_id": GOOGLE_CLIENT_ID, "redirect_uri": origin + GOOGLE_CALLBACK,
2780
+ "client_id": app["clientId"], "redirect_uri": origin + GOOGLE_CALLBACK,
2645
2781
  "response_type": "code", "scope": GOOGLE_SCOPE, "state": state,
2646
2782
  "access_type": "offline", "prompt": "consent", "include_granted_scopes": "true",
2647
2783
  }), None
@@ -2661,7 +2797,7 @@ class Connections:
2661
2797
  which never include the code or a token."""
2662
2798
  state = (query.get("state") or [""])[0]
2663
2799
  with self.lock:
2664
- lapses, origin = self.states.pop(state, (0, ""))
2800
+ lapses, origin, app = self.states.pop(state, (0, "", None))
2665
2801
  if lapses <= time.time():
2666
2802
  return "that sign-in had lapsed or was not this lab's; sign in again"
2667
2803
  if query.get("error"):
@@ -2671,7 +2807,7 @@ class Connections:
2671
2807
  return "Google sent no code back"
2672
2808
  try:
2673
2809
  got = self._call(GOOGLE_TOKEN_URL, {
2674
- "code": code, "client_id": GOOGLE_CLIENT_ID, "client_secret": GOOGLE_CLIENT_SECRET,
2810
+ "code": code, "client_id": app["clientId"], "client_secret": app["clientSecret"],
2675
2811
  "redirect_uri": origin + GOOGLE_CALLBACK, "grant_type": "authorization_code"})
2676
2812
  except (OSError, ValueError) as e:
2677
2813
  return f"the code could not be exchanged ({type(e).__name__})"
@@ -2689,6 +2825,12 @@ class Connections:
2689
2825
  time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())))
2690
2826
  return None
2691
2827
 
2828
+ def forget(self):
2829
+ """Drops the grant without a word to Google: the client it was made
2830
+ with is no longer this lab's, so the grant is no use to it."""
2831
+ with self.store.lock, closing(sqlite3.connect(self.store.path)) as db, db:
2832
+ db.execute("DELETE FROM connections WHERE id = 'google'")
2833
+
2692
2834
  def sign_out(self):
2693
2835
  """Forgets the grant, and asks Google to revoke it; a revoke that
2694
2836
  fails still forgets it here, which is what signing out means."""
@@ -3589,17 +3731,17 @@ def worker_destinations(run: dict):
3589
3731
  base = api_base(str(conn.get("url") or "")) or api_base(OLLAMA)
3590
3732
  why = allowed(base, "")
3591
3733
  if why:
3592
- return None, f"Setup profile {name}: {why}"
3734
+ return None, f"Target profile {name}: {why}"
3593
3735
  profile = next((p for p in stored if isinstance(p, dict) and p.get("id") == pid), None)
3594
3736
  if profile is None:
3595
- return None, f"Setup profile {name} not found"
3737
+ return None, f"Target profile {name} not found"
3596
3738
  key = str(profile.get("key") or "").strip()
3597
3739
  if key:
3598
3740
  if not header_safe(key):
3599
- return None, f"Setup profile {name} has a key that cannot go in a header"
3741
+ return None, f"Target profile {name} has a key that cannot go in a header"
3600
3742
  why = allowed(base, key)
3601
3743
  if why:
3602
- return None, f"Setup profile {name}: {why}"
3744
+ return None, f"Target profile {name}: {why}"
3603
3745
  env[key_var(pid)] = key
3604
3746
  return env, None
3605
3747
 
@@ -3618,12 +3760,17 @@ def worker_destinations(run: dict):
3618
3760
  # 7: chains are jobs: the field is `jobs` and each job's type is "job".
3619
3761
  # 8: a job is its steps; the content is job 1's Attach Content step.
3620
3762
  # 9: a test is Metrics; a Single Test or a Graded set is read converted.
3621
- PIPELINE_VERSION = 9
3763
+ # 10: a job's steps are its stages, and each scenario is a target whose own
3764
+ # step in each job is what it sends there (docs/pipeline-model.md §16).
3765
+ PIPELINE_VERSION = 10
3622
3766
  # What a stored run may be: the current version, and the ones evals-core.ts's
3623
- # upgradePipeline reads. A new submission is always the current one.
3624
- READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9)
3625
- SCENARIO_CAP = 4
3626
- RUN_FIELDS = ("version", "id", "name", "jobs", "scenarios", "tests", "profiles", "comment", "plugins")
3767
+ # upgradePipeline reads. A new submission is upgraded to the current one.
3768
+ READABLE_VERSIONS = (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
3769
+ TARGET_CAP = 4
3770
+ # Target steps whose words the Prompt library does not record as a use: they
3771
+ # ask no model (evals-core.ts's STEP_TYPES.echo).
3772
+ UNRECORDED_STEPS = {"echo"}
3773
+ RUN_FIELDS = ("version", "id", "name", "jobs", "targets", "tests", "profiles", "comment", "plugins")
3627
3774
 
3628
3775
 
3629
3776
  def content_of(doc):
@@ -3823,16 +3970,20 @@ def upgrade_run(run):
3823
3970
  return up, None
3824
3971
 
3825
3972
 
3826
- def answers_from_item(run, sc, cell):
3827
- """Whether job 1 of a scenario is answered from its item's own text --
3828
- a local type (Echo) over a Source or Text -- so its prompt may be blank:
3829
- evals-core.ts's rule, mirrored. Over Prompt only the prompt is the reply."""
3830
- content = content_of(run)
3831
- if not isinstance(content, dict) or content.get("type") not in ("source", "text"):
3832
- return False
3833
- ref = cell.get("profile") or sc.get("profile")
3834
- conn = (run.get("profiles") or {}).get(ref.get("id")) if isinstance(ref, dict) else None
3835
- return isinstance(conn, dict) and conn.get("type") in LOCAL_CONNECTIONS
3973
+ def cells_of(run):
3974
+ """Every target's step in every job of a run, as (i, target, k, step):
3975
+ version 10's targets and their steps, or an earlier version's scenarios
3976
+ and their cells -- what the Prompt library records a use from, whichever
3977
+ version a stored run is."""
3978
+ if not isinstance(run, dict):
3979
+ return
3980
+ targets = run.get("targets") if isinstance(run.get("targets"), list) else run.get("scenarios")
3981
+ for i, t in enumerate(targets if isinstance(targets, list) else []):
3982
+ if not isinstance(t, dict):
3983
+ continue
3984
+ steps = t.get("steps") if isinstance(t.get("steps"), list) else t.get("stages")
3985
+ for k, step in enumerate(steps if isinstance(steps, list) else []):
3986
+ yield i, t, k, step
3836
3987
 
3837
3988
 
3838
3989
  def run_problems(run):
@@ -3877,36 +4028,41 @@ def run_problems(run):
3877
4028
  bad.append(f"{at} has to be an object")
3878
4029
  continue
3879
4030
  connection_problems(pid, conn, at, bad)
3880
- scenarios = run.get("scenarios")
3881
- if not isinstance(scenarios, list) or not 1 <= len(scenarios) <= SCENARIO_CAP:
3882
- return bad + [f"a run needs between one and {SCENARIO_CAP} scenarios"]
3883
- ids = [sc.get("id") for sc in scenarios if isinstance(sc, dict)]
3884
- for i, sc in enumerate(scenarios):
3885
- at = f"scenario {i + 1}"
3886
- if not isinstance(sc, dict):
4031
+ targets = run.get("targets")
4032
+ if not isinstance(targets, list) or not 1 <= len(targets) <= TARGET_CAP:
4033
+ return bad + [f"a run needs between one and {TARGET_CAP} targets"]
4034
+ ids = [t.get("id") for t in targets if isinstance(t, dict)]
4035
+ for i, t in enumerate(targets):
4036
+ at = f"target {i + 1}"
4037
+ if not isinstance(t, dict):
3887
4038
  bad.append(f"{at} has to be an object")
3888
4039
  continue
3889
4040
  # The id the Prompt library records a use under (version 4).
3890
- if not isinstance(sc.get("id"), str) or not sc["id"].strip():
4041
+ if not isinstance(t.get("id"), str) or not t["id"].strip():
3891
4042
  bad.append(f"{at} has no id")
3892
- elif ids.count(sc["id"]) > 1:
3893
- bad.append(f"{at} has the id of another scenario")
3894
- stages = sc.get("stages")
3895
- if not isinstance(stages, list) or len(stages) != len(jobs):
3896
- bad.append(f"{at} has to have one prompt per job")
4043
+ elif ids.count(t["id"]) > 1:
4044
+ bad.append(f"{at} has the id of another target")
4045
+ steps = t.get("steps")
4046
+ if not isinstance(steps, list) or len(steps) != len(jobs):
4047
+ bad.append(f"{at} has to have one step per job")
3897
4048
  continue
3898
- refs = [sc.get("profile")] + [c.get("profile") for c in stages
3899
- if isinstance(c, dict) and c.get("profile") is not None]
3900
- for k, cell in enumerate(stages):
3901
- if not isinstance(cell, dict) or not isinstance(cell.get("prompt"), str) or (
3902
- not cell["prompt"].strip() and not (k == 0 and answers_from_item(run, sc, cell))):
4049
+ # A target with no profile is one whose steps ask none (Echo) or each
4050
+ # name their own; which steps need one is the core's to judge.
4051
+ refs = ([t["profile"]] if t.get("profile") is not None else []) + [
4052
+ st.get("profile") for st in steps if isinstance(st, dict) and st.get("profile") is not None]
4053
+ # Whether the words may be blank -- Echo answering from the item --
4054
+ # is the core's to judge, and the worker refuses the run in its words.
4055
+ for k, step in enumerate(steps):
4056
+ if not isinstance(step, dict) or not isinstance(step.get("type"), str):
4057
+ bad.append(f"{at}, job {k + 1} has to name what it sends")
4058
+ elif not isinstance(step.get("prompt"), str):
3903
4059
  bad.append(f"{at}, job {k + 1} has no prompt")
3904
- elif cell.get("from") is not None and not (
3905
- isinstance(cell["from"], dict) and isinstance(cell["from"].get("id"), str)):
4060
+ elif step.get("from") is not None and not (
4061
+ isinstance(step["from"], dict) and isinstance(step["from"].get("id"), str)):
3906
4062
  bad.append(f"{at}, job {k + 1} names the prompt it was picked from without an id")
3907
4063
  for ref in refs:
3908
4064
  if not isinstance(ref, dict) or ref.get("id") not in table:
3909
- bad.append(f"{at} names a Setup profile the run does not carry")
4065
+ bad.append(f"{at} names a Target profile the run does not carry")
3910
4066
  content = content_of(run)
3911
4067
  kind = content.get("type") if isinstance(content, dict) else None
3912
4068
  if kind == "source":
@@ -3946,10 +4102,11 @@ if DATA_DIR:
3946
4102
  PLUGINS = Plugins(STORE)
3947
4103
  CONNECTIONS = Connections(STORE)
3948
4104
  MICROSOFT = MicrosoftApp(STORE)
4105
+ GOOGLE = GoogleApp(STORE)
3949
4106
  QUEUE = Queue(STORE)
3950
4107
  QUEUE.prompts = PROMPTS
3951
4108
  else:
3952
- STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = QUEUE = None
4109
+ STORE = SOURCES = PROMPTS = DATASETS = PACKS = PLUGINS = CONNECTIONS = MICROSOFT = GOOGLE = QUEUE = None
3953
4110
 
3954
4111
 
3955
4112
  class NoRedirects(urllib.request.HTTPRedirectHandler):
@@ -4105,7 +4262,7 @@ class Handler(BaseHTTPRequestHandler):
4105
4262
  # And the Microsoft app a flow is read with, when there is one.
4106
4263
  # The app may be entered in Setup unless the deployment names
4107
4264
  # its own, and only in a lab with a store to keep it in.
4108
- return self._json(200, {"scenarioCap": SCENARIO_CAP, "microsoft": microsoft_config(),
4265
+ return self._json(200, {"targetCap": TARGET_CAP, "microsoft": microsoft_config(),
4109
4266
  "microsoftEditable": MICROSOFT is not None and not M365_CLIENT_ID})
4110
4267
  if path == "/api/state":
4111
4268
  if STORE is None:
@@ -4161,6 +4318,10 @@ class Handler(BaseHTTPRequestHandler):
4161
4318
  if CONNECTIONS is None:
4162
4319
  return self._send(404, b"not found", "text/plain")
4163
4320
  return self._json(200, {"connections": CONNECTIONS.list()})
4321
+ if path == "/api/google":
4322
+ if GOOGLE is None:
4323
+ return self._send(404, b"not found", "text/plain")
4324
+ return self._json(200, {"google": google_public(google_app()), "editable": not GOOGLE_CLIENT_ID})
4164
4325
  if path == GOOGLE_CALLBACK:
4165
4326
  return self._google_callback()
4166
4327
  if path.startswith("/plugins/"):
@@ -4381,6 +4542,17 @@ class Handler(BaseHTTPRequestHandler):
4381
4542
  parts = self.path.split("?", 1)[0].split("/")
4382
4543
  if len(parts) == 4 and parts[:3] == ["", "api", "prompts"] and PROMPTS is not None:
4383
4544
  return self._prompts_put(parts[3])
4545
+ if parts == ["", "api", "google"] and GOOGLE is not None:
4546
+ if GOOGLE_CLIENT_ID:
4547
+ return self._json(409, {"error": "this lab's Google app is its deployment's (GOOGLE_CLIENT_ID)"})
4548
+ _, err, changed = GOOGLE.set(self._payload())
4549
+ if err:
4550
+ return self._json(err[0], {"error": err[1]})
4551
+ # A grant is its client's: one made with another client cannot be
4552
+ # refreshed with this one, so it goes rather than failing later.
4553
+ if changed and CONNECTIONS is not None:
4554
+ CONNECTIONS.forget()
4555
+ return self._json(200, {"google": google_public(google_app()), "editable": True})
4384
4556
  if parts == ["", "api", "microsoft"] and MICROSOFT is not None:
4385
4557
  if M365_CLIENT_ID:
4386
4558
  return self._json(409, {"error": "this lab's Microsoft app is its deployment's (M365_CLIENT_ID)"})
@@ -5065,6 +5237,10 @@ def main():
5065
5237
  if not (DEMO / "manifest.json").is_file():
5066
5238
  raise SystemExit(f"no {DEMO / 'manifest.json'} -- the demo pack sits beside server.py, "
5067
5239
  "and the image copies it there")
5240
+ # A published app the build wrote and that cannot be read is a broken
5241
+ # package, not one without the app: said at once, not at Sign in.
5242
+ if PUBLIC_GOOGLE_PROBLEM:
5243
+ raise SystemExit(PUBLIC_GOOGLE_PROBLEM)
5068
5244
  print(f"prompt-lab on {HOST}:{PORT} -> {OLLAMA} by default", flush=True)
5069
5245
  print(f"samples: {SAMPLES}", flush=True)
5070
5246
  print(f"page: {WEB_DIST}", flush=True)