evals-lab 0.4.1 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -275,9 +275,41 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
275
275
 
276
276
 
277
277
 
278
- /** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
279
- older document held (upgradePipeline reads it converted). */
280
-
278
+ /** A private group (docs/pipeline-model.md §17): an eval group's body held
279
+ in the pipeline rather than the Library -- what a version-12 Metrics
280
+ eval upgrades to when it is not a plain link. `casesFrom` is the Library
281
+ group whose cases, at its newest version, join this group's Every item
282
+ under this group's scoring; only the upgrade writes one. */
283
+
284
+
285
+
286
+
287
+
288
+ /** A Library eval group by reference; a run's copy says the version it
289
+ graded with: the body's fingerprint, and its number. */
290
+
291
+
292
+ /** An eval, from version 13: a link to an eval group (§17). `group` names a
293
+ Library group, followed at its newest version (`pin: null`) or pinned at
294
+ one; a private group is `own`, with `group: null`. A run's copy carries
295
+ the version it graded with on the reference (`version`, the body's
296
+ fingerprint, and `n`), and the lab's grader where the group names none. */
297
+
298
+
299
+
300
+
301
+
302
+
303
+
304
+
305
+ /** An eval, from version 13: a link to an eval group. A Metrics eval
306
+ (version 9 to 12), a Single Test or a Graded set is what an older
307
+ document held (upgradePipeline reads it converted). */
308
+
309
+
310
+ /** The rule a pipeline's linked groups pass by, together: every one, or at
311
+ least `count` of them. */
312
+
281
313
 
282
314
  /** A pipeline's evals, in the order they read a run. */
283
315
 
@@ -293,11 +325,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
293
325
 
294
326
 
295
327
 
328
+
296
329
 
297
330
 
298
331
  /** A Setup profile as a run carries it: request settings, never the key. */
299
332
 
300
333
 
334
+
335
+
301
336
 
302
337
 
303
338
 
@@ -338,6 +373,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
338
373
 
339
374
 
340
375
 
376
+
341
377
 
342
378
 
343
379
 
@@ -364,11 +400,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
364
400
 
365
401
 
366
402
 
403
+ /** A Target profile as an import or export matches it: its slug, where it
404
+ has one -- withSlugs gives the rest theirs. */
405
+
406
+
367
407
  /** What an import matches references against: { id, name } each. */
368
408
 
369
-
409
+
370
410
 
371
-
411
+
372
412
 
373
413
 
374
414
 
@@ -735,16 +775,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
735
775
 
736
776
  // ---- the registries ----
737
777
 
738
- /** An option a modifier or an eval type exposes for editing. */
739
-
740
-
741
-
778
+ /** An option a modifier or an eval type exposes for editing. `hint` says
779
+ what it does in one short line, under its control, where the label
780
+ cannot: a switch's effect, not why it exists. */
781
+
782
+
783
+
742
784
 
743
-
744
-
745
-
785
+
786
+
787
+
746
788
 
747
-
789
+
748
790
 
749
791
  /** What an output kind made of a reply. */
750
792
 
@@ -817,8 +859,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
817
859
 
818
860
 
819
861
  /** What validation looks references up in: `profiles(id)` and `sources(id)`
820
- answer with what the id names, or nothing; `datasets` is the datasets the
821
- lab holds, as `GET /api/datasets` lists them. Each is optional, and a
862
+ answer with what the id names, or nothing; `groups` is the eval groups
863
+ the lab holds. Each is optional, and a
822
864
  reference nothing is given to look up is not checked. */
823
865
 
824
866
 
@@ -826,7 +868,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
826
868
 
827
869
 
828
870
 
829
-
871
+
872
+
873
+
874
+
830
875
 
831
876
 
832
877
  /** Lookups, and whether it is a run document being validated. */
@@ -904,6 +949,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
904
949
 
905
950
 
906
951
 
952
+
953
+
954
+
907
955
 
908
956
 
909
957
 
@@ -917,6 +965,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
917
965
 
918
966
 
919
967
 
968
+
969
+
970
+
971
+
972
+
973
+
974
+
920
975
 
921
976
 
922
977
  /** A job's stages, in the order they run (pipeline-model §16). A target's
@@ -2687,7 +2742,10 @@ function applyModifiers (list , kind ,
2687
2742
  // scenario is a target, whose own step in each job is what it sends there.
2688
2743
  // 11: `tests` are `evals`: the key renames and nothing in an eval changes,
2689
2744
  // so a stored result's scores, keyed by eval id, read as they did.
2690
- const PIPELINE_VERSION = 12 ;
2745
+ // 12: a Contains metric's Ignore case holds item by item too.
2746
+ // 13: an eval is a link to an eval group, or a group of the pipeline's own,
2747
+ // and the document has an overall pass rule (pipeline-model §17).
2748
+ const PIPELINE_VERSION = 13 ;
2691
2749
 
2692
2750
  // Plain objects, so an entry is added by assignment and a reader never needs
2693
2751
  // to know which registered it.
@@ -2780,11 +2838,13 @@ OUTPUT_KINDS.text = {
2780
2838
 
2781
2839
  // ---- the connection a profile reference resolves to ------------------------
2782
2840
  // What a run carries of a Setup profile: everything a request needs and never
2783
- // the key. A key follows the profile's id into $EVAL_API_KEY_<ID>, which the
2784
- // server sets from its profiles store and the runner reads, so a name is only
2785
- // a label and two profiles can share one. #46: the profile carries `type`,
2786
- // and the run's resolved copy carries it too (docs/pipeline-model.md §7).
2787
- const CONNECTION_FIELDS = ["name", "url", "model", "type", "temperature",
2841
+ // the key. A key follows the profile's slug into $EVALSLAB_API_KEY_<SLUG> -- or,
2842
+ // in a run document from before slugs, its id into $EVALSLAB_API_KEY_<ID> --
2843
+ // which the server sets from its profiles store and the runner reads, so a
2844
+ // name is only a label and two profiles can share one. #46: the profile
2845
+ // carries `type`, and the run's resolved copy carries it too
2846
+ // (docs/pipeline-model.md §7).
2847
+ const CONNECTION_FIELDS = ["name", "slug", "url", "model", "type", "temperature",
2788
2848
  "px", "format", "quality", "options"];
2789
2849
  const SETTING_KEYS = [...new Set(Object.values(CONNECTION_TYPES).flatMap(t => t.settings.map(s => s.key)))];
2790
2850
  const OPTION_FIELDS = ["seed", "nPredict", ...DECODING_KEYS];
@@ -2794,9 +2854,69 @@ const IMAGE_FORMATS = ["image/jpeg", "image/webp", "image/png
2794
2854
  const PROFILE_ID = /^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?$/;
2795
2855
  const LOOKS_LIKE_A_KEY = /^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$/i;
2796
2856
 
2797
- /** The variable a profile's key is read from: pm1x8k2q is EVAL_API_KEY_PM1X8K2Q. */
2798
- const keyVar = (id ) => "EVAL_API_KEY_"
2799
- + String(id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
2857
+ /** The variable a profile's key is read from: its slug's, when it has one --
2858
+ anthropic is EVALSLAB_API_KEY_ANTHROPIC -- else its id's, as a run document
2859
+ from before slugs has it: pm1x8k2q is EVALSLAB_API_KEY_PM1X8K2Q. */
2860
+ const keyVar = (id , slug ) => "EVALSLAB_API_KEY_"
2861
+ + String(slug || id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
2862
+
2863
+ // ---- a profile's slug (#253) ------------------------------------------------
2864
+ // What a profile is called outside the lab: in a pipeline file and in the
2865
+ // name of the variable CI keeps its key in. An id (pm1x9c0r) means nothing to
2866
+ // whoever sets that secret; a slug (anthropic) does, and stays put when the
2867
+ // lab it came from does not. Lowercase letters and digits, hyphens between,
2868
+ // so its variable is spelt one way and two slugs never spell the same one.
2869
+ const SLUG = /^[a-z0-9]+(?:-[a-z0-9]+)*$/;
2870
+ const SLUG_MAX = 32;
2871
+
2872
+ /** [text] as a slug: lowercase, every other run of characters one hyphen,
2873
+ cut to SLUG_MAX. "" when nothing of it is a letter or a digit. */
2874
+ function slugOf(text ) {
2875
+ return String(text ?? "").toLowerCase().normalize("NFKD").replace(/[\u0300-\u036f]/g, "")
2876
+ .replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, SLUG_MAX).replace(/-+$/, "");
2877
+ }
2878
+
2879
+ /** [base] made unique among [taken]: anthropic, then anthropic-2, -3. */
2880
+ function uniqueSlug(base , taken ) {
2881
+ const root = base || "profile";
2882
+ if (!taken.has(root)) return root;
2883
+ for (let n = 2; ; n++) {
2884
+ const tail = `-${n}`;
2885
+ const s = root.slice(0, SLUG_MAX - tail.length).replace(/-+$/, "") + tail;
2886
+ if (!taken.has(s)) return s;
2887
+ }
2888
+ }
2889
+
2890
+ /**
2891
+ * [list] with every profile holding a slug of its own: one it holds already
2892
+ * is kept, tidied into a slug (Setup's field holds what was typed, a
2893
+ * trailing hyphen and all), while no profile before it holds the same; any
2894
+ * other is made from its name, made unique. The slugs held are settled
2895
+ * first, so a slug minted for one profile never takes another's. The same
2896
+ * array when nothing changed, so a reader can tell a store that needs
2897
+ * writing back from one that does not.
2898
+ */
2899
+ function withSlugs (list ) {
2900
+ const taken = new Set ();
2901
+ const held = list.map(p => {
2902
+ const s = slugOf(p?.slug);
2903
+ if (!s || taken.has(s)) return "";
2904
+ taken.add(s);
2905
+ return s;
2906
+ });
2907
+ let changed = false;
2908
+ const out = list.map((p, i) => {
2909
+ const slug = held[i] || uniqueSlug(slugOf(p?.name), taken);
2910
+ taken.add(slug);
2911
+ if (slug === p?.slug) return p;
2912
+ changed = true;
2913
+ return { ...p, slug };
2914
+ });
2915
+ return changed ? out : list ;
2916
+ }
2917
+
2918
+ /** Whether [s] is a slug as a profile may hold one. */
2919
+ const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
2800
2920
 
2801
2921
  /** A stored profile as a run carries it: its request settings, no key. */
2802
2922
  function connectionOf(p ) {
@@ -2807,6 +2927,7 @@ function connectionOf(p ) {
2807
2927
  // the options bag holds the type's own settings only.
2808
2928
  for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
2809
2929
  const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
2930
+ if (isSlug(p.slug)) out.slug = p.slug;
2810
2931
  for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
2811
2932
  if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
2812
2933
  if (Object.keys(options).length) out.options = options;
@@ -2863,7 +2984,7 @@ function onlyFields(obj , at , allowed , bad
2863
2984
  for (const k of Object.keys(obj)) {
2864
2985
  if (allowed.includes(k)) continue;
2865
2986
  bad.push(LOOKS_LIKE_A_KEY.test(k)
2866
- ? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVAL_API_KEY_<ID>"} supplies it`
2987
+ ? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVALSLAB_API_KEY_<ID>"} supplies it`
2867
2988
  : `${at} has "${k}", which is not a pipeline field`);
2868
2989
  }
2869
2990
  }
@@ -3026,7 +3147,7 @@ LEGACY_TESTS.graded = {
3026
3147
  if (!isRef(d) || (d.version != null && !isStr(d.version))) {
3027
3148
  return void bad.push("a graded eval has to name its dataset as { id, name }");
3028
3149
  }
3029
- if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
3150
+ if (ctx.groups && !ctx.groups.some(x => x.id === d.id)) bad.push(`Eval group ${d.name || d.id} not found`);
3030
3151
  },
3031
3152
  };
3032
3153
 
@@ -3169,7 +3290,10 @@ function readRun(list , run ) {
3169
3290
  });
3170
3291
  }
3171
3292
 
3172
- EVAL_TYPES.metrics = {
3293
+ // The eval of versions 9 to 12, read only to upgrade (LEGACY_TESTS.metrics.
3294
+ // toGroup) and kept whole as the reference groups-check.js and
3295
+ // pipeline-check.js hold the eval group to.
3296
+ LEGACY_TESTS.metrics = {
3173
3297
  label: "Metrics",
3174
3298
  description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3175
3299
  fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
@@ -3187,7 +3311,7 @@ EVAL_TYPES.metrics = {
3187
3311
  // The dataset whose cases a case metric reads, and whose own metrics join.
3188
3312
  if (t.dataset != null) {
3189
3313
  if (!isRef(t.dataset) || (t.dataset.version != null && !isStr(t.dataset.version))) bad.push("the Metrics name their dataset as { id, name }");
3190
- else if (ctx.datasets && !ctx.datasets.some(x => x.id === t.dataset.id)) bad.push(`Dataset ${t.dataset.name || t.dataset.id} not found`);
3314
+ else if (ctx.groups && !ctx.groups.some(x => x.id === t.dataset.id)) bad.push(`Eval group ${t.dataset.name || t.dataset.id} not found`);
3191
3315
  }
3192
3316
  if (!SCORING_MODES.includes(t.mode)) bad.push(`the Metrics are scored ${SCORING_MODES.join(" or ")}`);
3193
3317
  if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
@@ -3228,6 +3352,162 @@ EVAL_TYPES.metrics = {
3228
3352
  metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3229
3353
  };
3230
3354
 
3355
+ /** A version-12 Metrics eval as the version-13 eval it reads as (§17): a
3356
+ link to its dataset, followed at the newest version, where it named one
3357
+ and held nothing of its own -- no metrics, scored All, the lab's grader,
3358
+ over each item -- and a private group holding its metrics, scoring and
3359
+ grader otherwise, with its dataset's cases (`casesFrom`). An eval over the
3360
+ whole run never read its dataset's cases, so its group takes none. The
3361
+ id, name and Continue on failure are kept, so a stored score keyed by
3362
+ the eval's id is the link's. */
3363
+ function groupOfMetrics(t ) {
3364
+ const { id, name, continueOnFailure } = t;
3365
+ const common = { id, ...(name !== undefined ? { name } : {}), type: "group", continueOnFailure };
3366
+ const metrics = Array.isArray(t.metrics) ? t.metrics : [];
3367
+ const run = t.over === "run";
3368
+ if (isRef(t.dataset) && !metrics.length && t.mode === "all" && t.grader == null && !run) {
3369
+ return { ...common, group: { ...t.dataset }, pin: null };
3370
+ }
3371
+ const own = { version: DATASET_BODY_VERSION, source: null, scoring: { mode: t.mode, threshold: t.threshold ?? null },
3372
+ grader: t.grader ?? null, every: run ? [] : metrics, run: run ? metrics : [], cases: [],
3373
+ casesFrom: isRef(t.dataset) && !run ? { ...t.dataset } : null };
3374
+ return { ...common, group: null, pin: null, own };
3375
+ }
3376
+ LEGACY_TESTS.metrics.toGroup = groupOfMetrics;
3377
+
3378
+ /** The group an eval reads by: a private group's own body, or a link's
3379
+ Library group as the runner hands it ([more].group). A link whose body
3380
+ nothing hands over reads as a Library group upgraded from version 6 does
3381
+ -- no metrics of its own, scored All -- which is what every one was
3382
+ until a body could hold more. */
3383
+ function groupOf(t , more = {}) {
3384
+ if (isObj(t.own)) return t.own ;
3385
+ const body = isRef(t.group) ? more.group?.(t.group) : undefined;
3386
+ return body ?? { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
3387
+ every: [], run: [], cases: [] };
3388
+ }
3389
+
3390
+ /** The Library group whose cases an eval reads, where it reads one: a
3391
+ link's group, or a private group's `casesFrom`. */
3392
+ function casesRef(t ) {
3393
+ if (!isObj(t) || t.type !== "group") return null;
3394
+ if (isRef(t.group)) return t.group ;
3395
+ return isObj(t.own) && isRef(t.own.casesFrom) ? t.own.casesFrom : null;
3396
+ }
3397
+
3398
+ /** The metrics a private group shows as its rules: its Whole run where it
3399
+ reads the run, its Every item otherwise. */
3400
+ const ownList = (t ) => (isObj(t.own) ? (wholeRunGroup(t) ? t.own.run : t.own.every) ?? [] : []);
3401
+
3402
+ /** A private group read over the whole run: Whole run metrics, and nothing
3403
+ read item by item. A group holding both is read item by item until the
3404
+ runner reports a group's Whole run beside its items (#233). */
3405
+ const wholeRunGroup = (t ) => isObj(t.own) && Array.isArray(t.own.run) && t.own.run.length > 0
3406
+ && !(Array.isArray(t.own.every) && t.own.every.length) && !(Array.isArray(t.own.cases) && t.own.cases.length) && !t.own.casesFrom;
3407
+
3408
+ // A link to an eval group (docs/pipeline-model.md §17): a Library group by
3409
+ // reference, or a private group held here (`own`). It reads each item as its
3410
+ // group does (readGroup), or -- a private group of Whole run metrics alone --
3411
+ // the whole run (readGroupRun).
3412
+ EVAL_TYPES.group = {
3413
+ label: "Eval group",
3414
+ description: "Checks of each reply, or of the whole run, from an eval group: linked from the Library, or this pipeline's own.",
3415
+ fields: ["type", "group", "pin", "own", "grader"],
3416
+ // A metric that reads terms (a case) reads only a kind that yields them.
3417
+ accepts: t => ([...ownList(t)].some((m ) => METRICS[m?.type]?.needsTerms)
3418
+ ? Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms) : null),
3419
+ defaults: () => ({ type: "group", group: null, pin: null,
3420
+ own: { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
3421
+ every: [], run: [], cases: [], casesFrom: null } }),
3422
+ validate(t, ctx, bad){
3423
+ const g = t.group, own = t.own;
3424
+ const named = (ref , what ) => {
3425
+ const n = isObj(ref) ? ref.n : undefined;
3426
+ if (!isRef(ref) || (ref.version != null && !isStr(ref.version)) || (n != null && !Number.isInteger(n))) {
3427
+ return void bad.push(`${what} names its eval group as { id, name }`);
3428
+ }
3429
+ onlyFields(ref, what, ["id", "name", "version", "n"], bad);
3430
+ const held = ctx.groups?.find(x => x.id === ref.id);
3431
+ if (ctx.groups && !held) bad.push(`Eval group ${ref.name || ref.id} not found`);
3432
+ return held;
3433
+ };
3434
+ if (t.pin != null && !(Number.isInteger(t.pin) && t.pin > 0)) bad.push("a pin is a group's version: a whole number from 1");
3435
+ if (t.grader != null && !isRef(t.grader)) bad.push("the link names its grader as { id, name }");
3436
+ if (g != null) {
3437
+ if (own != null) return void bad.push("an eval links a Library group or holds its own, not both");
3438
+ const held = named(g, "the link");
3439
+ if (held && t.pin != null && typeof held.versions === "number" && t.pin > held.versions) {
3440
+ bad.push(`${g.name || g.id} has no version ${t.pin}`);
3441
+ }
3442
+ return;
3443
+ }
3444
+ if (!isObj(own)) return void bad.push("an eval links a Library group, or holds its own");
3445
+ if (t.pin != null) bad.push("a group of this pipeline's own versions with it, so it is never pinned");
3446
+ if (own.version !== DATASET_BODY_VERSION) bad.push(`the group's own body is version ${DATASET_BODY_VERSION}`);
3447
+ for (const key of ["scoring", "every", "run", "cases"]) if (own[key] == null) bad.push(`the group's own body has no ${key}`);
3448
+ onlyFields(own, "the group's own body", ["version", "source", "scoring", "grader", "every", "run", "cases", "casesFrom"], bad);
3449
+ if (bad.length) return;
3450
+ bad.push(...validateEvals(own));
3451
+ if (own.casesFrom != null) named(own.casesFrom, "Cases from");
3452
+ // Read item by item or over the run: one section at a time, until a
3453
+ // group's Whole run is reported beside its items (#233).
3454
+ if (own.run.length && !wholeRunGroup(t)) bad.push("the group's own body reads each item or the whole run, not both");
3455
+ // The group's own grader, or the lab's.
3456
+ const grader = isRef(own.grader) ? own.grader : isRef(ctx.grader) ? ctx.grader : null;
3457
+ if (gradedIn(own.every) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
3458
+ // A grader is asked words, and needs a model to ask.
3459
+ const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3460
+ if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
3461
+ else if (p) {
3462
+ const type = CONNECTION_TYPES[typeOf(p )];
3463
+ if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
3464
+ else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
3465
+ }
3466
+ },
3467
+ rules: t => ownList(t).map((m , x ) => ({
3468
+ key: `m${x}`, label: (isStr(m.metric) && m.metric) || METRICS[m.type]?.label || m.type,
3469
+ want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3470
+ .filter(Boolean).join(" ") })),
3471
+ profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
3472
+ // A run carries the lab's grader where the group names none and may ask
3473
+ // one: a model-graded metric of its own, or a case's. A link's group is
3474
+ // the Library's, so the run's copy of the link carries it.
3475
+ resolve(t, ctx){
3476
+ if (!isRef(ctx.grader)) return t;
3477
+ const grader = { id: ctx.grader.id, name: ctx.grader.name };
3478
+ if (isObj(t.own)) {
3479
+ return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
3480
+ }
3481
+ return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
3482
+ },
3483
+ wholeRun: wholeRunGroup,
3484
+ verdict: (t, ress, kind) => {
3485
+ const own = t.own ;
3486
+ return wholeRunVerdict(own.run, ress, kind, own.scoring.mode, own.scoring.threshold);
3487
+ },
3488
+ // Rule m{x} is the group's own metric x, read in that place on every item.
3489
+ ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3490
+ : ownList(t).length === 1 ? score.pass : null),
3491
+ // Every item, with its case where the group reads the Library group's
3492
+ // cases and the item has one there.
3493
+ read: async (t, kase, res, more) => {
3494
+ const group = groupOf(t, more);
3495
+ const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
3496
+ const ref = casesRef(t);
3497
+ // The item's case in this eval's own group: resolved from the group whose
3498
+ // cases it reads where the runner names the item (grading several groups
3499
+ // at once, --groups), else the case handed in (a single group).
3500
+ const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
3501
+ return readGroup({ ...group, grader } , ref ? own : null, res, more);
3502
+ },
3503
+ };
3504
+
3505
+ /** The case for an item in a group's body, by the name its file has, or null:
3506
+ a non-todo case whose item matches, as a Library group reads one. */
3507
+ function caseIn(body , item ) {
3508
+ return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
3509
+ }
3510
+
3231
3511
  // ---- the lab's own scorers, as metrics ----------------------------------------
3232
3512
  // What the Single Test checks, one metric each, so an eval of it converts to
3233
3513
  // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
@@ -3638,9 +3918,15 @@ function targetsOf(doc
3638
3918
  return Array.isArray(doc?.targets) ? doc .targets : [];
3639
3919
  }
3640
3920
 
3641
- /** Target [i]'s name, or the number it has always shown. */
3921
+ /** Target [i]'s letter, A for the first: the same in every job, so a
3922
+ Target reads as one column through them. A number past Z. */
3923
+ function targetLetter(i ) {
3924
+ return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
3925
+ }
3926
+
3927
+ /** Target [i]'s name, or its letter's: "Target A". */
3642
3928
  function targetLabel(doc , i ) {
3643
- return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
3929
+ return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
3644
3930
  }
3645
3931
 
3646
3932
  /** What target [i] sends in job [k]: its step there. */
@@ -3849,11 +4135,16 @@ function evalsOf(doc )
3849
4135
  return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3850
4136
  }
3851
4137
 
3852
- /** The dataset a document's evals grade against, where one does: a run
3853
- grades against one (validatePipeline says so), so the first names it. */
3854
- function evalsDataset(doc ) {
3855
- const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3856
- return t && "dataset" in t ? t.dataset : null;
4138
+ /** The Library group whose cases a document's evals read, where one does:
4139
+ a link's group, or a private group's Cases from -- or, in a stored run
4140
+ from before version 13, a Metrics eval's dataset. A run grades against
4141
+ one (validatePipeline says so), so the first names it. */
4142
+ function evalsDataset(doc ) {
4143
+ for (const t of evalsOf(doc) ) {
4144
+ const ref = casesRef(t) ?? (isObj(t) && isRef(t.dataset) ? t.dataset : null);
4145
+ if (ref) return ref;
4146
+ }
4147
+ return null;
3857
4148
  }
3858
4149
 
3859
4150
  STEP_TYPES.evals = {
@@ -3882,15 +4173,14 @@ STEP_TYPES.evals = {
3882
4173
  + `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
3883
4174
  }
3884
4175
  });
3885
- // A run is handed one dataset's body to grade against (server-side-runs §4).
3886
- const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3887
- if (named.size > 1) bad.push("the evals grade against one dataset at a time");
4176
+ // The worker reads a body per group now (#233), so the evals may grade
4177
+ // against several Library groups at once; the queue keeps each body.
3888
4178
  },
3889
4179
  };
3890
4180
 
3891
4181
  // ---- the document -------------------------------------------------------------
3892
4182
 
3893
- const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
4183
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals", "pass"];
3894
4184
  // What resolving adds, and nothing else: the profiles it resolved to and the
3895
4185
  // run's own comment, which belongs to the run and never to the pipeline.
3896
4186
  const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
@@ -3999,6 +4289,12 @@ function newEval(type , name = "", fields = {})
3999
4289
  /** Version 6 to 7: chains are jobs. The field renames in place -- so an
4000
4290
  upgraded document reads the same, key for key -- and every job's `type`
4001
4291
  becomes `"job"`, so an older document reads as one of today's everywhere. */
4292
+ /** A new eval linking the Library eval group [group], followed at its
4293
+ newest version. */
4294
+ function newLink(group , name = "") {
4295
+ return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
4296
+ }
4297
+
4002
4298
  function jobsOf(next ) {
4003
4299
  next.version = PIPELINE_VERSION;
4004
4300
  if (Array.isArray(next.chains)) {
@@ -4113,6 +4409,26 @@ function caseAsWritten(next ) {
4113
4409
  return next;
4114
4410
  }
4115
4411
 
4412
+ /** Version 12 to 13: each Metrics eval is a link to an eval group or a
4413
+ group of the pipeline's own (groupOfMetrics), and the document passes
4414
+ when every linked group does. Pure: it needs nothing but the document,
4415
+ so every reader reads the same result. Every version before 13 ends
4416
+ here. */
4417
+ function groupsOf(next ) {
4418
+ next.version = PIPELINE_VERSION;
4419
+ if (Array.isArray(next.evals)) {
4420
+ next.evals = next.evals.map((t ) => (isObj(t) && t.type === "metrics" ? groupOfMetrics(t) : t));
4421
+ }
4422
+ if (!("pass" in next)) {
4423
+ // After the evals, where a reader of the document looks for it.
4424
+ const at = Object.keys(next).indexOf("evals");
4425
+ const entries = Object.entries(next);
4426
+ entries.splice(at < 0 ? entries.length : at + 1, 0, ["pass", { mode: "all" }]);
4427
+ return Object.fromEntries(entries);
4428
+ }
4429
+ return next;
4430
+ }
4431
+
4116
4432
  function upgradePipeline (doc , ctx = {}) {
4117
4433
  // A current document is read as it is, but for a profile reference the Runs
4118
4434
  // tab saved whole (see profileRefs), which is cut back, and a step on a
@@ -4123,24 +4439,26 @@ function upgradePipeline (doc , ctx = {}) {
4123
4439
  out = clone(out) ;
4124
4440
  profileRefs(out);
4125
4441
  }
4126
- return localSteps(withoutCaseMetric(out), ctx) ;
4442
+ return localSteps(out, ctx) ;
4443
+ }
4444
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
4445
+ if (doc.version === 11 || doc.version === 12) {
4446
+ return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
4127
4447
  }
4128
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
4129
- if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
4130
- // Every version before 10 reads as version 9 first, then as 10, then as 11
4131
- // and 12.
4448
+ // Every version before 10 reads as version 9 first, then as 10, then as 11,
4449
+ // 12 and 13.
4132
4450
  // A version-10 document is cut as a current one was (profileRefs); the
4133
4451
  // earlier ones are cut on their way through nineOf.
4134
4452
  const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4135
- return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
4453
+ return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
4136
4454
  }
4137
4455
 
4138
4456
  /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
4139
4457
  are metrics now (dataset body version 5), which an eval naming the
4140
4458
  dataset adds for each item, so the case's own metrics carry what that
4141
- metric scored. Read so at every version, the current one included, as a
4142
- document saved before it went still holds it. The same document where
4143
- none does. */
4459
+ metric scored. Read so at every version up to 12, as a document saved
4460
+ before it went still holds it; version 13 holds no Metrics eval. The
4461
+ same document where none does. */
4144
4462
  function withoutCaseMetric(doc ) {
4145
4463
  const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
4146
4464
  if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
@@ -4229,7 +4547,7 @@ function tokenMappingsFromV1(set ) {
4229
4547
  function blankPipeline(opts = {}) {
4230
4548
  const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
4231
4549
  return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
4232
- jobs: [job], targets: [], evals: [] };
4550
+ jobs: [job], targets: [], evals: [], pass: { mode: "all" } };
4233
4551
  }
4234
4552
 
4235
4553
  /**
@@ -4368,6 +4686,7 @@ function validatePipeline(input , ctx = {}) {
4368
4686
  }
4369
4687
  });
4370
4688
  STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4689
+ passProblems(doc.pass, Array.isArray(doc.evals) ? doc.evals.length : 0, bad);
4371
4690
  if (bad.length) return bad;
4372
4691
 
4373
4692
  // Asked before anything is sent, so a misspelt token costs nothing and
@@ -4382,16 +4701,45 @@ function validatePipeline(input , ctx = {}) {
4382
4701
  return bad;
4383
4702
  }
4384
4703
 
4704
+ /** The overall pass rule: every linked group, or at least [count] of the
4705
+ [n] the pipeline has. */
4706
+ function passProblems(pass , n , bad ) {
4707
+ if (!isObj(pass) || (pass.mode !== "all" && pass.mode !== "atLeast")) {
4708
+ return void bad.push("pass has to be { mode: all } or { mode: atLeast, count }");
4709
+ }
4710
+ onlyFields(pass, "pass", pass.mode === "all" ? ["mode"] : ["mode", "count"], bad);
4711
+ if (pass.mode === "atLeast" && !(Number.isInteger(pass.count) && pass.count > 0)) bad.push("pass at least has to count a whole number of groups from 1");
4712
+ else if (pass.mode === "atLeast" && pass.count > n) bad.push(`pass asks for ${pass.count} groups, and the pipeline has ${n}`);
4713
+ }
4714
+
4715
+ /**
4716
+ * What [doc] may still be right about, as sentences: a link to a group made
4717
+ * for another Source than the pipeline's content (§17, decision 2). The run
4718
+ * grades the cases whose items it holds, and the rest read Missing.
4719
+ */
4720
+ function pipelineWarnings(doc , ctx = {}) {
4721
+ if (!isObj(doc) || !Array.isArray(doc.evals) || !ctx.groups) return [];
4722
+ const content = contentOf(doc ) ;
4723
+ const here = content?.type === "source" && isRef(content.ref) ? content.ref.id : null;
4724
+ const warn = [];
4725
+ doc.evals.forEach((t , j ) => {
4726
+ const ref = casesRef(t);
4727
+ const src = ref && ctx.groups .find(x => x.id === ref.id)?.source;
4728
+ if (src && src.id !== here) warn.push(`${evalLabel(doc , j)}: ${ref .name || ref .id} grades ${src.name || src.id}`);
4729
+ });
4730
+ return warn;
4731
+ }
4732
+
4385
4733
  /** A run document's profiles table: id → connection, keyless, spellable. */
4386
4734
  function profilesProblems(profiles , bad ) {
4387
4735
  if (!isObj(profiles)) return void bad.push("profiles has to be an object of id → connection");
4388
4736
  const spelt = new Map ();
4389
4737
  for (const [id, conn] of Object.entries(profiles)) {
4390
4738
  if (!PROFILE_ID.test(id)) {
4391
- bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVAL_API_KEY_<ID>`);
4739
+ bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVALSLAB_API_KEY_<ID>`);
4392
4740
  continue;
4393
4741
  }
4394
- const from = keyVar(id);
4742
+ const from = keyVar(id, isObj(conn) && isStr(conn.slug) ? conn.slug : null);
4395
4743
  if (spelt.has(from)) bad.push(`profiles ${spelt.get(from)} and ${id} both take their key from $${from}`);
4396
4744
  spelt.set(from, id);
4397
4745
  const at = `profile ${id}`;
@@ -4408,6 +4756,9 @@ function connectionProblems(conn , at , from
4408
4756
  bad.push(`${at}: type has to be one of ${Object.keys(CONNECTION_TYPES).join(", ")}`);
4409
4757
  }
4410
4758
  if (!isStr(conn.name)) bad.push(`${at}: name has to be text`);
4759
+ if (conn.slug != null && !isSlug(conn.slug)) {
4760
+ bad.push(`${at}: slug has to be lowercase letters and digits, with hyphens between, at most ${SLUG_MAX}`);
4761
+ }
4411
4762
  const url = conn.url ?? "";
4412
4763
  if (!isStr(url)) bad.push(`${at}: url has to be text, blank for the lab's Ollama`);
4413
4764
  else if (url.trim()) {
@@ -4477,7 +4828,8 @@ function profileIds(doc )
4477
4828
  * the Source's file list as it reads now, `ctx.comment` is the run's own.
4478
4829
  */
4479
4830
  function resolvePipeline(doc ,
4480
- ctx = {}) {
4831
+ ctx
4832
+ = {}) {
4481
4833
  const run = clone(doc) ;
4482
4834
  // What the lab supplies an eval -- its grader -- before the profiles it
4483
4835
  // asks are carried.
@@ -4489,8 +4841,18 @@ function resolvePipeline(doc ,
4489
4841
  }
4490
4842
  const content = contentOf(run) ;
4491
4843
  if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4492
- if (ctx.datasetVersion) {
4493
- for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4844
+ // Each Library group the evals read, as the version it grades with: the
4845
+ // body's fingerprint and, where the lab numbers them, `n` -- a pinned
4846
+ // link's pin.
4847
+ if (ctx.groupVersion) {
4848
+ for (const t of run.evals || []) {
4849
+ const ref = casesRef(t);
4850
+ const at = ref && ctx.groupVersion(ref.id);
4851
+ if (!ref || !at) continue;
4852
+ ref.version = at.version;
4853
+ const n = isObj(t) && Number.isInteger(t.pin) ? t.pin : at.n;
4854
+ if (n != null) ref.n = n;
4855
+ }
4494
4856
  }
4495
4857
  if (isStr(ctx.comment)) run.comment = ctx.comment;
4496
4858
  return run;
@@ -4504,7 +4866,12 @@ function pipelineOfRun(run , ctx = {}) {
4504
4866
  delete doc.plugins;
4505
4867
  const content = contentOf(doc) ;
4506
4868
  if (content) { delete content.files; delete content.revs; }
4507
- for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
4869
+ // What the run recorded of each group it read is the run's. The grader
4870
+ // it carried stays, as a Metrics eval's did: a bundle writes it down.
4871
+ for (const t of doc.evals || []) {
4872
+ const ref = casesRef(t);
4873
+ if (ref) { delete ref.version; delete ref.n; }
4874
+ }
4508
4875
  return doc;
4509
4876
  }
4510
4877
 
@@ -4517,9 +4884,35 @@ function pipelineOfRun(run , ctx = {}) {
4517
4884
 
4518
4885
 
4519
4886
  /** [doc] as YAML text, the pipeline's file form -- references only, so an
4520
- export carries no key, no URL, no file list and no file bytes. */
4521
- function pipelineToYaml(doc ) {
4522
- return yaml.dump(doc, { lineWidth: -1, noRefs: true });
4887
+ export carries no key, no URL, no file list and no file bytes. A Target
4888
+ profile [ctx] holds is named by its slug, `{ slug, name }`, rather than
4889
+ by this lab's id (#253): the slug is what another lab, or CI's
4890
+ $EVALSLAB_API_KEY_<SLUG>, knows it by. */
4891
+ function pipelineToYaml(doc , ctx = {}) {
4892
+ const profiles = withSlugs(ctx.profiles ?? []);
4893
+ if (!profiles.length) return yaml.dump(doc, { lineWidth: -1, noRefs: true });
4894
+ const out = clone(doc) ;
4895
+ const bySlug = (ref ) => {
4896
+ const p = isRef(ref) ? profiles.find(x => x.id === ref.id) : null;
4897
+ return p ? { slug: p.slug, name: ref.name ?? p.name } : ref;
4898
+ };
4899
+ eachProfileRef(out, bySlug);
4900
+ return yaml.dump(out, { lineWidth: -1, noRefs: true });
4901
+ }
4902
+
4903
+ /** Every Target profile reference in [doc] -- each target's, each of its
4904
+ steps', each eval's grader -- replaced, in place, by [f] of it. */
4905
+ function eachProfileRef(doc , f ) {
4906
+ for (const t of targetsOf(doc)) {
4907
+ if (t.profile) t.profile = f(t.profile);
4908
+ for (const st of Array.isArray(t.steps) ? t.steps : []) {
4909
+ if (st?.profile) st.profile = f(st.profile);
4910
+ }
4911
+ }
4912
+ for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4913
+ if (isObj(t) && t.grader) t.grader = f(t.grader);
4914
+ if (isObj(t?.own) && t.own.grader) t.own.grader = f(t.own.grader);
4915
+ }
4523
4916
  }
4524
4917
 
4525
4918
  /**
@@ -4527,7 +4920,7 @@ function pipelineToYaml(doc ) {
4527
4920
  * name (pipeline-model §5), so a pipeline written in another lab -- ids that
4528
4921
  * mean nothing here -- finds its Source and Setup profiles by the names they
4529
4922
  * were exported under. [ctx]'s lists are what the lab holds:
4530
- * `{ profiles, sources, datasets }`, each `{ id, name }[]`. Returns
4923
+ * `{ profiles, sources, groups }`, each `{ id, name }[]`. Returns
4531
4924
  * `{ doc, missing }`, where `missing` is one sentence per reference neither
4532
4925
  * id nor name matched, or `{ error }` when the document is refused: not a
4533
4926
  * version the lab reads, or not something the lab can edit and run as it
@@ -4538,14 +4931,14 @@ function importPipeline(input , ctx = {}) {
4538
4931
  const doc = upgradePipeline(input, ctx);
4539
4932
  const v = versionProblem(doc);
4540
4933
  if (v) return { error: v };
4541
- const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], datasets = ctx.datasets ?? [];
4934
+ const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], groups = ctx.groups ?? [];
4542
4935
  const next = clone(doc) ;
4543
4936
  const missing = [], seen = new Set ();
4544
- const lists = { profile: profiles, source: sources, dataset: datasets };
4937
+ const lists = { profile: profiles, source: sources, group: groups };
4545
4938
  const words = {
4546
- profile: (ref ) => `Target profile ${ref.name || ref.id} not found`,
4939
+ profile: (ref ) => `Target profile ${ref.slug || ref.name || ref.id} not found`,
4547
4940
  source: (ref ) => `Source ${ref.name || ref.id} not found`,
4548
- dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
4941
+ group: (ref ) => `Eval group ${ref.name || ref.id} not found`,
4549
4942
  };
4550
4943
  const remap = (ref , kind ) => {
4551
4944
  const hit = lists[kind].find(x => x.id === ref.id) || lists[kind].find(x => x.name === ref.name);
@@ -4554,16 +4947,34 @@ function importPipeline(input , ctx = {}) {
4554
4947
  if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
4555
4948
  return ref;
4556
4949
  };
4950
+ // A profile is named by its slug in a file this lab writes (#253), and by
4951
+ // its id in one written before slugs: a slug is matched by slug, then by
4952
+ // name. One this lab holds under neither is kept as a reference the run
4953
+ // is blocked on, the slug standing in for an id.
4954
+ const slugged = withSlugs(profiles);
4955
+ const remapProfile = (ref ) => {
4956
+ if (!isObj(ref) || !isStr(ref.slug) || isRef(ref)) return isRef(ref) ? remap(ref, "profile") : ref;
4957
+ const hit = slugged.find(x => x.slug === ref.slug) || slugged.find(x => isStr(ref.name) && x.name === ref.name);
4958
+ if (hit) return { id: hit.id, name: hit.name };
4959
+ const sentence = words.profile(ref );
4960
+ if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
4961
+ return { id: ref.slug, name: isStr(ref.name) && ref.name ? ref.name : ref.slug };
4962
+ };
4557
4963
  const content = contentOf(next) ;
4558
4964
  if (content?.type === "source") content.ref = remap(content.ref, "source");
4559
4965
  for (const t of targetsOf(next)) {
4560
- if (t.profile) t.profile = remap(t.profile, "profile");
4966
+ if (t.profile) t.profile = remapProfile(t.profile);
4561
4967
  for (const st of Array.isArray(t.steps) ? t.steps : []) {
4562
- if (st?.profile) st.profile = remap(st.profile, "profile");
4968
+ if (st?.profile) st.profile = remapProfile(st.profile);
4563
4969
  }
4564
4970
  }
4565
4971
  for (const t of Array.isArray(next.evals) ? next.evals : []) {
4566
- if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4972
+ if (isRef(t?.group)) t.group = remap(t.group, "group");
4973
+ else if (isObj(t?.own) && isRef(t.own.casesFrom)) t.own.casesFrom = remap(t.own.casesFrom, "group");
4974
+ // The grader is a profile like any other, matched the same way: a
4975
+ // private group's own, or the one a run's link carries.
4976
+ if (isObj(t?.own) && t.own.grader) t.own.grader = remapProfile(t.own.grader);
4977
+ if (isObj(t) && t.grader) t.grader = remapProfile(t.grader);
4567
4978
  }
4568
4979
  // Its references are this lab's now, so a step asking this lab's Echo
4569
4980
  // profile is read as an Echo step (upgradePipeline did it for ids that
@@ -4597,6 +5008,8 @@ function mintIds(doc ) {
4597
5008
  for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4598
5009
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4599
5010
  }
5011
+ // Nor need it say how its linked groups pass together: all of them.
5012
+ if (!("pass" in doc)) doc.pass = { mode: "all" };
4600
5013
  return doc;
4601
5014
  }
4602
5015
 
@@ -4610,6 +5023,224 @@ function yamlToPipeline(text , ctx ) {
4610
5023
  return importPipeline(doc, ctx);
4611
5024
  }
4612
5025
 
5026
+ // ---- JUnit (#251) --------------------------------------------------------------
5027
+ //
5028
+ // What a CI test panel reads, from the worker's report and from a lab's
5029
+ // stored run alike (`evals-lab run --lab`, #256), so both write one format.
5030
+ // A suite is one target's eval, a case one item: a failure carries the reason
5031
+ // the item failed, an item that never ran is an error, and one the eval had
5032
+ // nothing to read of is skipped.
5033
+
5034
+ /** One JUnit test suite: a target's eval, and a case per item it read. */
5035
+
5036
+
5037
+
5038
+
5039
+
5040
+ // XML 1.0 has no way to write most control characters, escaped or not, and a
5041
+ // model's reply can hold any of them.
5042
+ const xmlText = (v ) => String(v).replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFE\uFFFF]/g, "")
5043
+ .replace(/[&<>"']/g, c => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", '"': "&quot;", "'": "&apos;" })[c] );
5044
+
5045
+ /** [suites] as JUnit XML, under the name of what ran them. */
5046
+ function junitXml(suites , runner ) {
5047
+ const count = (cases , k ) => cases.filter(c => c[k] != null).length;
5048
+ const body = suites.map(st => {
5049
+ const cases = st.cases.map(c => ` <testcase classname="${xmlText(st.name)}" name="${xmlText(c.name)}">`
5050
+ + (c.failure != null ? `<failure message="${xmlText(c.failure)}"/>`
5051
+ : c.error != null ? `<error message="${xmlText(c.error)}"/>`
5052
+ : c.skipped != null ? `<skipped message="${xmlText(c.skipped)}"/>` : "")
5053
+ + "</testcase>");
5054
+ return ` <testsuite name="${xmlText(st.name)}" tests="${st.cases.length}" failures="${count(st.cases, "failure")}" `
5055
+ + `errors="${count(st.cases, "error")}" skipped="${count(st.cases, "skipped")}">\n`
5056
+ + cases.map(c => c + "\n").join("") + " </testsuite>\n";
5057
+ });
5058
+ const all = suites.flatMap(st => st.cases);
5059
+ return `<?xml version="1.0" encoding="UTF-8"?>\n<testsuites name="${xmlText(runner)}" tests="${all.length}" `
5060
+ + `failures="${count(all, "failure")}" errors="${count(all, "error")}" skipped="${count(all, "skipped")}">\n`
5061
+ + body.join("") + "</testsuites>\n";
5062
+ }
5063
+
5064
+ // ---- a pipeline as a CI bundle (#254) --------------------------------------
5065
+ //
5066
+ // Export for CI writes a pipeline as files a repository keeps and
5067
+ // `evals-lab run` runs with no lab around it (docs/pipeline-yaml.md, "Export
5068
+ // for CI"): the pipeline's YAML, its Target profiles by slug, and each
5069
+ // dataset it grades against as the Library's export of it. The server adds
5070
+ // the plugins and, when asked, the Source's items, and zips the lot. Read
5071
+ // back, the files are the run document the page would have submitted, with
5072
+ // each profile keyed by its slug rather than this lab's id -- the slug is
5073
+ // what the key's $EVALSLAB_API_KEY_<SLUG> is spelt from either way.
5074
+
5075
+ /** The text files of a bundle, by path inside it. */
5076
+
5077
+
5078
+ /** What a bundle is read back as: the run document, and each dataset it
5079
+ grades against, by the id its evals name it by. */
5080
+
5081
+
5082
+
5083
+
5084
+
5085
+ const BUNDLE_PIPELINE = "pipeline.yaml";
5086
+ const BUNDLE_PROFILES = "profiles.yaml";
5087
+ const BUNDLE_DATASET = /^datasets\/([a-z0-9]+(?:-[a-z0-9]+)*)\.json$/;
5088
+
5089
+ /**
5090
+ * [doc] as a bundle's text files: `pipeline.yaml`, `profiles.yaml` and
5091
+ * `datasets/<slug>.json`. [ctx.profiles] is Setup's list -- keys and all,
5092
+ * none of which is written -- [ctx.grader] the lab's grader, written into
5093
+ * each eval that would have asked it, and [ctx.datasets] each graded
5094
+ * dataset by id, as `{ name, body }`. `{ error }` when a dataset or a
5095
+ * profile the pipeline names is not given, since a bundle without it runs
5096
+ * something else.
5097
+ */
5098
+ function exportBundle(doc , ctx
5099
+ = {})
5100
+ {
5101
+ const profiles = withSlugs((ctx.profiles ?? []) );
5102
+ // What the lab would supply at submit, written down: no lab supplies it later.
5103
+ const out = clone(doc) ;
5104
+ out.evals = (Array.isArray(out.evals) ? out.evals : [])
5105
+ .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
5106
+ const table = {};
5107
+ for (const id of profileIds(out)) {
5108
+ const p = profiles.find(x => x.id === id);
5109
+ if (!p) return { error: `Target profile ${id} not found` };
5110
+ const { slug, ...flat } = connectionSettings(connectionOf(p) );
5111
+ table[slug ] = flat;
5112
+ }
5113
+ const files = {};
5114
+ const taken = new Set (), written = new Set ();
5115
+ for (const t of evalsOf(out) ) {
5116
+ // A link's group, or a private group's Cases from: the reference itself.
5117
+ const ref = casesRef(t);
5118
+ if (!ref) continue;
5119
+ const d = ctx.datasets?.[ref.id];
5120
+ if (!d) return { error: `Dataset ${ref.name || ref.id} not found` };
5121
+ // Named as the dataset is now, since the name is what a reader joins on.
5122
+ ref.name = d.name;
5123
+ if (written.has(ref.id)) continue;
5124
+ const slug = uniqueSlug(slugOf(d.name) || "dataset", taken);
5125
+ taken.add(slug);
5126
+ written.add(ref.id);
5127
+ // The Library's Export of it, so the file imports into any lab too. Its
5128
+ // name is what joins it back to the eval naming it: a lab's dataset
5129
+ // names are its own.
5130
+ files[`datasets/${slug}.json`] = JSON.stringify({ format: "evals-lab/dataset", version: DATASET_BODY_VERSION,
5131
+ dataset: { name: d.name, body: d.body } }, null, 2) + "\n";
5132
+ }
5133
+ files[BUNDLE_PIPELINE] = pipelineToYaml(out, { profiles });
5134
+ files[BUNDLE_PROFILES] = yaml.dump(table, { lineWidth: -1, noRefs: true });
5135
+ const bad = bundleProblems(files, profiles.map(p => p.key));
5136
+ return bad.length ? { error: bad[0] } : { files };
5137
+ }
5138
+
5139
+ /**
5140
+ * Why [files] cannot leave the lab: a field named like a key anywhere in
5141
+ * them, or any of [keys] -- Setup's own -- written anywhere at all. Empty
5142
+ * when nothing key-shaped is in them. The server asks the same of the zip
5143
+ * it is handed (server.py `bundle_problems`).
5144
+ */
5145
+ function bundleProblems(files , keys = []) {
5146
+ const bad = [];
5147
+ const named = (v , at ) => {
5148
+ if (Array.isArray(v)) return v.forEach(x => named(x, at));
5149
+ if (!isObj(v)) return;
5150
+ for (const [k, x] of Object.entries(v)) {
5151
+ if (LOOKS_LIKE_A_KEY.test(k)) bad.push(`${at} holds a field named ${k}, and a key never leaves the lab`);
5152
+ named(x, at);
5153
+ }
5154
+ };
5155
+ for (const [path, text] of Object.entries(files)) {
5156
+ for (const key of keys) {
5157
+ if (isStr(key) && key.trim().length >= 4 && text.includes(key.trim())) {
5158
+ bad.push(`${path} holds a Target profile's key, and a key never leaves the lab`);
5159
+ break;
5160
+ }
5161
+ }
5162
+ let doc ;
5163
+ try { doc = path.endsWith(".json") ? JSON.parse(text) : yaml.load(text); } catch { continue; }
5164
+ named(doc, path);
5165
+ }
5166
+ return [...new Set(bad)];
5167
+ }
5168
+
5169
+ /**
5170
+ * A bundle's [files] as the run they describe: `pipeline.yaml` read as an
5171
+ * import would read it, against `profiles.yaml` and the `datasets/` it
5172
+ * carries, then resolved as the page resolves one at submit. [ctx.items] is
5173
+ * the Source's file list -- the names in the bundle's `items/`, or a
5174
+ * directory CI names; with none, each case's item is the list, so an item
5175
+ * nothing supplies is a missing item rather than an item never asked.
5176
+ * `{ error }` names the first thing that keeps it from running: a profile
5177
+ * or a dataset the files do not hold, a key in them, or a document the lab
5178
+ * would refuse. The caller registers the bundle's plugins first.
5179
+ */
5180
+ function readBundle(files , ctx = {}) {
5181
+ if (!isStr(files[BUNDLE_PIPELINE])) return { error: `the bundle has no ${BUNDLE_PIPELINE}` };
5182
+ let table = {};
5183
+ try { table = isStr(files[BUNDLE_PROFILES]) ? yaml.load(files[BUNDLE_PROFILES] ) ?? {} : {}; }
5184
+ catch { return { error: `${BUNDLE_PROFILES} is not YAML` }; }
5185
+ if (!isObj(table)) return { error: `${BUNDLE_PROFILES} maps each profile's slug to its settings` };
5186
+ const keyed = bundleProblems(files);
5187
+ if (keyed.length) return { error: keyed[0] };
5188
+ const profiles = [];
5189
+ for (const [slug, p] of Object.entries(table)) {
5190
+ if (!isSlug(slug)) return { error: `${BUNDLE_PROFILES}: ${JSON.stringify(slug)} is not a slug` };
5191
+ if (!isObj(p)) return { error: `${BUNDLE_PROFILES}: ${slug} has to hold its settings` };
5192
+ if (!CONNECTION_TYPES[typeOf(p )]) {
5193
+ return { error: `${BUNDLE_PROFILES}: ${slug} is of type ${p.type}, which this lab does not have` };
5194
+ }
5195
+ profiles.push({ ...p, id: slug, slug, name: isStr(p.name) && p.name ? p.name : slug });
5196
+ }
5197
+ // Each dataset by its name, which is what joins it to the eval naming it.
5198
+ const datasets = [];
5199
+ for (const [path, text] of Object.entries(files)) {
5200
+ if (!path.startsWith("datasets/")) continue;
5201
+ if (!BUNDLE_DATASET.test(path)) return { error: `${path} is not named datasets/<slug>.json` };
5202
+ let doc ;
5203
+ try { doc = JSON.parse(text); } catch { return { error: `${path} is not JSON` }; }
5204
+ if (!isObj(doc) || doc.format !== "evals-lab/dataset" || !isObj(doc.dataset)) {
5205
+ return { error: `${path} is not a dataset's export` };
5206
+ }
5207
+ if (!(Number.isInteger(doc.version) && doc.version >= 1 && doc.version <= DATASET_BODY_VERSION)) {
5208
+ return { error: `${path} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to ${DATASET_BODY_VERSION}` };
5209
+ }
5210
+ const body = upgradeDatasetBody(doc.dataset.body) ;
5211
+ if (!isObj(body) || !Array.isArray(body.cases)) return { error: `${path} holds no cases` };
5212
+ datasets.push({ name: String(doc.dataset.name ?? ""), body: body });
5213
+ }
5214
+ let raw ;
5215
+ try { raw = yaml.load(files[BUNDLE_PIPELINE] ); }
5216
+ catch { return { error: `${BUNDLE_PIPELINE} is not YAML` }; }
5217
+ // The bundle stands in for the Source: whatever it names, the items are
5218
+ // the ones handed over.
5219
+ const content = isObj(raw) ? contentOf(upgradePipeline(raw)) : null;
5220
+ const refs = [];
5221
+ // Read as this version, whichever wrote the bundle: a link's group, or a
5222
+ // group of its own's Cases from.
5223
+ for (const t of isObj(raw) ? evalsOf(upgradePipeline(raw)) : []) {
5224
+ const ref = casesRef(t);
5225
+ if (!ref) continue;
5226
+ if (datasets.some(d => d.name === ref.name)) refs.push({ id: ref.id, name: ref.name });
5227
+ }
5228
+ const imported = importPipeline(raw, { profiles, groups: refs,
5229
+ sources: content?.type === "source" && isRef(content.ref) ? [content.ref] : [] });
5230
+ if ("error" in imported) return imported;
5231
+ if (imported.missing.length) return { error: `${imported.missing[0]}: the bundle does not hold it` };
5232
+ const byId = {};
5233
+ for (const r of refs) byId[r.id] = datasets.find(d => d.name === r.name) .body;
5234
+ const caseItems = [...new Set(Object.values(byId).flatMap(b => gradedSetFrom(b ).filter(c => !c.todo).map(caseItem)).filter(Boolean))];
5235
+ const run = resolvePipeline(imported.doc, {
5236
+ profiles: id => profiles.find(p => p.id === id) ?? null,
5237
+ files: ctx.items ? [...ctx.items].sort() : caseItems,
5238
+ });
5239
+ const bad = validatePipeline(run, { groups: refs });
5240
+ if (bad.length) return { error: bad[0] };
5241
+ return { run, datasets: byId };
5242
+ }
5243
+
4613
5244
  /**
4614
5245
  * Scenario [i] of a run document as runPipeline and a transport take it:
4615
5246
  * each stage's wording, kind, image flag and modifiers, the token set of
@@ -4779,22 +5410,40 @@ function scenarioEvals(run , i , items
4779
5410
  });
4780
5411
  }
4781
5412
 
5413
+ /** One eval group's verdict over a scenario (§17): a whole-run group's own
5414
+ verdict, or a per-item group's "every item it read passed"; null where it
5415
+ was skipped, or had nothing to read yet. */
5416
+ function groupPass(o ) {
5417
+ if (o.skipped) return null;
5418
+ if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
5419
+ return o.ran ? o.passed === o.ran : null;
5420
+ }
5421
+
4782
5422
  /** A scenario's pass or fail over every eval that read it: null where none
4783
5423
  has anything to say yet. */
4784
5424
  function scenarioPasses(outcomes ) {
4785
- let said = false;
4786
- for (const o of outcomes) {
4787
- if (o.skipped) continue;
4788
- if (o.whole) {
4789
- if (!o.verdict?.ran) continue;
4790
- said = true;
4791
- if (!o.verdict.pass) return false;
4792
- } else if (o.ran) {
4793
- said = true;
4794
- if (o.passed < o.ran) return false;
4795
- }
4796
- }
4797
- return said ? true : null;
5425
+ const passes = outcomes.map(groupPass);
5426
+ if (!passes.some((p) => p != null)) return null;
5427
+ return !passes.includes(false);
5428
+ }
5429
+
5430
+ /** A scenario's overall verdict under a run's pass rule (§17): each linked
5431
+ group's verdict folded together the way `pass` says -- every group passes,
5432
+ or at least a number of them. Null where no group has a verdict yet, so a
5433
+ run with no evals, or one still grading, reads as it does without a rule. */
5434
+ function overallVerdict(outcomes , pass ) {
5435
+ const passes = outcomes.map(groupPass);
5436
+ if (!passes.some((p) => p != null)) return null;
5437
+ return passVerdict(pass, passes);
5438
+ }
5439
+
5440
+ /** Whether a run passes a Target under its overall pass rule (§17): `all`
5441
+ passes when no linked group fails, `atLeast` when at least `count` pass.
5442
+ [passes] is each group's pass (true), fail (false), or neither (null: it
5443
+ graded nothing, or an earlier group skipped it). */
5444
+ function passVerdict(pass , passes ) {
5445
+ if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
5446
+ return !passes.includes(false);
4798
5447
  }
4799
5448
 
4800
5449
  /**
@@ -4995,15 +5644,15 @@ export {
4995
5644
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
4996
5645
  PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4997
5646
  SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4998
- registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
5647
+ registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
4999
5648
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5000
5649
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5001
5650
  contentOf, withContent, replyOf,
5002
- targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5651
+ targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5003
5652
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
5004
- blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5005
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
5006
- evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
5007
- pipelineToYaml, importPipeline, yamlToPipeline,
5653
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5654
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
5655
+ evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
5656
+ pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
5008
5657
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
5009
5658
  };