evals-lab 0.4.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -275,9 +275,41 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
275
275
 
276
276
 
277
277
 
278
- /** An eval, from version 9: Metrics. A Single Test or a Graded set is what an
279
- older document held (upgradePipeline reads it converted). */
280
-
278
+ /** A private group (docs/pipeline-model.md §17): an eval group's body held
279
+ in the pipeline rather than the Library -- what a version-12 Metrics
280
+ eval upgrades to when it is not a plain link. `casesFrom` is the Library
281
+ group whose cases, at its newest version, join this group's Every item
282
+ under this group's scoring; only the upgrade writes one. */
283
+
284
+
285
+
286
+
287
+
288
+ /** A Library eval group by reference; a run's copy says the version it
289
+ graded with: the body's fingerprint, and its number. */
290
+
291
+
292
+ /** An eval, from version 13: a link to an eval group (§17). `group` names a
293
+ Library group, followed at its newest version (`pin: null`) or pinned at
294
+ one; a private group is `own`, with `group: null`. A run's copy carries
295
+ the version it graded with on the reference (`version`, the body's
296
+ fingerprint, and `n`), and the lab's grader where the group names none. */
297
+
298
+
299
+
300
+
301
+
302
+
303
+
304
+
305
+ /** An eval, from version 13: a link to an eval group. A Metrics eval
306
+ (version 9 to 12), a Single Test or a Graded set is what an older
307
+ document held (upgradePipeline reads it converted). */
308
+
309
+
310
+ /** The rule a pipeline's linked groups pass by, together: every one, or at
311
+ least `count` of them. */
312
+
281
313
 
282
314
  /** A pipeline's evals, in the order they read a run. */
283
315
 
@@ -293,11 +325,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
293
325
 
294
326
 
295
327
 
328
+
296
329
 
297
330
 
298
331
  /** A Setup profile as a run carries it: request settings, never the key. */
299
332
 
300
333
 
334
+
335
+
301
336
 
302
337
 
303
338
 
@@ -338,6 +373,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
338
373
 
339
374
 
340
375
 
376
+
341
377
 
342
378
 
343
379
 
@@ -364,11 +400,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
364
400
 
365
401
 
366
402
 
403
+ /** A Target profile as an import or export matches it: its slug, where it
404
+ has one -- withSlugs gives the rest theirs. */
405
+
406
+
367
407
  /** What an import matches references against: { id, name } each. */
368
408
 
369
-
409
+
370
410
 
371
-
411
+
372
412
 
373
413
 
374
414
 
@@ -817,8 +857,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
817
857
 
818
858
 
819
859
  /** What validation looks references up in: `profiles(id)` and `sources(id)`
820
- answer with what the id names, or nothing; `datasets` is the datasets the
821
- lab holds, as `GET /api/datasets` lists them. Each is optional, and a
860
+ answer with what the id names, or nothing; `groups` is the eval groups
861
+ the lab holds. Each is optional, and a
822
862
  reference nothing is given to look up is not checked. */
823
863
 
824
864
 
@@ -826,7 +866,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
826
866
 
827
867
 
828
868
 
829
-
869
+
870
+
871
+
872
+
830
873
 
831
874
 
832
875
  /** Lookups, and whether it is a run document being validated. */
@@ -904,6 +947,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
904
947
 
905
948
 
906
949
 
950
+
951
+
952
+
907
953
 
908
954
 
909
955
 
@@ -917,6 +963,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
917
963
 
918
964
 
919
965
 
966
+
967
+
968
+
920
969
 
921
970
 
922
971
  /** A job's stages, in the order they run (pipeline-model §16). A target's
@@ -2687,7 +2736,10 @@ function applyModifiers (list , kind ,
2687
2736
  // scenario is a target, whose own step in each job is what it sends there.
2688
2737
  // 11: `tests` are `evals`: the key renames and nothing in an eval changes,
2689
2738
  // so a stored result's scores, keyed by eval id, read as they did.
2690
- const PIPELINE_VERSION = 12 ;
2739
+ // 12: a Contains metric's Ignore case holds item by item too.
2740
+ // 13: an eval is a link to an eval group, or a group of the pipeline's own,
2741
+ // and the document has an overall pass rule (pipeline-model §17).
2742
+ const PIPELINE_VERSION = 13 ;
2691
2743
 
2692
2744
  // Plain objects, so an entry is added by assignment and a reader never needs
2693
2745
  // to know which registered it.
@@ -2780,11 +2832,13 @@ OUTPUT_KINDS.text = {
2780
2832
 
2781
2833
  // ---- the connection a profile reference resolves to ------------------------
2782
2834
  // What a run carries of a Setup profile: everything a request needs and never
2783
- // the key. A key follows the profile's id into $EVAL_API_KEY_<ID>, which the
2784
- // server sets from its profiles store and the runner reads, so a name is only
2785
- // a label and two profiles can share one. #46: the profile carries `type`,
2786
- // and the run's resolved copy carries it too (docs/pipeline-model.md §7).
2787
- const CONNECTION_FIELDS = ["name", "url", "model", "type", "temperature",
2835
+ // the key. A key follows the profile's slug into $EVALSLAB_API_KEY_<SLUG> -- or,
2836
+ // in a run document from before slugs, its id into $EVALSLAB_API_KEY_<ID> --
2837
+ // which the server sets from its profiles store and the runner reads, so a
2838
+ // name is only a label and two profiles can share one. #46: the profile
2839
+ // carries `type`, and the run's resolved copy carries it too
2840
+ // (docs/pipeline-model.md §7).
2841
+ const CONNECTION_FIELDS = ["name", "slug", "url", "model", "type", "temperature",
2788
2842
  "px", "format", "quality", "options"];
2789
2843
  const SETTING_KEYS = [...new Set(Object.values(CONNECTION_TYPES).flatMap(t => t.settings.map(s => s.key)))];
2790
2844
  const OPTION_FIELDS = ["seed", "nPredict", ...DECODING_KEYS];
@@ -2794,9 +2848,69 @@ const IMAGE_FORMATS = ["image/jpeg", "image/webp", "image/png
2794
2848
  const PROFILE_ID = /^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?$/;
2795
2849
  const LOOKS_LIKE_A_KEY = /^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$/i;
2796
2850
 
2797
- /** The variable a profile's key is read from: pm1x8k2q is EVAL_API_KEY_PM1X8K2Q. */
2798
- const keyVar = (id ) => "EVAL_API_KEY_"
2799
- + String(id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
2851
+ /** The variable a profile's key is read from: its slug's, when it has one --
2852
+ anthropic is EVALSLAB_API_KEY_ANTHROPIC -- else its id's, as a run document
2853
+ from before slugs has it: pm1x8k2q is EVALSLAB_API_KEY_PM1X8K2Q. */
2854
+ const keyVar = (id , slug ) => "EVALSLAB_API_KEY_"
2855
+ + String(slug || id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
2856
+
2857
+ // ---- a profile's slug (#253) ------------------------------------------------
2858
+ // What a profile is called outside the lab: in a pipeline file and in the
2859
+ // name of the variable CI keeps its key in. An id (pm1x9c0r) means nothing to
2860
+ // whoever sets that secret; a slug (anthropic) does, and stays put when the
2861
+ // lab it came from does not. Lowercase letters and digits, hyphens between,
2862
+ // so its variable is spelt one way and two slugs never spell the same one.
2863
+ const SLUG = /^[a-z0-9]+(?:-[a-z0-9]+)*$/;
2864
+ const SLUG_MAX = 32;
2865
+
2866
+ /** [text] as a slug: lowercase, every other run of characters one hyphen,
2867
+ cut to SLUG_MAX. "" when nothing of it is a letter or a digit. */
2868
+ function slugOf(text ) {
2869
+ return String(text ?? "").toLowerCase().normalize("NFKD").replace(/[\u0300-\u036f]/g, "")
2870
+ .replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, SLUG_MAX).replace(/-+$/, "");
2871
+ }
2872
+
2873
+ /** [base] made unique among [taken]: anthropic, then anthropic-2, -3. */
2874
+ function uniqueSlug(base , taken ) {
2875
+ const root = base || "profile";
2876
+ if (!taken.has(root)) return root;
2877
+ for (let n = 2; ; n++) {
2878
+ const tail = `-${n}`;
2879
+ const s = root.slice(0, SLUG_MAX - tail.length).replace(/-+$/, "") + tail;
2880
+ if (!taken.has(s)) return s;
2881
+ }
2882
+ }
2883
+
2884
+ /**
2885
+ * [list] with every profile holding a slug of its own: one it holds already
2886
+ * is kept, tidied into a slug (Setup's field holds what was typed, a
2887
+ * trailing hyphen and all), while no profile before it holds the same; any
2888
+ * other is made from its name, made unique. The slugs held are settled
2889
+ * first, so a slug minted for one profile never takes another's. The same
2890
+ * array when nothing changed, so a reader can tell a store that needs
2891
+ * writing back from one that does not.
2892
+ */
2893
+ function withSlugs (list ) {
2894
+ const taken = new Set ();
2895
+ const held = list.map(p => {
2896
+ const s = slugOf(p?.slug);
2897
+ if (!s || taken.has(s)) return "";
2898
+ taken.add(s);
2899
+ return s;
2900
+ });
2901
+ let changed = false;
2902
+ const out = list.map((p, i) => {
2903
+ const slug = held[i] || uniqueSlug(slugOf(p?.name), taken);
2904
+ taken.add(slug);
2905
+ if (slug === p?.slug) return p;
2906
+ changed = true;
2907
+ return { ...p, slug };
2908
+ });
2909
+ return changed ? out : list ;
2910
+ }
2911
+
2912
+ /** Whether [s] is a slug as a profile may hold one. */
2913
+ const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
2800
2914
 
2801
2915
  /** A stored profile as a run carries it: its request settings, no key. */
2802
2916
  function connectionOf(p ) {
@@ -2807,6 +2921,7 @@ function connectionOf(p ) {
2807
2921
  // the options bag holds the type's own settings only.
2808
2922
  for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
2809
2923
  const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
2924
+ if (isSlug(p.slug)) out.slug = p.slug;
2810
2925
  for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
2811
2926
  if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
2812
2927
  if (Object.keys(options).length) out.options = options;
@@ -2863,7 +2978,7 @@ function onlyFields(obj , at , allowed , bad
2863
2978
  for (const k of Object.keys(obj)) {
2864
2979
  if (allowed.includes(k)) continue;
2865
2980
  bad.push(LOOKS_LIKE_A_KEY.test(k)
2866
- ? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVAL_API_KEY_<ID>"} supplies it`
2981
+ ? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVALSLAB_API_KEY_<ID>"} supplies it`
2867
2982
  : `${at} has "${k}", which is not a pipeline field`);
2868
2983
  }
2869
2984
  }
@@ -3026,7 +3141,7 @@ LEGACY_TESTS.graded = {
3026
3141
  if (!isRef(d) || (d.version != null && !isStr(d.version))) {
3027
3142
  return void bad.push("a graded eval has to name its dataset as { id, name }");
3028
3143
  }
3029
- if (ctx.datasets && !ctx.datasets.some(x => x.id === d.id)) bad.push(`Dataset ${d.name || d.id} not found`);
3144
+ if (ctx.groups && !ctx.groups.some(x => x.id === d.id)) bad.push(`Eval group ${d.name || d.id} not found`);
3030
3145
  },
3031
3146
  };
3032
3147
 
@@ -3169,7 +3284,10 @@ function readRun(list , run ) {
3169
3284
  });
3170
3285
  }
3171
3286
 
3172
- EVAL_TYPES.metrics = {
3287
+ // The eval of versions 9 to 12, read only to upgrade (LEGACY_TESTS.metrics.
3288
+ // toGroup) and kept whole as the reference groups-check.js and
3289
+ // pipeline-check.js hold the eval group to.
3290
+ LEGACY_TESTS.metrics = {
3173
3291
  label: "Metrics",
3174
3292
  description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
3175
3293
  fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
@@ -3187,7 +3305,7 @@ EVAL_TYPES.metrics = {
3187
3305
  // The dataset whose cases a case metric reads, and whose own metrics join.
3188
3306
  if (t.dataset != null) {
3189
3307
  if (!isRef(t.dataset) || (t.dataset.version != null && !isStr(t.dataset.version))) bad.push("the Metrics name their dataset as { id, name }");
3190
- else if (ctx.datasets && !ctx.datasets.some(x => x.id === t.dataset.id)) bad.push(`Dataset ${t.dataset.name || t.dataset.id} not found`);
3308
+ else if (ctx.groups && !ctx.groups.some(x => x.id === t.dataset.id)) bad.push(`Eval group ${t.dataset.name || t.dataset.id} not found`);
3191
3309
  }
3192
3310
  if (!SCORING_MODES.includes(t.mode)) bad.push(`the Metrics are scored ${SCORING_MODES.join(" or ")}`);
3193
3311
  if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
@@ -3228,6 +3346,151 @@ EVAL_TYPES.metrics = {
3228
3346
  metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
3229
3347
  };
3230
3348
 
3349
+ /** A version-12 Metrics eval as the version-13 eval it reads as (§17): a
3350
+ link to its dataset, followed at the newest version, where it named one
3351
+ and held nothing of its own -- no metrics, scored All, the lab's grader,
3352
+ over each item -- and a private group holding its metrics, scoring and
3353
+ grader otherwise, with its dataset's cases (`casesFrom`). An eval over the
3354
+ whole run never read its dataset's cases, so its group takes none. The
3355
+ id, name and Continue on failure are kept, so a stored score keyed by
3356
+ the eval's id is the link's. */
3357
+ function groupOfMetrics(t ) {
3358
+ const { id, name, continueOnFailure } = t;
3359
+ const common = { id, ...(name !== undefined ? { name } : {}), type: "group", continueOnFailure };
3360
+ const metrics = Array.isArray(t.metrics) ? t.metrics : [];
3361
+ const run = t.over === "run";
3362
+ if (isRef(t.dataset) && !metrics.length && t.mode === "all" && t.grader == null && !run) {
3363
+ return { ...common, group: { ...t.dataset }, pin: null };
3364
+ }
3365
+ const own = { version: DATASET_BODY_VERSION, source: null, scoring: { mode: t.mode, threshold: t.threshold ?? null },
3366
+ grader: t.grader ?? null, every: run ? [] : metrics, run: run ? metrics : [], cases: [],
3367
+ casesFrom: isRef(t.dataset) && !run ? { ...t.dataset } : null };
3368
+ return { ...common, group: null, pin: null, own };
3369
+ }
3370
+ LEGACY_TESTS.metrics.toGroup = groupOfMetrics;
3371
+
3372
+ /** The group an eval reads by: a private group's own body, or a link's
3373
+ Library group as the runner hands it ([more].group). A link whose body
3374
+ nothing hands over reads as a Library group upgraded from version 6 does
3375
+ -- no metrics of its own, scored All -- which is what every one was
3376
+ until a body could hold more. */
3377
+ function groupOf(t , more = {}) {
3378
+ if (isObj(t.own)) return t.own ;
3379
+ const body = isRef(t.group) ? more.group?.(t.group) : undefined;
3380
+ return body ?? { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
3381
+ every: [], run: [], cases: [] };
3382
+ }
3383
+
3384
+ /** The Library group whose cases an eval reads, where it reads one: a
3385
+ link's group, or a private group's `casesFrom`. */
3386
+ function casesRef(t ) {
3387
+ if (!isObj(t) || t.type !== "group") return null;
3388
+ if (isRef(t.group)) return t.group ;
3389
+ return isObj(t.own) && isRef(t.own.casesFrom) ? t.own.casesFrom : null;
3390
+ }
3391
+
3392
+ /** The metrics a private group shows as its rules: its Whole run where it
3393
+ reads the run, its Every item otherwise. */
3394
+ const ownList = (t ) => (isObj(t.own) ? (wholeRunGroup(t) ? t.own.run : t.own.every) ?? [] : []);
3395
+
3396
+ /** A private group read over the whole run: Whole run metrics, and nothing
3397
+ read item by item. A group holding both is read item by item until the
3398
+ runner reports a group's Whole run beside its items (#233). */
3399
+ const wholeRunGroup = (t ) => isObj(t.own) && Array.isArray(t.own.run) && t.own.run.length > 0
3400
+ && !(Array.isArray(t.own.every) && t.own.every.length) && !(Array.isArray(t.own.cases) && t.own.cases.length) && !t.own.casesFrom;
3401
+
3402
+ // A link to an eval group (docs/pipeline-model.md §17): a Library group by
3403
+ // reference, or a private group held here (`own`). It reads each item as its
3404
+ // group does (readGroup), or -- a private group of Whole run metrics alone --
3405
+ // the whole run (readGroupRun).
3406
+ EVAL_TYPES.group = {
3407
+ label: "Eval group",
3408
+ description: "Checks of each reply, or of the whole run, from an eval group: linked from the Library, or this pipeline's own.",
3409
+ fields: ["type", "group", "pin", "own", "grader"],
3410
+ // A metric that reads terms (a case) reads only a kind that yields them.
3411
+ accepts: t => ([...ownList(t)].some((m ) => METRICS[m?.type]?.needsTerms)
3412
+ ? Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms) : null),
3413
+ defaults: () => ({ type: "group", group: null, pin: null,
3414
+ own: { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
3415
+ every: [], run: [], cases: [], casesFrom: null } }),
3416
+ validate(t, ctx, bad){
3417
+ const g = t.group, own = t.own;
3418
+ const named = (ref , what ) => {
3419
+ const n = isObj(ref) ? ref.n : undefined;
3420
+ if (!isRef(ref) || (ref.version != null && !isStr(ref.version)) || (n != null && !Number.isInteger(n))) {
3421
+ return void bad.push(`${what} names its eval group as { id, name }`);
3422
+ }
3423
+ onlyFields(ref, what, ["id", "name", "version", "n"], bad);
3424
+ const held = ctx.groups?.find(x => x.id === ref.id);
3425
+ if (ctx.groups && !held) bad.push(`Eval group ${ref.name || ref.id} not found`);
3426
+ return held;
3427
+ };
3428
+ if (t.pin != null && !(Number.isInteger(t.pin) && t.pin > 0)) bad.push("a pin is a group's version: a whole number from 1");
3429
+ if (t.grader != null && !isRef(t.grader)) bad.push("the link names its grader as { id, name }");
3430
+ if (g != null) {
3431
+ if (own != null) return void bad.push("an eval links a Library group or holds its own, not both");
3432
+ const held = named(g, "the link");
3433
+ if (held && t.pin != null && typeof held.versions === "number" && t.pin > held.versions) {
3434
+ bad.push(`${g.name || g.id} has no version ${t.pin}`);
3435
+ }
3436
+ return;
3437
+ }
3438
+ if (!isObj(own)) return void bad.push("an eval links a Library group, or holds its own");
3439
+ if (t.pin != null) bad.push("a group of this pipeline's own versions with it, so it is never pinned");
3440
+ if (own.version !== DATASET_BODY_VERSION) bad.push(`the group's own body is version ${DATASET_BODY_VERSION}`);
3441
+ for (const key of ["scoring", "every", "run", "cases"]) if (own[key] == null) bad.push(`the group's own body has no ${key}`);
3442
+ onlyFields(own, "the group's own body", ["version", "source", "scoring", "grader", "every", "run", "cases", "casesFrom"], bad);
3443
+ if (bad.length) return;
3444
+ bad.push(...validateEvals(own));
3445
+ if (own.casesFrom != null) named(own.casesFrom, "Cases from");
3446
+ // Read item by item or over the run: one section at a time, until a
3447
+ // group's Whole run is reported beside its items (#233).
3448
+ if (own.run.length && !wholeRunGroup(t)) bad.push("the group's own body reads each item or the whole run, not both");
3449
+ // The group's own grader, or the lab's.
3450
+ const grader = isRef(own.grader) ? own.grader : isRef(ctx.grader) ? ctx.grader : null;
3451
+ if (gradedIn(own.every) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
3452
+ // A grader is asked words, and needs a model to ask.
3453
+ const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
3454
+ if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
3455
+ else if (p) {
3456
+ const type = CONNECTION_TYPES[typeOf(p )];
3457
+ if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
3458
+ else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
3459
+ }
3460
+ },
3461
+ rules: t => ownList(t).map((m , x ) => ({
3462
+ key: `m${x}`, label: (isStr(m.metric) && m.metric) || METRICS[m.type]?.label || m.type,
3463
+ want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
3464
+ .filter(Boolean).join(" ") })),
3465
+ profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
3466
+ // A run carries the lab's grader where the group names none and may ask
3467
+ // one: a model-graded metric of its own, or a case's. A link's group is
3468
+ // the Library's, so the run's copy of the link carries it.
3469
+ resolve(t, ctx){
3470
+ if (!isRef(ctx.grader)) return t;
3471
+ const grader = { id: ctx.grader.id, name: ctx.grader.name };
3472
+ if (isObj(t.own)) {
3473
+ return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
3474
+ }
3475
+ return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
3476
+ },
3477
+ wholeRun: wholeRunGroup,
3478
+ verdict: (t, ress, kind) => {
3479
+ const own = t.own ;
3480
+ return wholeRunVerdict(own.run, ress, kind, own.scoring.mode, own.scoring.threshold);
3481
+ },
3482
+ // Rule m{x} is the group's own metric x, read in that place on every item.
3483
+ ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
3484
+ : ownList(t).length === 1 ? score.pass : null),
3485
+ // Every item, with its case where the group reads the Library group's
3486
+ // cases and the item has one there.
3487
+ read: async (t, kase, res, more) => {
3488
+ const group = groupOf(t, more);
3489
+ const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
3490
+ return readGroup({ ...group, grader } , casesRef(t) ? kase : null, res, more);
3491
+ },
3492
+ };
3493
+
3231
3494
  // ---- the lab's own scorers, as metrics ----------------------------------------
3232
3495
  // What the Single Test checks, one metric each, so an eval of it converts to
3233
3496
  // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
@@ -3849,11 +4112,16 @@ function evalsOf(doc )
3849
4112
  return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
3850
4113
  }
3851
4114
 
3852
- /** The dataset a document's evals grade against, where one does: a run
3853
- grades against one (validatePipeline says so), so the first names it. */
3854
- function evalsDataset(doc ) {
3855
- const t = evalsOf(doc).find((x ) => isObj(x) && isRef(x.dataset));
3856
- return t && "dataset" in t ? t.dataset : null;
4115
+ /** The Library group whose cases a document's evals read, where one does:
4116
+ a link's group, or a private group's Cases from -- or, in a stored run
4117
+ from before version 13, a Metrics eval's dataset. A run grades against
4118
+ one (validatePipeline says so), so the first names it. */
4119
+ function evalsDataset(doc ) {
4120
+ for (const t of evalsOf(doc) ) {
4121
+ const ref = casesRef(t) ?? (isObj(t) && isRef(t.dataset) ? t.dataset : null);
4122
+ if (ref) return ref;
4123
+ }
4124
+ return null;
3857
4125
  }
3858
4126
 
3859
4127
  STEP_TYPES.evals = {
@@ -3882,15 +4150,17 @@ STEP_TYPES.evals = {
3882
4150
  + `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
3883
4151
  }
3884
4152
  });
3885
- // A run is handed one dataset's body to grade against (server-side-runs §4).
3886
- const named = new Set(evals.filter((t ) => isObj(t) && isRef(t.dataset)).map((t ) => t.dataset.id));
3887
- if (named.size > 1) bad.push("the evals grade against one dataset at a time");
4153
+ // The worker is handed one Library group's body to grade against
4154
+ // (server-side-runs §4) until it reads the body the queue keeps per
4155
+ // group (#233).
4156
+ const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
4157
+ if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
3888
4158
  },
3889
4159
  };
3890
4160
 
3891
4161
  // ---- the document -------------------------------------------------------------
3892
4162
 
3893
- const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
4163
+ const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals", "pass"];
3894
4164
  // What resolving adds, and nothing else: the profiles it resolved to and the
3895
4165
  // run's own comment, which belongs to the run and never to the pipeline.
3896
4166
  const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
@@ -3999,6 +4269,12 @@ function newEval(type , name = "", fields = {})
3999
4269
  /** Version 6 to 7: chains are jobs. The field renames in place -- so an
4000
4270
  upgraded document reads the same, key for key -- and every job's `type`
4001
4271
  becomes `"job"`, so an older document reads as one of today's everywhere. */
4272
+ /** A new eval linking the Library eval group [group], followed at its
4273
+ newest version. */
4274
+ function newLink(group , name = "") {
4275
+ return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
4276
+ }
4277
+
4002
4278
  function jobsOf(next ) {
4003
4279
  next.version = PIPELINE_VERSION;
4004
4280
  if (Array.isArray(next.chains)) {
@@ -4113,6 +4389,26 @@ function caseAsWritten(next ) {
4113
4389
  return next;
4114
4390
  }
4115
4391
 
4392
+ /** Version 12 to 13: each Metrics eval is a link to an eval group or a
4393
+ group of the pipeline's own (groupOfMetrics), and the document passes
4394
+ when every linked group does. Pure: it needs nothing but the document,
4395
+ so every reader reads the same result. Every version before 13 ends
4396
+ here. */
4397
+ function groupsOf(next ) {
4398
+ next.version = PIPELINE_VERSION;
4399
+ if (Array.isArray(next.evals)) {
4400
+ next.evals = next.evals.map((t ) => (isObj(t) && t.type === "metrics" ? groupOfMetrics(t) : t));
4401
+ }
4402
+ if (!("pass" in next)) {
4403
+ // After the evals, where a reader of the document looks for it.
4404
+ const at = Object.keys(next).indexOf("evals");
4405
+ const entries = Object.entries(next);
4406
+ entries.splice(at < 0 ? entries.length : at + 1, 0, ["pass", { mode: "all" }]);
4407
+ return Object.fromEntries(entries);
4408
+ }
4409
+ return next;
4410
+ }
4411
+
4116
4412
  function upgradePipeline (doc , ctx = {}) {
4117
4413
  // A current document is read as it is, but for a profile reference the Runs
4118
4414
  // tab saved whole (see profileRefs), which is cut back, and a step on a
@@ -4123,24 +4419,26 @@ function upgradePipeline (doc , ctx = {}) {
4123
4419
  out = clone(out) ;
4124
4420
  profileRefs(out);
4125
4421
  }
4126
- return localSteps(withoutCaseMetric(out), ctx) ;
4422
+ return localSteps(out, ctx) ;
4423
+ }
4424
+ if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
4425
+ if (doc.version === 11 || doc.version === 12) {
4426
+ return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
4127
4427
  }
4128
- if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11].includes(doc.version )) return doc;
4129
- if (doc.version === 11) return localSteps(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) ))), ctx) ;
4130
- // Every version before 10 reads as version 9 first, then as 10, then as 11
4131
- // and 12.
4428
+ // Every version before 10 reads as version 9 first, then as 10, then as 11,
4429
+ // 12 and 13.
4132
4430
  // A version-10 document is cut as a current one was (profileRefs); the
4133
4431
  // earlier ones are cut on their way through nineOf.
4134
4432
  const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
4135
- return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
4433
+ return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
4136
4434
  }
4137
4435
 
4138
4436
  /** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
4139
4437
  are metrics now (dataset body version 5), which an eval naming the
4140
4438
  dataset adds for each item, so the case's own metrics carry what that
4141
- metric scored. Read so at every version, the current one included, as a
4142
- document saved before it went still holds it. The same document where
4143
- none does. */
4439
+ metric scored. Read so at every version up to 12, as a document saved
4440
+ before it went still holds it; version 13 holds no Metrics eval. The
4441
+ same document where none does. */
4144
4442
  function withoutCaseMetric(doc ) {
4145
4443
  const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
4146
4444
  if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
@@ -4229,7 +4527,7 @@ function tokenMappingsFromV1(set ) {
4229
4527
  function blankPipeline(opts = {}) {
4230
4528
  const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
4231
4529
  return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
4232
- jobs: [job], targets: [], evals: [] };
4530
+ jobs: [job], targets: [], evals: [], pass: { mode: "all" } };
4233
4531
  }
4234
4532
 
4235
4533
  /**
@@ -4368,6 +4666,7 @@ function validatePipeline(input , ctx = {}) {
4368
4666
  }
4369
4667
  });
4370
4668
  STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
4669
+ passProblems(doc.pass, Array.isArray(doc.evals) ? doc.evals.length : 0, bad);
4371
4670
  if (bad.length) return bad;
4372
4671
 
4373
4672
  // Asked before anything is sent, so a misspelt token costs nothing and
@@ -4382,16 +4681,45 @@ function validatePipeline(input , ctx = {}) {
4382
4681
  return bad;
4383
4682
  }
4384
4683
 
4684
+ /** The overall pass rule: every linked group, or at least [count] of the
4685
+ [n] the pipeline has. */
4686
+ function passProblems(pass , n , bad ) {
4687
+ if (!isObj(pass) || (pass.mode !== "all" && pass.mode !== "atLeast")) {
4688
+ return void bad.push("pass has to be { mode: all } or { mode: atLeast, count }");
4689
+ }
4690
+ onlyFields(pass, "pass", pass.mode === "all" ? ["mode"] : ["mode", "count"], bad);
4691
+ if (pass.mode === "atLeast" && !(Number.isInteger(pass.count) && pass.count > 0)) bad.push("pass at least has to count a whole number of groups from 1");
4692
+ else if (pass.mode === "atLeast" && pass.count > n) bad.push(`pass asks for ${pass.count} groups, and the pipeline has ${n}`);
4693
+ }
4694
+
4695
+ /**
4696
+ * What [doc] may still be right about, as sentences: a link to a group made
4697
+ * for another Source than the pipeline's content (§17, decision 2). The run
4698
+ * grades the cases whose items it holds, and the rest read Missing.
4699
+ */
4700
+ function pipelineWarnings(doc , ctx = {}) {
4701
+ if (!isObj(doc) || !Array.isArray(doc.evals) || !ctx.groups) return [];
4702
+ const content = contentOf(doc ) ;
4703
+ const here = content?.type === "source" && isRef(content.ref) ? content.ref.id : null;
4704
+ const warn = [];
4705
+ doc.evals.forEach((t , j ) => {
4706
+ const ref = casesRef(t);
4707
+ const src = ref && ctx.groups .find(x => x.id === ref.id)?.source;
4708
+ if (src && src.id !== here) warn.push(`${evalLabel(doc , j)}: ${ref .name || ref .id} grades ${src.name || src.id}`);
4709
+ });
4710
+ return warn;
4711
+ }
4712
+
4385
4713
  /** A run document's profiles table: id → connection, keyless, spellable. */
4386
4714
  function profilesProblems(profiles , bad ) {
4387
4715
  if (!isObj(profiles)) return void bad.push("profiles has to be an object of id → connection");
4388
4716
  const spelt = new Map ();
4389
4717
  for (const [id, conn] of Object.entries(profiles)) {
4390
4718
  if (!PROFILE_ID.test(id)) {
4391
- bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVAL_API_KEY_<ID>`);
4719
+ bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVALSLAB_API_KEY_<ID>`);
4392
4720
  continue;
4393
4721
  }
4394
- const from = keyVar(id);
4722
+ const from = keyVar(id, isObj(conn) && isStr(conn.slug) ? conn.slug : null);
4395
4723
  if (spelt.has(from)) bad.push(`profiles ${spelt.get(from)} and ${id} both take their key from $${from}`);
4396
4724
  spelt.set(from, id);
4397
4725
  const at = `profile ${id}`;
@@ -4408,6 +4736,9 @@ function connectionProblems(conn , at , from
4408
4736
  bad.push(`${at}: type has to be one of ${Object.keys(CONNECTION_TYPES).join(", ")}`);
4409
4737
  }
4410
4738
  if (!isStr(conn.name)) bad.push(`${at}: name has to be text`);
4739
+ if (conn.slug != null && !isSlug(conn.slug)) {
4740
+ bad.push(`${at}: slug has to be lowercase letters and digits, with hyphens between, at most ${SLUG_MAX}`);
4741
+ }
4411
4742
  const url = conn.url ?? "";
4412
4743
  if (!isStr(url)) bad.push(`${at}: url has to be text, blank for the lab's Ollama`);
4413
4744
  else if (url.trim()) {
@@ -4477,7 +4808,8 @@ function profileIds(doc )
4477
4808
  * the Source's file list as it reads now, `ctx.comment` is the run's own.
4478
4809
  */
4479
4810
  function resolvePipeline(doc ,
4480
- ctx = {}) {
4811
+ ctx
4812
+ = {}) {
4481
4813
  const run = clone(doc) ;
4482
4814
  // What the lab supplies an eval -- its grader -- before the profiles it
4483
4815
  // asks are carried.
@@ -4489,8 +4821,18 @@ function resolvePipeline(doc ,
4489
4821
  }
4490
4822
  const content = contentOf(run) ;
4491
4823
  if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
4492
- if (ctx.datasetVersion) {
4493
- for (const t of run.evals || []) if (isRef(t.dataset)) t.dataset.version = ctx.datasetVersion;
4824
+ // Each Library group the evals read, as the version it grades with: the
4825
+ // body's fingerprint and, where the lab numbers them, `n` -- a pinned
4826
+ // link's pin.
4827
+ if (ctx.groupVersion) {
4828
+ for (const t of run.evals || []) {
4829
+ const ref = casesRef(t);
4830
+ const at = ref && ctx.groupVersion(ref.id);
4831
+ if (!ref || !at) continue;
4832
+ ref.version = at.version;
4833
+ const n = isObj(t) && Number.isInteger(t.pin) ? t.pin : at.n;
4834
+ if (n != null) ref.n = n;
4835
+ }
4494
4836
  }
4495
4837
  if (isStr(ctx.comment)) run.comment = ctx.comment;
4496
4838
  return run;
@@ -4504,7 +4846,12 @@ function pipelineOfRun(run , ctx = {}) {
4504
4846
  delete doc.plugins;
4505
4847
  const content = contentOf(doc) ;
4506
4848
  if (content) { delete content.files; delete content.revs; }
4507
- for (const t of doc.evals || []) if (isObj(t.dataset)) delete t.dataset.version;
4849
+ // What the run recorded of each group it read is the run's. The grader
4850
+ // it carried stays, as a Metrics eval's did: a bundle writes it down.
4851
+ for (const t of doc.evals || []) {
4852
+ const ref = casesRef(t);
4853
+ if (ref) { delete ref.version; delete ref.n; }
4854
+ }
4508
4855
  return doc;
4509
4856
  }
4510
4857
 
@@ -4517,9 +4864,35 @@ function pipelineOfRun(run , ctx = {}) {
4517
4864
 
4518
4865
 
4519
4866
  /** [doc] as YAML text, the pipeline's file form -- references only, so an
4520
- export carries no key, no URL, no file list and no file bytes. */
4521
- function pipelineToYaml(doc ) {
4522
- return yaml.dump(doc, { lineWidth: -1, noRefs: true });
4867
+ export carries no key, no URL, no file list and no file bytes. A Target
4868
+ profile [ctx] holds is named by its slug, `{ slug, name }`, rather than
4869
+ by this lab's id (#253): the slug is what another lab, or CI's
4870
+ $EVALSLAB_API_KEY_<SLUG>, knows it by. */
4871
+ function pipelineToYaml(doc , ctx = {}) {
4872
+ const profiles = withSlugs(ctx.profiles ?? []);
4873
+ if (!profiles.length) return yaml.dump(doc, { lineWidth: -1, noRefs: true });
4874
+ const out = clone(doc) ;
4875
+ const bySlug = (ref ) => {
4876
+ const p = isRef(ref) ? profiles.find(x => x.id === ref.id) : null;
4877
+ return p ? { slug: p.slug, name: ref.name ?? p.name } : ref;
4878
+ };
4879
+ eachProfileRef(out, bySlug);
4880
+ return yaml.dump(out, { lineWidth: -1, noRefs: true });
4881
+ }
4882
+
4883
+ /** Every Target profile reference in [doc] -- each target's, each of its
4884
+ steps', each eval's grader -- replaced, in place, by [f] of it. */
4885
+ function eachProfileRef(doc , f ) {
4886
+ for (const t of targetsOf(doc)) {
4887
+ if (t.profile) t.profile = f(t.profile);
4888
+ for (const st of Array.isArray(t.steps) ? t.steps : []) {
4889
+ if (st?.profile) st.profile = f(st.profile);
4890
+ }
4891
+ }
4892
+ for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4893
+ if (isObj(t) && t.grader) t.grader = f(t.grader);
4894
+ if (isObj(t?.own) && t.own.grader) t.own.grader = f(t.own.grader);
4895
+ }
4523
4896
  }
4524
4897
 
4525
4898
  /**
@@ -4527,7 +4900,7 @@ function pipelineToYaml(doc ) {
4527
4900
  * name (pipeline-model §5), so a pipeline written in another lab -- ids that
4528
4901
  * mean nothing here -- finds its Source and Setup profiles by the names they
4529
4902
  * were exported under. [ctx]'s lists are what the lab holds:
4530
- * `{ profiles, sources, datasets }`, each `{ id, name }[]`. Returns
4903
+ * `{ profiles, sources, groups }`, each `{ id, name }[]`. Returns
4531
4904
  * `{ doc, missing }`, where `missing` is one sentence per reference neither
4532
4905
  * id nor name matched, or `{ error }` when the document is refused: not a
4533
4906
  * version the lab reads, or not something the lab can edit and run as it
@@ -4538,14 +4911,14 @@ function importPipeline(input , ctx = {}) {
4538
4911
  const doc = upgradePipeline(input, ctx);
4539
4912
  const v = versionProblem(doc);
4540
4913
  if (v) return { error: v };
4541
- const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], datasets = ctx.datasets ?? [];
4914
+ const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], groups = ctx.groups ?? [];
4542
4915
  const next = clone(doc) ;
4543
4916
  const missing = [], seen = new Set ();
4544
- const lists = { profile: profiles, source: sources, dataset: datasets };
4917
+ const lists = { profile: profiles, source: sources, group: groups };
4545
4918
  const words = {
4546
- profile: (ref ) => `Target profile ${ref.name || ref.id} not found`,
4919
+ profile: (ref ) => `Target profile ${ref.slug || ref.name || ref.id} not found`,
4547
4920
  source: (ref ) => `Source ${ref.name || ref.id} not found`,
4548
- dataset: (ref ) => `Dataset ${ref.name || ref.id} not found`,
4921
+ group: (ref ) => `Eval group ${ref.name || ref.id} not found`,
4549
4922
  };
4550
4923
  const remap = (ref , kind ) => {
4551
4924
  const hit = lists[kind].find(x => x.id === ref.id) || lists[kind].find(x => x.name === ref.name);
@@ -4554,16 +4927,34 @@ function importPipeline(input , ctx = {}) {
4554
4927
  if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
4555
4928
  return ref;
4556
4929
  };
4930
+ // A profile is named by its slug in a file this lab writes (#253), and by
4931
+ // its id in one written before slugs: a slug is matched by slug, then by
4932
+ // name. One this lab holds under neither is kept as a reference the run
4933
+ // is blocked on, the slug standing in for an id.
4934
+ const slugged = withSlugs(profiles);
4935
+ const remapProfile = (ref ) => {
4936
+ if (!isObj(ref) || !isStr(ref.slug) || isRef(ref)) return isRef(ref) ? remap(ref, "profile") : ref;
4937
+ const hit = slugged.find(x => x.slug === ref.slug) || slugged.find(x => isStr(ref.name) && x.name === ref.name);
4938
+ if (hit) return { id: hit.id, name: hit.name };
4939
+ const sentence = words.profile(ref );
4940
+ if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
4941
+ return { id: ref.slug, name: isStr(ref.name) && ref.name ? ref.name : ref.slug };
4942
+ };
4557
4943
  const content = contentOf(next) ;
4558
4944
  if (content?.type === "source") content.ref = remap(content.ref, "source");
4559
4945
  for (const t of targetsOf(next)) {
4560
- if (t.profile) t.profile = remap(t.profile, "profile");
4946
+ if (t.profile) t.profile = remapProfile(t.profile);
4561
4947
  for (const st of Array.isArray(t.steps) ? t.steps : []) {
4562
- if (st?.profile) st.profile = remap(st.profile, "profile");
4948
+ if (st?.profile) st.profile = remapProfile(st.profile);
4563
4949
  }
4564
4950
  }
4565
4951
  for (const t of Array.isArray(next.evals) ? next.evals : []) {
4566
- if (isRef(t?.dataset)) t.dataset = remap(t.dataset, "dataset");
4952
+ if (isRef(t?.group)) t.group = remap(t.group, "group");
4953
+ else if (isObj(t?.own) && isRef(t.own.casesFrom)) t.own.casesFrom = remap(t.own.casesFrom, "group");
4954
+ // The grader is a profile like any other, matched the same way: a
4955
+ // private group's own, or the one a run's link carries.
4956
+ if (isObj(t?.own) && t.own.grader) t.own.grader = remapProfile(t.own.grader);
4957
+ if (isObj(t) && t.grader) t.grader = remapProfile(t.grader);
4567
4958
  }
4568
4959
  // Its references are this lab's now, so a step asking this lab's Echo
4569
4960
  // profile is read as an Echo step (upgradePipeline did it for ids that
@@ -4597,6 +4988,8 @@ function mintIds(doc ) {
4597
4988
  for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
4598
4989
  if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
4599
4990
  }
4991
+ // Nor need it say how its linked groups pass together: all of them.
4992
+ if (!("pass" in doc)) doc.pass = { mode: "all" };
4600
4993
  return doc;
4601
4994
  }
4602
4995
 
@@ -4610,6 +5003,224 @@ function yamlToPipeline(text , ctx ) {
4610
5003
  return importPipeline(doc, ctx);
4611
5004
  }
4612
5005
 
5006
+ // ---- JUnit (#251) --------------------------------------------------------------
5007
+ //
5008
+ // What a CI test panel reads, from the worker's report and from a lab's
5009
+ // stored run alike (`evals-lab run --lab`, #256), so both write one format.
5010
+ // A suite is one target's eval, a case one item: a failure carries the reason
5011
+ // the item failed, an item that never ran is an error, and one the eval had
5012
+ // nothing to read of is skipped.
5013
+
5014
+ /** One JUnit test suite: a target's eval, and a case per item it read. */
5015
+
5016
+
5017
+
5018
+
5019
+
5020
+ // XML 1.0 has no way to write most control characters, escaped or not, and a
5021
+ // model's reply can hold any of them.
5022
+ const xmlText = (v ) => String(v).replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFE\uFFFF]/g, "")
5023
+ .replace(/[&<>"']/g, c => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", '"': "&quot;", "'": "&apos;" })[c] );
5024
+
5025
+ /** [suites] as JUnit XML, under the name of what ran them. */
5026
+ function junitXml(suites , runner ) {
5027
+ const count = (cases , k ) => cases.filter(c => c[k] != null).length;
5028
+ const body = suites.map(st => {
5029
+ const cases = st.cases.map(c => ` <testcase classname="${xmlText(st.name)}" name="${xmlText(c.name)}">`
5030
+ + (c.failure != null ? `<failure message="${xmlText(c.failure)}"/>`
5031
+ : c.error != null ? `<error message="${xmlText(c.error)}"/>`
5032
+ : c.skipped != null ? `<skipped message="${xmlText(c.skipped)}"/>` : "")
5033
+ + "</testcase>");
5034
+ return ` <testsuite name="${xmlText(st.name)}" tests="${st.cases.length}" failures="${count(st.cases, "failure")}" `
5035
+ + `errors="${count(st.cases, "error")}" skipped="${count(st.cases, "skipped")}">\n`
5036
+ + cases.map(c => c + "\n").join("") + " </testsuite>\n";
5037
+ });
5038
+ const all = suites.flatMap(st => st.cases);
5039
+ return `<?xml version="1.0" encoding="UTF-8"?>\n<testsuites name="${xmlText(runner)}" tests="${all.length}" `
5040
+ + `failures="${count(all, "failure")}" errors="${count(all, "error")}" skipped="${count(all, "skipped")}">\n`
5041
+ + body.join("") + "</testsuites>\n";
5042
+ }
5043
+
5044
+ // ---- a pipeline as a CI bundle (#254) --------------------------------------
5045
+ //
5046
+ // Export for CI writes a pipeline as files a repository keeps and
5047
+ // `evals-lab run` runs with no lab around it (docs/pipeline-yaml.md, "Export
5048
+ // for CI"): the pipeline's YAML, its Target profiles by slug, and each
5049
+ // dataset it grades against as the Library's export of it. The server adds
5050
+ // the plugins and, when asked, the Source's items, and zips the lot. Read
5051
+ // back, the files are the run document the page would have submitted, with
5052
+ // each profile keyed by its slug rather than this lab's id -- the slug is
5053
+ // what the key's $EVALSLAB_API_KEY_<SLUG> is spelt from either way.
5054
+
5055
+ /** The text files of a bundle, by path inside it. */
5056
+
5057
+
5058
+ /** What a bundle is read back as: the run document, and each dataset it
5059
+ grades against, by the id its evals name it by. */
5060
+
5061
+
5062
+
5063
+
5064
+
5065
+ const BUNDLE_PIPELINE = "pipeline.yaml";
5066
+ const BUNDLE_PROFILES = "profiles.yaml";
5067
+ const BUNDLE_DATASET = /^datasets\/([a-z0-9]+(?:-[a-z0-9]+)*)\.json$/;
5068
+
5069
+ /**
5070
+ * [doc] as a bundle's text files: `pipeline.yaml`, `profiles.yaml` and
5071
+ * `datasets/<slug>.json`. [ctx.profiles] is Setup's list -- keys and all,
5072
+ * none of which is written -- [ctx.grader] the lab's grader, written into
5073
+ * each eval that would have asked it, and [ctx.datasets] each graded
5074
+ * dataset by id, as `{ name, body }`. `{ error }` when a dataset or a
5075
+ * profile the pipeline names is not given, since a bundle without it runs
5076
+ * something else.
5077
+ */
5078
+ function exportBundle(doc , ctx
5079
+ = {})
5080
+ {
5081
+ const profiles = withSlugs((ctx.profiles ?? []) );
5082
+ // What the lab would supply at submit, written down: no lab supplies it later.
5083
+ const out = clone(doc) ;
5084
+ out.evals = (Array.isArray(out.evals) ? out.evals : [])
5085
+ .map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
5086
+ const table = {};
5087
+ for (const id of profileIds(out)) {
5088
+ const p = profiles.find(x => x.id === id);
5089
+ if (!p) return { error: `Target profile ${id} not found` };
5090
+ const { slug, ...flat } = connectionSettings(connectionOf(p) );
5091
+ table[slug ] = flat;
5092
+ }
5093
+ const files = {};
5094
+ const taken = new Set (), written = new Set ();
5095
+ for (const t of evalsOf(out) ) {
5096
+ // A link's group, or a private group's Cases from: the reference itself.
5097
+ const ref = casesRef(t);
5098
+ if (!ref) continue;
5099
+ const d = ctx.datasets?.[ref.id];
5100
+ if (!d) return { error: `Dataset ${ref.name || ref.id} not found` };
5101
+ // Named as the dataset is now, since the name is what a reader joins on.
5102
+ ref.name = d.name;
5103
+ if (written.has(ref.id)) continue;
5104
+ const slug = uniqueSlug(slugOf(d.name) || "dataset", taken);
5105
+ taken.add(slug);
5106
+ written.add(ref.id);
5107
+ // The Library's Export of it, so the file imports into any lab too. Its
5108
+ // name is what joins it back to the eval naming it: a lab's dataset
5109
+ // names are its own.
5110
+ files[`datasets/${slug}.json`] = JSON.stringify({ format: "evals-lab/dataset", version: DATASET_BODY_VERSION,
5111
+ dataset: { name: d.name, body: d.body } }, null, 2) + "\n";
5112
+ }
5113
+ files[BUNDLE_PIPELINE] = pipelineToYaml(out, { profiles });
5114
+ files[BUNDLE_PROFILES] = yaml.dump(table, { lineWidth: -1, noRefs: true });
5115
+ const bad = bundleProblems(files, profiles.map(p => p.key));
5116
+ return bad.length ? { error: bad[0] } : { files };
5117
+ }
5118
+
5119
+ /**
5120
+ * Why [files] cannot leave the lab: a field named like a key anywhere in
5121
+ * them, or any of [keys] -- Setup's own -- written anywhere at all. Empty
5122
+ * when nothing key-shaped is in them. The server asks the same of the zip
5123
+ * it is handed (server.py `bundle_problems`).
5124
+ */
5125
+ function bundleProblems(files , keys = []) {
5126
+ const bad = [];
5127
+ const named = (v , at ) => {
5128
+ if (Array.isArray(v)) return v.forEach(x => named(x, at));
5129
+ if (!isObj(v)) return;
5130
+ for (const [k, x] of Object.entries(v)) {
5131
+ if (LOOKS_LIKE_A_KEY.test(k)) bad.push(`${at} holds a field named ${k}, and a key never leaves the lab`);
5132
+ named(x, at);
5133
+ }
5134
+ };
5135
+ for (const [path, text] of Object.entries(files)) {
5136
+ for (const key of keys) {
5137
+ if (isStr(key) && key.trim().length >= 4 && text.includes(key.trim())) {
5138
+ bad.push(`${path} holds a Target profile's key, and a key never leaves the lab`);
5139
+ break;
5140
+ }
5141
+ }
5142
+ let doc ;
5143
+ try { doc = path.endsWith(".json") ? JSON.parse(text) : yaml.load(text); } catch { continue; }
5144
+ named(doc, path);
5145
+ }
5146
+ return [...new Set(bad)];
5147
+ }
5148
+
5149
+ /**
5150
+ * A bundle's [files] as the run they describe: `pipeline.yaml` read as an
5151
+ * import would read it, against `profiles.yaml` and the `datasets/` it
5152
+ * carries, then resolved as the page resolves one at submit. [ctx.items] is
5153
+ * the Source's file list -- the names in the bundle's `items/`, or a
5154
+ * directory CI names; with none, each case's item is the list, so an item
5155
+ * nothing supplies is a missing item rather than an item never asked.
5156
+ * `{ error }` names the first thing that keeps it from running: a profile
5157
+ * or a dataset the files do not hold, a key in them, or a document the lab
5158
+ * would refuse. The caller registers the bundle's plugins first.
5159
+ */
5160
+ function readBundle(files , ctx = {}) {
5161
+ if (!isStr(files[BUNDLE_PIPELINE])) return { error: `the bundle has no ${BUNDLE_PIPELINE}` };
5162
+ let table = {};
5163
+ try { table = isStr(files[BUNDLE_PROFILES]) ? yaml.load(files[BUNDLE_PROFILES] ) ?? {} : {}; }
5164
+ catch { return { error: `${BUNDLE_PROFILES} is not YAML` }; }
5165
+ if (!isObj(table)) return { error: `${BUNDLE_PROFILES} maps each profile's slug to its settings` };
5166
+ const keyed = bundleProblems(files);
5167
+ if (keyed.length) return { error: keyed[0] };
5168
+ const profiles = [];
5169
+ for (const [slug, p] of Object.entries(table)) {
5170
+ if (!isSlug(slug)) return { error: `${BUNDLE_PROFILES}: ${JSON.stringify(slug)} is not a slug` };
5171
+ if (!isObj(p)) return { error: `${BUNDLE_PROFILES}: ${slug} has to hold its settings` };
5172
+ if (!CONNECTION_TYPES[typeOf(p )]) {
5173
+ return { error: `${BUNDLE_PROFILES}: ${slug} is of type ${p.type}, which this lab does not have` };
5174
+ }
5175
+ profiles.push({ ...p, id: slug, slug, name: isStr(p.name) && p.name ? p.name : slug });
5176
+ }
5177
+ // Each dataset by its name, which is what joins it to the eval naming it.
5178
+ const datasets = [];
5179
+ for (const [path, text] of Object.entries(files)) {
5180
+ if (!path.startsWith("datasets/")) continue;
5181
+ if (!BUNDLE_DATASET.test(path)) return { error: `${path} is not named datasets/<slug>.json` };
5182
+ let doc ;
5183
+ try { doc = JSON.parse(text); } catch { return { error: `${path} is not JSON` }; }
5184
+ if (!isObj(doc) || doc.format !== "evals-lab/dataset" || !isObj(doc.dataset)) {
5185
+ return { error: `${path} is not a dataset's export` };
5186
+ }
5187
+ if (!(Number.isInteger(doc.version) && doc.version >= 1 && doc.version <= DATASET_BODY_VERSION)) {
5188
+ return { error: `${path} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to ${DATASET_BODY_VERSION}` };
5189
+ }
5190
+ const body = upgradeDatasetBody(doc.dataset.body) ;
5191
+ if (!isObj(body) || !Array.isArray(body.cases)) return { error: `${path} holds no cases` };
5192
+ datasets.push({ name: String(doc.dataset.name ?? ""), body: body });
5193
+ }
5194
+ let raw ;
5195
+ try { raw = yaml.load(files[BUNDLE_PIPELINE] ); }
5196
+ catch { return { error: `${BUNDLE_PIPELINE} is not YAML` }; }
5197
+ // The bundle stands in for the Source: whatever it names, the items are
5198
+ // the ones handed over.
5199
+ const content = isObj(raw) ? contentOf(upgradePipeline(raw)) : null;
5200
+ const refs = [];
5201
+ // Read as this version, whichever wrote the bundle: a link's group, or a
5202
+ // group of its own's Cases from.
5203
+ for (const t of isObj(raw) ? evalsOf(upgradePipeline(raw)) : []) {
5204
+ const ref = casesRef(t);
5205
+ if (!ref) continue;
5206
+ if (datasets.some(d => d.name === ref.name)) refs.push({ id: ref.id, name: ref.name });
5207
+ }
5208
+ const imported = importPipeline(raw, { profiles, groups: refs,
5209
+ sources: content?.type === "source" && isRef(content.ref) ? [content.ref] : [] });
5210
+ if ("error" in imported) return imported;
5211
+ if (imported.missing.length) return { error: `${imported.missing[0]}: the bundle does not hold it` };
5212
+ const byId = {};
5213
+ for (const r of refs) byId[r.id] = datasets.find(d => d.name === r.name) .body;
5214
+ const caseItems = [...new Set(Object.values(byId).flatMap(b => gradedSetFrom(b ).filter(c => !c.todo).map(caseItem)).filter(Boolean))];
5215
+ const run = resolvePipeline(imported.doc, {
5216
+ profiles: id => profiles.find(p => p.id === id) ?? null,
5217
+ files: ctx.items ? [...ctx.items].sort() : caseItems,
5218
+ });
5219
+ const bad = validatePipeline(run, { groups: refs });
5220
+ if (bad.length) return { error: bad[0] };
5221
+ return { run, datasets: byId };
5222
+ }
5223
+
4613
5224
  /**
4614
5225
  * Scenario [i] of a run document as runPipeline and a transport take it:
4615
5226
  * each stage's wording, kind, image flag and modifiers, the token set of
@@ -4995,15 +5606,15 @@ export {
4995
5606
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
4996
5607
  PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
4997
5608
  SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
4998
- registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
5609
+ registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
4999
5610
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5000
5611
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5001
5612
  contentOf, withContent, replyOf,
5002
5613
  targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5003
5614
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
5004
- blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5615
+ blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5005
5616
  scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
5006
- evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
5007
- pipelineToYaml, importPipeline, yamlToPipeline,
5617
+ evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
5618
+ pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
5008
5619
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
5009
5620
  };