evals-lab 0.4.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/README.md +125 -28
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +531 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +672 -61
- package/lab/run-evals.js +195 -57
- package/lab/server.py +681 -127
- package/lab/web/dist/assets/gallery-BFf9vis6.js +3 -0
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/main-DeeRLWnO.css +1 -0
- package/lab/web/dist/assets/main-LT0U2TYF.js +21 -0
- package/lab/web/dist/assets/tokens-CCEtCZtQ.js +59 -0
- package/lab/web/dist/assets/tokens-lq45aAPS.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-BkZTEix2.js +0 -21
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/evals-core.mjs
CHANGED
|
@@ -275,9 +275,41 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
275
275
|
|
|
276
276
|
|
|
277
277
|
|
|
278
|
-
/**
|
|
279
|
-
|
|
280
|
-
|
|
278
|
+
/** A private group (docs/pipeline-model.md §17): an eval group's body held
|
|
279
|
+
in the pipeline rather than the Library -- what a version-12 Metrics
|
|
280
|
+
eval upgrades to when it is not a plain link. `casesFrom` is the Library
|
|
281
|
+
group whose cases, at its newest version, join this group's Every item
|
|
282
|
+
under this group's scoring; only the upgrade writes one. */
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
/** A Library eval group by reference; a run's copy says the version it
|
|
289
|
+
graded with: the body's fingerprint, and its number. */
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
/** An eval, from version 13: a link to an eval group (§17). `group` names a
|
|
293
|
+
Library group, followed at its newest version (`pin: null`) or pinned at
|
|
294
|
+
one; a private group is `own`, with `group: null`. A run's copy carries
|
|
295
|
+
the version it graded with on the reference (`version`, the body's
|
|
296
|
+
fingerprint, and `n`), and the lab's grader where the group names none. */
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
/** An eval, from version 13: a link to an eval group. A Metrics eval
|
|
306
|
+
(version 9 to 12), a Single Test or a Graded set is what an older
|
|
307
|
+
document held (upgradePipeline reads it converted). */
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
/** The rule a pipeline's linked groups pass by, together: every one, or at
|
|
311
|
+
least `count` of them. */
|
|
312
|
+
|
|
281
313
|
|
|
282
314
|
/** A pipeline's evals, in the order they read a run. */
|
|
283
315
|
|
|
@@ -293,11 +325,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
293
325
|
|
|
294
326
|
|
|
295
327
|
|
|
328
|
+
|
|
296
329
|
|
|
297
330
|
|
|
298
331
|
/** A Setup profile as a run carries it: request settings, never the key. */
|
|
299
332
|
|
|
300
333
|
|
|
334
|
+
|
|
335
|
+
|
|
301
336
|
|
|
302
337
|
|
|
303
338
|
|
|
@@ -338,6 +373,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
338
373
|
|
|
339
374
|
|
|
340
375
|
|
|
376
|
+
|
|
341
377
|
|
|
342
378
|
|
|
343
379
|
|
|
@@ -364,11 +400,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
364
400
|
|
|
365
401
|
|
|
366
402
|
|
|
403
|
+
/** A Target profile as an import or export matches it: its slug, where it
|
|
404
|
+
has one -- withSlugs gives the rest theirs. */
|
|
405
|
+
|
|
406
|
+
|
|
367
407
|
/** What an import matches references against: { id, name } each. */
|
|
368
408
|
|
|
369
|
-
|
|
409
|
+
|
|
370
410
|
|
|
371
|
-
|
|
411
|
+
|
|
372
412
|
|
|
373
413
|
|
|
374
414
|
|
|
@@ -817,8 +857,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
817
857
|
|
|
818
858
|
|
|
819
859
|
/** What validation looks references up in: `profiles(id)` and `sources(id)`
|
|
820
|
-
answer with what the id names, or nothing; `
|
|
821
|
-
lab holds
|
|
860
|
+
answer with what the id names, or nothing; `groups` is the eval groups
|
|
861
|
+
the lab holds. Each is optional, and a
|
|
822
862
|
reference nothing is given to look up is not checked. */
|
|
823
863
|
|
|
824
864
|
|
|
@@ -826,7 +866,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
826
866
|
|
|
827
867
|
|
|
828
868
|
|
|
829
|
-
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
|
|
830
873
|
|
|
831
874
|
|
|
832
875
|
/** Lookups, and whether it is a run document being validated. */
|
|
@@ -904,6 +947,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
904
947
|
|
|
905
948
|
|
|
906
949
|
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
|
|
907
953
|
|
|
908
954
|
|
|
909
955
|
|
|
@@ -917,6 +963,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
917
963
|
|
|
918
964
|
|
|
919
965
|
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
|
|
920
969
|
|
|
921
970
|
|
|
922
971
|
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
@@ -2687,7 +2736,10 @@ function applyModifiers (list , kind ,
|
|
|
2687
2736
|
// scenario is a target, whose own step in each job is what it sends there.
|
|
2688
2737
|
// 11: `tests` are `evals`: the key renames and nothing in an eval changes,
|
|
2689
2738
|
// so a stored result's scores, keyed by eval id, read as they did.
|
|
2690
|
-
|
|
2739
|
+
// 12: a Contains metric's Ignore case holds item by item too.
|
|
2740
|
+
// 13: an eval is a link to an eval group, or a group of the pipeline's own,
|
|
2741
|
+
// and the document has an overall pass rule (pipeline-model §17).
|
|
2742
|
+
const PIPELINE_VERSION = 13 ;
|
|
2691
2743
|
|
|
2692
2744
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2693
2745
|
// to know which registered it.
|
|
@@ -2780,11 +2832,13 @@ OUTPUT_KINDS.text = {
|
|
|
2780
2832
|
|
|
2781
2833
|
// ---- the connection a profile reference resolves to ------------------------
|
|
2782
2834
|
// What a run carries of a Setup profile: everything a request needs and never
|
|
2783
|
-
// the key. A key follows the profile's
|
|
2784
|
-
//
|
|
2785
|
-
//
|
|
2786
|
-
//
|
|
2787
|
-
|
|
2835
|
+
// the key. A key follows the profile's slug into $EVALSLAB_API_KEY_<SLUG> -- or,
|
|
2836
|
+
// in a run document from before slugs, its id into $EVALSLAB_API_KEY_<ID> --
|
|
2837
|
+
// which the server sets from its profiles store and the runner reads, so a
|
|
2838
|
+
// name is only a label and two profiles can share one. #46: the profile
|
|
2839
|
+
// carries `type`, and the run's resolved copy carries it too
|
|
2840
|
+
// (docs/pipeline-model.md §7).
|
|
2841
|
+
const CONNECTION_FIELDS = ["name", "slug", "url", "model", "type", "temperature",
|
|
2788
2842
|
"px", "format", "quality", "options"];
|
|
2789
2843
|
const SETTING_KEYS = [...new Set(Object.values(CONNECTION_TYPES).flatMap(t => t.settings.map(s => s.key)))];
|
|
2790
2844
|
const OPTION_FIELDS = ["seed", "nPredict", ...DECODING_KEYS];
|
|
@@ -2794,9 +2848,69 @@ const IMAGE_FORMATS = ["image/jpeg", "image/webp", "image/png
|
|
|
2794
2848
|
const PROFILE_ID = /^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?$/;
|
|
2795
2849
|
const LOOKS_LIKE_A_KEY = /^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$/i;
|
|
2796
2850
|
|
|
2797
|
-
/** The variable a profile's key is read from:
|
|
2798
|
-
|
|
2799
|
-
|
|
2851
|
+
/** The variable a profile's key is read from: its slug's, when it has one --
|
|
2852
|
+
anthropic is EVALSLAB_API_KEY_ANTHROPIC -- else its id's, as a run document
|
|
2853
|
+
from before slugs has it: pm1x8k2q is EVALSLAB_API_KEY_PM1X8K2Q. */
|
|
2854
|
+
const keyVar = (id , slug ) => "EVALSLAB_API_KEY_"
|
|
2855
|
+
+ String(slug || id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
|
|
2856
|
+
|
|
2857
|
+
// ---- a profile's slug (#253) ------------------------------------------------
|
|
2858
|
+
// What a profile is called outside the lab: in a pipeline file and in the
|
|
2859
|
+
// name of the variable CI keeps its key in. An id (pm1x9c0r) means nothing to
|
|
2860
|
+
// whoever sets that secret; a slug (anthropic) does, and stays put when the
|
|
2861
|
+
// lab it came from does not. Lowercase letters and digits, hyphens between,
|
|
2862
|
+
// so its variable is spelt one way and two slugs never spell the same one.
|
|
2863
|
+
const SLUG = /^[a-z0-9]+(?:-[a-z0-9]+)*$/;
|
|
2864
|
+
const SLUG_MAX = 32;
|
|
2865
|
+
|
|
2866
|
+
/** [text] as a slug: lowercase, every other run of characters one hyphen,
|
|
2867
|
+
cut to SLUG_MAX. "" when nothing of it is a letter or a digit. */
|
|
2868
|
+
function slugOf(text ) {
|
|
2869
|
+
return String(text ?? "").toLowerCase().normalize("NFKD").replace(/[\u0300-\u036f]/g, "")
|
|
2870
|
+
.replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, SLUG_MAX).replace(/-+$/, "");
|
|
2871
|
+
}
|
|
2872
|
+
|
|
2873
|
+
/** [base] made unique among [taken]: anthropic, then anthropic-2, -3. */
|
|
2874
|
+
function uniqueSlug(base , taken ) {
|
|
2875
|
+
const root = base || "profile";
|
|
2876
|
+
if (!taken.has(root)) return root;
|
|
2877
|
+
for (let n = 2; ; n++) {
|
|
2878
|
+
const tail = `-${n}`;
|
|
2879
|
+
const s = root.slice(0, SLUG_MAX - tail.length).replace(/-+$/, "") + tail;
|
|
2880
|
+
if (!taken.has(s)) return s;
|
|
2881
|
+
}
|
|
2882
|
+
}
|
|
2883
|
+
|
|
2884
|
+
/**
|
|
2885
|
+
* [list] with every profile holding a slug of its own: one it holds already
|
|
2886
|
+
* is kept, tidied into a slug (Setup's field holds what was typed, a
|
|
2887
|
+
* trailing hyphen and all), while no profile before it holds the same; any
|
|
2888
|
+
* other is made from its name, made unique. The slugs held are settled
|
|
2889
|
+
* first, so a slug minted for one profile never takes another's. The same
|
|
2890
|
+
* array when nothing changed, so a reader can tell a store that needs
|
|
2891
|
+
* writing back from one that does not.
|
|
2892
|
+
*/
|
|
2893
|
+
function withSlugs (list ) {
|
|
2894
|
+
const taken = new Set ();
|
|
2895
|
+
const held = list.map(p => {
|
|
2896
|
+
const s = slugOf(p?.slug);
|
|
2897
|
+
if (!s || taken.has(s)) return "";
|
|
2898
|
+
taken.add(s);
|
|
2899
|
+
return s;
|
|
2900
|
+
});
|
|
2901
|
+
let changed = false;
|
|
2902
|
+
const out = list.map((p, i) => {
|
|
2903
|
+
const slug = held[i] || uniqueSlug(slugOf(p?.name), taken);
|
|
2904
|
+
taken.add(slug);
|
|
2905
|
+
if (slug === p?.slug) return p;
|
|
2906
|
+
changed = true;
|
|
2907
|
+
return { ...p, slug };
|
|
2908
|
+
});
|
|
2909
|
+
return changed ? out : list ;
|
|
2910
|
+
}
|
|
2911
|
+
|
|
2912
|
+
/** Whether [s] is a slug as a profile may hold one. */
|
|
2913
|
+
const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
|
|
2800
2914
|
|
|
2801
2915
|
/** A stored profile as a run carries it: its request settings, no key. */
|
|
2802
2916
|
function connectionOf(p ) {
|
|
@@ -2807,6 +2921,7 @@ function connectionOf(p ) {
|
|
|
2807
2921
|
// the options bag holds the type's own settings only.
|
|
2808
2922
|
for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
|
|
2809
2923
|
const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
|
|
2924
|
+
if (isSlug(p.slug)) out.slug = p.slug;
|
|
2810
2925
|
for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
|
|
2811
2926
|
if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
|
|
2812
2927
|
if (Object.keys(options).length) out.options = options;
|
|
@@ -2863,7 +2978,7 @@ function onlyFields(obj , at , allowed , bad
|
|
|
2863
2978
|
for (const k of Object.keys(obj)) {
|
|
2864
2979
|
if (allowed.includes(k)) continue;
|
|
2865
2980
|
bad.push(LOOKS_LIKE_A_KEY.test(k)
|
|
2866
|
-
? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "
|
|
2981
|
+
? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVALSLAB_API_KEY_<ID>"} supplies it`
|
|
2867
2982
|
: `${at} has "${k}", which is not a pipeline field`);
|
|
2868
2983
|
}
|
|
2869
2984
|
}
|
|
@@ -3026,7 +3141,7 @@ LEGACY_TESTS.graded = {
|
|
|
3026
3141
|
if (!isRef(d) || (d.version != null && !isStr(d.version))) {
|
|
3027
3142
|
return void bad.push("a graded eval has to name its dataset as { id, name }");
|
|
3028
3143
|
}
|
|
3029
|
-
if (ctx.
|
|
3144
|
+
if (ctx.groups && !ctx.groups.some(x => x.id === d.id)) bad.push(`Eval group ${d.name || d.id} not found`);
|
|
3030
3145
|
},
|
|
3031
3146
|
};
|
|
3032
3147
|
|
|
@@ -3169,7 +3284,10 @@ function readRun(list , run ) {
|
|
|
3169
3284
|
});
|
|
3170
3285
|
}
|
|
3171
3286
|
|
|
3172
|
-
|
|
3287
|
+
// The eval of versions 9 to 12, read only to upgrade (LEGACY_TESTS.metrics.
|
|
3288
|
+
// toGroup) and kept whole as the reference groups-check.js and
|
|
3289
|
+
// pipeline-check.js hold the eval group to.
|
|
3290
|
+
LEGACY_TESTS.metrics = {
|
|
3173
3291
|
label: "Metrics",
|
|
3174
3292
|
description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
|
|
3175
3293
|
fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
|
|
@@ -3187,7 +3305,7 @@ EVAL_TYPES.metrics = {
|
|
|
3187
3305
|
// The dataset whose cases a case metric reads, and whose own metrics join.
|
|
3188
3306
|
if (t.dataset != null) {
|
|
3189
3307
|
if (!isRef(t.dataset) || (t.dataset.version != null && !isStr(t.dataset.version))) bad.push("the Metrics name their dataset as { id, name }");
|
|
3190
|
-
else if (ctx.
|
|
3308
|
+
else if (ctx.groups && !ctx.groups.some(x => x.id === t.dataset.id)) bad.push(`Eval group ${t.dataset.name || t.dataset.id} not found`);
|
|
3191
3309
|
}
|
|
3192
3310
|
if (!SCORING_MODES.includes(t.mode)) bad.push(`the Metrics are scored ${SCORING_MODES.join(" or ")}`);
|
|
3193
3311
|
if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
|
|
@@ -3228,6 +3346,151 @@ EVAL_TYPES.metrics = {
|
|
|
3228
3346
|
metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
|
|
3229
3347
|
};
|
|
3230
3348
|
|
|
3349
|
+
/** A version-12 Metrics eval as the version-13 eval it reads as (§17): a
|
|
3350
|
+
link to its dataset, followed at the newest version, where it named one
|
|
3351
|
+
and held nothing of its own -- no metrics, scored All, the lab's grader,
|
|
3352
|
+
over each item -- and a private group holding its metrics, scoring and
|
|
3353
|
+
grader otherwise, with its dataset's cases (`casesFrom`). An eval over the
|
|
3354
|
+
whole run never read its dataset's cases, so its group takes none. The
|
|
3355
|
+
id, name and Continue on failure are kept, so a stored score keyed by
|
|
3356
|
+
the eval's id is the link's. */
|
|
3357
|
+
function groupOfMetrics(t ) {
|
|
3358
|
+
const { id, name, continueOnFailure } = t;
|
|
3359
|
+
const common = { id, ...(name !== undefined ? { name } : {}), type: "group", continueOnFailure };
|
|
3360
|
+
const metrics = Array.isArray(t.metrics) ? t.metrics : [];
|
|
3361
|
+
const run = t.over === "run";
|
|
3362
|
+
if (isRef(t.dataset) && !metrics.length && t.mode === "all" && t.grader == null && !run) {
|
|
3363
|
+
return { ...common, group: { ...t.dataset }, pin: null };
|
|
3364
|
+
}
|
|
3365
|
+
const own = { version: DATASET_BODY_VERSION, source: null, scoring: { mode: t.mode, threshold: t.threshold ?? null },
|
|
3366
|
+
grader: t.grader ?? null, every: run ? [] : metrics, run: run ? metrics : [], cases: [],
|
|
3367
|
+
casesFrom: isRef(t.dataset) && !run ? { ...t.dataset } : null };
|
|
3368
|
+
return { ...common, group: null, pin: null, own };
|
|
3369
|
+
}
|
|
3370
|
+
LEGACY_TESTS.metrics.toGroup = groupOfMetrics;
|
|
3371
|
+
|
|
3372
|
+
/** The group an eval reads by: a private group's own body, or a link's
|
|
3373
|
+
Library group as the runner hands it ([more].group). A link whose body
|
|
3374
|
+
nothing hands over reads as a Library group upgraded from version 6 does
|
|
3375
|
+
-- no metrics of its own, scored All -- which is what every one was
|
|
3376
|
+
until a body could hold more. */
|
|
3377
|
+
function groupOf(t , more = {}) {
|
|
3378
|
+
if (isObj(t.own)) return t.own ;
|
|
3379
|
+
const body = isRef(t.group) ? more.group?.(t.group) : undefined;
|
|
3380
|
+
return body ?? { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
|
|
3381
|
+
every: [], run: [], cases: [] };
|
|
3382
|
+
}
|
|
3383
|
+
|
|
3384
|
+
/** The Library group whose cases an eval reads, where it reads one: a
|
|
3385
|
+
link's group, or a private group's `casesFrom`. */
|
|
3386
|
+
function casesRef(t ) {
|
|
3387
|
+
if (!isObj(t) || t.type !== "group") return null;
|
|
3388
|
+
if (isRef(t.group)) return t.group ;
|
|
3389
|
+
return isObj(t.own) && isRef(t.own.casesFrom) ? t.own.casesFrom : null;
|
|
3390
|
+
}
|
|
3391
|
+
|
|
3392
|
+
/** The metrics a private group shows as its rules: its Whole run where it
|
|
3393
|
+
reads the run, its Every item otherwise. */
|
|
3394
|
+
const ownList = (t ) => (isObj(t.own) ? (wholeRunGroup(t) ? t.own.run : t.own.every) ?? [] : []);
|
|
3395
|
+
|
|
3396
|
+
/** A private group read over the whole run: Whole run metrics, and nothing
|
|
3397
|
+
read item by item. A group holding both is read item by item until the
|
|
3398
|
+
runner reports a group's Whole run beside its items (#233). */
|
|
3399
|
+
const wholeRunGroup = (t ) => isObj(t.own) && Array.isArray(t.own.run) && t.own.run.length > 0
|
|
3400
|
+
&& !(Array.isArray(t.own.every) && t.own.every.length) && !(Array.isArray(t.own.cases) && t.own.cases.length) && !t.own.casesFrom;
|
|
3401
|
+
|
|
3402
|
+
// A link to an eval group (docs/pipeline-model.md §17): a Library group by
|
|
3403
|
+
// reference, or a private group held here (`own`). It reads each item as its
|
|
3404
|
+
// group does (readGroup), or -- a private group of Whole run metrics alone --
|
|
3405
|
+
// the whole run (readGroupRun).
|
|
3406
|
+
EVAL_TYPES.group = {
|
|
3407
|
+
label: "Eval group",
|
|
3408
|
+
description: "Checks of each reply, or of the whole run, from an eval group: linked from the Library, or this pipeline's own.",
|
|
3409
|
+
fields: ["type", "group", "pin", "own", "grader"],
|
|
3410
|
+
// A metric that reads terms (a case) reads only a kind that yields them.
|
|
3411
|
+
accepts: t => ([...ownList(t)].some((m ) => METRICS[m?.type]?.needsTerms)
|
|
3412
|
+
? Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms) : null),
|
|
3413
|
+
defaults: () => ({ type: "group", group: null, pin: null,
|
|
3414
|
+
own: { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
|
|
3415
|
+
every: [], run: [], cases: [], casesFrom: null } }),
|
|
3416
|
+
validate(t, ctx, bad){
|
|
3417
|
+
const g = t.group, own = t.own;
|
|
3418
|
+
const named = (ref , what ) => {
|
|
3419
|
+
const n = isObj(ref) ? ref.n : undefined;
|
|
3420
|
+
if (!isRef(ref) || (ref.version != null && !isStr(ref.version)) || (n != null && !Number.isInteger(n))) {
|
|
3421
|
+
return void bad.push(`${what} names its eval group as { id, name }`);
|
|
3422
|
+
}
|
|
3423
|
+
onlyFields(ref, what, ["id", "name", "version", "n"], bad);
|
|
3424
|
+
const held = ctx.groups?.find(x => x.id === ref.id);
|
|
3425
|
+
if (ctx.groups && !held) bad.push(`Eval group ${ref.name || ref.id} not found`);
|
|
3426
|
+
return held;
|
|
3427
|
+
};
|
|
3428
|
+
if (t.pin != null && !(Number.isInteger(t.pin) && t.pin > 0)) bad.push("a pin is a group's version: a whole number from 1");
|
|
3429
|
+
if (t.grader != null && !isRef(t.grader)) bad.push("the link names its grader as { id, name }");
|
|
3430
|
+
if (g != null) {
|
|
3431
|
+
if (own != null) return void bad.push("an eval links a Library group or holds its own, not both");
|
|
3432
|
+
const held = named(g, "the link");
|
|
3433
|
+
if (held && t.pin != null && typeof held.versions === "number" && t.pin > held.versions) {
|
|
3434
|
+
bad.push(`${g.name || g.id} has no version ${t.pin}`);
|
|
3435
|
+
}
|
|
3436
|
+
return;
|
|
3437
|
+
}
|
|
3438
|
+
if (!isObj(own)) return void bad.push("an eval links a Library group, or holds its own");
|
|
3439
|
+
if (t.pin != null) bad.push("a group of this pipeline's own versions with it, so it is never pinned");
|
|
3440
|
+
if (own.version !== DATASET_BODY_VERSION) bad.push(`the group's own body is version ${DATASET_BODY_VERSION}`);
|
|
3441
|
+
for (const key of ["scoring", "every", "run", "cases"]) if (own[key] == null) bad.push(`the group's own body has no ${key}`);
|
|
3442
|
+
onlyFields(own, "the group's own body", ["version", "source", "scoring", "grader", "every", "run", "cases", "casesFrom"], bad);
|
|
3443
|
+
if (bad.length) return;
|
|
3444
|
+
bad.push(...validateEvals(own));
|
|
3445
|
+
if (own.casesFrom != null) named(own.casesFrom, "Cases from");
|
|
3446
|
+
// Read item by item or over the run: one section at a time, until a
|
|
3447
|
+
// group's Whole run is reported beside its items (#233).
|
|
3448
|
+
if (own.run.length && !wholeRunGroup(t)) bad.push("the group's own body reads each item or the whole run, not both");
|
|
3449
|
+
// The group's own grader, or the lab's.
|
|
3450
|
+
const grader = isRef(own.grader) ? own.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3451
|
+
if (gradedIn(own.every) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
|
|
3452
|
+
// A grader is asked words, and needs a model to ask.
|
|
3453
|
+
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3454
|
+
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
3455
|
+
else if (p) {
|
|
3456
|
+
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3457
|
+
if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
|
|
3458
|
+
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
|
|
3459
|
+
}
|
|
3460
|
+
},
|
|
3461
|
+
rules: t => ownList(t).map((m , x ) => ({
|
|
3462
|
+
key: `m${x}`, label: (isStr(m.metric) && m.metric) || METRICS[m.type]?.label || m.type,
|
|
3463
|
+
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3464
|
+
.filter(Boolean).join(" ") })),
|
|
3465
|
+
profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
|
|
3466
|
+
// A run carries the lab's grader where the group names none and may ask
|
|
3467
|
+
// one: a model-graded metric of its own, or a case's. A link's group is
|
|
3468
|
+
// the Library's, so the run's copy of the link carries it.
|
|
3469
|
+
resolve(t, ctx){
|
|
3470
|
+
if (!isRef(ctx.grader)) return t;
|
|
3471
|
+
const grader = { id: ctx.grader.id, name: ctx.grader.name };
|
|
3472
|
+
if (isObj(t.own)) {
|
|
3473
|
+
return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
|
|
3474
|
+
}
|
|
3475
|
+
return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
|
|
3476
|
+
},
|
|
3477
|
+
wholeRun: wholeRunGroup,
|
|
3478
|
+
verdict: (t, ress, kind) => {
|
|
3479
|
+
const own = t.own ;
|
|
3480
|
+
return wholeRunVerdict(own.run, ress, kind, own.scoring.mode, own.scoring.threshold);
|
|
3481
|
+
},
|
|
3482
|
+
// Rule m{x} is the group's own metric x, read in that place on every item.
|
|
3483
|
+
ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
|
|
3484
|
+
: ownList(t).length === 1 ? score.pass : null),
|
|
3485
|
+
// Every item, with its case where the group reads the Library group's
|
|
3486
|
+
// cases and the item has one there.
|
|
3487
|
+
read: async (t, kase, res, more) => {
|
|
3488
|
+
const group = groupOf(t, more);
|
|
3489
|
+
const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
|
|
3490
|
+
return readGroup({ ...group, grader } , casesRef(t) ? kase : null, res, more);
|
|
3491
|
+
},
|
|
3492
|
+
};
|
|
3493
|
+
|
|
3231
3494
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3232
3495
|
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3233
3496
|
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
@@ -3849,11 +4112,16 @@ function evalsOf(doc )
|
|
|
3849
4112
|
return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
|
|
3850
4113
|
}
|
|
3851
4114
|
|
|
3852
|
-
/** The
|
|
3853
|
-
|
|
3854
|
-
|
|
3855
|
-
|
|
3856
|
-
|
|
4115
|
+
/** The Library group whose cases a document's evals read, where one does:
|
|
4116
|
+
a link's group, or a private group's Cases from -- or, in a stored run
|
|
4117
|
+
from before version 13, a Metrics eval's dataset. A run grades against
|
|
4118
|
+
one (validatePipeline says so), so the first names it. */
|
|
4119
|
+
function evalsDataset(doc ) {
|
|
4120
|
+
for (const t of evalsOf(doc) ) {
|
|
4121
|
+
const ref = casesRef(t) ?? (isObj(t) && isRef(t.dataset) ? t.dataset : null);
|
|
4122
|
+
if (ref) return ref;
|
|
4123
|
+
}
|
|
4124
|
+
return null;
|
|
3857
4125
|
}
|
|
3858
4126
|
|
|
3859
4127
|
STEP_TYPES.evals = {
|
|
@@ -3882,15 +4150,17 @@ STEP_TYPES.evals = {
|
|
|
3882
4150
|
+ `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
|
|
3883
4151
|
}
|
|
3884
4152
|
});
|
|
3885
|
-
//
|
|
3886
|
-
|
|
3887
|
-
|
|
4153
|
+
// The worker is handed one Library group's body to grade against
|
|
4154
|
+
// (server-side-runs §4) until it reads the body the queue keeps per
|
|
4155
|
+
// group (#233).
|
|
4156
|
+
const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
|
|
4157
|
+
if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
|
|
3888
4158
|
},
|
|
3889
4159
|
};
|
|
3890
4160
|
|
|
3891
4161
|
// ---- the document -------------------------------------------------------------
|
|
3892
4162
|
|
|
3893
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
|
|
4163
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals", "pass"];
|
|
3894
4164
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3895
4165
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3896
4166
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3999,6 +4269,12 @@ function newEval(type , name = "", fields = {})
|
|
|
3999
4269
|
/** Version 6 to 7: chains are jobs. The field renames in place -- so an
|
|
4000
4270
|
upgraded document reads the same, key for key -- and every job's `type`
|
|
4001
4271
|
becomes `"job"`, so an older document reads as one of today's everywhere. */
|
|
4272
|
+
/** A new eval linking the Library eval group [group], followed at its
|
|
4273
|
+
newest version. */
|
|
4274
|
+
function newLink(group , name = "") {
|
|
4275
|
+
return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
|
|
4276
|
+
}
|
|
4277
|
+
|
|
4002
4278
|
function jobsOf(next ) {
|
|
4003
4279
|
next.version = PIPELINE_VERSION;
|
|
4004
4280
|
if (Array.isArray(next.chains)) {
|
|
@@ -4113,6 +4389,26 @@ function caseAsWritten(next ) {
|
|
|
4113
4389
|
return next;
|
|
4114
4390
|
}
|
|
4115
4391
|
|
|
4392
|
+
/** Version 12 to 13: each Metrics eval is a link to an eval group or a
|
|
4393
|
+
group of the pipeline's own (groupOfMetrics), and the document passes
|
|
4394
|
+
when every linked group does. Pure: it needs nothing but the document,
|
|
4395
|
+
so every reader reads the same result. Every version before 13 ends
|
|
4396
|
+
here. */
|
|
4397
|
+
function groupsOf(next ) {
|
|
4398
|
+
next.version = PIPELINE_VERSION;
|
|
4399
|
+
if (Array.isArray(next.evals)) {
|
|
4400
|
+
next.evals = next.evals.map((t ) => (isObj(t) && t.type === "metrics" ? groupOfMetrics(t) : t));
|
|
4401
|
+
}
|
|
4402
|
+
if (!("pass" in next)) {
|
|
4403
|
+
// After the evals, where a reader of the document looks for it.
|
|
4404
|
+
const at = Object.keys(next).indexOf("evals");
|
|
4405
|
+
const entries = Object.entries(next);
|
|
4406
|
+
entries.splice(at < 0 ? entries.length : at + 1, 0, ["pass", { mode: "all" }]);
|
|
4407
|
+
return Object.fromEntries(entries);
|
|
4408
|
+
}
|
|
4409
|
+
return next;
|
|
4410
|
+
}
|
|
4411
|
+
|
|
4116
4412
|
function upgradePipeline (doc , ctx = {}) {
|
|
4117
4413
|
// A current document is read as it is, but for a profile reference the Runs
|
|
4118
4414
|
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
@@ -4123,24 +4419,26 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
4123
4419
|
out = clone(out) ;
|
|
4124
4420
|
profileRefs(out);
|
|
4125
4421
|
}
|
|
4126
|
-
return localSteps(
|
|
4422
|
+
return localSteps(out, ctx) ;
|
|
4423
|
+
}
|
|
4424
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
|
|
4425
|
+
if (doc.version === 11 || doc.version === 12) {
|
|
4426
|
+
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
|
|
4127
4427
|
}
|
|
4128
|
-
|
|
4129
|
-
|
|
4130
|
-
// Every version before 10 reads as version 9 first, then as 10, then as 11
|
|
4131
|
-
// and 12.
|
|
4428
|
+
// Every version before 10 reads as version 9 first, then as 10, then as 11,
|
|
4429
|
+
// 12 and 13.
|
|
4132
4430
|
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4133
4431
|
// earlier ones are cut on their way through nineOf.
|
|
4134
4432
|
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4135
|
-
return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
|
|
4433
|
+
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
|
|
4136
4434
|
}
|
|
4137
4435
|
|
|
4138
4436
|
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
4139
4437
|
are metrics now (dataset body version 5), which an eval naming the
|
|
4140
4438
|
dataset adds for each item, so the case's own metrics carry what that
|
|
4141
|
-
metric scored. Read so at every version
|
|
4142
|
-
|
|
4143
|
-
none does. */
|
|
4439
|
+
metric scored. Read so at every version up to 12, as a document saved
|
|
4440
|
+
before it went still holds it; version 13 holds no Metrics eval. The
|
|
4441
|
+
same document where none does. */
|
|
4144
4442
|
function withoutCaseMetric(doc ) {
|
|
4145
4443
|
const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
|
|
4146
4444
|
if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
|
|
@@ -4229,7 +4527,7 @@ function tokenMappingsFromV1(set ) {
|
|
|
4229
4527
|
function blankPipeline(opts = {}) {
|
|
4230
4528
|
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
4231
4529
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
4232
|
-
jobs: [job], targets: [], evals: [] };
|
|
4530
|
+
jobs: [job], targets: [], evals: [], pass: { mode: "all" } };
|
|
4233
4531
|
}
|
|
4234
4532
|
|
|
4235
4533
|
/**
|
|
@@ -4368,6 +4666,7 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4368
4666
|
}
|
|
4369
4667
|
});
|
|
4370
4668
|
STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
4669
|
+
passProblems(doc.pass, Array.isArray(doc.evals) ? doc.evals.length : 0, bad);
|
|
4371
4670
|
if (bad.length) return bad;
|
|
4372
4671
|
|
|
4373
4672
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
@@ -4382,16 +4681,45 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4382
4681
|
return bad;
|
|
4383
4682
|
}
|
|
4384
4683
|
|
|
4684
|
+
/** The overall pass rule: every linked group, or at least [count] of the
|
|
4685
|
+
[n] the pipeline has. */
|
|
4686
|
+
function passProblems(pass , n , bad ) {
|
|
4687
|
+
if (!isObj(pass) || (pass.mode !== "all" && pass.mode !== "atLeast")) {
|
|
4688
|
+
return void bad.push("pass has to be { mode: all } or { mode: atLeast, count }");
|
|
4689
|
+
}
|
|
4690
|
+
onlyFields(pass, "pass", pass.mode === "all" ? ["mode"] : ["mode", "count"], bad);
|
|
4691
|
+
if (pass.mode === "atLeast" && !(Number.isInteger(pass.count) && pass.count > 0)) bad.push("pass at least has to count a whole number of groups from 1");
|
|
4692
|
+
else if (pass.mode === "atLeast" && pass.count > n) bad.push(`pass asks for ${pass.count} groups, and the pipeline has ${n}`);
|
|
4693
|
+
}
|
|
4694
|
+
|
|
4695
|
+
/**
|
|
4696
|
+
* What [doc] may still be right about, as sentences: a link to a group made
|
|
4697
|
+
* for another Source than the pipeline's content (§17, decision 2). The run
|
|
4698
|
+
* grades the cases whose items it holds, and the rest read Missing.
|
|
4699
|
+
*/
|
|
4700
|
+
function pipelineWarnings(doc , ctx = {}) {
|
|
4701
|
+
if (!isObj(doc) || !Array.isArray(doc.evals) || !ctx.groups) return [];
|
|
4702
|
+
const content = contentOf(doc ) ;
|
|
4703
|
+
const here = content?.type === "source" && isRef(content.ref) ? content.ref.id : null;
|
|
4704
|
+
const warn = [];
|
|
4705
|
+
doc.evals.forEach((t , j ) => {
|
|
4706
|
+
const ref = casesRef(t);
|
|
4707
|
+
const src = ref && ctx.groups .find(x => x.id === ref.id)?.source;
|
|
4708
|
+
if (src && src.id !== here) warn.push(`${evalLabel(doc , j)}: ${ref .name || ref .id} grades ${src.name || src.id}`);
|
|
4709
|
+
});
|
|
4710
|
+
return warn;
|
|
4711
|
+
}
|
|
4712
|
+
|
|
4385
4713
|
/** A run document's profiles table: id → connection, keyless, spellable. */
|
|
4386
4714
|
function profilesProblems(profiles , bad ) {
|
|
4387
4715
|
if (!isObj(profiles)) return void bad.push("profiles has to be an object of id → connection");
|
|
4388
4716
|
const spelt = new Map ();
|
|
4389
4717
|
for (const [id, conn] of Object.entries(profiles)) {
|
|
4390
4718
|
if (!PROFILE_ID.test(id)) {
|
|
4391
|
-
bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from
|
|
4719
|
+
bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVALSLAB_API_KEY_<ID>`);
|
|
4392
4720
|
continue;
|
|
4393
4721
|
}
|
|
4394
|
-
const from = keyVar(id);
|
|
4722
|
+
const from = keyVar(id, isObj(conn) && isStr(conn.slug) ? conn.slug : null);
|
|
4395
4723
|
if (spelt.has(from)) bad.push(`profiles ${spelt.get(from)} and ${id} both take their key from $${from}`);
|
|
4396
4724
|
spelt.set(from, id);
|
|
4397
4725
|
const at = `profile ${id}`;
|
|
@@ -4408,6 +4736,9 @@ function connectionProblems(conn , at , from
|
|
|
4408
4736
|
bad.push(`${at}: type has to be one of ${Object.keys(CONNECTION_TYPES).join(", ")}`);
|
|
4409
4737
|
}
|
|
4410
4738
|
if (!isStr(conn.name)) bad.push(`${at}: name has to be text`);
|
|
4739
|
+
if (conn.slug != null && !isSlug(conn.slug)) {
|
|
4740
|
+
bad.push(`${at}: slug has to be lowercase letters and digits, with hyphens between, at most ${SLUG_MAX}`);
|
|
4741
|
+
}
|
|
4411
4742
|
const url = conn.url ?? "";
|
|
4412
4743
|
if (!isStr(url)) bad.push(`${at}: url has to be text, blank for the lab's Ollama`);
|
|
4413
4744
|
else if (url.trim()) {
|
|
@@ -4477,7 +4808,8 @@ function profileIds(doc )
|
|
|
4477
4808
|
* the Source's file list as it reads now, `ctx.comment` is the run's own.
|
|
4478
4809
|
*/
|
|
4479
4810
|
function resolvePipeline(doc ,
|
|
4480
|
-
ctx
|
|
4811
|
+
ctx
|
|
4812
|
+
= {}) {
|
|
4481
4813
|
const run = clone(doc) ;
|
|
4482
4814
|
// What the lab supplies an eval -- its grader -- before the profiles it
|
|
4483
4815
|
// asks are carried.
|
|
@@ -4489,8 +4821,18 @@ function resolvePipeline(doc ,
|
|
|
4489
4821
|
}
|
|
4490
4822
|
const content = contentOf(run) ;
|
|
4491
4823
|
if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
|
|
4492
|
-
|
|
4493
|
-
|
|
4824
|
+
// Each Library group the evals read, as the version it grades with: the
|
|
4825
|
+
// body's fingerprint and, where the lab numbers them, `n` -- a pinned
|
|
4826
|
+
// link's pin.
|
|
4827
|
+
if (ctx.groupVersion) {
|
|
4828
|
+
for (const t of run.evals || []) {
|
|
4829
|
+
const ref = casesRef(t);
|
|
4830
|
+
const at = ref && ctx.groupVersion(ref.id);
|
|
4831
|
+
if (!ref || !at) continue;
|
|
4832
|
+
ref.version = at.version;
|
|
4833
|
+
const n = isObj(t) && Number.isInteger(t.pin) ? t.pin : at.n;
|
|
4834
|
+
if (n != null) ref.n = n;
|
|
4835
|
+
}
|
|
4494
4836
|
}
|
|
4495
4837
|
if (isStr(ctx.comment)) run.comment = ctx.comment;
|
|
4496
4838
|
return run;
|
|
@@ -4504,7 +4846,12 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4504
4846
|
delete doc.plugins;
|
|
4505
4847
|
const content = contentOf(doc) ;
|
|
4506
4848
|
if (content) { delete content.files; delete content.revs; }
|
|
4507
|
-
|
|
4849
|
+
// What the run recorded of each group it read is the run's. The grader
|
|
4850
|
+
// it carried stays, as a Metrics eval's did: a bundle writes it down.
|
|
4851
|
+
for (const t of doc.evals || []) {
|
|
4852
|
+
const ref = casesRef(t);
|
|
4853
|
+
if (ref) { delete ref.version; delete ref.n; }
|
|
4854
|
+
}
|
|
4508
4855
|
return doc;
|
|
4509
4856
|
}
|
|
4510
4857
|
|
|
@@ -4517,9 +4864,35 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4517
4864
|
|
|
4518
4865
|
|
|
4519
4866
|
/** [doc] as YAML text, the pipeline's file form -- references only, so an
|
|
4520
|
-
export carries no key, no URL, no file list and no file bytes.
|
|
4521
|
-
|
|
4522
|
-
|
|
4867
|
+
export carries no key, no URL, no file list and no file bytes. A Target
|
|
4868
|
+
profile [ctx] holds is named by its slug, `{ slug, name }`, rather than
|
|
4869
|
+
by this lab's id (#253): the slug is what another lab, or CI's
|
|
4870
|
+
$EVALSLAB_API_KEY_<SLUG>, knows it by. */
|
|
4871
|
+
function pipelineToYaml(doc , ctx = {}) {
|
|
4872
|
+
const profiles = withSlugs(ctx.profiles ?? []);
|
|
4873
|
+
if (!profiles.length) return yaml.dump(doc, { lineWidth: -1, noRefs: true });
|
|
4874
|
+
const out = clone(doc) ;
|
|
4875
|
+
const bySlug = (ref ) => {
|
|
4876
|
+
const p = isRef(ref) ? profiles.find(x => x.id === ref.id) : null;
|
|
4877
|
+
return p ? { slug: p.slug, name: ref.name ?? p.name } : ref;
|
|
4878
|
+
};
|
|
4879
|
+
eachProfileRef(out, bySlug);
|
|
4880
|
+
return yaml.dump(out, { lineWidth: -1, noRefs: true });
|
|
4881
|
+
}
|
|
4882
|
+
|
|
4883
|
+
/** Every Target profile reference in [doc] -- each target's, each of its
|
|
4884
|
+
steps', each eval's grader -- replaced, in place, by [f] of it. */
|
|
4885
|
+
function eachProfileRef(doc , f ) {
|
|
4886
|
+
for (const t of targetsOf(doc)) {
|
|
4887
|
+
if (t.profile) t.profile = f(t.profile);
|
|
4888
|
+
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4889
|
+
if (st?.profile) st.profile = f(st.profile);
|
|
4890
|
+
}
|
|
4891
|
+
}
|
|
4892
|
+
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4893
|
+
if (isObj(t) && t.grader) t.grader = f(t.grader);
|
|
4894
|
+
if (isObj(t?.own) && t.own.grader) t.own.grader = f(t.own.grader);
|
|
4895
|
+
}
|
|
4523
4896
|
}
|
|
4524
4897
|
|
|
4525
4898
|
/**
|
|
@@ -4527,7 +4900,7 @@ function pipelineToYaml(doc ) {
|
|
|
4527
4900
|
* name (pipeline-model §5), so a pipeline written in another lab -- ids that
|
|
4528
4901
|
* mean nothing here -- finds its Source and Setup profiles by the names they
|
|
4529
4902
|
* were exported under. [ctx]'s lists are what the lab holds:
|
|
4530
|
-
* `{ profiles, sources,
|
|
4903
|
+
* `{ profiles, sources, groups }`, each `{ id, name }[]`. Returns
|
|
4531
4904
|
* `{ doc, missing }`, where `missing` is one sentence per reference neither
|
|
4532
4905
|
* id nor name matched, or `{ error }` when the document is refused: not a
|
|
4533
4906
|
* version the lab reads, or not something the lab can edit and run as it
|
|
@@ -4538,14 +4911,14 @@ function importPipeline(input , ctx = {}) {
|
|
|
4538
4911
|
const doc = upgradePipeline(input, ctx);
|
|
4539
4912
|
const v = versionProblem(doc);
|
|
4540
4913
|
if (v) return { error: v };
|
|
4541
|
-
const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [],
|
|
4914
|
+
const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], groups = ctx.groups ?? [];
|
|
4542
4915
|
const next = clone(doc) ;
|
|
4543
4916
|
const missing = [], seen = new Set ();
|
|
4544
|
-
const lists = { profile: profiles, source: sources,
|
|
4917
|
+
const lists = { profile: profiles, source: sources, group: groups };
|
|
4545
4918
|
const words = {
|
|
4546
|
-
profile: (ref
|
|
4919
|
+
profile: (ref ) => `Target profile ${ref.slug || ref.name || ref.id} not found`,
|
|
4547
4920
|
source: (ref ) => `Source ${ref.name || ref.id} not found`,
|
|
4548
|
-
|
|
4921
|
+
group: (ref ) => `Eval group ${ref.name || ref.id} not found`,
|
|
4549
4922
|
};
|
|
4550
4923
|
const remap = (ref , kind ) => {
|
|
4551
4924
|
const hit = lists[kind].find(x => x.id === ref.id) || lists[kind].find(x => x.name === ref.name);
|
|
@@ -4554,16 +4927,34 @@ function importPipeline(input , ctx = {}) {
|
|
|
4554
4927
|
if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
|
|
4555
4928
|
return ref;
|
|
4556
4929
|
};
|
|
4930
|
+
// A profile is named by its slug in a file this lab writes (#253), and by
|
|
4931
|
+
// its id in one written before slugs: a slug is matched by slug, then by
|
|
4932
|
+
// name. One this lab holds under neither is kept as a reference the run
|
|
4933
|
+
// is blocked on, the slug standing in for an id.
|
|
4934
|
+
const slugged = withSlugs(profiles);
|
|
4935
|
+
const remapProfile = (ref ) => {
|
|
4936
|
+
if (!isObj(ref) || !isStr(ref.slug) || isRef(ref)) return isRef(ref) ? remap(ref, "profile") : ref;
|
|
4937
|
+
const hit = slugged.find(x => x.slug === ref.slug) || slugged.find(x => isStr(ref.name) && x.name === ref.name);
|
|
4938
|
+
if (hit) return { id: hit.id, name: hit.name };
|
|
4939
|
+
const sentence = words.profile(ref );
|
|
4940
|
+
if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
|
|
4941
|
+
return { id: ref.slug, name: isStr(ref.name) && ref.name ? ref.name : ref.slug };
|
|
4942
|
+
};
|
|
4557
4943
|
const content = contentOf(next) ;
|
|
4558
4944
|
if (content?.type === "source") content.ref = remap(content.ref, "source");
|
|
4559
4945
|
for (const t of targetsOf(next)) {
|
|
4560
|
-
if (t.profile) t.profile =
|
|
4946
|
+
if (t.profile) t.profile = remapProfile(t.profile);
|
|
4561
4947
|
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4562
|
-
if (st?.profile) st.profile =
|
|
4948
|
+
if (st?.profile) st.profile = remapProfile(st.profile);
|
|
4563
4949
|
}
|
|
4564
4950
|
}
|
|
4565
4951
|
for (const t of Array.isArray(next.evals) ? next.evals : []) {
|
|
4566
|
-
if (isRef(t?.
|
|
4952
|
+
if (isRef(t?.group)) t.group = remap(t.group, "group");
|
|
4953
|
+
else if (isObj(t?.own) && isRef(t.own.casesFrom)) t.own.casesFrom = remap(t.own.casesFrom, "group");
|
|
4954
|
+
// The grader is a profile like any other, matched the same way: a
|
|
4955
|
+
// private group's own, or the one a run's link carries.
|
|
4956
|
+
if (isObj(t?.own) && t.own.grader) t.own.grader = remapProfile(t.own.grader);
|
|
4957
|
+
if (isObj(t) && t.grader) t.grader = remapProfile(t.grader);
|
|
4567
4958
|
}
|
|
4568
4959
|
// Its references are this lab's now, so a step asking this lab's Echo
|
|
4569
4960
|
// profile is read as an Echo step (upgradePipeline did it for ids that
|
|
@@ -4597,6 +4988,8 @@ function mintIds(doc ) {
|
|
|
4597
4988
|
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4598
4989
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4599
4990
|
}
|
|
4991
|
+
// Nor need it say how its linked groups pass together: all of them.
|
|
4992
|
+
if (!("pass" in doc)) doc.pass = { mode: "all" };
|
|
4600
4993
|
return doc;
|
|
4601
4994
|
}
|
|
4602
4995
|
|
|
@@ -4610,6 +5003,224 @@ function yamlToPipeline(text , ctx ) {
|
|
|
4610
5003
|
return importPipeline(doc, ctx);
|
|
4611
5004
|
}
|
|
4612
5005
|
|
|
5006
|
+
// ---- JUnit (#251) --------------------------------------------------------------
|
|
5007
|
+
//
|
|
5008
|
+
// What a CI test panel reads, from the worker's report and from a lab's
|
|
5009
|
+
// stored run alike (`evals-lab run --lab`, #256), so both write one format.
|
|
5010
|
+
// A suite is one target's eval, a case one item: a failure carries the reason
|
|
5011
|
+
// the item failed, an item that never ran is an error, and one the eval had
|
|
5012
|
+
// nothing to read of is skipped.
|
|
5013
|
+
|
|
5014
|
+
/** One JUnit test suite: a target's eval, and a case per item it read. */
|
|
5015
|
+
|
|
5016
|
+
|
|
5017
|
+
|
|
5018
|
+
|
|
5019
|
+
|
|
5020
|
+
// XML 1.0 has no way to write most control characters, escaped or not, and a
|
|
5021
|
+
// model's reply can hold any of them.
|
|
5022
|
+
const xmlText = (v ) => String(v).replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFE\uFFFF]/g, "")
|
|
5023
|
+
.replace(/[&<>"']/g, c => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" })[c] );
|
|
5024
|
+
|
|
5025
|
+
/** [suites] as JUnit XML, under the name of what ran them. */
|
|
5026
|
+
function junitXml(suites , runner ) {
|
|
5027
|
+
const count = (cases , k ) => cases.filter(c => c[k] != null).length;
|
|
5028
|
+
const body = suites.map(st => {
|
|
5029
|
+
const cases = st.cases.map(c => ` <testcase classname="${xmlText(st.name)}" name="${xmlText(c.name)}">`
|
|
5030
|
+
+ (c.failure != null ? `<failure message="${xmlText(c.failure)}"/>`
|
|
5031
|
+
: c.error != null ? `<error message="${xmlText(c.error)}"/>`
|
|
5032
|
+
: c.skipped != null ? `<skipped message="${xmlText(c.skipped)}"/>` : "")
|
|
5033
|
+
+ "</testcase>");
|
|
5034
|
+
return ` <testsuite name="${xmlText(st.name)}" tests="${st.cases.length}" failures="${count(st.cases, "failure")}" `
|
|
5035
|
+
+ `errors="${count(st.cases, "error")}" skipped="${count(st.cases, "skipped")}">\n`
|
|
5036
|
+
+ cases.map(c => c + "\n").join("") + " </testsuite>\n";
|
|
5037
|
+
});
|
|
5038
|
+
const all = suites.flatMap(st => st.cases);
|
|
5039
|
+
return `<?xml version="1.0" encoding="UTF-8"?>\n<testsuites name="${xmlText(runner)}" tests="${all.length}" `
|
|
5040
|
+
+ `failures="${count(all, "failure")}" errors="${count(all, "error")}" skipped="${count(all, "skipped")}">\n`
|
|
5041
|
+
+ body.join("") + "</testsuites>\n";
|
|
5042
|
+
}
|
|
5043
|
+
|
|
5044
|
+
// ---- a pipeline as a CI bundle (#254) --------------------------------------
|
|
5045
|
+
//
|
|
5046
|
+
// Export for CI writes a pipeline as files a repository keeps and
|
|
5047
|
+
// `evals-lab run` runs with no lab around it (docs/pipeline-yaml.md, "Export
|
|
5048
|
+
// for CI"): the pipeline's YAML, its Target profiles by slug, and each
|
|
5049
|
+
// dataset it grades against as the Library's export of it. The server adds
|
|
5050
|
+
// the plugins and, when asked, the Source's items, and zips the lot. Read
|
|
5051
|
+
// back, the files are the run document the page would have submitted, with
|
|
5052
|
+
// each profile keyed by its slug rather than this lab's id -- the slug is
|
|
5053
|
+
// what the key's $EVALSLAB_API_KEY_<SLUG> is spelt from either way.
|
|
5054
|
+
|
|
5055
|
+
/** The text files of a bundle, by path inside it. */
|
|
5056
|
+
|
|
5057
|
+
|
|
5058
|
+
/** What a bundle is read back as: the run document, and each dataset it
|
|
5059
|
+
grades against, by the id its evals name it by. */
|
|
5060
|
+
|
|
5061
|
+
|
|
5062
|
+
|
|
5063
|
+
|
|
5064
|
+
|
|
5065
|
+
const BUNDLE_PIPELINE = "pipeline.yaml";
|
|
5066
|
+
const BUNDLE_PROFILES = "profiles.yaml";
|
|
5067
|
+
const BUNDLE_DATASET = /^datasets\/([a-z0-9]+(?:-[a-z0-9]+)*)\.json$/;
|
|
5068
|
+
|
|
5069
|
+
/**
|
|
5070
|
+
* [doc] as a bundle's text files: `pipeline.yaml`, `profiles.yaml` and
|
|
5071
|
+
* `datasets/<slug>.json`. [ctx.profiles] is Setup's list -- keys and all,
|
|
5072
|
+
* none of which is written -- [ctx.grader] the lab's grader, written into
|
|
5073
|
+
* each eval that would have asked it, and [ctx.datasets] each graded
|
|
5074
|
+
* dataset by id, as `{ name, body }`. `{ error }` when a dataset or a
|
|
5075
|
+
* profile the pipeline names is not given, since a bundle without it runs
|
|
5076
|
+
* something else.
|
|
5077
|
+
*/
|
|
5078
|
+
function exportBundle(doc , ctx
|
|
5079
|
+
= {})
|
|
5080
|
+
{
|
|
5081
|
+
const profiles = withSlugs((ctx.profiles ?? []) );
|
|
5082
|
+
// What the lab would supply at submit, written down: no lab supplies it later.
|
|
5083
|
+
const out = clone(doc) ;
|
|
5084
|
+
out.evals = (Array.isArray(out.evals) ? out.evals : [])
|
|
5085
|
+
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
|
|
5086
|
+
const table = {};
|
|
5087
|
+
for (const id of profileIds(out)) {
|
|
5088
|
+
const p = profiles.find(x => x.id === id);
|
|
5089
|
+
if (!p) return { error: `Target profile ${id} not found` };
|
|
5090
|
+
const { slug, ...flat } = connectionSettings(connectionOf(p) );
|
|
5091
|
+
table[slug ] = flat;
|
|
5092
|
+
}
|
|
5093
|
+
const files = {};
|
|
5094
|
+
const taken = new Set (), written = new Set ();
|
|
5095
|
+
for (const t of evalsOf(out) ) {
|
|
5096
|
+
// A link's group, or a private group's Cases from: the reference itself.
|
|
5097
|
+
const ref = casesRef(t);
|
|
5098
|
+
if (!ref) continue;
|
|
5099
|
+
const d = ctx.datasets?.[ref.id];
|
|
5100
|
+
if (!d) return { error: `Dataset ${ref.name || ref.id} not found` };
|
|
5101
|
+
// Named as the dataset is now, since the name is what a reader joins on.
|
|
5102
|
+
ref.name = d.name;
|
|
5103
|
+
if (written.has(ref.id)) continue;
|
|
5104
|
+
const slug = uniqueSlug(slugOf(d.name) || "dataset", taken);
|
|
5105
|
+
taken.add(slug);
|
|
5106
|
+
written.add(ref.id);
|
|
5107
|
+
// The Library's Export of it, so the file imports into any lab too. Its
|
|
5108
|
+
// name is what joins it back to the eval naming it: a lab's dataset
|
|
5109
|
+
// names are its own.
|
|
5110
|
+
files[`datasets/${slug}.json`] = JSON.stringify({ format: "evals-lab/dataset", version: DATASET_BODY_VERSION,
|
|
5111
|
+
dataset: { name: d.name, body: d.body } }, null, 2) + "\n";
|
|
5112
|
+
}
|
|
5113
|
+
files[BUNDLE_PIPELINE] = pipelineToYaml(out, { profiles });
|
|
5114
|
+
files[BUNDLE_PROFILES] = yaml.dump(table, { lineWidth: -1, noRefs: true });
|
|
5115
|
+
const bad = bundleProblems(files, profiles.map(p => p.key));
|
|
5116
|
+
return bad.length ? { error: bad[0] } : { files };
|
|
5117
|
+
}
|
|
5118
|
+
|
|
5119
|
+
/**
|
|
5120
|
+
* Why [files] cannot leave the lab: a field named like a key anywhere in
|
|
5121
|
+
* them, or any of [keys] -- Setup's own -- written anywhere at all. Empty
|
|
5122
|
+
* when nothing key-shaped is in them. The server asks the same of the zip
|
|
5123
|
+
* it is handed (server.py `bundle_problems`).
|
|
5124
|
+
*/
|
|
5125
|
+
function bundleProblems(files , keys = []) {
|
|
5126
|
+
const bad = [];
|
|
5127
|
+
const named = (v , at ) => {
|
|
5128
|
+
if (Array.isArray(v)) return v.forEach(x => named(x, at));
|
|
5129
|
+
if (!isObj(v)) return;
|
|
5130
|
+
for (const [k, x] of Object.entries(v)) {
|
|
5131
|
+
if (LOOKS_LIKE_A_KEY.test(k)) bad.push(`${at} holds a field named ${k}, and a key never leaves the lab`);
|
|
5132
|
+
named(x, at);
|
|
5133
|
+
}
|
|
5134
|
+
};
|
|
5135
|
+
for (const [path, text] of Object.entries(files)) {
|
|
5136
|
+
for (const key of keys) {
|
|
5137
|
+
if (isStr(key) && key.trim().length >= 4 && text.includes(key.trim())) {
|
|
5138
|
+
bad.push(`${path} holds a Target profile's key, and a key never leaves the lab`);
|
|
5139
|
+
break;
|
|
5140
|
+
}
|
|
5141
|
+
}
|
|
5142
|
+
let doc ;
|
|
5143
|
+
try { doc = path.endsWith(".json") ? JSON.parse(text) : yaml.load(text); } catch { continue; }
|
|
5144
|
+
named(doc, path);
|
|
5145
|
+
}
|
|
5146
|
+
return [...new Set(bad)];
|
|
5147
|
+
}
|
|
5148
|
+
|
|
5149
|
+
/**
|
|
5150
|
+
* A bundle's [files] as the run they describe: `pipeline.yaml` read as an
|
|
5151
|
+
* import would read it, against `profiles.yaml` and the `datasets/` it
|
|
5152
|
+
* carries, then resolved as the page resolves one at submit. [ctx.items] is
|
|
5153
|
+
* the Source's file list -- the names in the bundle's `items/`, or a
|
|
5154
|
+
* directory CI names; with none, each case's item is the list, so an item
|
|
5155
|
+
* nothing supplies is a missing item rather than an item never asked.
|
|
5156
|
+
* `{ error }` names the first thing that keeps it from running: a profile
|
|
5157
|
+
* or a dataset the files do not hold, a key in them, or a document the lab
|
|
5158
|
+
* would refuse. The caller registers the bundle's plugins first.
|
|
5159
|
+
*/
|
|
5160
|
+
function readBundle(files , ctx = {}) {
|
|
5161
|
+
if (!isStr(files[BUNDLE_PIPELINE])) return { error: `the bundle has no ${BUNDLE_PIPELINE}` };
|
|
5162
|
+
let table = {};
|
|
5163
|
+
try { table = isStr(files[BUNDLE_PROFILES]) ? yaml.load(files[BUNDLE_PROFILES] ) ?? {} : {}; }
|
|
5164
|
+
catch { return { error: `${BUNDLE_PROFILES} is not YAML` }; }
|
|
5165
|
+
if (!isObj(table)) return { error: `${BUNDLE_PROFILES} maps each profile's slug to its settings` };
|
|
5166
|
+
const keyed = bundleProblems(files);
|
|
5167
|
+
if (keyed.length) return { error: keyed[0] };
|
|
5168
|
+
const profiles = [];
|
|
5169
|
+
for (const [slug, p] of Object.entries(table)) {
|
|
5170
|
+
if (!isSlug(slug)) return { error: `${BUNDLE_PROFILES}: ${JSON.stringify(slug)} is not a slug` };
|
|
5171
|
+
if (!isObj(p)) return { error: `${BUNDLE_PROFILES}: ${slug} has to hold its settings` };
|
|
5172
|
+
if (!CONNECTION_TYPES[typeOf(p )]) {
|
|
5173
|
+
return { error: `${BUNDLE_PROFILES}: ${slug} is of type ${p.type}, which this lab does not have` };
|
|
5174
|
+
}
|
|
5175
|
+
profiles.push({ ...p, id: slug, slug, name: isStr(p.name) && p.name ? p.name : slug });
|
|
5176
|
+
}
|
|
5177
|
+
// Each dataset by its name, which is what joins it to the eval naming it.
|
|
5178
|
+
const datasets = [];
|
|
5179
|
+
for (const [path, text] of Object.entries(files)) {
|
|
5180
|
+
if (!path.startsWith("datasets/")) continue;
|
|
5181
|
+
if (!BUNDLE_DATASET.test(path)) return { error: `${path} is not named datasets/<slug>.json` };
|
|
5182
|
+
let doc ;
|
|
5183
|
+
try { doc = JSON.parse(text); } catch { return { error: `${path} is not JSON` }; }
|
|
5184
|
+
if (!isObj(doc) || doc.format !== "evals-lab/dataset" || !isObj(doc.dataset)) {
|
|
5185
|
+
return { error: `${path} is not a dataset's export` };
|
|
5186
|
+
}
|
|
5187
|
+
if (!(Number.isInteger(doc.version) && doc.version >= 1 && doc.version <= DATASET_BODY_VERSION)) {
|
|
5188
|
+
return { error: `${path} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to ${DATASET_BODY_VERSION}` };
|
|
5189
|
+
}
|
|
5190
|
+
const body = upgradeDatasetBody(doc.dataset.body) ;
|
|
5191
|
+
if (!isObj(body) || !Array.isArray(body.cases)) return { error: `${path} holds no cases` };
|
|
5192
|
+
datasets.push({ name: String(doc.dataset.name ?? ""), body: body });
|
|
5193
|
+
}
|
|
5194
|
+
let raw ;
|
|
5195
|
+
try { raw = yaml.load(files[BUNDLE_PIPELINE] ); }
|
|
5196
|
+
catch { return { error: `${BUNDLE_PIPELINE} is not YAML` }; }
|
|
5197
|
+
// The bundle stands in for the Source: whatever it names, the items are
|
|
5198
|
+
// the ones handed over.
|
|
5199
|
+
const content = isObj(raw) ? contentOf(upgradePipeline(raw)) : null;
|
|
5200
|
+
const refs = [];
|
|
5201
|
+
// Read as this version, whichever wrote the bundle: a link's group, or a
|
|
5202
|
+
// group of its own's Cases from.
|
|
5203
|
+
for (const t of isObj(raw) ? evalsOf(upgradePipeline(raw)) : []) {
|
|
5204
|
+
const ref = casesRef(t);
|
|
5205
|
+
if (!ref) continue;
|
|
5206
|
+
if (datasets.some(d => d.name === ref.name)) refs.push({ id: ref.id, name: ref.name });
|
|
5207
|
+
}
|
|
5208
|
+
const imported = importPipeline(raw, { profiles, groups: refs,
|
|
5209
|
+
sources: content?.type === "source" && isRef(content.ref) ? [content.ref] : [] });
|
|
5210
|
+
if ("error" in imported) return imported;
|
|
5211
|
+
if (imported.missing.length) return { error: `${imported.missing[0]}: the bundle does not hold it` };
|
|
5212
|
+
const byId = {};
|
|
5213
|
+
for (const r of refs) byId[r.id] = datasets.find(d => d.name === r.name) .body;
|
|
5214
|
+
const caseItems = [...new Set(Object.values(byId).flatMap(b => gradedSetFrom(b ).filter(c => !c.todo).map(caseItem)).filter(Boolean))];
|
|
5215
|
+
const run = resolvePipeline(imported.doc, {
|
|
5216
|
+
profiles: id => profiles.find(p => p.id === id) ?? null,
|
|
5217
|
+
files: ctx.items ? [...ctx.items].sort() : caseItems,
|
|
5218
|
+
});
|
|
5219
|
+
const bad = validatePipeline(run, { groups: refs });
|
|
5220
|
+
if (bad.length) return { error: bad[0] };
|
|
5221
|
+
return { run, datasets: byId };
|
|
5222
|
+
}
|
|
5223
|
+
|
|
4613
5224
|
/**
|
|
4614
5225
|
* Scenario [i] of a run document as runPipeline and a transport take it:
|
|
4615
5226
|
* each stage's wording, kind, image flag and modifiers, the token set of
|
|
@@ -4995,15 +5606,15 @@ export {
|
|
|
4995
5606
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
4996
5607
|
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
|
|
4997
5608
|
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
4998
|
-
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
5609
|
+
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
4999
5610
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5000
5611
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5001
5612
|
contentOf, withContent, replyOf,
|
|
5002
5613
|
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5003
5614
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5004
|
-
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5615
|
+
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5005
5616
|
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
5006
|
-
evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
|
|
5007
|
-
pipelineToYaml, importPipeline, yamlToPipeline,
|
|
5617
|
+
evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
|
|
5618
|
+
pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
|
|
5008
5619
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
5009
5620
|
};
|