evals-lab 0.4.1 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +81 -0
- package/README.md +125 -28
- package/bin/evals-lab.js +9 -1
- package/bin/run.js +542 -0
- package/lab/VERSION +1 -1
- package/lab/demo/pipelines/demo-1.json +8 -9
- package/lab/demo/pipelines/demo-2.json +8 -9
- package/lab/evals-core.mjs +735 -86
- package/lab/kinds/list.mjs +2 -1
- package/lab/run-evals.js +354 -99
- package/lab/server.py +711 -135
- package/lab/web/dist/assets/{gallery-DFeJkfUw.css → gallery-B_-TH0F-.css} +1 -1
- package/lab/web/dist/assets/gallery-D_MxkfLj.js +3 -0
- package/lab/web/dist/assets/main-DDoeU6hq.css +1 -0
- package/lab/web/dist/assets/main-nz6Q4jVm.js +21 -0
- package/lab/web/dist/assets/tokens-Bl8cOqkf.js +59 -0
- package/lab/web/dist/assets/tokens-DCHW9ru7.css +1 -0
- package/lab/web/dist/gallery.html +4 -4
- package/lab/web/dist/index.html +4 -4
- package/package.json +1 -1
- package/lab/web/dist/assets/gallery-B-7oyY37.js +0 -3
- package/lab/web/dist/assets/main-BkZTEix2.js +0 -21
- package/lab/web/dist/assets/main-C_b7QoTv.css +0 -1
- package/lab/web/dist/assets/tokens-BEn6_hVz.css +0 -1
- package/lab/web/dist/assets/tokens-DLRdTFGY.js +0 -55
package/lab/evals-core.mjs
CHANGED
|
@@ -275,9 +275,41 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
275
275
|
|
|
276
276
|
|
|
277
277
|
|
|
278
|
-
/**
|
|
279
|
-
|
|
280
|
-
|
|
278
|
+
/** A private group (docs/pipeline-model.md §17): an eval group's body held
|
|
279
|
+
in the pipeline rather than the Library -- what a version-12 Metrics
|
|
280
|
+
eval upgrades to when it is not a plain link. `casesFrom` is the Library
|
|
281
|
+
group whose cases, at its newest version, join this group's Every item
|
|
282
|
+
under this group's scoring; only the upgrade writes one. */
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
/** A Library eval group by reference; a run's copy says the version it
|
|
289
|
+
graded with: the body's fingerprint, and its number. */
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
/** An eval, from version 13: a link to an eval group (§17). `group` names a
|
|
293
|
+
Library group, followed at its newest version (`pin: null`) or pinned at
|
|
294
|
+
one; a private group is `own`, with `group: null`. A run's copy carries
|
|
295
|
+
the version it graded with on the reference (`version`, the body's
|
|
296
|
+
fingerprint, and `n`), and the lab's grader where the group names none. */
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
/** An eval, from version 13: a link to an eval group. A Metrics eval
|
|
306
|
+
(version 9 to 12), a Single Test or a Graded set is what an older
|
|
307
|
+
document held (upgradePipeline reads it converted). */
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
/** The rule a pipeline's linked groups pass by, together: every one, or at
|
|
311
|
+
least `count` of them. */
|
|
312
|
+
|
|
281
313
|
|
|
282
314
|
/** A pipeline's evals, in the order they read a run. */
|
|
283
315
|
|
|
@@ -293,11 +325,14 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
293
325
|
|
|
294
326
|
|
|
295
327
|
|
|
328
|
+
|
|
296
329
|
|
|
297
330
|
|
|
298
331
|
/** A Setup profile as a run carries it: request settings, never the key. */
|
|
299
332
|
|
|
300
333
|
|
|
334
|
+
|
|
335
|
+
|
|
301
336
|
|
|
302
337
|
|
|
303
338
|
|
|
@@ -338,6 +373,7 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
338
373
|
|
|
339
374
|
|
|
340
375
|
|
|
376
|
+
|
|
341
377
|
|
|
342
378
|
|
|
343
379
|
|
|
@@ -364,11 +400,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
364
400
|
|
|
365
401
|
|
|
366
402
|
|
|
403
|
+
/** A Target profile as an import or export matches it: its slug, where it
|
|
404
|
+
has one -- withSlugs gives the rest theirs. */
|
|
405
|
+
|
|
406
|
+
|
|
367
407
|
/** What an import matches references against: { id, name } each. */
|
|
368
408
|
|
|
369
|
-
|
|
409
|
+
|
|
370
410
|
|
|
371
|
-
|
|
411
|
+
|
|
372
412
|
|
|
373
413
|
|
|
374
414
|
|
|
@@ -735,16 +775,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
735
775
|
|
|
736
776
|
// ---- the registries ----
|
|
737
777
|
|
|
738
|
-
/** An option a modifier or an eval type exposes for editing.
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
778
|
+
/** An option a modifier or an eval type exposes for editing. `hint` says
|
|
779
|
+
what it does in one short line, under its control, where the label
|
|
780
|
+
cannot: a switch's effect, not why it exists. */
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
|
|
742
784
|
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
|
|
746
788
|
|
|
747
|
-
|
|
789
|
+
|
|
748
790
|
|
|
749
791
|
/** What an output kind made of a reply. */
|
|
750
792
|
|
|
@@ -817,8 +859,8 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
817
859
|
|
|
818
860
|
|
|
819
861
|
/** What validation looks references up in: `profiles(id)` and `sources(id)`
|
|
820
|
-
answer with what the id names, or nothing; `
|
|
821
|
-
lab holds
|
|
862
|
+
answer with what the id names, or nothing; `groups` is the eval groups
|
|
863
|
+
the lab holds. Each is optional, and a
|
|
822
864
|
reference nothing is given to look up is not checked. */
|
|
823
865
|
|
|
824
866
|
|
|
@@ -826,7 +868,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
826
868
|
|
|
827
869
|
|
|
828
870
|
|
|
829
|
-
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
|
|
830
875
|
|
|
831
876
|
|
|
832
877
|
/** Lookups, and whether it is a run document being validated. */
|
|
@@ -904,6 +949,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
904
949
|
|
|
905
950
|
|
|
906
951
|
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
907
955
|
|
|
908
956
|
|
|
909
957
|
|
|
@@ -917,6 +965,13 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
|
|
|
917
965
|
|
|
918
966
|
|
|
919
967
|
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
|
|
973
|
+
|
|
974
|
+
|
|
920
975
|
|
|
921
976
|
|
|
922
977
|
/** A job's stages, in the order they run (pipeline-model §16). A target's
|
|
@@ -2687,7 +2742,10 @@ function applyModifiers (list , kind ,
|
|
|
2687
2742
|
// scenario is a target, whose own step in each job is what it sends there.
|
|
2688
2743
|
// 11: `tests` are `evals`: the key renames and nothing in an eval changes,
|
|
2689
2744
|
// so a stored result's scores, keyed by eval id, read as they did.
|
|
2690
|
-
|
|
2745
|
+
// 12: a Contains metric's Ignore case holds item by item too.
|
|
2746
|
+
// 13: an eval is a link to an eval group, or a group of the pipeline's own,
|
|
2747
|
+
// and the document has an overall pass rule (pipeline-model §17).
|
|
2748
|
+
const PIPELINE_VERSION = 13 ;
|
|
2691
2749
|
|
|
2692
2750
|
// Plain objects, so an entry is added by assignment and a reader never needs
|
|
2693
2751
|
// to know which registered it.
|
|
@@ -2780,11 +2838,13 @@ OUTPUT_KINDS.text = {
|
|
|
2780
2838
|
|
|
2781
2839
|
// ---- the connection a profile reference resolves to ------------------------
|
|
2782
2840
|
// What a run carries of a Setup profile: everything a request needs and never
|
|
2783
|
-
// the key. A key follows the profile's
|
|
2784
|
-
//
|
|
2785
|
-
//
|
|
2786
|
-
//
|
|
2787
|
-
|
|
2841
|
+
// the key. A key follows the profile's slug into $EVALSLAB_API_KEY_<SLUG> -- or,
|
|
2842
|
+
// in a run document from before slugs, its id into $EVALSLAB_API_KEY_<ID> --
|
|
2843
|
+
// which the server sets from its profiles store and the runner reads, so a
|
|
2844
|
+
// name is only a label and two profiles can share one. #46: the profile
|
|
2845
|
+
// carries `type`, and the run's resolved copy carries it too
|
|
2846
|
+
// (docs/pipeline-model.md §7).
|
|
2847
|
+
const CONNECTION_FIELDS = ["name", "slug", "url", "model", "type", "temperature",
|
|
2788
2848
|
"px", "format", "quality", "options"];
|
|
2789
2849
|
const SETTING_KEYS = [...new Set(Object.values(CONNECTION_TYPES).flatMap(t => t.settings.map(s => s.key)))];
|
|
2790
2850
|
const OPTION_FIELDS = ["seed", "nPredict", ...DECODING_KEYS];
|
|
@@ -2794,9 +2854,69 @@ const IMAGE_FORMATS = ["image/jpeg", "image/webp", "image/png
|
|
|
2794
2854
|
const PROFILE_ID = /^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)?$/;
|
|
2795
2855
|
const LOOKS_LIKE_A_KEY = /^(?:api[-_]?)?key$|^(?:authorization|bearer|secret|password|token)$/i;
|
|
2796
2856
|
|
|
2797
|
-
/** The variable a profile's key is read from:
|
|
2798
|
-
|
|
2799
|
-
|
|
2857
|
+
/** The variable a profile's key is read from: its slug's, when it has one --
|
|
2858
|
+
anthropic is EVALSLAB_API_KEY_ANTHROPIC -- else its id's, as a run document
|
|
2859
|
+
from before slugs has it: pm1x8k2q is EVALSLAB_API_KEY_PM1X8K2Q. */
|
|
2860
|
+
const keyVar = (id , slug ) => "EVALSLAB_API_KEY_"
|
|
2861
|
+
+ String(slug || id).toUpperCase().replace(/[^A-Z0-9]+/g, "_").replace(/^_+|_+$/g, "");
|
|
2862
|
+
|
|
2863
|
+
// ---- a profile's slug (#253) ------------------------------------------------
|
|
2864
|
+
// What a profile is called outside the lab: in a pipeline file and in the
|
|
2865
|
+
// name of the variable CI keeps its key in. An id (pm1x9c0r) means nothing to
|
|
2866
|
+
// whoever sets that secret; a slug (anthropic) does, and stays put when the
|
|
2867
|
+
// lab it came from does not. Lowercase letters and digits, hyphens between,
|
|
2868
|
+
// so its variable is spelt one way and two slugs never spell the same one.
|
|
2869
|
+
const SLUG = /^[a-z0-9]+(?:-[a-z0-9]+)*$/;
|
|
2870
|
+
const SLUG_MAX = 32;
|
|
2871
|
+
|
|
2872
|
+
/** [text] as a slug: lowercase, every other run of characters one hyphen,
|
|
2873
|
+
cut to SLUG_MAX. "" when nothing of it is a letter or a digit. */
|
|
2874
|
+
function slugOf(text ) {
|
|
2875
|
+
return String(text ?? "").toLowerCase().normalize("NFKD").replace(/[\u0300-\u036f]/g, "")
|
|
2876
|
+
.replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, SLUG_MAX).replace(/-+$/, "");
|
|
2877
|
+
}
|
|
2878
|
+
|
|
2879
|
+
/** [base] made unique among [taken]: anthropic, then anthropic-2, -3. */
|
|
2880
|
+
function uniqueSlug(base , taken ) {
|
|
2881
|
+
const root = base || "profile";
|
|
2882
|
+
if (!taken.has(root)) return root;
|
|
2883
|
+
for (let n = 2; ; n++) {
|
|
2884
|
+
const tail = `-${n}`;
|
|
2885
|
+
const s = root.slice(0, SLUG_MAX - tail.length).replace(/-+$/, "") + tail;
|
|
2886
|
+
if (!taken.has(s)) return s;
|
|
2887
|
+
}
|
|
2888
|
+
}
|
|
2889
|
+
|
|
2890
|
+
/**
|
|
2891
|
+
* [list] with every profile holding a slug of its own: one it holds already
|
|
2892
|
+
* is kept, tidied into a slug (Setup's field holds what was typed, a
|
|
2893
|
+
* trailing hyphen and all), while no profile before it holds the same; any
|
|
2894
|
+
* other is made from its name, made unique. The slugs held are settled
|
|
2895
|
+
* first, so a slug minted for one profile never takes another's. The same
|
|
2896
|
+
* array when nothing changed, so a reader can tell a store that needs
|
|
2897
|
+
* writing back from one that does not.
|
|
2898
|
+
*/
|
|
2899
|
+
function withSlugs (list ) {
|
|
2900
|
+
const taken = new Set ();
|
|
2901
|
+
const held = list.map(p => {
|
|
2902
|
+
const s = slugOf(p?.slug);
|
|
2903
|
+
if (!s || taken.has(s)) return "";
|
|
2904
|
+
taken.add(s);
|
|
2905
|
+
return s;
|
|
2906
|
+
});
|
|
2907
|
+
let changed = false;
|
|
2908
|
+
const out = list.map((p, i) => {
|
|
2909
|
+
const slug = held[i] || uniqueSlug(slugOf(p?.name), taken);
|
|
2910
|
+
taken.add(slug);
|
|
2911
|
+
if (slug === p?.slug) return p;
|
|
2912
|
+
changed = true;
|
|
2913
|
+
return { ...p, slug };
|
|
2914
|
+
});
|
|
2915
|
+
return changed ? out : list ;
|
|
2916
|
+
}
|
|
2917
|
+
|
|
2918
|
+
/** Whether [s] is a slug as a profile may hold one. */
|
|
2919
|
+
const isSlug = (s ) => isStr(s) && SLUG.test(s) && s.length <= SLUG_MAX;
|
|
2800
2920
|
|
|
2801
2921
|
/** A stored profile as a run carries it: its request settings, no key. */
|
|
2802
2922
|
function connectionOf(p ) {
|
|
@@ -2807,6 +2927,7 @@ function connectionOf(p ) {
|
|
|
2807
2927
|
// the options bag holds the type's own settings only.
|
|
2808
2928
|
for (const k of settings) if (k !== "temperature" && String(p[k] ?? "").trim()) options[k] = p[k];
|
|
2809
2929
|
const out = { name: p.name || "unnamed", url: p.url || "", model: p.model || "", type };
|
|
2930
|
+
if (isSlug(p.slug)) out.slug = p.slug;
|
|
2810
2931
|
for (const k of ["px", "format", "quality"] ) if (p[k] != null && p[k] !== "") out[k] = p[k] ;
|
|
2811
2932
|
if (String(p.temperature ?? "").trim()) out.temperature = p.temperature ;
|
|
2812
2933
|
if (Object.keys(options).length) out.options = options;
|
|
@@ -2863,7 +2984,7 @@ function onlyFields(obj , at , allowed , bad
|
|
|
2863
2984
|
for (const k of Object.keys(obj)) {
|
|
2864
2985
|
if (allowed.includes(k)) continue;
|
|
2865
2986
|
bad.push(LOOKS_LIKE_A_KEY.test(k)
|
|
2866
|
-
? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "
|
|
2987
|
+
? `${at} carries "${k}", and a key never goes in a pipeline — $${keyFrom || "EVALSLAB_API_KEY_<ID>"} supplies it`
|
|
2867
2988
|
: `${at} has "${k}", which is not a pipeline field`);
|
|
2868
2989
|
}
|
|
2869
2990
|
}
|
|
@@ -3026,7 +3147,7 @@ LEGACY_TESTS.graded = {
|
|
|
3026
3147
|
if (!isRef(d) || (d.version != null && !isStr(d.version))) {
|
|
3027
3148
|
return void bad.push("a graded eval has to name its dataset as { id, name }");
|
|
3028
3149
|
}
|
|
3029
|
-
if (ctx.
|
|
3150
|
+
if (ctx.groups && !ctx.groups.some(x => x.id === d.id)) bad.push(`Eval group ${d.name || d.id} not found`);
|
|
3030
3151
|
},
|
|
3031
3152
|
};
|
|
3032
3153
|
|
|
@@ -3169,7 +3290,10 @@ function readRun(list , run ) {
|
|
|
3169
3290
|
});
|
|
3170
3291
|
}
|
|
3171
3292
|
|
|
3172
|
-
|
|
3293
|
+
// The eval of versions 9 to 12, read only to upgrade (LEGACY_TESTS.metrics.
|
|
3294
|
+
// toGroup) and kept whole as the reference groups-check.js and
|
|
3295
|
+
// pipeline-check.js hold the eval group to.
|
|
3296
|
+
LEGACY_TESTS.metrics = {
|
|
3173
3297
|
label: "Metrics",
|
|
3174
3298
|
description: "Checks of each reply, or of the whole run, scored all-must-pass or in weighted points.",
|
|
3175
3299
|
fields: ["type", "metrics", "mode", "threshold", "grader", "dataset", "over"],
|
|
@@ -3187,7 +3311,7 @@ EVAL_TYPES.metrics = {
|
|
|
3187
3311
|
// The dataset whose cases a case metric reads, and whose own metrics join.
|
|
3188
3312
|
if (t.dataset != null) {
|
|
3189
3313
|
if (!isRef(t.dataset) || (t.dataset.version != null && !isStr(t.dataset.version))) bad.push("the Metrics name their dataset as { id, name }");
|
|
3190
|
-
else if (ctx.
|
|
3314
|
+
else if (ctx.groups && !ctx.groups.some(x => x.id === t.dataset.id)) bad.push(`Eval group ${t.dataset.name || t.dataset.id} not found`);
|
|
3191
3315
|
}
|
|
3192
3316
|
if (!SCORING_MODES.includes(t.mode)) bad.push(`the Metrics are scored ${SCORING_MODES.join(" or ")}`);
|
|
3193
3317
|
if (t.mode === "weighted" && typeof t.threshold !== "number") bad.push("weighted Metrics need a threshold: the points an item has to reach");
|
|
@@ -3228,6 +3352,162 @@ EVAL_TYPES.metrics = {
|
|
|
3228
3352
|
metricInput(res, kase, more), graderCtx(t, more), t.mode, t.threshold)),
|
|
3229
3353
|
};
|
|
3230
3354
|
|
|
3355
|
+
/** A version-12 Metrics eval as the version-13 eval it reads as (§17): a
|
|
3356
|
+
link to its dataset, followed at the newest version, where it named one
|
|
3357
|
+
and held nothing of its own -- no metrics, scored All, the lab's grader,
|
|
3358
|
+
over each item -- and a private group holding its metrics, scoring and
|
|
3359
|
+
grader otherwise, with its dataset's cases (`casesFrom`). An eval over the
|
|
3360
|
+
whole run never read its dataset's cases, so its group takes none. The
|
|
3361
|
+
id, name and Continue on failure are kept, so a stored score keyed by
|
|
3362
|
+
the eval's id is the link's. */
|
|
3363
|
+
function groupOfMetrics(t ) {
|
|
3364
|
+
const { id, name, continueOnFailure } = t;
|
|
3365
|
+
const common = { id, ...(name !== undefined ? { name } : {}), type: "group", continueOnFailure };
|
|
3366
|
+
const metrics = Array.isArray(t.metrics) ? t.metrics : [];
|
|
3367
|
+
const run = t.over === "run";
|
|
3368
|
+
if (isRef(t.dataset) && !metrics.length && t.mode === "all" && t.grader == null && !run) {
|
|
3369
|
+
return { ...common, group: { ...t.dataset }, pin: null };
|
|
3370
|
+
}
|
|
3371
|
+
const own = { version: DATASET_BODY_VERSION, source: null, scoring: { mode: t.mode, threshold: t.threshold ?? null },
|
|
3372
|
+
grader: t.grader ?? null, every: run ? [] : metrics, run: run ? metrics : [], cases: [],
|
|
3373
|
+
casesFrom: isRef(t.dataset) && !run ? { ...t.dataset } : null };
|
|
3374
|
+
return { ...common, group: null, pin: null, own };
|
|
3375
|
+
}
|
|
3376
|
+
LEGACY_TESTS.metrics.toGroup = groupOfMetrics;
|
|
3377
|
+
|
|
3378
|
+
/** The group an eval reads by: a private group's own body, or a link's
|
|
3379
|
+
Library group as the runner hands it ([more].group). A link whose body
|
|
3380
|
+
nothing hands over reads as a Library group upgraded from version 6 does
|
|
3381
|
+
-- no metrics of its own, scored All -- which is what every one was
|
|
3382
|
+
until a body could hold more. */
|
|
3383
|
+
function groupOf(t , more = {}) {
|
|
3384
|
+
if (isObj(t.own)) return t.own ;
|
|
3385
|
+
const body = isRef(t.group) ? more.group?.(t.group) : undefined;
|
|
3386
|
+
return body ?? { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
|
|
3387
|
+
every: [], run: [], cases: [] };
|
|
3388
|
+
}
|
|
3389
|
+
|
|
3390
|
+
/** The Library group whose cases an eval reads, where it reads one: a
|
|
3391
|
+
link's group, or a private group's `casesFrom`. */
|
|
3392
|
+
function casesRef(t ) {
|
|
3393
|
+
if (!isObj(t) || t.type !== "group") return null;
|
|
3394
|
+
if (isRef(t.group)) return t.group ;
|
|
3395
|
+
return isObj(t.own) && isRef(t.own.casesFrom) ? t.own.casesFrom : null;
|
|
3396
|
+
}
|
|
3397
|
+
|
|
3398
|
+
/** The metrics a private group shows as its rules: its Whole run where it
|
|
3399
|
+
reads the run, its Every item otherwise. */
|
|
3400
|
+
const ownList = (t ) => (isObj(t.own) ? (wholeRunGroup(t) ? t.own.run : t.own.every) ?? [] : []);
|
|
3401
|
+
|
|
3402
|
+
/** A private group read over the whole run: Whole run metrics, and nothing
|
|
3403
|
+
read item by item. A group holding both is read item by item until the
|
|
3404
|
+
runner reports a group's Whole run beside its items (#233). */
|
|
3405
|
+
const wholeRunGroup = (t ) => isObj(t.own) && Array.isArray(t.own.run) && t.own.run.length > 0
|
|
3406
|
+
&& !(Array.isArray(t.own.every) && t.own.every.length) && !(Array.isArray(t.own.cases) && t.own.cases.length) && !t.own.casesFrom;
|
|
3407
|
+
|
|
3408
|
+
// A link to an eval group (docs/pipeline-model.md §17): a Library group by
|
|
3409
|
+
// reference, or a private group held here (`own`). It reads each item as its
|
|
3410
|
+
// group does (readGroup), or -- a private group of Whole run metrics alone --
|
|
3411
|
+
// the whole run (readGroupRun).
|
|
3412
|
+
EVAL_TYPES.group = {
|
|
3413
|
+
label: "Eval group",
|
|
3414
|
+
description: "Checks of each reply, or of the whole run, from an eval group: linked from the Library, or this pipeline's own.",
|
|
3415
|
+
fields: ["type", "group", "pin", "own", "grader"],
|
|
3416
|
+
// A metric that reads terms (a case) reads only a kind that yields them.
|
|
3417
|
+
accepts: t => ([...ownList(t)].some((m ) => METRICS[m?.type]?.needsTerms)
|
|
3418
|
+
? Object.keys(OUTPUT_KINDS).filter(k => OUTPUT_KINDS[k] .terms) : null),
|
|
3419
|
+
defaults: () => ({ type: "group", group: null, pin: null,
|
|
3420
|
+
own: { version: DATASET_BODY_VERSION, source: null, scoring: { mode: "all", threshold: null }, grader: null,
|
|
3421
|
+
every: [], run: [], cases: [], casesFrom: null } }),
|
|
3422
|
+
validate(t, ctx, bad){
|
|
3423
|
+
const g = t.group, own = t.own;
|
|
3424
|
+
const named = (ref , what ) => {
|
|
3425
|
+
const n = isObj(ref) ? ref.n : undefined;
|
|
3426
|
+
if (!isRef(ref) || (ref.version != null && !isStr(ref.version)) || (n != null && !Number.isInteger(n))) {
|
|
3427
|
+
return void bad.push(`${what} names its eval group as { id, name }`);
|
|
3428
|
+
}
|
|
3429
|
+
onlyFields(ref, what, ["id", "name", "version", "n"], bad);
|
|
3430
|
+
const held = ctx.groups?.find(x => x.id === ref.id);
|
|
3431
|
+
if (ctx.groups && !held) bad.push(`Eval group ${ref.name || ref.id} not found`);
|
|
3432
|
+
return held;
|
|
3433
|
+
};
|
|
3434
|
+
if (t.pin != null && !(Number.isInteger(t.pin) && t.pin > 0)) bad.push("a pin is a group's version: a whole number from 1");
|
|
3435
|
+
if (t.grader != null && !isRef(t.grader)) bad.push("the link names its grader as { id, name }");
|
|
3436
|
+
if (g != null) {
|
|
3437
|
+
if (own != null) return void bad.push("an eval links a Library group or holds its own, not both");
|
|
3438
|
+
const held = named(g, "the link");
|
|
3439
|
+
if (held && t.pin != null && typeof held.versions === "number" && t.pin > held.versions) {
|
|
3440
|
+
bad.push(`${g.name || g.id} has no version ${t.pin}`);
|
|
3441
|
+
}
|
|
3442
|
+
return;
|
|
3443
|
+
}
|
|
3444
|
+
if (!isObj(own)) return void bad.push("an eval links a Library group, or holds its own");
|
|
3445
|
+
if (t.pin != null) bad.push("a group of this pipeline's own versions with it, so it is never pinned");
|
|
3446
|
+
if (own.version !== DATASET_BODY_VERSION) bad.push(`the group's own body is version ${DATASET_BODY_VERSION}`);
|
|
3447
|
+
for (const key of ["scoring", "every", "run", "cases"]) if (own[key] == null) bad.push(`the group's own body has no ${key}`);
|
|
3448
|
+
onlyFields(own, "the group's own body", ["version", "source", "scoring", "grader", "every", "run", "cases", "casesFrom"], bad);
|
|
3449
|
+
if (bad.length) return;
|
|
3450
|
+
bad.push(...validateEvals(own));
|
|
3451
|
+
if (own.casesFrom != null) named(own.casesFrom, "Cases from");
|
|
3452
|
+
// Read item by item or over the run: one section at a time, until a
|
|
3453
|
+
// group's Whole run is reported beside its items (#233).
|
|
3454
|
+
if (own.run.length && !wholeRunGroup(t)) bad.push("the group's own body reads each item or the whole run, not both");
|
|
3455
|
+
// The group's own grader, or the lab's.
|
|
3456
|
+
const grader = isRef(own.grader) ? own.grader : isRef(ctx.grader) ? ctx.grader : null;
|
|
3457
|
+
if (gradedIn(own.every) && !grader) bad.push("a model-graded metric needs a grader: pick one on the eval, or make a Target profile the lab's grader");
|
|
3458
|
+
// A grader is asked words, and needs a model to ask.
|
|
3459
|
+
const p = grader && ctx.profiles ? ctx.profiles(grader.id) : null;
|
|
3460
|
+
if (grader && ctx.profiles && !p) bad.push(`Target profile ${grader.name || grader.id} not found`);
|
|
3461
|
+
else if (p) {
|
|
3462
|
+
const type = CONNECTION_TYPES[typeOf(p )];
|
|
3463
|
+
if (type && !(type.answers ?? ["prompt"]).includes("prompt")) bad.push(`the grader has to be asked words, and ${type.label} is not`);
|
|
3464
|
+
else if (!type?.local && !String((p ).model || "").trim()) bad.push("no model on the grader's Target profile");
|
|
3465
|
+
}
|
|
3466
|
+
},
|
|
3467
|
+
rules: t => ownList(t).map((m , x ) => ({
|
|
3468
|
+
key: `m${x}`, label: (isStr(m.metric) && m.metric) || METRICS[m.type]?.label || m.type,
|
|
3469
|
+
want: [m.not ? "not" : "", metricSummary(m), typeof m.weight === "number" && m.weight !== 1 ? `×${m.weight}` : ""]
|
|
3470
|
+
.filter(Boolean).join(" ") })),
|
|
3471
|
+
profiles: t => (isObj(t.own) && isRef(t.own.grader) ? [t.own.grader] : isRef(t.grader) ? [t.grader] : []),
|
|
3472
|
+
// A run carries the lab's grader where the group names none and may ask
|
|
3473
|
+
// one: a model-graded metric of its own, or a case's. A link's group is
|
|
3474
|
+
// the Library's, so the run's copy of the link carries it.
|
|
3475
|
+
resolve(t, ctx){
|
|
3476
|
+
if (!isRef(ctx.grader)) return t;
|
|
3477
|
+
const grader = { id: ctx.grader.id, name: ctx.grader.name };
|
|
3478
|
+
if (isObj(t.own)) {
|
|
3479
|
+
return !isRef(t.own.grader) && (gradedIn(t.own.every) || isRef(t.own.casesFrom)) ? { ...t, own: { ...t.own, grader } } : t;
|
|
3480
|
+
}
|
|
3481
|
+
return isRef(t.group) && !isRef(t.grader) ? { ...t, grader } : t;
|
|
3482
|
+
},
|
|
3483
|
+
wholeRun: wholeRunGroup,
|
|
3484
|
+
verdict: (t, ress, kind) => {
|
|
3485
|
+
const own = t.own ;
|
|
3486
|
+
return wholeRunVerdict(own.run, ress, kind, own.scoring.mode, own.scoring.threshold);
|
|
3487
|
+
},
|
|
3488
|
+
// Rule m{x} is the group's own metric x, read in that place on every item.
|
|
3489
|
+
ruleOf: (score, key, t) => (score.metrics ? score.metrics[Number(key.slice(1))]?.pass ?? null
|
|
3490
|
+
: ownList(t).length === 1 ? score.pass : null),
|
|
3491
|
+
// Every item, with its case where the group reads the Library group's
|
|
3492
|
+
// cases and the item has one there.
|
|
3493
|
+
read: async (t, kase, res, more) => {
|
|
3494
|
+
const group = groupOf(t, more);
|
|
3495
|
+
const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
|
|
3496
|
+
const ref = casesRef(t);
|
|
3497
|
+
// The item's case in this eval's own group: resolved from the group whose
|
|
3498
|
+
// cases it reads where the runner names the item (grading several groups
|
|
3499
|
+
// at once, --groups), else the case handed in (a single group).
|
|
3500
|
+
const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
|
|
3501
|
+
return readGroup({ ...group, grader } , ref ? own : null, res, more);
|
|
3502
|
+
},
|
|
3503
|
+
};
|
|
3504
|
+
|
|
3505
|
+
/** The case for an item in a group's body, by the name its file has, or null:
|
|
3506
|
+
a non-todo case whose item matches, as a Library group reads one. */
|
|
3507
|
+
function caseIn(body , item ) {
|
|
3508
|
+
return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
|
|
3509
|
+
}
|
|
3510
|
+
|
|
3231
3511
|
// ---- the lab's own scorers, as metrics ----------------------------------------
|
|
3232
3512
|
// What the Single Test checks, one metric each, so an eval of it converts to
|
|
3233
3513
|
// Metrics that read a run exactly as it did (metrics-parity-check.js). Here
|
|
@@ -3638,9 +3918,15 @@ function targetsOf(doc
|
|
|
3638
3918
|
return Array.isArray(doc?.targets) ? doc .targets : [];
|
|
3639
3919
|
}
|
|
3640
3920
|
|
|
3641
|
-
/** Target [i]'s
|
|
3921
|
+
/** Target [i]'s letter, A for the first: the same in every job, so a
|
|
3922
|
+
Target reads as one column through them. A number past Z. */
|
|
3923
|
+
function targetLetter(i ) {
|
|
3924
|
+
return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
|
|
3925
|
+
}
|
|
3926
|
+
|
|
3927
|
+
/** Target [i]'s name, or its letter's: "Target A". */
|
|
3642
3928
|
function targetLabel(doc , i ) {
|
|
3643
|
-
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i
|
|
3929
|
+
return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
|
|
3644
3930
|
}
|
|
3645
3931
|
|
|
3646
3932
|
/** What target [i] sends in job [k]: its step there. */
|
|
@@ -3849,11 +4135,16 @@ function evalsOf(doc )
|
|
|
3849
4135
|
return isObj(t) ? [{ id: "t1", name: "", ...t, continueOnFailure: true } ] : [];
|
|
3850
4136
|
}
|
|
3851
4137
|
|
|
3852
|
-
/** The
|
|
3853
|
-
|
|
3854
|
-
|
|
3855
|
-
|
|
3856
|
-
|
|
4138
|
+
/** The Library group whose cases a document's evals read, where one does:
|
|
4139
|
+
a link's group, or a private group's Cases from -- or, in a stored run
|
|
4140
|
+
from before version 13, a Metrics eval's dataset. A run grades against
|
|
4141
|
+
one (validatePipeline says so), so the first names it. */
|
|
4142
|
+
function evalsDataset(doc ) {
|
|
4143
|
+
for (const t of evalsOf(doc) ) {
|
|
4144
|
+
const ref = casesRef(t) ?? (isObj(t) && isRef(t.dataset) ? t.dataset : null);
|
|
4145
|
+
if (ref) return ref;
|
|
4146
|
+
}
|
|
4147
|
+
return null;
|
|
3857
4148
|
}
|
|
3858
4149
|
|
|
3859
4150
|
STEP_TYPES.evals = {
|
|
@@ -3882,15 +4173,14 @@ STEP_TYPES.evals = {
|
|
|
3882
4173
|
+ `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
|
|
3883
4174
|
}
|
|
3884
4175
|
});
|
|
3885
|
-
//
|
|
3886
|
-
|
|
3887
|
-
if (named.size > 1) bad.push("the evals grade against one dataset at a time");
|
|
4176
|
+
// The worker reads a body per group now (#233), so the evals may grade
|
|
4177
|
+
// against several Library groups at once; the queue keeps each body.
|
|
3888
4178
|
},
|
|
3889
4179
|
};
|
|
3890
4180
|
|
|
3891
4181
|
// ---- the document -------------------------------------------------------------
|
|
3892
4182
|
|
|
3893
|
-
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals"];
|
|
4183
|
+
const PIPELINE_FIELDS = ["version", "id", "name", "jobs", "targets", "evals", "pass"];
|
|
3894
4184
|
// What resolving adds, and nothing else: the profiles it resolved to and the
|
|
3895
4185
|
// run's own comment, which belongs to the run and never to the pipeline.
|
|
3896
4186
|
const RUN_FIELDS = [...PIPELINE_FIELDS, "profiles", "comment", "plugins"];
|
|
@@ -3999,6 +4289,12 @@ function newEval(type , name = "", fields = {})
|
|
|
3999
4289
|
/** Version 6 to 7: chains are jobs. The field renames in place -- so an
|
|
4000
4290
|
upgraded document reads the same, key for key -- and every job's `type`
|
|
4001
4291
|
becomes `"job"`, so an older document reads as one of today's everywhere. */
|
|
4292
|
+
/** A new eval linking the Library eval group [group], followed at its
|
|
4293
|
+
newest version. */
|
|
4294
|
+
function newLink(group , name = "") {
|
|
4295
|
+
return { type: "group", id: newId(), name, continueOnFailure: true, group: { id: group.id, name: group.name }, pin: null };
|
|
4296
|
+
}
|
|
4297
|
+
|
|
4002
4298
|
function jobsOf(next ) {
|
|
4003
4299
|
next.version = PIPELINE_VERSION;
|
|
4004
4300
|
if (Array.isArray(next.chains)) {
|
|
@@ -4113,6 +4409,26 @@ function caseAsWritten(next ) {
|
|
|
4113
4409
|
return next;
|
|
4114
4410
|
}
|
|
4115
4411
|
|
|
4412
|
+
/** Version 12 to 13: each Metrics eval is a link to an eval group or a
|
|
4413
|
+
group of the pipeline's own (groupOfMetrics), and the document passes
|
|
4414
|
+
when every linked group does. Pure: it needs nothing but the document,
|
|
4415
|
+
so every reader reads the same result. Every version before 13 ends
|
|
4416
|
+
here. */
|
|
4417
|
+
function groupsOf(next ) {
|
|
4418
|
+
next.version = PIPELINE_VERSION;
|
|
4419
|
+
if (Array.isArray(next.evals)) {
|
|
4420
|
+
next.evals = next.evals.map((t ) => (isObj(t) && t.type === "metrics" ? groupOfMetrics(t) : t));
|
|
4421
|
+
}
|
|
4422
|
+
if (!("pass" in next)) {
|
|
4423
|
+
// After the evals, where a reader of the document looks for it.
|
|
4424
|
+
const at = Object.keys(next).indexOf("evals");
|
|
4425
|
+
const entries = Object.entries(next);
|
|
4426
|
+
entries.splice(at < 0 ? entries.length : at + 1, 0, ["pass", { mode: "all" }]);
|
|
4427
|
+
return Object.fromEntries(entries);
|
|
4428
|
+
}
|
|
4429
|
+
return next;
|
|
4430
|
+
}
|
|
4431
|
+
|
|
4116
4432
|
function upgradePipeline (doc , ctx = {}) {
|
|
4117
4433
|
// A current document is read as it is, but for a profile reference the Runs
|
|
4118
4434
|
// tab saved whole (see profileRefs), which is cut back, and a step on a
|
|
@@ -4123,24 +4439,26 @@ function upgradePipeline (doc , ctx = {}) {
|
|
|
4123
4439
|
out = clone(out) ;
|
|
4124
4440
|
profileRefs(out);
|
|
4125
4441
|
}
|
|
4126
|
-
return localSteps(
|
|
4442
|
+
return localSteps(out, ctx) ;
|
|
4443
|
+
}
|
|
4444
|
+
if (!isObj(doc) || ![1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12].includes(doc.version )) return doc;
|
|
4445
|
+
if (doc.version === 11 || doc.version === 12) {
|
|
4446
|
+
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(cutRefs(clone(doc) )))), ctx) ;
|
|
4127
4447
|
}
|
|
4128
|
-
|
|
4129
|
-
|
|
4130
|
-
// Every version before 10 reads as version 9 first, then as 10, then as 11
|
|
4131
|
-
// and 12.
|
|
4448
|
+
// Every version before 10 reads as version 9 first, then as 10, then as 11,
|
|
4449
|
+
// 12 and 13.
|
|
4132
4450
|
// A version-10 document is cut as a current one was (profileRefs); the
|
|
4133
4451
|
// earlier ones are cut on their way through nineOf.
|
|
4134
4452
|
const ten = doc.version === 10 ? cutRefs(clone(doc) ) : targetsFromScenarios(nineOf(clone(doc) , ctx));
|
|
4135
|
-
return localSteps(withoutCaseMetric(caseAsWritten(evalsKey(ten))), ctx) ;
|
|
4453
|
+
return localSteps(groupsOf(withoutCaseMetric(caseAsWritten(evalsKey(ten)))), ctx) ;
|
|
4136
4454
|
}
|
|
4137
4455
|
|
|
4138
4456
|
/** [doc] with no Metrics eval holding the `case` metric: a dataset's cases
|
|
4139
4457
|
are metrics now (dataset body version 5), which an eval naming the
|
|
4140
4458
|
dataset adds for each item, so the case's own metrics carry what that
|
|
4141
|
-
metric scored. Read so at every version
|
|
4142
|
-
|
|
4143
|
-
none does. */
|
|
4459
|
+
metric scored. Read so at every version up to 12, as a document saved
|
|
4460
|
+
before it went still holds it; version 13 holds no Metrics eval. The
|
|
4461
|
+
same document where none does. */
|
|
4144
4462
|
function withoutCaseMetric(doc ) {
|
|
4145
4463
|
const has = (t ) => isObj(t) && Array.isArray(t.metrics) && t.metrics.some((m ) => isObj(m) && m.type === "case");
|
|
4146
4464
|
if (!Array.isArray(doc.evals) || !doc.evals.some(has)) return doc;
|
|
@@ -4229,7 +4547,7 @@ function tokenMappingsFromV1(set ) {
|
|
|
4229
4547
|
function blankPipeline(opts = {}) {
|
|
4230
4548
|
const job = withImage(jobDefaults(opts.kind, opts.tokens), true);
|
|
4231
4549
|
return { version: PIPELINE_VERSION, id: newId(), name: opts.name || "",
|
|
4232
|
-
jobs: [job], targets: [], evals: [] };
|
|
4550
|
+
jobs: [job], targets: [], evals: [], pass: { mode: "all" } };
|
|
4233
4551
|
}
|
|
4234
4552
|
|
|
4235
4553
|
/**
|
|
@@ -4368,6 +4686,7 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4368
4686
|
}
|
|
4369
4687
|
});
|
|
4370
4688
|
STEP_TYPES.evals .validate(doc.evals, c, bad, outOf(doc.jobs.at(-1)).kind, doc);
|
|
4689
|
+
passProblems(doc.pass, Array.isArray(doc.evals) ? doc.evals.length : 0, bad);
|
|
4371
4690
|
if (bad.length) return bad;
|
|
4372
4691
|
|
|
4373
4692
|
// Asked before anything is sent, so a misspelt token costs nothing and
|
|
@@ -4382,16 +4701,45 @@ function validatePipeline(input , ctx = {}) {
|
|
|
4382
4701
|
return bad;
|
|
4383
4702
|
}
|
|
4384
4703
|
|
|
4704
|
+
/** The overall pass rule: every linked group, or at least [count] of the
|
|
4705
|
+
[n] the pipeline has. */
|
|
4706
|
+
function passProblems(pass , n , bad ) {
|
|
4707
|
+
if (!isObj(pass) || (pass.mode !== "all" && pass.mode !== "atLeast")) {
|
|
4708
|
+
return void bad.push("pass has to be { mode: all } or { mode: atLeast, count }");
|
|
4709
|
+
}
|
|
4710
|
+
onlyFields(pass, "pass", pass.mode === "all" ? ["mode"] : ["mode", "count"], bad);
|
|
4711
|
+
if (pass.mode === "atLeast" && !(Number.isInteger(pass.count) && pass.count > 0)) bad.push("pass at least has to count a whole number of groups from 1");
|
|
4712
|
+
else if (pass.mode === "atLeast" && pass.count > n) bad.push(`pass asks for ${pass.count} groups, and the pipeline has ${n}`);
|
|
4713
|
+
}
|
|
4714
|
+
|
|
4715
|
+
/**
|
|
4716
|
+
* What [doc] may still be right about, as sentences: a link to a group made
|
|
4717
|
+
* for another Source than the pipeline's content (§17, decision 2). The run
|
|
4718
|
+
* grades the cases whose items it holds, and the rest read Missing.
|
|
4719
|
+
*/
|
|
4720
|
+
function pipelineWarnings(doc , ctx = {}) {
|
|
4721
|
+
if (!isObj(doc) || !Array.isArray(doc.evals) || !ctx.groups) return [];
|
|
4722
|
+
const content = contentOf(doc ) ;
|
|
4723
|
+
const here = content?.type === "source" && isRef(content.ref) ? content.ref.id : null;
|
|
4724
|
+
const warn = [];
|
|
4725
|
+
doc.evals.forEach((t , j ) => {
|
|
4726
|
+
const ref = casesRef(t);
|
|
4727
|
+
const src = ref && ctx.groups .find(x => x.id === ref.id)?.source;
|
|
4728
|
+
if (src && src.id !== here) warn.push(`${evalLabel(doc , j)}: ${ref .name || ref .id} grades ${src.name || src.id}`);
|
|
4729
|
+
});
|
|
4730
|
+
return warn;
|
|
4731
|
+
}
|
|
4732
|
+
|
|
4385
4733
|
/** A run document's profiles table: id → connection, keyless, spellable. */
|
|
4386
4734
|
function profilesProblems(profiles , bad ) {
|
|
4387
4735
|
if (!isObj(profiles)) return void bad.push("profiles has to be an object of id → connection");
|
|
4388
4736
|
const spelt = new Map ();
|
|
4389
4737
|
for (const [id, conn] of Object.entries(profiles)) {
|
|
4390
4738
|
if (!PROFILE_ID.test(id)) {
|
|
4391
|
-
bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from
|
|
4739
|
+
bad.push(`profile id "${id}" has to be letters and digits, with at most one hyphen: its key comes from EVALSLAB_API_KEY_<ID>`);
|
|
4392
4740
|
continue;
|
|
4393
4741
|
}
|
|
4394
|
-
const from = keyVar(id);
|
|
4742
|
+
const from = keyVar(id, isObj(conn) && isStr(conn.slug) ? conn.slug : null);
|
|
4395
4743
|
if (spelt.has(from)) bad.push(`profiles ${spelt.get(from)} and ${id} both take their key from $${from}`);
|
|
4396
4744
|
spelt.set(from, id);
|
|
4397
4745
|
const at = `profile ${id}`;
|
|
@@ -4408,6 +4756,9 @@ function connectionProblems(conn , at , from
|
|
|
4408
4756
|
bad.push(`${at}: type has to be one of ${Object.keys(CONNECTION_TYPES).join(", ")}`);
|
|
4409
4757
|
}
|
|
4410
4758
|
if (!isStr(conn.name)) bad.push(`${at}: name has to be text`);
|
|
4759
|
+
if (conn.slug != null && !isSlug(conn.slug)) {
|
|
4760
|
+
bad.push(`${at}: slug has to be lowercase letters and digits, with hyphens between, at most ${SLUG_MAX}`);
|
|
4761
|
+
}
|
|
4411
4762
|
const url = conn.url ?? "";
|
|
4412
4763
|
if (!isStr(url)) bad.push(`${at}: url has to be text, blank for the lab's Ollama`);
|
|
4413
4764
|
else if (url.trim()) {
|
|
@@ -4477,7 +4828,8 @@ function profileIds(doc )
|
|
|
4477
4828
|
* the Source's file list as it reads now, `ctx.comment` is the run's own.
|
|
4478
4829
|
*/
|
|
4479
4830
|
function resolvePipeline(doc ,
|
|
4480
|
-
ctx
|
|
4831
|
+
ctx
|
|
4832
|
+
= {}) {
|
|
4481
4833
|
const run = clone(doc) ;
|
|
4482
4834
|
// What the lab supplies an eval -- its grader -- before the profiles it
|
|
4483
4835
|
// asks are carried.
|
|
@@ -4489,8 +4841,18 @@ function resolvePipeline(doc ,
|
|
|
4489
4841
|
}
|
|
4490
4842
|
const content = contentOf(run) ;
|
|
4491
4843
|
if (content?.type === "source") content.files = CONTENT_TYPES.source .expand (content, ctx.files);
|
|
4492
|
-
|
|
4493
|
-
|
|
4844
|
+
// Each Library group the evals read, as the version it grades with: the
|
|
4845
|
+
// body's fingerprint and, where the lab numbers them, `n` -- a pinned
|
|
4846
|
+
// link's pin.
|
|
4847
|
+
if (ctx.groupVersion) {
|
|
4848
|
+
for (const t of run.evals || []) {
|
|
4849
|
+
const ref = casesRef(t);
|
|
4850
|
+
const at = ref && ctx.groupVersion(ref.id);
|
|
4851
|
+
if (!ref || !at) continue;
|
|
4852
|
+
ref.version = at.version;
|
|
4853
|
+
const n = isObj(t) && Number.isInteger(t.pin) ? t.pin : at.n;
|
|
4854
|
+
if (n != null) ref.n = n;
|
|
4855
|
+
}
|
|
4494
4856
|
}
|
|
4495
4857
|
if (isStr(ctx.comment)) run.comment = ctx.comment;
|
|
4496
4858
|
return run;
|
|
@@ -4504,7 +4866,12 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4504
4866
|
delete doc.plugins;
|
|
4505
4867
|
const content = contentOf(doc) ;
|
|
4506
4868
|
if (content) { delete content.files; delete content.revs; }
|
|
4507
|
-
|
|
4869
|
+
// What the run recorded of each group it read is the run's. The grader
|
|
4870
|
+
// it carried stays, as a Metrics eval's did: a bundle writes it down.
|
|
4871
|
+
for (const t of doc.evals || []) {
|
|
4872
|
+
const ref = casesRef(t);
|
|
4873
|
+
if (ref) { delete ref.version; delete ref.n; }
|
|
4874
|
+
}
|
|
4508
4875
|
return doc;
|
|
4509
4876
|
}
|
|
4510
4877
|
|
|
@@ -4517,9 +4884,35 @@ function pipelineOfRun(run , ctx = {}) {
|
|
|
4517
4884
|
|
|
4518
4885
|
|
|
4519
4886
|
/** [doc] as YAML text, the pipeline's file form -- references only, so an
|
|
4520
|
-
export carries no key, no URL, no file list and no file bytes.
|
|
4521
|
-
|
|
4522
|
-
|
|
4887
|
+
export carries no key, no URL, no file list and no file bytes. A Target
|
|
4888
|
+
profile [ctx] holds is named by its slug, `{ slug, name }`, rather than
|
|
4889
|
+
by this lab's id (#253): the slug is what another lab, or CI's
|
|
4890
|
+
$EVALSLAB_API_KEY_<SLUG>, knows it by. */
|
|
4891
|
+
function pipelineToYaml(doc , ctx = {}) {
|
|
4892
|
+
const profiles = withSlugs(ctx.profiles ?? []);
|
|
4893
|
+
if (!profiles.length) return yaml.dump(doc, { lineWidth: -1, noRefs: true });
|
|
4894
|
+
const out = clone(doc) ;
|
|
4895
|
+
const bySlug = (ref ) => {
|
|
4896
|
+
const p = isRef(ref) ? profiles.find(x => x.id === ref.id) : null;
|
|
4897
|
+
return p ? { slug: p.slug, name: ref.name ?? p.name } : ref;
|
|
4898
|
+
};
|
|
4899
|
+
eachProfileRef(out, bySlug);
|
|
4900
|
+
return yaml.dump(out, { lineWidth: -1, noRefs: true });
|
|
4901
|
+
}
|
|
4902
|
+
|
|
4903
|
+
/** Every Target profile reference in [doc] -- each target's, each of its
|
|
4904
|
+
steps', each eval's grader -- replaced, in place, by [f] of it. */
|
|
4905
|
+
function eachProfileRef(doc , f ) {
|
|
4906
|
+
for (const t of targetsOf(doc)) {
|
|
4907
|
+
if (t.profile) t.profile = f(t.profile);
|
|
4908
|
+
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4909
|
+
if (st?.profile) st.profile = f(st.profile);
|
|
4910
|
+
}
|
|
4911
|
+
}
|
|
4912
|
+
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4913
|
+
if (isObj(t) && t.grader) t.grader = f(t.grader);
|
|
4914
|
+
if (isObj(t?.own) && t.own.grader) t.own.grader = f(t.own.grader);
|
|
4915
|
+
}
|
|
4523
4916
|
}
|
|
4524
4917
|
|
|
4525
4918
|
/**
|
|
@@ -4527,7 +4920,7 @@ function pipelineToYaml(doc ) {
|
|
|
4527
4920
|
* name (pipeline-model §5), so a pipeline written in another lab -- ids that
|
|
4528
4921
|
* mean nothing here -- finds its Source and Setup profiles by the names they
|
|
4529
4922
|
* were exported under. [ctx]'s lists are what the lab holds:
|
|
4530
|
-
* `{ profiles, sources,
|
|
4923
|
+
* `{ profiles, sources, groups }`, each `{ id, name }[]`. Returns
|
|
4531
4924
|
* `{ doc, missing }`, where `missing` is one sentence per reference neither
|
|
4532
4925
|
* id nor name matched, or `{ error }` when the document is refused: not a
|
|
4533
4926
|
* version the lab reads, or not something the lab can edit and run as it
|
|
@@ -4538,14 +4931,14 @@ function importPipeline(input , ctx = {}) {
|
|
|
4538
4931
|
const doc = upgradePipeline(input, ctx);
|
|
4539
4932
|
const v = versionProblem(doc);
|
|
4540
4933
|
if (v) return { error: v };
|
|
4541
|
-
const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [],
|
|
4934
|
+
const profiles = ctx.profiles ?? [], sources = ctx.sources ?? [], groups = ctx.groups ?? [];
|
|
4542
4935
|
const next = clone(doc) ;
|
|
4543
4936
|
const missing = [], seen = new Set ();
|
|
4544
|
-
const lists = { profile: profiles, source: sources,
|
|
4937
|
+
const lists = { profile: profiles, source: sources, group: groups };
|
|
4545
4938
|
const words = {
|
|
4546
|
-
profile: (ref
|
|
4939
|
+
profile: (ref ) => `Target profile ${ref.slug || ref.name || ref.id} not found`,
|
|
4547
4940
|
source: (ref ) => `Source ${ref.name || ref.id} not found`,
|
|
4548
|
-
|
|
4941
|
+
group: (ref ) => `Eval group ${ref.name || ref.id} not found`,
|
|
4549
4942
|
};
|
|
4550
4943
|
const remap = (ref , kind ) => {
|
|
4551
4944
|
const hit = lists[kind].find(x => x.id === ref.id) || lists[kind].find(x => x.name === ref.name);
|
|
@@ -4554,16 +4947,34 @@ function importPipeline(input , ctx = {}) {
|
|
|
4554
4947
|
if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
|
|
4555
4948
|
return ref;
|
|
4556
4949
|
};
|
|
4950
|
+
// A profile is named by its slug in a file this lab writes (#253), and by
|
|
4951
|
+
// its id in one written before slugs: a slug is matched by slug, then by
|
|
4952
|
+
// name. One this lab holds under neither is kept as a reference the run
|
|
4953
|
+
// is blocked on, the slug standing in for an id.
|
|
4954
|
+
const slugged = withSlugs(profiles);
|
|
4955
|
+
const remapProfile = (ref ) => {
|
|
4956
|
+
if (!isObj(ref) || !isStr(ref.slug) || isRef(ref)) return isRef(ref) ? remap(ref, "profile") : ref;
|
|
4957
|
+
const hit = slugged.find(x => x.slug === ref.slug) || slugged.find(x => isStr(ref.name) && x.name === ref.name);
|
|
4958
|
+
if (hit) return { id: hit.id, name: hit.name };
|
|
4959
|
+
const sentence = words.profile(ref );
|
|
4960
|
+
if (!seen.has(sentence)) { seen.add(sentence); missing.push(sentence); }
|
|
4961
|
+
return { id: ref.slug, name: isStr(ref.name) && ref.name ? ref.name : ref.slug };
|
|
4962
|
+
};
|
|
4557
4963
|
const content = contentOf(next) ;
|
|
4558
4964
|
if (content?.type === "source") content.ref = remap(content.ref, "source");
|
|
4559
4965
|
for (const t of targetsOf(next)) {
|
|
4560
|
-
if (t.profile) t.profile =
|
|
4966
|
+
if (t.profile) t.profile = remapProfile(t.profile);
|
|
4561
4967
|
for (const st of Array.isArray(t.steps) ? t.steps : []) {
|
|
4562
|
-
if (st?.profile) st.profile =
|
|
4968
|
+
if (st?.profile) st.profile = remapProfile(st.profile);
|
|
4563
4969
|
}
|
|
4564
4970
|
}
|
|
4565
4971
|
for (const t of Array.isArray(next.evals) ? next.evals : []) {
|
|
4566
|
-
if (isRef(t?.
|
|
4972
|
+
if (isRef(t?.group)) t.group = remap(t.group, "group");
|
|
4973
|
+
else if (isObj(t?.own) && isRef(t.own.casesFrom)) t.own.casesFrom = remap(t.own.casesFrom, "group");
|
|
4974
|
+
// The grader is a profile like any other, matched the same way: a
|
|
4975
|
+
// private group's own, or the one a run's link carries.
|
|
4976
|
+
if (isObj(t?.own) && t.own.grader) t.own.grader = remapProfile(t.own.grader);
|
|
4977
|
+
if (isObj(t) && t.grader) t.grader = remapProfile(t.grader);
|
|
4567
4978
|
}
|
|
4568
4979
|
// Its references are this lab's now, so a step asking this lab's Echo
|
|
4569
4980
|
// profile is read as an Echo step (upgradePipeline did it for ids that
|
|
@@ -4597,6 +5008,8 @@ function mintIds(doc ) {
|
|
|
4597
5008
|
for (const t of Array.isArray(doc.evals) ? doc.evals : []) {
|
|
4598
5009
|
if (isObj(t) && (!isStr(t.id) || !t.id)) t.id = newId();
|
|
4599
5010
|
}
|
|
5011
|
+
// Nor need it say how its linked groups pass together: all of them.
|
|
5012
|
+
if (!("pass" in doc)) doc.pass = { mode: "all" };
|
|
4600
5013
|
return doc;
|
|
4601
5014
|
}
|
|
4602
5015
|
|
|
@@ -4610,6 +5023,224 @@ function yamlToPipeline(text , ctx ) {
|
|
|
4610
5023
|
return importPipeline(doc, ctx);
|
|
4611
5024
|
}
|
|
4612
5025
|
|
|
5026
|
+
// ---- JUnit (#251) --------------------------------------------------------------
|
|
5027
|
+
//
|
|
5028
|
+
// What a CI test panel reads, from the worker's report and from a lab's
|
|
5029
|
+
// stored run alike (`evals-lab run --lab`, #256), so both write one format.
|
|
5030
|
+
// A suite is one target's eval, a case one item: a failure carries the reason
|
|
5031
|
+
// the item failed, an item that never ran is an error, and one the eval had
|
|
5032
|
+
// nothing to read of is skipped.
|
|
5033
|
+
|
|
5034
|
+
/** One JUnit test suite: a target's eval, and a case per item it read. */
|
|
5035
|
+
|
|
5036
|
+
|
|
5037
|
+
|
|
5038
|
+
|
|
5039
|
+
|
|
5040
|
+
// XML 1.0 has no way to write most control characters, escaped or not, and a
|
|
5041
|
+
// model's reply can hold any of them.
|
|
5042
|
+
const xmlText = (v ) => String(v).replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFE\uFFFF]/g, "")
|
|
5043
|
+
.replace(/[&<>"']/g, c => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" })[c] );
|
|
5044
|
+
|
|
5045
|
+
/** [suites] as JUnit XML, under the name of what ran them. */
|
|
5046
|
+
function junitXml(suites , runner ) {
|
|
5047
|
+
const count = (cases , k ) => cases.filter(c => c[k] != null).length;
|
|
5048
|
+
const body = suites.map(st => {
|
|
5049
|
+
const cases = st.cases.map(c => ` <testcase classname="${xmlText(st.name)}" name="${xmlText(c.name)}">`
|
|
5050
|
+
+ (c.failure != null ? `<failure message="${xmlText(c.failure)}"/>`
|
|
5051
|
+
: c.error != null ? `<error message="${xmlText(c.error)}"/>`
|
|
5052
|
+
: c.skipped != null ? `<skipped message="${xmlText(c.skipped)}"/>` : "")
|
|
5053
|
+
+ "</testcase>");
|
|
5054
|
+
return ` <testsuite name="${xmlText(st.name)}" tests="${st.cases.length}" failures="${count(st.cases, "failure")}" `
|
|
5055
|
+
+ `errors="${count(st.cases, "error")}" skipped="${count(st.cases, "skipped")}">\n`
|
|
5056
|
+
+ cases.map(c => c + "\n").join("") + " </testsuite>\n";
|
|
5057
|
+
});
|
|
5058
|
+
const all = suites.flatMap(st => st.cases);
|
|
5059
|
+
return `<?xml version="1.0" encoding="UTF-8"?>\n<testsuites name="${xmlText(runner)}" tests="${all.length}" `
|
|
5060
|
+
+ `failures="${count(all, "failure")}" errors="${count(all, "error")}" skipped="${count(all, "skipped")}">\n`
|
|
5061
|
+
+ body.join("") + "</testsuites>\n";
|
|
5062
|
+
}
|
|
5063
|
+
|
|
5064
|
+
// ---- a pipeline as a CI bundle (#254) --------------------------------------
|
|
5065
|
+
//
|
|
5066
|
+
// Export for CI writes a pipeline as files a repository keeps and
|
|
5067
|
+
// `evals-lab run` runs with no lab around it (docs/pipeline-yaml.md, "Export
|
|
5068
|
+
// for CI"): the pipeline's YAML, its Target profiles by slug, and each
|
|
5069
|
+
// dataset it grades against as the Library's export of it. The server adds
|
|
5070
|
+
// the plugins and, when asked, the Source's items, and zips the lot. Read
|
|
5071
|
+
// back, the files are the run document the page would have submitted, with
|
|
5072
|
+
// each profile keyed by its slug rather than this lab's id -- the slug is
|
|
5073
|
+
// what the key's $EVALSLAB_API_KEY_<SLUG> is spelt from either way.
|
|
5074
|
+
|
|
5075
|
+
/** The text files of a bundle, by path inside it. */
|
|
5076
|
+
|
|
5077
|
+
|
|
5078
|
+
/** What a bundle is read back as: the run document, and each dataset it
|
|
5079
|
+
grades against, by the id its evals name it by. */
|
|
5080
|
+
|
|
5081
|
+
|
|
5082
|
+
|
|
5083
|
+
|
|
5084
|
+
|
|
5085
|
+
const BUNDLE_PIPELINE = "pipeline.yaml";
|
|
5086
|
+
const BUNDLE_PROFILES = "profiles.yaml";
|
|
5087
|
+
const BUNDLE_DATASET = /^datasets\/([a-z0-9]+(?:-[a-z0-9]+)*)\.json$/;
|
|
5088
|
+
|
|
5089
|
+
/**
|
|
5090
|
+
* [doc] as a bundle's text files: `pipeline.yaml`, `profiles.yaml` and
|
|
5091
|
+
* `datasets/<slug>.json`. [ctx.profiles] is Setup's list -- keys and all,
|
|
5092
|
+
* none of which is written -- [ctx.grader] the lab's grader, written into
|
|
5093
|
+
* each eval that would have asked it, and [ctx.datasets] each graded
|
|
5094
|
+
* dataset by id, as `{ name, body }`. `{ error }` when a dataset or a
|
|
5095
|
+
* profile the pipeline names is not given, since a bundle without it runs
|
|
5096
|
+
* something else.
|
|
5097
|
+
*/
|
|
5098
|
+
function exportBundle(doc , ctx
|
|
5099
|
+
= {})
|
|
5100
|
+
{
|
|
5101
|
+
const profiles = withSlugs((ctx.profiles ?? []) );
|
|
5102
|
+
// What the lab would supply at submit, written down: no lab supplies it later.
|
|
5103
|
+
const out = clone(doc) ;
|
|
5104
|
+
out.evals = (Array.isArray(out.evals) ? out.evals : [])
|
|
5105
|
+
.map((t ) => EVAL_TYPES[t?.type]?.resolve?.(t, { grader: ctx.grader ?? null }) ?? t);
|
|
5106
|
+
const table = {};
|
|
5107
|
+
for (const id of profileIds(out)) {
|
|
5108
|
+
const p = profiles.find(x => x.id === id);
|
|
5109
|
+
if (!p) return { error: `Target profile ${id} not found` };
|
|
5110
|
+
const { slug, ...flat } = connectionSettings(connectionOf(p) );
|
|
5111
|
+
table[slug ] = flat;
|
|
5112
|
+
}
|
|
5113
|
+
const files = {};
|
|
5114
|
+
const taken = new Set (), written = new Set ();
|
|
5115
|
+
for (const t of evalsOf(out) ) {
|
|
5116
|
+
// A link's group, or a private group's Cases from: the reference itself.
|
|
5117
|
+
const ref = casesRef(t);
|
|
5118
|
+
if (!ref) continue;
|
|
5119
|
+
const d = ctx.datasets?.[ref.id];
|
|
5120
|
+
if (!d) return { error: `Dataset ${ref.name || ref.id} not found` };
|
|
5121
|
+
// Named as the dataset is now, since the name is what a reader joins on.
|
|
5122
|
+
ref.name = d.name;
|
|
5123
|
+
if (written.has(ref.id)) continue;
|
|
5124
|
+
const slug = uniqueSlug(slugOf(d.name) || "dataset", taken);
|
|
5125
|
+
taken.add(slug);
|
|
5126
|
+
written.add(ref.id);
|
|
5127
|
+
// The Library's Export of it, so the file imports into any lab too. Its
|
|
5128
|
+
// name is what joins it back to the eval naming it: a lab's dataset
|
|
5129
|
+
// names are its own.
|
|
5130
|
+
files[`datasets/${slug}.json`] = JSON.stringify({ format: "evals-lab/dataset", version: DATASET_BODY_VERSION,
|
|
5131
|
+
dataset: { name: d.name, body: d.body } }, null, 2) + "\n";
|
|
5132
|
+
}
|
|
5133
|
+
files[BUNDLE_PIPELINE] = pipelineToYaml(out, { profiles });
|
|
5134
|
+
files[BUNDLE_PROFILES] = yaml.dump(table, { lineWidth: -1, noRefs: true });
|
|
5135
|
+
const bad = bundleProblems(files, profiles.map(p => p.key));
|
|
5136
|
+
return bad.length ? { error: bad[0] } : { files };
|
|
5137
|
+
}
|
|
5138
|
+
|
|
5139
|
+
/**
|
|
5140
|
+
* Why [files] cannot leave the lab: a field named like a key anywhere in
|
|
5141
|
+
* them, or any of [keys] -- Setup's own -- written anywhere at all. Empty
|
|
5142
|
+
* when nothing key-shaped is in them. The server asks the same of the zip
|
|
5143
|
+
* it is handed (server.py `bundle_problems`).
|
|
5144
|
+
*/
|
|
5145
|
+
function bundleProblems(files , keys = []) {
|
|
5146
|
+
const bad = [];
|
|
5147
|
+
const named = (v , at ) => {
|
|
5148
|
+
if (Array.isArray(v)) return v.forEach(x => named(x, at));
|
|
5149
|
+
if (!isObj(v)) return;
|
|
5150
|
+
for (const [k, x] of Object.entries(v)) {
|
|
5151
|
+
if (LOOKS_LIKE_A_KEY.test(k)) bad.push(`${at} holds a field named ${k}, and a key never leaves the lab`);
|
|
5152
|
+
named(x, at);
|
|
5153
|
+
}
|
|
5154
|
+
};
|
|
5155
|
+
for (const [path, text] of Object.entries(files)) {
|
|
5156
|
+
for (const key of keys) {
|
|
5157
|
+
if (isStr(key) && key.trim().length >= 4 && text.includes(key.trim())) {
|
|
5158
|
+
bad.push(`${path} holds a Target profile's key, and a key never leaves the lab`);
|
|
5159
|
+
break;
|
|
5160
|
+
}
|
|
5161
|
+
}
|
|
5162
|
+
let doc ;
|
|
5163
|
+
try { doc = path.endsWith(".json") ? JSON.parse(text) : yaml.load(text); } catch { continue; }
|
|
5164
|
+
named(doc, path);
|
|
5165
|
+
}
|
|
5166
|
+
return [...new Set(bad)];
|
|
5167
|
+
}
|
|
5168
|
+
|
|
5169
|
+
/**
|
|
5170
|
+
* A bundle's [files] as the run they describe: `pipeline.yaml` read as an
|
|
5171
|
+
* import would read it, against `profiles.yaml` and the `datasets/` it
|
|
5172
|
+
* carries, then resolved as the page resolves one at submit. [ctx.items] is
|
|
5173
|
+
* the Source's file list -- the names in the bundle's `items/`, or a
|
|
5174
|
+
* directory CI names; with none, each case's item is the list, so an item
|
|
5175
|
+
* nothing supplies is a missing item rather than an item never asked.
|
|
5176
|
+
* `{ error }` names the first thing that keeps it from running: a profile
|
|
5177
|
+
* or a dataset the files do not hold, a key in them, or a document the lab
|
|
5178
|
+
* would refuse. The caller registers the bundle's plugins first.
|
|
5179
|
+
*/
|
|
5180
|
+
function readBundle(files , ctx = {}) {
|
|
5181
|
+
if (!isStr(files[BUNDLE_PIPELINE])) return { error: `the bundle has no ${BUNDLE_PIPELINE}` };
|
|
5182
|
+
let table = {};
|
|
5183
|
+
try { table = isStr(files[BUNDLE_PROFILES]) ? yaml.load(files[BUNDLE_PROFILES] ) ?? {} : {}; }
|
|
5184
|
+
catch { return { error: `${BUNDLE_PROFILES} is not YAML` }; }
|
|
5185
|
+
if (!isObj(table)) return { error: `${BUNDLE_PROFILES} maps each profile's slug to its settings` };
|
|
5186
|
+
const keyed = bundleProblems(files);
|
|
5187
|
+
if (keyed.length) return { error: keyed[0] };
|
|
5188
|
+
const profiles = [];
|
|
5189
|
+
for (const [slug, p] of Object.entries(table)) {
|
|
5190
|
+
if (!isSlug(slug)) return { error: `${BUNDLE_PROFILES}: ${JSON.stringify(slug)} is not a slug` };
|
|
5191
|
+
if (!isObj(p)) return { error: `${BUNDLE_PROFILES}: ${slug} has to hold its settings` };
|
|
5192
|
+
if (!CONNECTION_TYPES[typeOf(p )]) {
|
|
5193
|
+
return { error: `${BUNDLE_PROFILES}: ${slug} is of type ${p.type}, which this lab does not have` };
|
|
5194
|
+
}
|
|
5195
|
+
profiles.push({ ...p, id: slug, slug, name: isStr(p.name) && p.name ? p.name : slug });
|
|
5196
|
+
}
|
|
5197
|
+
// Each dataset by its name, which is what joins it to the eval naming it.
|
|
5198
|
+
const datasets = [];
|
|
5199
|
+
for (const [path, text] of Object.entries(files)) {
|
|
5200
|
+
if (!path.startsWith("datasets/")) continue;
|
|
5201
|
+
if (!BUNDLE_DATASET.test(path)) return { error: `${path} is not named datasets/<slug>.json` };
|
|
5202
|
+
let doc ;
|
|
5203
|
+
try { doc = JSON.parse(text); } catch { return { error: `${path} is not JSON` }; }
|
|
5204
|
+
if (!isObj(doc) || doc.format !== "evals-lab/dataset" || !isObj(doc.dataset)) {
|
|
5205
|
+
return { error: `${path} is not a dataset's export` };
|
|
5206
|
+
}
|
|
5207
|
+
if (!(Number.isInteger(doc.version) && doc.version >= 1 && doc.version <= DATASET_BODY_VERSION)) {
|
|
5208
|
+
return { error: `${path} is version ${JSON.stringify(doc.version)}, and this reads versions 1 to ${DATASET_BODY_VERSION}` };
|
|
5209
|
+
}
|
|
5210
|
+
const body = upgradeDatasetBody(doc.dataset.body) ;
|
|
5211
|
+
if (!isObj(body) || !Array.isArray(body.cases)) return { error: `${path} holds no cases` };
|
|
5212
|
+
datasets.push({ name: String(doc.dataset.name ?? ""), body: body });
|
|
5213
|
+
}
|
|
5214
|
+
let raw ;
|
|
5215
|
+
try { raw = yaml.load(files[BUNDLE_PIPELINE] ); }
|
|
5216
|
+
catch { return { error: `${BUNDLE_PIPELINE} is not YAML` }; }
|
|
5217
|
+
// The bundle stands in for the Source: whatever it names, the items are
|
|
5218
|
+
// the ones handed over.
|
|
5219
|
+
const content = isObj(raw) ? contentOf(upgradePipeline(raw)) : null;
|
|
5220
|
+
const refs = [];
|
|
5221
|
+
// Read as this version, whichever wrote the bundle: a link's group, or a
|
|
5222
|
+
// group of its own's Cases from.
|
|
5223
|
+
for (const t of isObj(raw) ? evalsOf(upgradePipeline(raw)) : []) {
|
|
5224
|
+
const ref = casesRef(t);
|
|
5225
|
+
if (!ref) continue;
|
|
5226
|
+
if (datasets.some(d => d.name === ref.name)) refs.push({ id: ref.id, name: ref.name });
|
|
5227
|
+
}
|
|
5228
|
+
const imported = importPipeline(raw, { profiles, groups: refs,
|
|
5229
|
+
sources: content?.type === "source" && isRef(content.ref) ? [content.ref] : [] });
|
|
5230
|
+
if ("error" in imported) return imported;
|
|
5231
|
+
if (imported.missing.length) return { error: `${imported.missing[0]}: the bundle does not hold it` };
|
|
5232
|
+
const byId = {};
|
|
5233
|
+
for (const r of refs) byId[r.id] = datasets.find(d => d.name === r.name) .body;
|
|
5234
|
+
const caseItems = [...new Set(Object.values(byId).flatMap(b => gradedSetFrom(b ).filter(c => !c.todo).map(caseItem)).filter(Boolean))];
|
|
5235
|
+
const run = resolvePipeline(imported.doc, {
|
|
5236
|
+
profiles: id => profiles.find(p => p.id === id) ?? null,
|
|
5237
|
+
files: ctx.items ? [...ctx.items].sort() : caseItems,
|
|
5238
|
+
});
|
|
5239
|
+
const bad = validatePipeline(run, { groups: refs });
|
|
5240
|
+
if (bad.length) return { error: bad[0] };
|
|
5241
|
+
return { run, datasets: byId };
|
|
5242
|
+
}
|
|
5243
|
+
|
|
4613
5244
|
/**
|
|
4614
5245
|
* Scenario [i] of a run document as runPipeline and a transport take it:
|
|
4615
5246
|
* each stage's wording, kind, image flag and modifiers, the token set of
|
|
@@ -4779,22 +5410,40 @@ function scenarioEvals(run , i , items
|
|
|
4779
5410
|
});
|
|
4780
5411
|
}
|
|
4781
5412
|
|
|
5413
|
+
/** One eval group's verdict over a scenario (§17): a whole-run group's own
|
|
5414
|
+
verdict, or a per-item group's "every item it read passed"; null where it
|
|
5415
|
+
was skipped, or had nothing to read yet. */
|
|
5416
|
+
function groupPass(o ) {
|
|
5417
|
+
if (o.skipped) return null;
|
|
5418
|
+
if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
|
|
5419
|
+
return o.ran ? o.passed === o.ran : null;
|
|
5420
|
+
}
|
|
5421
|
+
|
|
4782
5422
|
/** A scenario's pass or fail over every eval that read it: null where none
|
|
4783
5423
|
has anything to say yet. */
|
|
4784
5424
|
function scenarioPasses(outcomes ) {
|
|
4785
|
-
|
|
4786
|
-
|
|
4787
|
-
|
|
4788
|
-
|
|
4789
|
-
|
|
4790
|
-
|
|
4791
|
-
|
|
4792
|
-
|
|
4793
|
-
|
|
4794
|
-
|
|
4795
|
-
|
|
4796
|
-
|
|
4797
|
-
return
|
|
5425
|
+
const passes = outcomes.map(groupPass);
|
|
5426
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5427
|
+
return !passes.includes(false);
|
|
5428
|
+
}
|
|
5429
|
+
|
|
5430
|
+
/** A scenario's overall verdict under a run's pass rule (§17): each linked
|
|
5431
|
+
group's verdict folded together the way `pass` says -- every group passes,
|
|
5432
|
+
or at least a number of them. Null where no group has a verdict yet, so a
|
|
5433
|
+
run with no evals, or one still grading, reads as it does without a rule. */
|
|
5434
|
+
function overallVerdict(outcomes , pass ) {
|
|
5435
|
+
const passes = outcomes.map(groupPass);
|
|
5436
|
+
if (!passes.some((p) => p != null)) return null;
|
|
5437
|
+
return passVerdict(pass, passes);
|
|
5438
|
+
}
|
|
5439
|
+
|
|
5440
|
+
/** Whether a run passes a Target under its overall pass rule (§17): `all`
|
|
5441
|
+
passes when no linked group fails, `atLeast` when at least `count` pass.
|
|
5442
|
+
[passes] is each group's pass (true), fail (false), or neither (null: it
|
|
5443
|
+
graded nothing, or an earlier group skipped it). */
|
|
5444
|
+
function passVerdict(pass , passes ) {
|
|
5445
|
+
if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
|
|
5446
|
+
return !passes.includes(false);
|
|
4798
5447
|
}
|
|
4799
5448
|
|
|
4800
5449
|
/**
|
|
@@ -4995,15 +5644,15 @@ export {
|
|
|
4995
5644
|
COUNT_OP_LABEL, LENGTH_OP_LABEL,
|
|
4996
5645
|
PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
|
|
4997
5646
|
SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
|
|
4998
|
-
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, connectionOf, connectionSettings,
|
|
5647
|
+
registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
|
|
4999
5648
|
connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
|
|
5000
5649
|
CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
|
|
5001
5650
|
contentOf, withContent, replyOf,
|
|
5002
|
-
targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5651
|
+
targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
|
|
5003
5652
|
tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
|
|
5004
|
-
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5005
|
-
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
|
|
5006
|
-
evalLabel, evalsOf, evalsDataset, isWholeRun, EVAL_FIELDS,
|
|
5007
|
-
pipelineToYaml, importPipeline, yamlToPipeline,
|
|
5653
|
+
blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
|
|
5654
|
+
scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
|
|
5655
|
+
evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
|
|
5656
|
+
pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
|
|
5008
5657
|
pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,
|
|
5009
5658
|
};
|