evals-lab 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -35,6 +35,8 @@
35
35
  import yaml from "./js-yaml.mjs";
36
36
 
37
37
  import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
38
+ import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
39
+ import { flowApi, } from "./flows/flowApi.mjs";
38
40
  import registerMetrics from "./metrics/builtin.mjs";
39
41
  import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
40
42
 
@@ -543,6 +545,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
543
545
 
544
546
 
545
547
 
548
+
549
+
550
+
551
+
552
+
553
+
546
554
 
547
555
 
548
556
 
@@ -646,9 +654,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
646
654
 
647
655
 
648
656
 
649
-
650
-
651
-
657
+
658
+
659
+
652
660
 
653
661
 
654
662
 
@@ -662,6 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
662
670
 
663
671
 
664
672
 
673
+
674
+
675
+
676
+
665
677
 
666
678
 
667
679
  /** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
@@ -775,16 +787,18 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
775
787
 
776
788
  // ---- the registries ----
777
789
 
778
- /** An option a modifier or an eval type exposes for editing. */
779
-
780
-
781
-
790
+ /** An option a modifier or an eval type exposes for editing. `hint` says
791
+ what it does in one short line, under its control, where the label
792
+ cannot: a switch's effect, not why it exists. */
793
+
794
+
795
+
782
796
 
783
-
784
-
785
-
797
+
798
+
799
+
786
800
 
787
-
801
+
788
802
 
789
803
  /** What an output kind made of a reply. */
790
804
 
@@ -878,11 +892,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
878
892
 
879
893
 
880
894
  /** A kind of Source: what one of its items is, and what the page and the
881
- runner do with it. A reader asks the entry; it never branches on an id. */
895
+ runner do with it. A reader asks the entry; it never branches on an id.
896
+ The row's `type` is the kind; a workflow kind's platform -- which flow
897
+ engine a row speaks to -- is `config.platform`, a WORKFLOW_PLATFORMS id. */
882
898
 
883
899
 
884
900
 
885
901
 
902
+
903
+
886
904
 
887
905
 
888
906
 
@@ -891,10 +909,68 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
891
909
 
892
910
 
893
911
 
912
+
913
+    
914
+
915
+
894
916
 
895
917
 
918
+
919
+
920
+
896
921
 
897
922
 
923
+ /** What a workflow of one platform -- a flow engine -- answers: the pieces
924
+ that were Power-Automate-specific, lifted behind the platform id so
925
+ generic code asks the entry and never names a platform. Power Automate is
926
+ the one built; a second (Logic App, n8n) is one more entry. */
927
+
928
+
929
+
930
+
931
+
932
+
933
+
934
+
935
+
936
+
937
+
938
+
939
+
940
+
941
+
942
+
943
+
944
+
945
+
946
+
947
+
948
+
949
+
950
+
951
+
952
+
953
+
954
+
955
+
956
+
957
+
958
+
959
+  
960
+
961
+
962
+ /** A platform's management API reader (flowApi.ts for Power Automate); its
963
+ shape is the platform's own, so generic code holds it opaquely. */
964
+
965
+
966
+ const WORKFLOW_PLATFORMS = Object.create(null);
967
+ /** The platform a workflow Source speaks to, from its `config.platform`;
968
+ undefined for a row that is not a workflow or names an unknown platform. */
969
+ const platformOf = (src ) => {
970
+ const id = src?.config?.platform;
971
+ return isStr(id) ? WORKFLOW_PLATFORMS[id] : undefined;
972
+ };
973
+
898
974
 
899
975
 
900
976
 
@@ -959,6 +1035,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
959
1035
  the item, and a grader, where the run has them. */
960
1036
 
961
1037
 
1038
+
1039
+
1040
+
962
1041
 
963
1042
 
964
1043
 
@@ -966,6 +1045,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
966
1045
 
967
1046
 
968
1047
 
1048
+
1049
+
1050
+
1051
+
969
1052
 
970
1053
 
971
1054
  /** A job's stages, in the order they run (pipeline-model §16). A target's
@@ -1053,16 +1136,43 @@ const SLOTS = ["content", "target", "responses"];
1053
1136
 
1054
1137
 
1055
1138
 
1056
- /** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
1057
- named its Source; 4 and earlier, neither. */
1058
- const DATASET_BODY_VERSION = 7 ;
1139
+ /** The dataset body's version: 8 reads the recorded-reply metric ids under
1140
+ their new names (#299); 7 is an eval group (§17); 6 marks itself; 5 named
1141
+ its Source; 4 and earlier, neither. */
1142
+ const DATASET_BODY_VERSION = 8 ;
1143
+
1144
+ /** The recorded-reply metric type ids renamed at version 8 (#299): the family
1145
+ read "production" before the Recorded target gave it a home. The ids are
1146
+ stored tokens inside eval group bodies and a pipeline's private group, so
1147
+ they are mapped wherever a body is read rather than rewritten in place. */
1148
+ const RECORDED_IDS = {
1149
+ "equals-production": "same-as-recorded",
1150
+ "fields-equal-production": "fields-equal-recorded",
1151
+ "same-parse-outcome": "same-parse-as-recorded",
1152
+ };
1153
+ const renameMetric = (m ) =>
1154
+ isObj(m) && isStr(m.type) && RECORDED_IDS[m.type] ? { ...m, type: RECORDED_IDS[m.type] } : m;
1155
+ const renameMetrics = (list ) =>
1156
+ Array.isArray(list) ? list.map(renameMetric) : list;
1157
+ /** A version-7 body (or a freshly made version-8 one) with its recorded-reply
1158
+ metric ids read under their version-8 names, at version 8. */
1159
+ function recordedIdsV7(b ) {
1160
+ return {
1161
+ ...b,
1162
+ version: DATASET_BODY_VERSION,
1163
+ ...(Array.isArray(b.every) ? { every: renameMetrics(b.every) } : {}),
1164
+ ...(Array.isArray(b.run) ? { run: renameMetrics(b.run) } : {}),
1165
+ ...(Array.isArray(b.cases) ? { cases: b.cases.map((c ) =>
1166
+ isObj(c) && Array.isArray(c.metrics) ? { ...c, metrics: renameMetrics(c.metrics) } : c) } : {}),
1167
+ } ;
1168
+ }
1059
1169
 
1060
1170
  /** The metrics whose Ignore case version 6 made mean what it says for a
1061
1171
  reply read as a list, as well as one read as text. */
1062
1172
  const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1063
1173
 
1064
1174
  /**
1065
- * [body] as this version of a dataset (7), from any earlier one. Version 1
1175
+ * [body] as this version of a dataset (8), from any earlier one. Version 1
1066
1176
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
1067
1177
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
1068
1178
  * Version 2 held `rules`, which clean a job's answer and so belong to the
@@ -1075,8 +1185,12 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1075
1185
  * a list's items ignoring case whatever a Contains metric's Ignore case
1076
1186
  * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1077
1187
  * the body says its version. Version 6 is a group's cases alone: version 7
1078
- * adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
1079
- * stored row is read that way rather than rewritten. Every reader of a
1188
+ * adds its scoring, grader, Every item and Whole run (`groupOfV6`). Version 8
1189
+ * reads the recorded-reply metric ids (`equals-production` and its two
1190
+ * siblings) under their new names (`RECORDED_IDS`, #299), in a group's own
1191
+ * metrics and its cases' -- a pipeline's private group is a group body, so
1192
+ * this covers it too. A stored row is read that way rather than rewritten.
1193
+ * Every reader of a
1080
1194
  * body calls this: the runner, the page, a run's kept copy. A reader that
1081
1195
  * needs a version-2 body's rules -- to upgrade a pipeline graded against
1082
1196
  * it -- takes them first (`datasetRules`). Pure: the same body gives the
@@ -1087,19 +1201,25 @@ function upgradeDatasetBody (body ) {
1087
1201
  if (!isObj(body)) return body;
1088
1202
  if (body.version === DATASET_BODY_VERSION) return body;
1089
1203
  const b = body ;
1090
- // A body saying any other version is one this lab does not read.
1091
- if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
1204
+ if ("version" in b) {
1205
+ // Version 7 is an eval group whose recorded-reply metrics are read under
1206
+ // their version-8 ids; version 6 is its cases alone. Any other version is
1207
+ // one this lab does not read.
1208
+ if (b.version === 7) return recordedIdsV7(b);
1209
+ if (b.version === 6) return recordedIdsV7(groupOfV6(b));
1210
+ return body;
1211
+ }
1092
1212
  // A body that names its Source, even as null, is version 5.
1093
1213
  if ("source" in b) {
1094
- return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
1214
+ return recordedIdsV7(groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) }));
1095
1215
  }
1096
1216
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
1097
1217
  // Not a body of any version -- a copy kept while a dataset was an overlay.
1098
1218
  if (!raw) return body;
1099
- return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
1219
+ return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
1100
1220
  }
1101
1221
 
1102
- /** A version-6 body as a version-7 eval group: scored All, the lab's
1222
+ /** A version-6 body as a version-8 eval group: scored All, the lab's
1103
1223
  grader, and no metrics of its own for every item or the whole run --
1104
1224
  what a Metrics eval naming the dataset with none of its own graded, which
1105
1225
  is what every Graded set converted to. */
@@ -1243,7 +1363,12 @@ function datasetRules(body ) {
1243
1363
 
1244
1364
 
1245
1365
 
1246
-
1366
+
1367
+
1368
+
1369
+
1370
+
1371
+
1247
1372
 
1248
1373
 
1249
1374
 
@@ -1614,12 +1739,33 @@ CONNECTION_TYPES.echo = {
1614
1739
  request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1615
1740
  };
1616
1741
 
1742
+ /** A call's recorded result as transport-level text -- what a model asked the
1743
+ same way would have come back with, before the flow reads it. The runner's
1744
+ replay branch reads it as the flow reads a reply (`httpReplyOf`); this is
1745
+ the fallback where there is no Read as to read it through. */
1746
+ function recordedRaw(record ) {
1747
+ const result = isObj(record) && isObj(record.result) ? record.result : null;
1748
+ if (!result) return "";
1749
+ return isStr(result.body) ? result.body : JSON.stringify(result.body ?? "");
1750
+ }
1751
+
1752
+ // Recorded (#299): a target that replays each call's recorded reply, sending
1753
+ // nothing and holding no key or model. The runner reads the recorded result as
1754
+ // the flow reads a reply; `local` is the no-Read-as fallback.
1755
+ CONNECTION_TYPES.recorded = {
1756
+ id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
1757
+ description: "Replays the reply production recorded for each call. Sends nothing.",
1758
+ local: (_item, _sent, record) => recordedRaw(record),
1759
+ request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1760
+ };
1761
+
1617
1762
  /** A local type's answer to one stage, as a model's would come back. An item
1618
1763
  with no text of its own (Prompt only) is answered from the prompt, and
1619
1764
  that reply is read against nothing -- the prompt is the reply, not an
1620
- instruction it could repeat. */
1621
- function localAnswer(conn , text , sent ) {
1622
- const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent);
1765
+ instruction it could repeat. A replaying type (Recorded) is handed the
1766
+ item's record to answer from. */
1767
+ function localAnswer(conn , text , sent , record ) {
1768
+ const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent, record);
1623
1769
  return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
1624
1770
  }
1625
1771
 
@@ -2991,24 +3137,54 @@ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
2991
3137
  SOURCE_TYPES.files = {
2992
3138
  label: "File Library",
2993
3139
  description: "A folder of images and text files you upload.",
3140
+ group: "Files",
2994
3141
  noun: "file",
2995
3142
  uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
2996
3143
  sendsImage: true,
3144
+ hasActions: false,
2997
3145
  itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
2998
3146
  };
2999
3147
 
3000
- // A Power Automate cloud flow (docs/power-automate.md): its items are
3001
- // records -- one call of a Power Automate step each: what its template read,
3002
- // the request production sent and what came back (flows/record.ts) -- and it
3003
- // keeps the flow's definition beside them. Records are imported from run
3004
- // history, written by hand or uploaded; nothing it holds is an image.
3005
- SOURCE_TYPES["power-automate"] = {
3006
- label: "Power Automate workflow",
3007
- description: "A cloud flow read from Microsoft 365, with records of its steps' calls from its run history.",
3148
+ // A workflow Source: a cloud flow, read through a platform (WORKFLOW_PLATFORMS,
3149
+ // below) its row names in `config.platform`. Its items are records -- one call
3150
+ // of one of the flow's steps each: what the step's template read, the request
3151
+ // production sent and what came back (flows/record.ts) -- and it keeps the
3152
+ // flow's definition beside them. The kind is generic; the platform owns what
3153
+ // is engine-specific, and the type's shown label is the platform's.
3154
+ SOURCE_TYPES.workflow = {
3155
+ label: "Workflow",
3156
+ description: "A cloud flow, with records of its steps' calls from its run history.",
3157
+ group: "Workflows",
3008
3158
  noun: "record",
3009
3159
  uploads: [".json"],
3010
3160
  sendsImage: false,
3161
+ hasActions: true,
3011
3162
  itemText: () => false,
3163
+ platforms: ["power-automate"], // platform: the WORKFLOW_PLATFORMS a workflow offers
3164
+ };
3165
+
3166
+ // Power Automate (docs/power-automate.md): the one workflow platform built.
3167
+ // Its pieces were the lab's Power-Automate-specific code -- flows/record.ts,
3168
+ // flows/wdl.ts and flowApi.ts -- gathered here so generic code calls the
3169
+ // entry, never a platform id.
3170
+ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapter's own id
3171
+ id: "power-automate", // platform: its id, the row's config.platform
3172
+ label: "Power Automate flow",
3173
+ nouns: { workflow: "flow", action: "action" },
3174
+ signIn: "microsoft",
3175
+ uploads: { ".json": "application/json; charset=utf-8" },
3176
+ keepsDefinition: true,
3177
+ api: token => flowApi(token),
3178
+ stepsOf: flowStepsOf,
3179
+ actionsOf: actionsByName,
3180
+ readerFor,
3181
+ evaluate,
3182
+ asText,
3183
+ readsOf,
3184
+ // Copy request… writes the body back as a flow's HTTP action holds it: JSON
3185
+ // whose strings keep Power Automate's own `@{…}` expressions untouched, so a
3186
+ // rebuilt body round-trips into the flow (docs/power-automate.md).
3187
+ render: body => JSON.stringify(body, null, 2),
3012
3188
  };
3013
3189
 
3014
3190
  CONTENT_TYPES.source = {
@@ -3188,7 +3364,11 @@ function metricInput(res , kase , more )
3188
3364
  const last = res.transcript?.at(-1);
3189
3365
  const text = last?.got ?? res.raw ?? "";
3190
3366
  return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
3191
- error: res.error ?? null, ms: res.ms ?? 0, production: more.production ?? null, kase };
3367
+ // A case that has been given its own expected reply ("Take B as
3368
+ // expected") is compared with that; otherwise the Source's recorded
3369
+ // reply for the item.
3370
+ error: res.error ?? null, ms: res.ms ?? 0,
3371
+ recorded: typeof kase?.recorded === "string" ? kase.recorded : (more.production ?? null), kase };
3192
3372
  }
3193
3373
 
3194
3374
  /** A whole run's replies as one input: every reply's text together, and
@@ -3202,7 +3382,7 @@ function runInput(ress , plain )
3202
3382
  terms.push(...(r.terms || []));
3203
3383
  ms += r.ms ?? 0;
3204
3384
  }
3205
- return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, production: null, kase: null };
3385
+ return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
3206
3386
  }
3207
3387
 
3208
3388
  /** One metric's reading of [input]: `of`, `not` and a failure all applied,
@@ -3487,10 +3667,21 @@ EVAL_TYPES.group = {
3487
3667
  read: async (t, kase, res, more) => {
3488
3668
  const group = groupOf(t, more);
3489
3669
  const grader = isRef(group.grader) ? group.grader : isRef(t.grader) ? t.grader : null;
3490
- return readGroup({ ...group, grader } , casesRef(t) ? kase : null, res, more);
3670
+ const ref = casesRef(t);
3671
+ // The item's case in this eval's own group: resolved from the group whose
3672
+ // cases it reads where the runner names the item (grading several groups
3673
+ // at once, --groups), else the case handed in (a single group).
3674
+ const own = ref && isStr(more.item) ? caseIn(more.group?.(ref), more.item) : kase;
3675
+ return readGroup({ ...group, grader } , ref ? own : null, res, more);
3491
3676
  },
3492
3677
  };
3493
3678
 
3679
+ /** The case for an item in a group's body, by the name its file has, or null:
3680
+ a non-todo case whose item matches, as a Library group reads one. */
3681
+ function caseIn(body , item ) {
3682
+ return (body?.cases ?? []).find(c => !c.todo && caseItem(c) === item) ?? null;
3683
+ }
3684
+
3494
3685
  // ---- the lab's own scorers, as metrics ----------------------------------------
3495
3686
  // What the Single Test checks, one metric each, so an eval of it converts to
3496
3687
  // Metrics that read a run exactly as it did (metrics-parity-check.js). Here
@@ -3612,10 +3803,38 @@ function readGroupRun(group , ress
3612
3803
  return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
3613
3804
  }
3614
3805
 
3806
+ /**
3807
+ * [body] with the cases named in [replies] expecting the reply given for
3808
+ * their item -- Results' "Take B as expected" (docs/workflow-sources.md §
3809
+ * Calls are graded cases). Each such case keeps everything else and gains
3810
+ * `recorded` (the reply the recorded-reply metrics compare against, read in
3811
+ * place of the Source's) and `todo: false`, so a later run grades the item
3812
+ * against that reply. A selected item the group has no case for, or one whose
3813
+ * reply is missing (Target B errored, or is the Recorded target), is left
3814
+ * untouched and named in `missing`; `taken` names the items whose expected was
3815
+ * written. Pure: the same body and replies give the same body, so Undo is the
3816
+ * body as it was.
3817
+ */
3818
+ function takeAsExpected(body , replies )
3819
+ {
3820
+ const byItem = new Map ();
3821
+ body.cases.forEach((c, i) => { const item = caseItem(c); if (item) byItem.set(item, i); });
3822
+ const cases = body.cases.slice();
3823
+ const taken = [], missing = [];
3824
+ for (const item of Object.keys(replies)) {
3825
+ const reply = replies[item];
3826
+ const at = byItem.get(item);
3827
+ if (at == null || typeof reply !== "string") { missing.push(item); continue; }
3828
+ cases[at] = { ...cases[at], recorded: reply, todo: false };
3829
+ taken.push(item);
3830
+ }
3831
+ return { body: { ...body, cases }, taken, missing };
3832
+ }
3833
+
3615
3834
  /** The grader a Metrics eval names, reached through what the runner hands it. */
3616
3835
  function graderCtx(t , more ) {
3617
3836
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3618
- return ask ? { ask } : {};
3837
+ return { ...(ask ? { ask } : {}), ...(more.recordedTarget ? { recordedTarget: true } : {}) };
3619
3838
  }
3620
3839
 
3621
3840
  /** A metric's options in a few words -- "watering", "^\\{" -- or nothing
@@ -3791,9 +4010,30 @@ STEP_TYPES.echo = {
3791
4010
  },
3792
4011
  };
3793
4012
 
3794
- /** What a local target step is answered as: Echo's own connection, with no
4013
+ // Recorded (#299): a target that replays each call's recorded reply, read as
4014
+ // the flow reads one. No profile, no prompt -- the record is the reply.
4015
+ STEP_TYPES.recorded = {
4016
+ label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
4017
+ description: "Replays the reply production recorded for each call. Sends nothing.",
4018
+ firstJobOnly: "a call's recorded reply is job 1's",
4019
+ in: "item", out: "text",
4020
+ apply: "runPipeline",
4021
+ // `prompt` only so the step satisfies the target-step shape; it is never
4022
+ // sent (the record is the reply), and the prompt rule exempts it below.
4023
+ fields: ["type", "prompt"],
4024
+ validate(){},
4025
+ };
4026
+
4027
+ /** What a local target step is answered as, by the connection type it stands
4028
+ in for (`replaces`): Echo's own connection, or Recorded's -- each with no
3795
4029
  address, key or model. */
3796
4030
  const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
4031
+ const RECORDED_CONNECTION = { name: "Recorded", url: "", model: "", type: "recorded" };
4032
+ const LOCAL_CONNECTIONS =
4033
+ { echo: ECHO_CONNECTION, recorded: RECORDED_CONNECTION };
4034
+ /** The connection a local target step stands in for, by its entry. */
4035
+ const localConnOf = (entry ) =>
4036
+ LOCAL_CONNECTIONS[entry?.replaces ?? ""] ?? ECHO_CONNECTION;
3797
4037
 
3798
4038
  // Responses: how a job's reply is read, in order -- possibly not at all.
3799
4039
 
@@ -3901,9 +4141,15 @@ function targetsOf(doc
3901
4141
  return Array.isArray(doc?.targets) ? doc .targets : [];
3902
4142
  }
3903
4143
 
3904
- /** Target [i]'s name, or the number it has always shown. */
4144
+ /** Target [i]'s letter, A for the first: the same in every job, so a
4145
+ Target reads as one column through them. A number past Z. */
4146
+ function targetLetter(i ) {
4147
+ return i < 26 ? String.fromCharCode(65 + i) : String(i + 1);
4148
+ }
4149
+
4150
+ /** Target [i]'s name, or its letter's: "Target A". */
3905
4151
  function targetLabel(doc , i ) {
3906
- return (targetsOf(doc)[i]?.name || "").trim() || `Target ${i + 1}`;
4152
+ return (targetsOf(doc)[i]?.name || "").trim() || `Target ${targetLetter(i)}`;
3907
4153
  }
3908
4154
 
3909
4155
  /** What target [i] sends in job [k]: its step there. */
@@ -3927,6 +4173,16 @@ function targetProfileOf(doc
3927
4173
  return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
3928
4174
  }
3929
4175
 
4176
+ /** Whether target [i] is a Recorded target: it replays each call's recorded
4177
+ reply rather than sending anything (its job-1 step answers locally with a
4178
+ connection that `replays`). A recorded-reply metric compares the recorded
4179
+ reply with itself there, so it says nothing (n/a). Asked through the
4180
+ registry, never a step id. */
4181
+ function recordedTargetOf(doc , i ) {
4182
+ const entry = targetEntryOf(doc, i, 0);
4183
+ return !!(entry?.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays);
4184
+ }
4185
+
3930
4186
  /** [doc] with target [i]'s step in job [k] changed: [change] merged into
3931
4187
  it, or the step [change] makes of it. */
3932
4188
  function withTarget (doc , i , k ,
@@ -4150,11 +4406,8 @@ STEP_TYPES.evals = {
4150
4406
  + `and the last job answers with ${OUTPUT_KINDS[lastKind] .noun || lastKind}`);
4151
4407
  }
4152
4408
  });
4153
- // The worker is handed one Library group's body to grade against
4154
- // (server-side-runs §4) until it reads the body the queue keeps per
4155
- // group (#233).
4156
- const named = new Set(evals.map(casesRef).filter(Boolean).map(r => r .id));
4157
- if (named.size > 1) bad.push("the evals grade against one Library eval group at a time");
4409
+ // The worker reads a body per group now (#233), so the evals may grade
4410
+ // against several Library groups at once; the queue keeps each body.
4158
4411
  },
4159
4412
  };
4160
4413
 
@@ -4621,11 +4874,16 @@ function validatePipeline(input , ctx = {}) {
4621
4874
  // what it is read against. Over Prompt only the prompt is the reply.
4622
4875
  // What target [i] asks in job [k]: Echo's own connection for a step
4623
4876
  // answered here, else its profile as the lookups hold it.
4624
- const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION
4877
+ const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k))
4625
4878
  : c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
4626
4879
  const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
4627
4880
  for (let k = 0; k < n; k++) {
4628
4881
  const i = targets.findIndex((_, i) => {
4882
+ const entry = targetEntryOf(doc, i, k);
4883
+ // A step with no prompt of its own, or a replaying target (Recorded,
4884
+ // which answers from the record), is never asked for one.
4885
+ if (!entry?.fields?.includes("prompt")) return false;
4886
+ if (entry.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays) return false;
4629
4887
  const words = targetStepOf(doc, i, k)?.prompt;
4630
4888
  return (!isStr(words) || !words.trim())
4631
4889
  && !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
@@ -4638,12 +4896,14 @@ function validatePipeline(input , ctx = {}) {
4638
4896
  const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
4639
4897
  .find(n => !/\.(?:txt|md|csv)$/i.test(n));
4640
4898
  targets.forEach((_, i) => {
4641
- const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION : lookup(targetProfileOf(doc, i, k) )));
4899
+ const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k)) : lookup(targetProfileOf(doc, i, k) )));
4642
4900
  conns.forEach((p , k ) => {
4643
4901
  if (!local(p)) return;
4644
- const label = CONNECTION_TYPES[typeOf(p )] .label;
4645
- if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
4646
- else if (nonText) bad.push(`${targetLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
4902
+ const entry = CONNECTION_TYPES[typeOf(p )] ;
4903
+ // A replaying target (Recorded) answers from the record, not the item
4904
+ // text, so the text rules do not apply -- only the job-1 one does.
4905
+ if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${entry.label} answers only job 1${entry.replays ? "" : ", with the item's own text"}`);
4906
+ else if (nonText && !entry.replays) bad.push(`${targetLabel(doc, i)}: ${entry.label} answers each text item with its own text, and ${nonText} is not text`);
4647
4907
  });
4648
4908
  // A connection answers what its target's step asks: words, or a whole request.
4649
4909
  conns.forEach((p , k ) => {
@@ -5241,7 +5501,7 @@ function stagesFor(run , i )
5241
5501
  if (targetEntryOf(run, i, k)?.local) {
5242
5502
  const id = targetProfileOf(run, i, k)?.id;
5243
5503
  const held = id != null ? run.profiles?.[id] : undefined;
5244
- return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...ECHO_CONNECTION };
5504
+ return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...localConnOf(targetEntryOf(run, i, k)) };
5245
5505
  }
5246
5506
  const id = targetProfileOf(run, i, k) .id;
5247
5507
  return { id, ...(run.profiles?.[id] || {}) };
@@ -5256,7 +5516,7 @@ function stagesFor(run , i )
5256
5516
  function scenarioProfile(run , i ) {
5257
5517
  const id = targetsOf(run)[i]?.profile?.id;
5258
5518
  // A target that asks nothing ran as Echo.
5259
- if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name: ECHO_CONNECTION.name, settings: { ...ECHO_CONNECTION } };
5519
+ if (id == null && targetEntryOf(run, i, 0)?.local) { const lc = localConnOf(targetEntryOf(run, i, 0)); return { id: "", name: lc.name, settings: { ...lc } }; }
5260
5520
  const conn = id != null ? run.profiles?.[id] : null;
5261
5521
  return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
5262
5522
  }
@@ -5317,9 +5577,9 @@ function itemScores(run , kase , res
5317
5577
  return Object.keys(out).length ? out : null;
5318
5578
  }
5319
5579
 
5320
- /** What production replied to an item, for a run's metrics: its last job's
5321
- call says, from the item's record, or nobody does. */
5322
- function productionOf(run , record ) {
5580
+ /** What production recorded as the reply to an item, for a run's metrics: its
5581
+ last job's call says, from the item's record, or nobody does. */
5582
+ function recordedReplyOf(run , record ) {
5323
5583
  const last = run.jobs.at(-1), flow = flowStepOf(last);
5324
5584
  return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
5325
5585
  }
@@ -5390,22 +5650,40 @@ function scenarioEvals(run , i , items
5390
5650
  });
5391
5651
  }
5392
5652
 
5653
+ /** One eval group's verdict over a scenario (§17): a whole-run group's own
5654
+ verdict, or a per-item group's "every item it read passed"; null where it
5655
+ was skipped, or had nothing to read yet. */
5656
+ function groupPass(o ) {
5657
+ if (o.skipped) return null;
5658
+ if (o.whole) return o.verdict?.ran ? o.verdict.pass : null;
5659
+ return o.ran ? o.passed === o.ran : null;
5660
+ }
5661
+
5393
5662
  /** A scenario's pass or fail over every eval that read it: null where none
5394
5663
  has anything to say yet. */
5395
5664
  function scenarioPasses(outcomes ) {
5396
- let said = false;
5397
- for (const o of outcomes) {
5398
- if (o.skipped) continue;
5399
- if (o.whole) {
5400
- if (!o.verdict?.ran) continue;
5401
- said = true;
5402
- if (!o.verdict.pass) return false;
5403
- } else if (o.ran) {
5404
- said = true;
5405
- if (o.passed < o.ran) return false;
5406
- }
5407
- }
5408
- return said ? true : null;
5665
+ const passes = outcomes.map(groupPass);
5666
+ if (!passes.some((p) => p != null)) return null;
5667
+ return !passes.includes(false);
5668
+ }
5669
+
5670
+ /** A scenario's overall verdict under a run's pass rule (§17): each linked
5671
+ group's verdict folded together the way `pass` says -- every group passes,
5672
+ or at least a number of them. Null where no group has a verdict yet, so a
5673
+ run with no evals, or one still grading, reads as it does without a rule. */
5674
+ function overallVerdict(outcomes , pass ) {
5675
+ const passes = outcomes.map(groupPass);
5676
+ if (!passes.some((p) => p != null)) return null;
5677
+ return passVerdict(pass, passes);
5678
+ }
5679
+
5680
+ /** Whether a run passes a Target under its overall pass rule (§17): `all`
5681
+ passes when no linked group fails, `atLeast` when at least `count` pass.
5682
+ [passes] is each group's pass (true), fail (false), or neither (null: it
5683
+ graded nothing, or an earlier group skipped it). */
5684
+ function passVerdict(pass , passes ) {
5685
+ if (pass?.mode === "atLeast") return passes.filter(p => p === true).length >= pass.count;
5686
+ return !passes.includes(false);
5409
5687
  }
5410
5688
 
5411
5689
  /**
@@ -5597,23 +5875,23 @@ registerMetrics({ registerKinds });
5597
5875
  export {
5598
5876
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
5599
5877
  loopReplyError, preparedSize,
5600
- termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
5878
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, takeAsExpected, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
5601
5879
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
5602
5880
  CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
5603
5881
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
5604
5882
  tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
5605
5883
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
5606
5884
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
5607
- PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
5608
- SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
5885
+ PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
5886
+ SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
5609
5887
  registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
5610
5888
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5611
5889
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5612
5890
  contentOf, withContent, replyOf,
5613
- targetsOf, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5891
+ targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
5614
5892
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
5615
5893
  blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5616
- scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, isSkipped, failedScore,
5894
+ scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,
5617
5895
  evalLabel, evalsOf, evalsDataset, casesRef, groupOf, groupOfMetrics, pipelineWarnings, isWholeRun, EVAL_FIELDS,
5618
5896
  pipelineToYaml, importPipeline, yamlToPipeline, exportBundle, readBundle, bundleProblems, junitXml,
5619
5897
  pluginHost, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher,