evals-lab 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -5,6 +5,46 @@ Newest first. Read a release's **Upgrade notes** before installing it: an
5
5
  upgrade can rewrite what the lab keeps in your data directory, and an older
6
6
  version cannot always read it back.
7
7
 
8
+ ## 0.7.0
9
+
10
+ ### Upgrade notes
11
+
12
+ - **Copy your data directory before upgrading** (see "User data" in the
13
+ README). An eval group 0.7.0 saves is dataset version 8, which 0.6.x
14
+ cannot read. To go back, reinstall 0.6.0 and restore the copy.
15
+ - On its first start, 0.7.0 moves everything the lab keeps into one
16
+ workspace, Default. Nothing on the page changes.
17
+ - Stored eval groups are not rewritten on upgrade: a version-7 group reads
18
+ as version 8, and is saved as version 8 at its next edit.
19
+ - Export JSON writes version 8, which 0.6.x refuses to import. Import reads
20
+ versions 1 to 8.
21
+ - The metrics that compare a reply with production's are now the Recorded
22
+ reply metrics. Groups that use them grade as before.
23
+
24
+ ### Added
25
+
26
+ - Test… on a Power Automate Source's action makes an eval group from the
27
+ Source's calls, graded against the replies production recorded, and a
28
+ pipeline to run it, then opens it in Runs.
29
+ - The Recorded target replays each call's recorded reply and sends nothing,
30
+ so a run compares an edited request with what production did.
31
+ - Adding a Power Automate Source imports the calls of its last five runs in
32
+ the background, while the tab stays open.
33
+ - Results of a run with a Recorded target: a table of each check with a
34
+ column per Target (n/a where a check cannot apply), a call that opens both
35
+ replies side by side, and Take B as expected, which makes the selected
36
+ calls expect Target B's reply.
37
+ - An HTTP Request target shows its words with the item's fields as chips.
38
+ Select prompt… and Save to library… use the Prompt library, and Copy
39
+ request… writes the request back for Power Automate, or as JSON.
40
+
41
+ ### Changed
42
+
43
+ - A Power Automate Source's records are its Calls, with Add call.
44
+ - Targets sit side by side on a wide screen. A job's Content and Responses
45
+ fold to one line until opened.
46
+ - A row with one action shows it as a button; two or more are in its ⋯ menu.
47
+
8
48
  ## 0.6.0
9
49
 
10
50
  ### Added
package/lab/VERSION CHANGED
@@ -1 +1 @@
1
- 0.6.0 (2026.10.05-430)
1
+ 0.7.0 (2026.10.06-456)
@@ -35,6 +35,8 @@
35
35
  import yaml from "./js-yaml.mjs";
36
36
 
37
37
  import { evaluate, asText, WdlError, } from "./flows/wdl.mjs";
38
+ import { stepsOf as flowStepsOf, actionsByName, readerFor, readsOf, } from "./flows/record.mjs";
39
+ import { flowApi, } from "./flows/flowApi.mjs";
38
40
  import registerMetrics from "./metrics/builtin.mjs";
39
41
  import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProblems, dropBy, rejectBy, matcher } from "./kinds/list.mjs";
40
42
 
@@ -543,6 +545,12 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
543
545
 
544
546
 
545
547
 
548
+
549
+
550
+
551
+
552
+
553
+
546
554
 
547
555
 
548
556
 
@@ -646,9 +654,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
646
654
 
647
655
 
648
656
 
649
-
650
-
651
-
657
+
658
+
659
+
652
660
 
653
661
 
654
662
 
@@ -662,6 +670,10 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
662
670
 
663
671
 
664
672
 
673
+
674
+
675
+
676
+
665
677
 
666
678
 
667
679
  /** A metric's reading: pass or fail, a score -- 0 to 1, or a count -- and why. */
@@ -880,11 +892,15 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
880
892
 
881
893
 
882
894
  /** A kind of Source: what one of its items is, and what the page and the
883
- runner do with it. A reader asks the entry; it never branches on an id. */
895
+ runner do with it. A reader asks the entry; it never branches on an id.
896
+ The row's `type` is the kind; a workflow kind's platform -- which flow
897
+ engine a row speaks to -- is `config.platform`, a WORKFLOW_PLATFORMS id. */
884
898
 
885
899
 
886
900
 
887
901
 
902
+
903
+
888
904
 
889
905
 
890
906
 
@@ -893,10 +909,68 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
893
909
 
894
910
 
895
911
 
912
+
913
+    
914
+
915
+
896
916
 
897
917
 
918
+
919
+
920
+
898
921
 
899
922
 
923
+ /** What a workflow of one platform -- a flow engine -- answers: the pieces
924
+ that were Power-Automate-specific, lifted behind the platform id so
925
+ generic code asks the entry and never names a platform. Power Automate is
926
+ the one built; a second (Logic App, n8n) is one more entry. */
927
+
928
+
929
+
930
+
931
+
932
+
933
+
934
+
935
+
936
+
937
+
938
+
939
+
940
+
941
+
942
+
943
+
944
+
945
+
946
+
947
+
948
+
949
+
950
+
951
+
952
+
953
+
954
+
955
+
956
+
957
+
958
+
959
+  
960
+
961
+
962
+ /** A platform's management API reader (flowApi.ts for Power Automate); its
963
+ shape is the platform's own, so generic code holds it opaquely. */
964
+
965
+
966
+ const WORKFLOW_PLATFORMS = Object.create(null);
967
+ /** The platform a workflow Source speaks to, from its `config.platform`;
968
+ undefined for a row that is not a workflow or names an unknown platform. */
969
+ const platformOf = (src ) => {
970
+ const id = src?.config?.platform;
971
+ return isStr(id) ? WORKFLOW_PLATFORMS[id] : undefined;
972
+ };
973
+
900
974
 
901
975
 
902
976
 
@@ -961,6 +1035,9 @@ import list, { ECHO, legacyTagsOut, splitItems, tidy, tryExamples, itemRulesProb
961
1035
  the item, and a grader, where the run has them. */
962
1036
 
963
1037
 
1038
+
1039
+
1040
+
964
1041
 
965
1042
 
966
1043
 
@@ -1059,16 +1136,43 @@ const SLOTS = ["content", "target", "responses"];
1059
1136
 
1060
1137
 
1061
1138
 
1062
- /** The dataset body's version: 7 is an eval group (§17); 6 marks itself; 5
1063
- named its Source; 4 and earlier, neither. */
1064
- const DATASET_BODY_VERSION = 7 ;
1139
+ /** The dataset body's version: 8 reads the recorded-reply metric ids under
1140
+ their new names (#299); 7 is an eval group (§17); 6 marks itself; 5 named
1141
+ its Source; 4 and earlier, neither. */
1142
+ const DATASET_BODY_VERSION = 8 ;
1143
+
1144
+ /** The recorded-reply metric type ids renamed at version 8 (#299): the family
1145
+ read "production" before the Recorded target gave it a home. The ids are
1146
+ stored tokens inside eval group bodies and a pipeline's private group, so
1147
+ they are mapped wherever a body is read rather than rewritten in place. */
1148
+ const RECORDED_IDS = {
1149
+ "equals-production": "same-as-recorded",
1150
+ "fields-equal-production": "fields-equal-recorded",
1151
+ "same-parse-outcome": "same-parse-as-recorded",
1152
+ };
1153
+ const renameMetric = (m ) =>
1154
+ isObj(m) && isStr(m.type) && RECORDED_IDS[m.type] ? { ...m, type: RECORDED_IDS[m.type] } : m;
1155
+ const renameMetrics = (list ) =>
1156
+ Array.isArray(list) ? list.map(renameMetric) : list;
1157
+ /** A version-7 body (or a freshly made version-8 one) with its recorded-reply
1158
+ metric ids read under their version-8 names, at version 8. */
1159
+ function recordedIdsV7(b ) {
1160
+ return {
1161
+ ...b,
1162
+ version: DATASET_BODY_VERSION,
1163
+ ...(Array.isArray(b.every) ? { every: renameMetrics(b.every) } : {}),
1164
+ ...(Array.isArray(b.run) ? { run: renameMetrics(b.run) } : {}),
1165
+ ...(Array.isArray(b.cases) ? { cases: b.cases.map((c ) =>
1166
+ isObj(c) && Array.isArray(c.metrics) ? { ...c, metrics: renameMetrics(c.metrics) } : c) } : {}),
1167
+ } ;
1168
+ }
1065
1169
 
1066
1170
  /** The metrics whose Ignore case version 6 made mean what it says for a
1067
1171
  reply read as a list, as well as one read as text. */
1068
1172
  const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1069
1173
 
1070
1174
  /**
1071
- * [body] as this version of a dataset (7), from any earlier one. Version 1
1175
+ * [body] as this version of a dataset (8), from any earlier one. Version 1
1072
1176
  * held `imageCases`, each with `minTags`/`maxTags`, and the parser's
1073
1177
  * `replays` and `conformance` (now fixtures/replays.json beside the checks).
1074
1178
  * Version 2 held `rules`, which clean a job's answer and so belong to the
@@ -1081,8 +1185,12 @@ const CASE_FOLDING = ["contains", "contains-all", "contains-any"];
1081
1185
  * a list's items ignoring case whatever a Contains metric's Ignore case
1082
1186
  * said, so each of its Contains metrics says Ignore case (`caseOfV5`), and
1083
1187
  * the body says its version. Version 6 is a group's cases alone: version 7
1084
- * adds its scoring, grader, Every item and Whole run (`groupOfV6`), and a
1085
- * stored row is read that way rather than rewritten. Every reader of a
1188
+ * adds its scoring, grader, Every item and Whole run (`groupOfV6`). Version 8
1189
+ * reads the recorded-reply metric ids (`equals-production` and its two
1190
+ * siblings) under their new names (`RECORDED_IDS`, #299), in a group's own
1191
+ * metrics and its cases' -- a pipeline's private group is a group body, so
1192
+ * this covers it too. A stored row is read that way rather than rewritten.
1193
+ * Every reader of a
1086
1194
  * body calls this: the runner, the page, a run's kept copy. A reader that
1087
1195
  * needs a version-2 body's rules -- to upgrade a pipeline graded against
1088
1196
  * it -- takes them first (`datasetRules`). Pure: the same body gives the
@@ -1093,19 +1201,25 @@ function upgradeDatasetBody (body ) {
1093
1201
  if (!isObj(body)) return body;
1094
1202
  if (body.version === DATASET_BODY_VERSION) return body;
1095
1203
  const b = body ;
1096
- // A body saying any other version is one this lab does not read.
1097
- if ("version" in b) return b.version === 6 ? groupOfV6(b) : body;
1204
+ if ("version" in b) {
1205
+ // Version 7 is an eval group whose recorded-reply metrics are read under
1206
+ // their version-8 ids; version 6 is its cases alone. Any other version is
1207
+ // one this lab does not read.
1208
+ if (b.version === 7) return recordedIdsV7(b);
1209
+ if (b.version === 6) return recordedIdsV7(groupOfV6(b));
1210
+ return body;
1211
+ }
1098
1212
  // A body that names its Source, even as null, is version 5.
1099
1213
  if ("source" in b) {
1100
- return groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) });
1214
+ return recordedIdsV7(groupOfV6({ ...b, ...(Array.isArray(b.cases) ? { cases: b.cases.map(caseOfV5) } : {}) }));
1101
1215
  }
1102
1216
  const raw = Array.isArray(b.cases) ? b.cases : Array.isArray(b.imageCases) ? b.imageCases : null;
1103
1217
  // Not a body of any version -- a copy kept while a dataset was an overlay.
1104
1218
  if (!raw) return body;
1105
- return groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) });
1219
+ return recordedIdsV7(groupOfV6({ source: null, cases: raw.map(caseOfV4).map(caseOfV5) }));
1106
1220
  }
1107
1221
 
1108
- /** A version-6 body as a version-7 eval group: scored All, the lab's
1222
+ /** A version-6 body as a version-8 eval group: scored All, the lab's
1109
1223
  grader, and no metrics of its own for every item or the whole run --
1110
1224
  what a Metrics eval naming the dataset with none of its own graded, which
1111
1225
  is what every Graded set converted to. */
@@ -1249,7 +1363,12 @@ function datasetRules(body ) {
1249
1363
 
1250
1364
 
1251
1365
 
1252
-
1366
+
1367
+
1368
+
1369
+
1370
+
1371
+
1253
1372
 
1254
1373
 
1255
1374
 
@@ -1620,12 +1739,33 @@ CONNECTION_TYPES.echo = {
1620
1739
  request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1621
1740
  };
1622
1741
 
1742
+ /** A call's recorded result as transport-level text -- what a model asked the
1743
+ same way would have come back with, before the flow reads it. The runner's
1744
+ replay branch reads it as the flow reads a reply (`httpReplyOf`); this is
1745
+ the fallback where there is no Read as to read it through. */
1746
+ function recordedRaw(record ) {
1747
+ const result = isObj(record) && isObj(record.result) ? record.result : null;
1748
+ if (!result) return "";
1749
+ return isStr(result.body) ? result.body : JSON.stringify(result.body ?? "");
1750
+ }
1751
+
1752
+ // Recorded (#299): a target that replays each call's recorded reply, sending
1753
+ // nothing and holding no key or model. The runner reads the recorded result as
1754
+ // the flow reads a reply; `local` is the no-Read-as fallback.
1755
+ CONNECTION_TYPES.recorded = {
1756
+ id: "recorded", label: "Recorded", settings: [], keyless: true, picker: false, replays: true,
1757
+ description: "Replays the reply production recorded for each call. Sends nothing.",
1758
+ local: (_item, _sent, record) => recordedRaw(record),
1759
+ request: sendsNothing, parseReply: sendsNothing, listModels: sendsNothing, parseModels: () => [],
1760
+ };
1761
+
1623
1762
  /** A local type's answer to one stage, as a model's would come back. An item
1624
1763
  with no text of its own (Prompt only) is answered from the prompt, and
1625
1764
  that reply is read against nothing -- the prompt is the reply, not an
1626
- instruction it could repeat. */
1627
- function localAnswer(conn , text , sent ) {
1628
- const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent);
1765
+ instruction it could repeat. A replaying type (Recorded) is handed the
1766
+ item's record to answer from. */
1767
+ function localAnswer(conn , text , sent , record ) {
1768
+ const raw = CONNECTION_TYPES[typeOf(conn)] .local ({ text }, sent, record);
1629
1769
  return { raw, ms: 0, conn: (conn.name ) ?? null, ...(text == null ? { readAgainst: "" } : {}) };
1630
1770
  }
1631
1771
 
@@ -2997,24 +3137,54 @@ const FILE_LIBRARY_TEXT = [".txt", ".md", ".csv"] ;
2997
3137
  SOURCE_TYPES.files = {
2998
3138
  label: "File Library",
2999
3139
  description: "A folder of images and text files you upload.",
3140
+ group: "Files",
3000
3141
  noun: "file",
3001
3142
  uploads: [".jpg", ".jpeg", ".png", ".heic", ".heif", ".dng", ".tif", ".tiff", ...FILE_LIBRARY_TEXT],
3002
3143
  sendsImage: true,
3144
+ hasActions: false,
3003
3145
  itemText: name => FILE_LIBRARY_TEXT.some(ext => name.toLowerCase().endsWith(ext)),
3004
3146
  };
3005
3147
 
3006
- // A Power Automate cloud flow (docs/power-automate.md): its items are
3007
- // records -- one call of a Power Automate step each: what its template read,
3008
- // the request production sent and what came back (flows/record.ts) -- and it
3009
- // keeps the flow's definition beside them. Records are imported from run
3010
- // history, written by hand or uploaded; nothing it holds is an image.
3011
- SOURCE_TYPES["power-automate"] = {
3012
- label: "Power Automate workflow",
3013
- description: "A cloud flow read from Microsoft 365, with records of its steps' calls from its run history.",
3148
+ // A workflow Source: a cloud flow, read through a platform (WORKFLOW_PLATFORMS,
3149
+ // below) its row names in `config.platform`. Its items are records -- one call
3150
+ // of one of the flow's steps each: what the step's template read, the request
3151
+ // production sent and what came back (flows/record.ts) -- and it keeps the
3152
+ // flow's definition beside them. The kind is generic; the platform owns what
3153
+ // is engine-specific, and the type's shown label is the platform's.
3154
+ SOURCE_TYPES.workflow = {
3155
+ label: "Workflow",
3156
+ description: "A cloud flow, with records of its steps' calls from its run history.",
3157
+ group: "Workflows",
3014
3158
  noun: "record",
3015
3159
  uploads: [".json"],
3016
3160
  sendsImage: false,
3161
+ hasActions: true,
3017
3162
  itemText: () => false,
3163
+ platforms: ["power-automate"], // platform: the WORKFLOW_PLATFORMS a workflow offers
3164
+ };
3165
+
3166
+ // Power Automate (docs/power-automate.md): the one workflow platform built.
3167
+ // Its pieces were the lab's Power-Automate-specific code -- flows/record.ts,
3168
+ // flows/wdl.ts and flowApi.ts -- gathered here so generic code calls the
3169
+ // entry, never a platform id.
3170
+ WORKFLOW_PLATFORMS["power-automate"] = { // platform: the Power Automate adapter's own id
3171
+ id: "power-automate", // platform: its id, the row's config.platform
3172
+ label: "Power Automate flow",
3173
+ nouns: { workflow: "flow", action: "action" },
3174
+ signIn: "microsoft",
3175
+ uploads: { ".json": "application/json; charset=utf-8" },
3176
+ keepsDefinition: true,
3177
+ api: token => flowApi(token),
3178
+ stepsOf: flowStepsOf,
3179
+ actionsOf: actionsByName,
3180
+ readerFor,
3181
+ evaluate,
3182
+ asText,
3183
+ readsOf,
3184
+ // Copy request… writes the body back as a flow's HTTP action holds it: JSON
3185
+ // whose strings keep Power Automate's own `@{…}` expressions untouched, so a
3186
+ // rebuilt body round-trips into the flow (docs/power-automate.md).
3187
+ render: body => JSON.stringify(body, null, 2),
3018
3188
  };
3019
3189
 
3020
3190
  CONTENT_TYPES.source = {
@@ -3194,7 +3364,11 @@ function metricInput(res , kase , more )
3194
3364
  const last = res.transcript?.at(-1);
3195
3365
  const text = last?.got ?? res.raw ?? "";
3196
3366
  return { text, said: last?.said ?? null, terms: res.terms ?? [], replies: [text], plain: !!more.plain,
3197
- error: res.error ?? null, ms: res.ms ?? 0, production: more.production ?? null, kase };
3367
+ // A case that has been given its own expected reply ("Take B as
3368
+ // expected") is compared with that; otherwise the Source's recorded
3369
+ // reply for the item.
3370
+ error: res.error ?? null, ms: res.ms ?? 0,
3371
+ recorded: typeof kase?.recorded === "string" ? kase.recorded : (more.production ?? null), kase };
3198
3372
  }
3199
3373
 
3200
3374
  /** A whole run's replies as one input: every reply's text together, and
@@ -3208,7 +3382,7 @@ function runInput(ress , plain )
3208
3382
  terms.push(...(r.terms || []));
3209
3383
  ms += r.ms ?? 0;
3210
3384
  }
3211
- return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, production: null, kase: null };
3385
+ return { text: replies.join("\n"), said: null, terms, replies, plain, error: null, ms, recorded: null, kase: null };
3212
3386
  }
3213
3387
 
3214
3388
  /** One metric's reading of [input]: `of`, `not` and a failure all applied,
@@ -3629,10 +3803,38 @@ function readGroupRun(group , ress
3629
3803
  return group.run.length ? wholeRunVerdict(group.run, ress, kind, group.scoring.mode, group.scoring.threshold) : null;
3630
3804
  }
3631
3805
 
3806
+ /**
3807
+ * [body] with the cases named in [replies] expecting the reply given for
3808
+ * their item -- Results' "Take B as expected" (docs/workflow-sources.md §
3809
+ * Calls are graded cases). Each such case keeps everything else and gains
3810
+ * `recorded` (the reply the recorded-reply metrics compare against, read in
3811
+ * place of the Source's) and `todo: false`, so a later run grades the item
3812
+ * against that reply. A selected item the group has no case for, or one whose
3813
+ * reply is missing (Target B errored, or is the Recorded target), is left
3814
+ * untouched and named in `missing`; `taken` names the items whose expected was
3815
+ * written. Pure: the same body and replies give the same body, so Undo is the
3816
+ * body as it was.
3817
+ */
3818
+ function takeAsExpected(body , replies )
3819
+ {
3820
+ const byItem = new Map ();
3821
+ body.cases.forEach((c, i) => { const item = caseItem(c); if (item) byItem.set(item, i); });
3822
+ const cases = body.cases.slice();
3823
+ const taken = [], missing = [];
3824
+ for (const item of Object.keys(replies)) {
3825
+ const reply = replies[item];
3826
+ const at = byItem.get(item);
3827
+ if (at == null || typeof reply !== "string") { missing.push(item); continue; }
3828
+ cases[at] = { ...cases[at], recorded: reply, todo: false };
3829
+ taken.push(item);
3830
+ }
3831
+ return { body: { ...body, cases }, taken, missing };
3832
+ }
3833
+
3632
3834
  /** The grader a Metrics eval names, reached through what the runner hands it. */
3633
3835
  function graderCtx(t , more ) {
3634
3836
  const ask = isRef(t.grader) ? more.grader?.(t.grader) : undefined;
3635
- return ask ? { ask } : {};
3837
+ return { ...(ask ? { ask } : {}), ...(more.recordedTarget ? { recordedTarget: true } : {}) };
3636
3838
  }
3637
3839
 
3638
3840
  /** A metric's options in a few words -- "watering", "^\\{" -- or nothing
@@ -3808,9 +4010,30 @@ STEP_TYPES.echo = {
3808
4010
  },
3809
4011
  };
3810
4012
 
3811
- /** What a local target step is answered as: Echo's own connection, with no
4013
+ // Recorded (#299): a target that replays each call's recorded reply, read as
4014
+ // the flow reads one. No profile, no prompt -- the record is the reply.
4015
+ STEP_TYPES.recorded = {
4016
+ label: "Recorded", slot: "target", asks: "nothing", local: true, replaces: "recorded",
4017
+ description: "Replays the reply production recorded for each call. Sends nothing.",
4018
+ firstJobOnly: "a call's recorded reply is job 1's",
4019
+ in: "item", out: "text",
4020
+ apply: "runPipeline",
4021
+ // `prompt` only so the step satisfies the target-step shape; it is never
4022
+ // sent (the record is the reply), and the prompt rule exempts it below.
4023
+ fields: ["type", "prompt"],
4024
+ validate(){},
4025
+ };
4026
+
4027
+ /** What a local target step is answered as, by the connection type it stands
4028
+ in for (`replaces`): Echo's own connection, or Recorded's -- each with no
3812
4029
  address, key or model. */
3813
4030
  const ECHO_CONNECTION = { name: "Echo", url: "", model: "", type: "echo" };
4031
+ const RECORDED_CONNECTION = { name: "Recorded", url: "", model: "", type: "recorded" };
4032
+ const LOCAL_CONNECTIONS =
4033
+ { echo: ECHO_CONNECTION, recorded: RECORDED_CONNECTION };
4034
+ /** The connection a local target step stands in for, by its entry. */
4035
+ const localConnOf = (entry ) =>
4036
+ LOCAL_CONNECTIONS[entry?.replaces ?? ""] ?? ECHO_CONNECTION;
3814
4037
 
3815
4038
  // Responses: how a job's reply is read, in order -- possibly not at all.
3816
4039
 
@@ -3950,6 +4173,16 @@ function targetProfileOf(doc
3950
4173
  return targetStepOf(doc, i, k)?.profile ?? targetsOf(doc)[i]?.profile ?? undefined;
3951
4174
  }
3952
4175
 
4176
+ /** Whether target [i] is a Recorded target: it replays each call's recorded
4177
+ reply rather than sending anything (its job-1 step answers locally with a
4178
+ connection that `replays`). A recorded-reply metric compares the recorded
4179
+ reply with itself there, so it says nothing (n/a). Asked through the
4180
+ registry, never a step id. */
4181
+ function recordedTargetOf(doc , i ) {
4182
+ const entry = targetEntryOf(doc, i, 0);
4183
+ return !!(entry?.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays);
4184
+ }
4185
+
3953
4186
  /** [doc] with target [i]'s step in job [k] changed: [change] merged into
3954
4187
  it, or the step [change] makes of it. */
3955
4188
  function withTarget (doc , i , k ,
@@ -4641,11 +4874,16 @@ function validatePipeline(input , ctx = {}) {
4641
4874
  // what it is read against. Over Prompt only the prompt is the reply.
4642
4875
  // What target [i] asks in job [k]: Echo's own connection for a step
4643
4876
  // answered here, else its profile as the lookups hold it.
4644
- const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION
4877
+ const askedOf = (i , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k))
4645
4878
  : c.profiles ? c.profiles(targetProfileOf(doc, i, k) .id) : null);
4646
4879
  const itemsHaveText = !!content && !CONTENT_TYPES[content.type]?.bare;
4647
4880
  for (let k = 0; k < n; k++) {
4648
4881
  const i = targets.findIndex((_, i) => {
4882
+ const entry = targetEntryOf(doc, i, k);
4883
+ // A step with no prompt of its own, or a replaying target (Recorded,
4884
+ // which answers from the record), is never asked for one.
4885
+ if (!entry?.fields?.includes("prompt")) return false;
4886
+ if (entry.local && CONNECTION_TYPES[localConnOf(entry).type]?.replays) return false;
4649
4887
  const words = targetStepOf(doc, i, k)?.prompt;
4650
4888
  return (!isStr(words) || !words.trim())
4651
4889
  && !(k === 0 && itemsHaveText && local(askedOf(i, 0)));
@@ -4658,12 +4896,14 @@ function validatePipeline(input , ctx = {}) {
4658
4896
  const nonText = (contentFiles || []).map(f => String((f )?.name ?? f))
4659
4897
  .find(n => !/\.(?:txt|md|csv)$/i.test(n));
4660
4898
  targets.forEach((_, i) => {
4661
- const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? ECHO_CONNECTION : lookup(targetProfileOf(doc, i, k) )));
4899
+ const conns = doc.jobs.map((_ , k ) => (targetEntryOf(doc, i, k)?.local ? localConnOf(targetEntryOf(doc, i, k)) : lookup(targetProfileOf(doc, i, k) )));
4662
4900
  conns.forEach((p , k ) => {
4663
4901
  if (!local(p)) return;
4664
- const label = CONNECTION_TYPES[typeOf(p )] .label;
4665
- if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${label} answers only job 1, with the item's own text`);
4666
- else if (nonText) bad.push(`${targetLabel(doc, i)}: ${label} answers each text item with its own text, and ${nonText} is not text`);
4902
+ const entry = CONNECTION_TYPES[typeOf(p )] ;
4903
+ // A replaying target (Recorded) answers from the record, not the item
4904
+ // text, so the text rules do not apply -- only the job-1 one does.
4905
+ if (k > 0) bad.push(`${jobLabel(doc, k)}, ${targetLabel(doc, i)}: ${entry.label} answers only job 1${entry.replays ? "" : ", with the item's own text"}`);
4906
+ else if (nonText && !entry.replays) bad.push(`${targetLabel(doc, i)}: ${entry.label} answers each text item with its own text, and ${nonText} is not text`);
4667
4907
  });
4668
4908
  // A connection answers what its target's step asks: words, or a whole request.
4669
4909
  conns.forEach((p , k ) => {
@@ -5261,7 +5501,7 @@ function stagesFor(run , i )
5261
5501
  if (targetEntryOf(run, i, k)?.local) {
5262
5502
  const id = targetProfileOf(run, i, k)?.id;
5263
5503
  const held = id != null ? run.profiles?.[id] : undefined;
5264
- return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...ECHO_CONNECTION };
5504
+ return held && CONNECTION_TYPES[held.type]?.local ? { id: id , ...held } : { id: "", ...localConnOf(targetEntryOf(run, i, k)) };
5265
5505
  }
5266
5506
  const id = targetProfileOf(run, i, k) .id;
5267
5507
  return { id, ...(run.profiles?.[id] || {}) };
@@ -5276,7 +5516,7 @@ function stagesFor(run , i )
5276
5516
  function scenarioProfile(run , i ) {
5277
5517
  const id = targetsOf(run)[i]?.profile?.id;
5278
5518
  // A target that asks nothing ran as Echo.
5279
- if (id == null && targetEntryOf(run, i, 0)?.local) return { id: "", name: ECHO_CONNECTION.name, settings: { ...ECHO_CONNECTION } };
5519
+ if (id == null && targetEntryOf(run, i, 0)?.local) { const lc = localConnOf(targetEntryOf(run, i, 0)); return { id: "", name: lc.name, settings: { ...lc } }; }
5280
5520
  const conn = id != null ? run.profiles?.[id] : null;
5281
5521
  return conn ? { id: id , name: conn.name, settings: connectionSettings(conn ) } : null;
5282
5522
  }
@@ -5337,9 +5577,9 @@ function itemScores(run , kase , res
5337
5577
  return Object.keys(out).length ? out : null;
5338
5578
  }
5339
5579
 
5340
- /** What production replied to an item, for a run's metrics: its last job's
5341
- call says, from the item's record, or nobody does. */
5342
- function productionOf(run , record ) {
5580
+ /** What production recorded as the reply to an item, for a run's metrics: its
5581
+ last job's call says, from the item's record, or nobody does. */
5582
+ function recordedReplyOf(run , record ) {
5343
5583
  const last = run.jobs.at(-1), flow = flowStepOf(last);
5344
5584
  return record && flow ? STEP_TYPES.flowStep .production ({ ...flow, readAs: readAsOf(last) }, record) : null;
5345
5585
  }
@@ -5635,20 +5875,20 @@ registerMetrics({ registerKinds });
5635
5875
  export {
5636
5876
  TOKEN_DEFAULTS, TOKEN_TYPES, tokenMapping, tokenNames, resolvePrompt, tokenSet, textPrompt, words,
5637
5877
  loopReplyError, preparedSize,
5638
- termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
5878
+ termIn, forbiddenIn, readCase, gradedSetFrom, caseItem, caseMetrics, metricLines, upgradeDatasetBody, takeAsExpected, readGroup, readGroupRun, datasetRules, upgradeResults, emptyTally, addToTally, isHostedUrl,
5639
5879
  SEED, REPLY_TOKENS_BEFORE, ANTHROPIC_MAX_TOKENS, asNumber, pinReplyTokens, mappingsFor,
5640
5880
  CONNECTION_TYPES, HTTP_APIS, httpApiFor, httpRequestOf, httpReplyOf, readFlowReply, WdlError, localAnswer, typeOf, profileType, convertProfile, splitOllama, DECODING_KEYS, apiBase,
5641
5881
  EDGE_448, budgetLabel, IMAGE_FORMATS, encoderQuality,
5642
5882
  tallyPercent, runPipeline, validateEvals, evalsWarnings, evalsJson, readList,
5643
5883
  scoreSingle, singleRules, parseCount, lengthHolds, countHolds, COUNT_OPS, LENGTH_OPS,
5644
5884
  COUNT_OP_LABEL, LENGTH_OP_LABEL,
5645
- PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, productionOf,
5646
- SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf,
5885
+ PIPELINE_VERSION, DATASET_BODY_VERSION, STEP_TYPES, CONTENT_TYPES, OUTPUT_KINDS, MODIFIERS, EVAL_TYPES, LEGACY_TESTS, METRICS, readMetrics, metricSummary, itemScoresAsync, recordedReplyOf,
5886
+ SOURCE_TYPES, DEFAULT_SOURCE_TYPE, sourceTypeOf, WORKFLOW_PLATFORMS, platformOf,
5647
5887
  registerKinds, defaultKind, jobDefaults, applyModifiers, keyVar, slugOf, uniqueSlug, isSlug, withSlugs, SLUG_MAX, connectionOf, connectionSettings,
5648
5888
  connectionRequest, connectionBase, connectionSend, connectionReply, connectionReplyProblem, connectionImage, connectionProblems,
5649
5889
  CONNECTION_FIELDS, SETTING_KEYS, OPTION_FIELDS, jobLabel, scenarioLabel, jobWords, versionProblem,
5650
5890
  contentOf, withContent, replyOf,
5651
- targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, withTarget,
5891
+ targetsOf, targetLetter, targetLabel, targetStepOf, targetEntryOf, jobTargetType, targetProfileOf, recordedTargetOf, withTarget,
5652
5892
  tokensOf, withTokens, sendsImage, withImage, flowStepOf, readAsOf, outOf, withOut, slotEntry, SLOTS,
5653
5893
  blankPipeline, upgradePipeline, fatProfileRef, newId, newEval, newLink, validatePipeline, resolvePipeline, pipelineOfRun, stagesFor, profileIds,
5654
5894
  scenarioProfile, lastKind, lastModifier, modifierSummary, itemScores, scenarioEvals, scenarioPasses, groupPass, overallVerdict, passVerdict, isSkipped, failedScore,