patchwork-os 1.2.0-beta.2.canary.819 → 1.2.0-beta.2.canary.821
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/recipeOrchestration.js +17 -14
- package/dist/recipeOrchestration.js.map +1 -1
- package/dist/recipeRoutes.js +11 -3
- package/dist/recipeRoutes.js.map +1 -1
- package/dist/recipes/chainedRunner.d.ts +12 -3
- package/dist/recipes/chainedRunner.js +17 -0
- package/dist/recipes/chainedRunner.js.map +1 -1
- package/dist/recipes/replayBoundary.d.ts +41 -0
- package/dist/recipes/replayBoundary.js +59 -0
- package/dist/recipes/replayBoundary.js.map +1 -0
- package/dist/recipes/replayRun.d.ts +53 -12
- package/dist/recipes/replayRun.js +113 -25
- package/dist/recipes/replayRun.js.map +1 -1
- package/dist/recipes/yamlRunner.d.ts +14 -1
- package/dist/recipes/yamlRunner.js +39 -0
- package/dist/recipes/yamlRunner.js.map +1 -1
- package/package.json +1 -1
|
@@ -3,24 +3,53 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Given an original `RecipeRun` (looked up from `RecipeRunLog`), build a
|
|
5
5
|
* `mockedOutputs` map from each step's captured `output` (VD-2) and
|
|
6
|
-
* re-run the recipe through
|
|
6
|
+
* re-run the recipe through the matching runner with every tool/agent
|
|
7
7
|
* execution short-circuited to those captured values.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* Evidence-only, by construction rather than by documentation. Until
|
|
10
|
+
* 2026-09-24 the header here promised "no external network calls, no
|
|
11
|
+
* write side effects" while any step WITHOUT a usable capture (missing
|
|
12
|
+
* `output`, a >8 KB truncation envelope, or a flat step with no `into:`)
|
|
13
|
+
* silently fell through to REAL execution — a real `http.post` went out
|
|
14
|
+
* and a deleted `file.write` target was recreated, with the list of
|
|
15
|
+
* unmocked steps reported only after the run. The invariant now enforced
|
|
16
|
+
* (see `replayBoundary.ts`):
|
|
12
17
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
18
|
+
* Missing replay evidence REDUCES what can be replayed; it never
|
|
19
|
+
* increases permission to execute.
|
|
20
|
+
*
|
|
21
|
+
* 1. Preflight: if any step lacks a usable capture the replay is REFUSED
|
|
22
|
+
* before a run starts, with `replay_refused_unmocked_step` naming
|
|
23
|
+
* every such step.
|
|
24
|
+
* 2. The run itself is started with `replayOnly: true`, so a step the
|
|
25
|
+
* preflight somehow missed is refused inside the runner instead of
|
|
26
|
+
* dispatched, and the dispatch seam (`executeStep` /
|
|
27
|
+
* `buildAgentExecutorDeps`) throws before any tool or agent is
|
|
28
|
+
* invoked. A replay's deps are structurally incapable of live
|
|
29
|
+
* dispatch.
|
|
30
|
+
* 3. After the run, any step carrying a boundary refusal forces
|
|
31
|
+
* `ok: false` — earlier mocked progress never reads as an unqualified
|
|
32
|
+
* success.
|
|
33
|
+
*
|
|
34
|
+
* The new run is logged with `triggerSource: "replay:<originalSeq>"` /
|
|
35
|
+
* `manualRunId: "replay-<originalSeq>"` so the audit trail is clear.
|
|
36
|
+
*
|
|
37
|
+
* Real-mode replay (write tools really fire) is deliberately NOT in this
|
|
38
|
+
* module, and there is no "continue live" fallback here either. It needs
|
|
39
|
+
* a confirmation UX, a kill-switch interaction, and possibly a
|
|
40
|
+
* connector-level read/write split. Ship separately after explicit user
|
|
41
|
+
* approval.
|
|
17
42
|
*/
|
|
18
43
|
import { runChainedRecipe } from "./chainedRunner.js";
|
|
44
|
+
import { isReplayRefusal, replayRefusalMessage } from "./replayBoundary.js";
|
|
19
45
|
import { buildChainedDeps, declaredRecipeEnv, runYamlRecipe, } from "./yamlRunner.js";
|
|
20
46
|
/**
|
|
21
47
|
* Build the `mockedOutputs` map. Truncated captures (>8 KB envelope from
|
|
22
48
|
* VD-2's `captureForRunlog`) are excluded — replaying with a `[truncated]`
|
|
23
49
|
* preview would be misleading. Steps without captures are excluded too.
|
|
50
|
+
*
|
|
51
|
+
* `null`, `""`, `0` and `false` are VALID captures (the tool really
|
|
52
|
+
* returned that) and are kept; only `undefined` means "nothing captured".
|
|
24
53
|
*/
|
|
25
54
|
export function buildMockedOutputs(originalRun) {
|
|
26
55
|
const outputs = new Map();
|
|
@@ -35,9 +64,7 @@ export function buildMockedOutputs(originalRun) {
|
|
|
35
64
|
}
|
|
36
65
|
// Skip the truncation envelope — replaying with a preview slice
|
|
37
66
|
// would be misleading.
|
|
38
|
-
if (out
|
|
39
|
-
typeof out === "object" &&
|
|
40
|
-
out["[truncated]"] === true) {
|
|
67
|
+
if (isTruncatedCapture(out)) {
|
|
41
68
|
unmocked.push(step.id);
|
|
42
69
|
continue;
|
|
43
70
|
}
|
|
@@ -45,16 +72,50 @@ export function buildMockedOutputs(originalRun) {
|
|
|
45
72
|
}
|
|
46
73
|
return { outputs, unmocked };
|
|
47
74
|
}
|
|
75
|
+
function isTruncatedCapture(out) {
|
|
76
|
+
return (out !== null &&
|
|
77
|
+
typeof out === "object" &&
|
|
78
|
+
out["[truncated]"] === true);
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Step ids in a finished run's `stepResults` whose error came from the
|
|
82
|
+
* replay boundary — a refusal the preflight did not pre-empt. Any such
|
|
83
|
+
* step disqualifies the run from reporting as a successful replay, even
|
|
84
|
+
* when the recipe's fail-open settings let the run complete.
|
|
85
|
+
*/
|
|
86
|
+
function refusedStepIds(stepResults) {
|
|
87
|
+
const ids = [];
|
|
88
|
+
for (const s of stepResults ?? []) {
|
|
89
|
+
if (s.error !== undefined && isReplayRefusal(s.error)) {
|
|
90
|
+
ids.push(s.id ?? "?");
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
return ids;
|
|
94
|
+
}
|
|
48
95
|
/**
|
|
49
96
|
* Fire a mocked replay of `originalRun` against `recipe`. The recipe
|
|
50
97
|
* argument is supplied by the caller (typically loaded fresh from disk
|
|
51
98
|
* by name) so an EDITED recipe can be replayed against captured
|
|
52
99
|
* outputs — that's the debugging value of replay.
|
|
100
|
+
*
|
|
101
|
+
* Refuses (no run started) when any step lacks a usable capture.
|
|
53
102
|
*/
|
|
54
103
|
export async function replayMockedRun(opts) {
|
|
55
104
|
const { originalRun, recipe, sourcePath, deps } = opts;
|
|
56
105
|
const { outputs, unmocked } = buildMockedOutputs(originalRun);
|
|
57
|
-
|
|
106
|
+
// Layer 1 — preflight. Explanatory: names every unmockable step at once.
|
|
107
|
+
if (unmocked.length > 0) {
|
|
108
|
+
return {
|
|
109
|
+
ok: false,
|
|
110
|
+
error: replayRefusalMessage(unmocked),
|
|
111
|
+
unmockedSteps: unmocked,
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
const chainedDeps = buildChainedDeps(
|
|
115
|
+
// Layer 3 — the deps a replay hands the runner cannot reach a live tool
|
|
116
|
+
// or agent: `replayOnly` rides on StepDeps into `executeStep` and
|
|
117
|
+
// `buildAgentExecutorDeps`, which throw `ReplayIncompleteError`.
|
|
118
|
+
{ ...deps.runnerDeps, replayOnly: true }, deps.runnerDeps.claudeCodeFn ??
|
|
58
119
|
(async () => {
|
|
59
120
|
return "";
|
|
60
121
|
}), recipe.name);
|
|
@@ -70,6 +131,9 @@ export async function replayMockedRun(opts) {
|
|
|
70
131
|
runLog: deps.runLog,
|
|
71
132
|
...(deps.activityLog !== undefined && { activityLog: deps.activityLog }),
|
|
72
133
|
mockedOutputs: outputs,
|
|
134
|
+
// Layer 2 — a step the preflight missed (e.g. one the edited recipe
|
|
135
|
+
// added) is refused in the runner's step loop, never dispatched.
|
|
136
|
+
replayOnly: true,
|
|
73
137
|
// BUG-4 fix: tag the new run's taskId so it's distinguishable from a
|
|
74
138
|
// fresh run. Searchable as `taskId LIKE 'replay:<seq>:%'`.
|
|
75
139
|
taskIdPrefix: `replay:${originalRun.seq}`,
|
|
@@ -81,19 +145,21 @@ export async function replayMockedRun(opts) {
|
|
|
81
145
|
// started after originalRun.doneAt.
|
|
82
146
|
const recent = deps.runLog.query({ recipe: recipe.name, limit: 5 });
|
|
83
147
|
const newRun = recent.find((r) => r.createdAt > originalRun.doneAt);
|
|
148
|
+
const refused = refusedStepIds([...result.stepResults].map(([id, r]) => ({ id, error: r.error })));
|
|
149
|
+
const ok = result.success && refused.length === 0;
|
|
150
|
+
const error = refused.length > 0 ? replayRefusalMessage(refused) : result.errorMessage;
|
|
84
151
|
return {
|
|
85
|
-
ok
|
|
152
|
+
ok,
|
|
86
153
|
...(newRun?.seq !== undefined && { newSeq: newRun.seq }),
|
|
87
154
|
result,
|
|
88
|
-
...(
|
|
89
|
-
...(
|
|
155
|
+
...(error !== undefined && { error }),
|
|
156
|
+
...(refused.length > 0 && { unmockedSteps: refused }),
|
|
90
157
|
};
|
|
91
158
|
}
|
|
92
159
|
catch (err) {
|
|
93
160
|
return {
|
|
94
161
|
ok: false,
|
|
95
162
|
error: err instanceof Error ? err.message : String(err),
|
|
96
|
-
...(unmocked.length > 0 && { unmockedSteps: unmocked }),
|
|
97
163
|
};
|
|
98
164
|
}
|
|
99
165
|
}
|
|
@@ -106,7 +172,10 @@ export async function replayMockedRun(opts) {
|
|
|
106
172
|
* positional id risks silently matching it to a DIFFERENT step after an
|
|
107
173
|
* edit; an explicit `into:` name is a stable, user-chosen identifier
|
|
108
174
|
* (the flat-recipe analogue of a chained recipe's mandatory `id:`) and is
|
|
109
|
-
* safe to replay against.
|
|
175
|
+
* safe to replay against. A positional step is therefore NOT mockable —
|
|
176
|
+
* and, since 2026-09-24, not executable under replay either: the replay
|
|
177
|
+
* is refused, naming the step. (The dashboard's `previewMockedReplay`
|
|
178
|
+
* in `registryDiff.ts` applies the same rule, reason `positional-id`.)
|
|
110
179
|
*/
|
|
111
180
|
const POSITIONAL_STEP_ID = /^step_\d+$/;
|
|
112
181
|
/**
|
|
@@ -116,6 +185,10 @@ const POSITIONAL_STEP_ID = /^step_\d+$/;
|
|
|
116
185
|
* JSON-stringified — matching what the original tool call's raw string
|
|
117
186
|
* result would have looked like before any downstream `{{template}}`
|
|
118
187
|
* substitution treated it as text.
|
|
188
|
+
*
|
|
189
|
+
* A captured `""` is a valid (empty) result and is kept; a captured
|
|
190
|
+
* `null` becomes the runner's `null` result via `?? null` at the mocked
|
|
191
|
+
* short-circuit. Only `undefined` means "nothing captured".
|
|
119
192
|
*/
|
|
120
193
|
export function buildFlatMockedOutputs(originalRun) {
|
|
121
194
|
const outputs = new Map();
|
|
@@ -125,7 +198,7 @@ export function buildFlatMockedOutputs(originalRun) {
|
|
|
125
198
|
continue;
|
|
126
199
|
// See POSITIONAL_STEP_ID's doc comment — a step with no explicit
|
|
127
200
|
// `into:` isn't safely replayable against a (possibly edited) recipe
|
|
128
|
-
// file, so
|
|
201
|
+
// file, so the replay is refused rather than the step run live.
|
|
129
202
|
if (POSITIONAL_STEP_ID.test(step.id)) {
|
|
130
203
|
unmocked.push(step.id);
|
|
131
204
|
continue;
|
|
@@ -135,9 +208,7 @@ export function buildFlatMockedOutputs(originalRun) {
|
|
|
135
208
|
unmocked.push(step.id);
|
|
136
209
|
continue;
|
|
137
210
|
}
|
|
138
|
-
if (out
|
|
139
|
-
typeof out === "object" &&
|
|
140
|
-
out["[truncated]"] === true) {
|
|
211
|
+
if (isTruncatedCapture(out)) {
|
|
141
212
|
unmocked.push(step.id);
|
|
142
213
|
continue;
|
|
143
214
|
}
|
|
@@ -159,34 +230,51 @@ export function buildFlatMockedOutputs(originalRun) {
|
|
|
159
230
|
* (`yaml:<recipe>:<startedAt>`) has no override seam, and manualRunId is
|
|
160
231
|
* already surfaced in the dashboard / `runs.jsonl`, so this is enough to
|
|
161
232
|
* distinguish a replay run from a real one without new plumbing.
|
|
233
|
+
*
|
|
234
|
+
* Refuses (no run started) when any step lacks a usable capture. Note
|
|
235
|
+
* that flat AGENT steps capture no `output`, so a flat recipe with an
|
|
236
|
+
* agent step is not replayable today — refused, never run live.
|
|
162
237
|
*/
|
|
163
238
|
export async function replayFlatMockedRun(opts) {
|
|
164
239
|
const { originalRun, recipe, deps } = opts;
|
|
165
240
|
const { outputs, unmocked } = buildFlatMockedOutputs(originalRun);
|
|
241
|
+
// Layer 1 — preflight.
|
|
242
|
+
if (unmocked.length > 0) {
|
|
243
|
+
return {
|
|
244
|
+
ok: false,
|
|
245
|
+
error: replayRefusalMessage(unmocked),
|
|
246
|
+
unmockedSteps: unmocked,
|
|
247
|
+
};
|
|
248
|
+
}
|
|
166
249
|
try {
|
|
167
250
|
const result = await runYamlRecipe(recipe, {
|
|
168
251
|
...deps.runnerDeps,
|
|
169
252
|
runLog: deps.runLog,
|
|
170
253
|
...(deps.activityLog !== undefined && { activityLog: deps.activityLog }),
|
|
171
254
|
mockedOutputs: outputs,
|
|
255
|
+
// Layers 2 + 3 — refused in the step loop, and the StepDeps built from
|
|
256
|
+
// these RunnerDeps throw at the dispatch seam.
|
|
257
|
+
replayOnly: true,
|
|
172
258
|
manualRunId: `replay-${originalRun.seq}`,
|
|
173
259
|
testMode: false,
|
|
174
260
|
});
|
|
175
261
|
const recent = deps.runLog.query({ recipe: recipe.name, limit: 5 });
|
|
176
262
|
const newRun = recent.find((r) => r.createdAt >= originalRun.doneAt);
|
|
263
|
+
const refused = refusedStepIds(result.stepResults);
|
|
264
|
+
const ok = !result.errorMessage && refused.length === 0;
|
|
265
|
+
const error = refused.length > 0 ? replayRefusalMessage(refused) : result.errorMessage;
|
|
177
266
|
return {
|
|
178
|
-
ok
|
|
267
|
+
ok,
|
|
179
268
|
...(newRun?.seq !== undefined && { newSeq: newRun.seq }),
|
|
180
269
|
result,
|
|
181
|
-
...(
|
|
182
|
-
...(
|
|
270
|
+
...(error !== undefined && { error }),
|
|
271
|
+
...(refused.length > 0 && { unmockedSteps: refused }),
|
|
183
272
|
};
|
|
184
273
|
}
|
|
185
274
|
catch (err) {
|
|
186
275
|
return {
|
|
187
276
|
ok: false,
|
|
188
277
|
error: err instanceof Error ? err.message : String(err),
|
|
189
|
-
...(unmocked.length > 0 && { unmockedSteps: unmocked }),
|
|
190
278
|
};
|
|
191
279
|
}
|
|
192
280
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"replayRun.js","sourceRoot":"","sources":["../../src/recipes/replayRun.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"replayRun.js","sourceRoot":"","sources":["../../src/recipes/replayRun.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAyCG;AAUH,OAAO,EAAE,gBAAgB,EAAE,MAAM,oBAAoB,CAAC;AACtD,OAAO,EAAE,eAAe,EAAE,oBAAoB,EAAE,MAAM,qBAAqB,CAAC;AAE5E,OAAO,EACL,gBAAgB,EAChB,iBAAiB,EACjB,aAAa,GACd,MAAM,iBAAiB,CAAC;AA2BzB;;;;;;;GAOG;AACH,MAAM,UAAU,kBAAkB,CAAC,WAAsB;IAIvD,MAAM,OAAO,GAAG,IAAI,GAAG,EAAmB,CAAC;IAC3C,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,KAAK,MAAM,IAAI,IAAI,WAAW,CAAC,WAAW,IAAI,EAAE,EAAE,CAAC;QACjD,IAAI,IAAI,CAAC,MAAM,KAAK,SAAS;YAAE,SAAS;QACxC,MAAM,GAAG,GAAG,IAAI,CAAC,MAAM,CAAC;QACxB,IAAI,GAAG,KAAK,SAAS,EAAE,CAAC;YACtB,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACvB,SAAS;QACX,CAAC;QACD,gEAAgE;QAChE,uBAAuB;QACvB,IAAI,kBAAkB,CAAC,GAAG,CAAC,EAAE,CAAC;YAC5B,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACvB,SAAS;QACX,CAAC;QACD,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,EAAE,GAAG,CAAC,CAAC;IAC5B,CAAC;IACD,OAAO,EAAE,OAAO,EAAE,QAAQ,EAAE,CAAC;AAC/B,CAAC;AAED,SAAS,kBAAkB,CAAC,GAAY;IACtC,OAAO,CACL,GAAG,KAAK,IAAI;QACZ,OAAO,GAAG,KAAK,QAAQ;QACtB,GAA+B,CAAC,aAAa,CAAC,KAAK,IAAI,CACzD,CAAC;AACJ,CAAC;AAED;;;;;GAKG;AACH,SAAS,cAAc,CACrB,WAA0E;IAE1E,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,WAAW,IAAI,EAAE,EAAE,CAAC;QAClC,IAAI,CAAC,CAAC,KAAK,KAAK,SAAS,IAAI,eAAe,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC;YACtD,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,IAAI,GAAG,CAAC,CAAC;QACxB,CAAC;IACH,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,CAAC,KAAK,UAAU,eAAe,CAAC,IAKrC;IACC,MAAM,EAAE,WAAW,EAAE,MAAM,EAAE,UAAU,EAAE,IAAI,EAAE,GAAG,IAAI,CAAC;IACvD,MAAM,EAAE,OAAO,EAAE,QAAQ,EAAE,GAAG,kBAAkB,CAAC,WAAW,CAAC,CAAC;IAE9D,yEAAyE;IACzE,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,OAAO;YACL,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,oBAAoB,CAAC,QAAQ,CAAC;YACrC,aAAa,EAAE,QAAQ;SACxB,CAAC;IACJ,CAAC;IAED,MAAM,WAAW,GAAkB,gBAAgB;IACjD,wEAAwE;IACxE,kEAAkE;IAClE,iEAAiE;IACjE,EAAE,GAAG,IAAI,CAAC,UAAU,EAAE,UAAU,EAAE,IAAI,EAAE,EACxC,IAAI,CAAC,UAAU,CAAC,YAAY;QAC1B,CAAC,KAAK,IAAI,EAAE;YACV,OAAO,EAAE,CAAC;QACZ,CAAC,CAAC,EACJ,MAAM,CAAC,IAAI,CACZ,CAAC;IAEF,MAAM,UAAU,GAAe;QAC7B,oEAAoE;QACpE,4EAA4E;QAC5E,kDAAkD;QAClD,GAAG,EAAE,EAAE,GAAG,iBAAiB,CAAC,MAAM,CAAC,EAAwC;QAC3E,cAAc,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,MAAM,CAAC,cAAc,IAAI,CAAC,CAAC;QACvD,QAAQ,EAAE,MAAM,CAAC,QAAQ,IAAI,CAAC;QAC9B,MAAM,EAAE,KAAK;QACb,GAAG,CAAC,UAAU,KAAK,SAAS,IAAI,EAAE,UAAU,EAAE,CAAC;QAC/C,MAAM,EAAE,IAAI,CAAC,MAAM;QACnB,GAAG,CAAC,IAAI,CAAC,WAAW,KAAK,SAAS,IAAI,EAAE,WAAW,EAAE,IAAI,CAAC,WAAW,EAAE,CAAC;QACxE,aAAa,EAAE,OAAO;QACtB,oEAAoE;QACpE,iEAAiE;QACjE,UAAU,EAAE,IAAI;QAChB,qEAAqE;QACrE,2DAA2D;QAC3D,YAAY,EAAE,UAAU,WAAW,CAAC,GAAG,EAAE;KAC1C,CAAC;IAEF,IAAI,CAAC;QACH,MAAM,MAAM,GAAG,MAAM,gBAAgB,CAAC,MAAM,EAAE,UAAU,EAAE,WAAW,CAAC,CAAC;QACvE,kEAAkE;QAClE,kEAAkE;QAClE,oCAAoC;QACpC,MAAM,MAAM,GAAG,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,MAAM,EAAE,MAAM,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC,CAAC;QACpE,MAAM,MAAM,GAAG,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,SAAS,GAAG,WAAW,CAAC,MAAM,CAAC,CAAC;QACpE,MAAM,OAAO,GAAG,cAAc,CAC5B,CAAC,GAAG,MAAM,CAAC,WAAW,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,CAAC,CAAC,CACnE,CAAC;QACF,MAAM,EAAE,GAAG,MAAM,CAAC,OAAO,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,CAAC;QAClD,MAAM,KAAK,GACT,OAAO,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,oBAAoB,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,YAAY,CAAC;QAC3E,OAAO;YACL,EAAE;YACF,GAAG,CAAC,MAAM,EAAE,GAAG,KAAK,SAAS,IAAI,EAAE,MAAM,EAAE,MAAM,CAAC,GAAG,EAAE,CAAC;YACxD,MAAM;YACN,GAAG,CAAC,KAAK,KAAK,SAAS,IAAI,EAAE,KAAK,EAAE,CAAC;YACrC,GAAG,CAAC,OAAO,CAAC,MAAM,GAAG,CAAC,IAAI,EAAE,aAAa,EAAE,OAAO,EAAE,CAAC;SACtD,CAAC;IACJ,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO;YACL,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,GAAG,CAAC;SACxD,CAAC;IACJ,CAAC;AACH,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,kBAAkB,GAAG,YAAY,CAAC;AAExC;;;;;;;;;;;GAWG;AACH,MAAM,UAAU,sBAAsB,CAAC,WAAsB;IAI3D,MAAM,OAAO,GAAG,IAAI,GAAG,EAAkB,CAAC;IAC1C,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,KAAK,MAAM,IAAI,IAAI,WAAW,CAAC,WAAW,IAAI,EAAE,EAAE,CAAC;QACjD,IAAI,IAAI,CAAC,MAAM,KAAK,SAAS;YAAE,SAAS;QACxC,iEAAiE;QACjE,qEAAqE;QACrE,gEAAgE;QAChE,IAAI,kBAAkB,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,CAAC;YACrC,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACvB,SAAS;QACX,CAAC;QACD,MAAM,GAAG,GAAG,IAAI,CAAC,MAAM,CAAC;QACxB,IAAI,GAAG,KAAK,SAAS,EAAE,CAAC;YACtB,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACvB,SAAS;QACX,CAAC;QACD,IAAI,kBAAkB,CAAC,GAAG,CAAC,EAAE,CAAC;YAC5B,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACvB,SAAS;QACX,CAAC;QACD,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,EAAE,OAAO,GAAG,KAAK,QAAQ,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC;IAC5E,CAAC;IACD,OAAO,EAAE,OAAO,EAAE,QAAQ,EAAE,CAAC;AAC/B,CAAC;AAaD;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,CAAC,KAAK,UAAU,mBAAmB,CAAC,IAIzC;IACC,MAAM,EAAE,WAAW,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,IAAI,CAAC;IAC3C,MAAM,EAAE,OAAO,EAAE,QAAQ,EAAE,GAAG,sBAAsB,CAAC,WAAW,CAAC,CAAC;IAElE,uBAAuB;IACvB,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,OAAO;YACL,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,oBAAoB,CAAC,QAAQ,CAAC;YACrC,aAAa,EAAE,QAAQ;SACxB,CAAC;IACJ,CAAC;IAED,IAAI,CAAC;QACH,MAAM,MAAM,GAAG,MAAM,aAAa,CAAC,MAAM,EAAE;YACzC,GAAG,IAAI,CAAC,UAAU;YAClB,MAAM,EAAE,IAAI,CAAC,MAAM;YACnB,GAAG,CAAC,IAAI,CAAC,WAAW,KAAK,SAAS,IAAI,EAAE,WAAW,EAAE,IAAI,CAAC,WAAW,EAAE,CAAC;YACxE,aAAa,EAAE,OAAO;YACtB,uEAAuE;YACvE,+CAA+C;YAC/C,UAAU,EAAE,IAAI;YAChB,WAAW,EAAE,UAAU,WAAW,CAAC,GAAG,EAAE;YACxC,QAAQ,EAAE,KAAK;SAChB,CAAC,CAAC;QACH,MAAM,MAAM,GAAG,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,MAAM,EAAE,MAAM,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC,CAAC;QACpE,MAAM,MAAM,GAAG,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,SAAS,IAAI,WAAW,CAAC,MAAM,CAAC,CAAC;QACrE,MAAM,OAAO,GAAG,cAAc,CAAC,MAAM,CAAC,WAAW,CAAC,CAAC;QACnD,MAAM,EAAE,GAAG,CAAC,MAAM,CAAC,YAAY,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,CAAC;QACxD,MAAM,KAAK,GACT,OAAO,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,oBAAoB,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,YAAY,CAAC;QAC3E,OAAO;YACL,EAAE;YACF,GAAG,CAAC,MAAM,EAAE,GAAG,KAAK,SAAS,IAAI,EAAE,MAAM,EAAE,MAAM,CAAC,GAAG,EAAE,CAAC;YACxD,MAAM;YACN,GAAG,CAAC,KAAK,KAAK,SAAS,IAAI,EAAE,KAAK,EAAE,CAAC;YACrC,GAAG,CAAC,OAAO,CAAC,MAAM,GAAG,CAAC,IAAI,EAAE,aAAa,EAAE,OAAO,EAAE,CAAC;SACtD,CAAC;IACJ,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO;YACL,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,GAAG,CAAC;SACxD,CAAC;IACJ,CAAC;AACH,CAAC"}
|
|
@@ -474,6 +474,17 @@ export interface RunnerDeps {
|
|
|
474
474
|
* sites below). Unset for a normal (non-replay) run.
|
|
475
475
|
*/
|
|
476
476
|
mockedOutputs?: Map<string, string>;
|
|
477
|
+
/**
|
|
478
|
+
* Mocked-replay boundary (see `replayBoundary.ts`). When true, the run is
|
|
479
|
+
* EVIDENCE-ONLY: a step whose id is absent from `mockedOutputs` is refused
|
|
480
|
+
* with `replay_refused_unmocked_step` instead of dispatched, and the
|
|
481
|
+
* dispatch seam itself (`executeStep`, `buildAgentExecutorDeps`) throws
|
|
482
|
+
* `ReplayIncompleteError` before any tool or agent is invoked. Set by
|
|
483
|
+
* `replayFlatMockedRun` / `replayMockedRun`; never for a live run. Absent
|
|
484
|
+
* ⇒ the historical fallthrough (a step not in the map runs for real),
|
|
485
|
+
* which the simulator (`simulateMockedRun.ts`) relies on with stub deps.
|
|
486
|
+
*/
|
|
487
|
+
replayOnly?: boolean;
|
|
477
488
|
}
|
|
478
489
|
export interface RunResult {
|
|
479
490
|
recipe: string;
|
|
@@ -576,8 +587,10 @@ export type StepResult = {
|
|
|
576
587
|
*/
|
|
577
588
|
output?: unknown;
|
|
578
589
|
};
|
|
579
|
-
export type StepDeps = Required<Omit<RunnerDeps, "now" | "logDir" | "recordFixturesDir" | "runLog" | "ledgerDir" | "manualRunId" | "activityLog" | "requireApprovalFn" | "governance" | "gateAutomatedRuns" | "agentDisallowedTools" | "signal" | "workerId" | "mockedOutputs">> & {
|
|
590
|
+
export type StepDeps = Required<Omit<RunnerDeps, "now" | "logDir" | "recordFixturesDir" | "runLog" | "ledgerDir" | "manualRunId" | "activityLog" | "requireApprovalFn" | "governance" | "gateAutomatedRuns" | "agentDisallowedTools" | "signal" | "workerId" | "mockedOutputs" | "replayOnly">> & {
|
|
580
591
|
workdir: string;
|
|
592
|
+
/** See `RunnerDeps.replayOnly`. Copied by `resolveStepDeps`. */
|
|
593
|
+
replayOnly?: boolean;
|
|
581
594
|
logDir?: string;
|
|
582
595
|
recordFixturesDir?: string;
|
|
583
596
|
runLog?: RecipeRunLog;
|
|
@@ -67,6 +67,7 @@ import { resolveLocalEndpoint, resolveLocalModel } from "./localSettings.js";
|
|
|
67
67
|
import { defaultDeprecationWarn, normalizeRecipeForRuntime, } from "./migrations/index.js";
|
|
68
68
|
import { costRouter } from "./pricing/costRouter.js";
|
|
69
69
|
import { loadPriceTable, costUsd as priceCostUsd, } from "./pricing/priceTable.js";
|
|
70
|
+
import { ReplayIncompleteError } from "./replayBoundary.js";
|
|
70
71
|
import { resolveRecipePath } from "./resolveRecipePath.js";
|
|
71
72
|
import { RunBudget } from "./runBudget.js";
|
|
72
73
|
import { recordAttemptRun } from "./runLedgers.js";
|
|
@@ -1698,6 +1699,13 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
1698
1699
|
// `RunBudget.reconcile` records a fail-open warning per driver per
|
|
1699
1700
|
// run and continues.
|
|
1700
1701
|
try {
|
|
1702
|
+
// Replay boundary (replayBoundary.ts, layer 2). Flat agent steps
|
|
1703
|
+
// capture no `output`, so under replay there is never a mock for
|
|
1704
|
+
// one — refuse here rather than let the catch below be the first
|
|
1705
|
+
// to hear about it from the seam.
|
|
1706
|
+
if (deps.replayOnly === true) {
|
|
1707
|
+
throw new ReplayIncompleteError([stepId], "agent step");
|
|
1708
|
+
}
|
|
1701
1709
|
// Phase 4: opt-in cost-aware routing. No-op (returns preferred) when
|
|
1702
1710
|
// the step has no `downshift` list or no USD cap is set.
|
|
1703
1711
|
const routed = resolveRouting({ driver: agentCfg.driver, model: agentCfg.model }, agentCfg.downshift, renderedPrompt, runBudget);
|
|
@@ -1991,6 +1999,16 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
1991
1999
|
if (deps.mockedOutputs?.has(stepId)) {
|
|
1992
2000
|
result = deps.mockedOutputs.get(stepId) ?? null;
|
|
1993
2001
|
}
|
|
2002
|
+
else if (deps.replayOnly === true) {
|
|
2003
|
+
// Replay boundary (replayBoundary.ts, layer 2): no capture for this
|
|
2004
|
+
// step ⇒ refuse, never dispatch. Recorded as a thrown step error so
|
|
2005
|
+
// the existing halt/fail-open bookkeeping below applies unchanged;
|
|
2006
|
+
// no retry, because there is nothing to retry against.
|
|
2007
|
+
const refusal = new ReplayIncompleteError([stepId], `tool ${step.tool ?? "?"}`);
|
|
2008
|
+
thrownError = refusal.message;
|
|
2009
|
+
thrownErrorCode = refusal.code;
|
|
2010
|
+
result = null;
|
|
2011
|
+
}
|
|
1994
2012
|
else {
|
|
1995
2013
|
for (let attempt = 0; attempt <= retryCount; attempt++) {
|
|
1996
2014
|
if (attempt > 0) {
|
|
@@ -2488,6 +2506,15 @@ rollbackAtDecision) {
|
|
|
2488
2506
|
if (!toolId) {
|
|
2489
2507
|
return null;
|
|
2490
2508
|
}
|
|
2509
|
+
// Replay boundary (replayBoundary.ts, layer 3). A mocked replay's StepDeps
|
|
2510
|
+
// carry `replayOnly`; reaching this seam under replay means no capture
|
|
2511
|
+
// short-circuited the step upstream, so the ONLY correct outcome is to
|
|
2512
|
+
// refuse. Checked before the registry lookup, the approval gate and the
|
|
2513
|
+
// policy check on purpose: none of those may convert "no evidence" into
|
|
2514
|
+
// "permitted to run".
|
|
2515
|
+
if (deps.replayOnly === true) {
|
|
2516
|
+
throw new ReplayIncompleteError([step.into ?? toolId], `tool ${toolId}`);
|
|
2517
|
+
}
|
|
2491
2518
|
// Check if tool is registered in the new registry
|
|
2492
2519
|
if (hasTool(toolId)) {
|
|
2493
2520
|
const tool = getTool(toolId);
|
|
@@ -2996,6 +3023,10 @@ function resolveStepDeps(deps, scope) {
|
|
|
2996
3023
|
: new WriteEffectLedger(),
|
|
2997
3024
|
workerId: deps.workerId,
|
|
2998
3025
|
recipeName: scope?.recipeName,
|
|
3026
|
+
// Replay boundary marker rides on StepDeps so the dispatch seam
|
|
3027
|
+
// (`executeStep`, `buildAgentExecutorDeps`) can refuse without the run
|
|
3028
|
+
// loop's help. Only ever set by the replay entrypoints.
|
|
3029
|
+
...(deps.replayOnly === true && { replayOnly: true }),
|
|
2999
3030
|
// Ephemeral rollback — same disk-availability gating as writeEffectLedger
|
|
3000
3031
|
// above (deliberately: both share the operator's --ledger-dir/--attempt
|
|
3001
3032
|
// inputs). No in-memory fallback: rollback only makes sense as a
|
|
@@ -3083,6 +3114,12 @@ function buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride,
|
|
|
3083
3114
|
* the run has started; the flat caller passes its local const directly.
|
|
3084
3115
|
*/
|
|
3085
3116
|
runTaskId) {
|
|
3117
|
+
// Replay boundary (replayBoundary.ts, layer 3) — the single place agent
|
|
3118
|
+
// executor deps are built for every dispatch site in this runner, so a
|
|
3119
|
+
// replay's StepDeps can never hand a driver to an agent step.
|
|
3120
|
+
if (stepDeps.replayOnly === true) {
|
|
3121
|
+
throw new ReplayIncompleteError(["agent"], "agent step");
|
|
3122
|
+
}
|
|
3086
3123
|
const claudeCliFn = claudeCodeFnOverride ?? stepDeps.claudeCodeFn;
|
|
3087
3124
|
return {
|
|
3088
3125
|
// ── ADR-0021 information boundary ───────────────────────────────────────
|
|
@@ -3954,6 +3991,8 @@ export async function dispatchRecipe(recipe, deps, seedContext = {}) {
|
|
|
3954
3991
|
runLog: deps.chainedOptions?.runLog,
|
|
3955
3992
|
activityLog: deps.chainedOptions?.activityLog,
|
|
3956
3993
|
mockedOutputs: deps.chainedOptions?.mockedOutputs,
|
|
3994
|
+
...((deps.replayOnly === true ||
|
|
3995
|
+
deps.chainedOptions?.replayOnly === true) && { replayOnly: true }),
|
|
3957
3996
|
taskIdPrefix: deps.chainedOptions?.taskIdPrefix,
|
|
3958
3997
|
// Parity (#850): forward the run-level budget, price table, and
|
|
3959
3998
|
// cancellation signal that the chained runner honours. Without these the
|