taskflow-core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/analyst.md +30 -0
- package/dist/agents/critic.md +31 -0
- package/dist/agents/doc-writer.md +43 -0
- package/dist/agents/executor-code.md +36 -0
- package/dist/agents/executor-fast.md +26 -0
- package/dist/agents/executor-ui.md +35 -0
- package/dist/agents/executor.md +29 -0
- package/dist/agents/final-arbiter.md +29 -0
- package/dist/agents/plan-arbiter.md +35 -0
- package/dist/agents/planner.md +30 -0
- package/dist/agents/recover.md +28 -0
- package/dist/agents/reviewer.md +37 -0
- package/dist/agents/risk-reviewer.md +37 -0
- package/dist/agents/scout.md +51 -0
- package/dist/agents/security-reviewer.md +39 -0
- package/dist/agents/test-engineer.md +31 -0
- package/dist/agents/verifier.md +29 -0
- package/dist/agents/visual-explorer.md +32 -0
- package/dist/agents.d.ts +55 -0
- package/dist/agents.d.ts.map +1 -0
- package/dist/agents.js +265 -0
- package/dist/agents.js.map +1 -0
- package/dist/cache.d.ts +57 -0
- package/dist/cache.d.ts.map +1 -0
- package/dist/cache.js +256 -0
- package/dist/cache.js.map +1 -0
- package/dist/compile.d.ts +37 -0
- package/dist/compile.d.ts.map +1 -0
- package/dist/compile.js +324 -0
- package/dist/compile.js.map +1 -0
- package/dist/context-store.d.ts +142 -0
- package/dist/context-store.d.ts.map +1 -0
- package/dist/context-store.js +383 -0
- package/dist/context-store.js.map +1 -0
- package/dist/detached-runner.d.ts +12 -0
- package/dist/detached-runner.d.ts.map +1 -0
- package/dist/detached-runner.js +68 -0
- package/dist/detached-runner.js.map +1 -0
- package/dist/flowir/hash.d.ts +51 -0
- package/dist/flowir/hash.d.ts.map +1 -0
- package/dist/flowir/hash.js +90 -0
- package/dist/flowir/hash.js.map +1 -0
- package/dist/flowir/index.d.ts +43 -0
- package/dist/flowir/index.d.ts.map +1 -0
- package/dist/flowir/index.js +62 -0
- package/dist/flowir/index.js.map +1 -0
- package/dist/flowir/meta.d.ts +103 -0
- package/dist/flowir/meta.d.ts.map +1 -0
- package/dist/flowir/meta.js +19 -0
- package/dist/flowir/meta.js.map +1 -0
- package/dist/flowir/phasefp.d.ts +55 -0
- package/dist/flowir/phasefp.d.ts.map +1 -0
- package/dist/flowir/phasefp.js +123 -0
- package/dist/flowir/phasefp.js.map +1 -0
- package/dist/flowir/translate.d.ts +37 -0
- package/dist/flowir/translate.d.ts.map +1 -0
- package/dist/flowir/translate.js +137 -0
- package/dist/flowir/translate.js.map +1 -0
- package/dist/frontmatter.d.ts +21 -0
- package/dist/frontmatter.d.ts.map +1 -0
- package/dist/frontmatter.js +107 -0
- package/dist/frontmatter.js.map +1 -0
- package/dist/host/runner-types.d.ts +98 -0
- package/dist/host/runner-types.d.ts.map +1 -0
- package/dist/host/runner-types.js +21 -0
- package/dist/host/runner-types.js.map +1 -0
- package/dist/index.d.ts +26 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +30 -0
- package/dist/index.js.map +1 -0
- package/dist/interpolate.d.ts +52 -0
- package/dist/interpolate.d.ts.map +1 -0
- package/dist/interpolate.js +418 -0
- package/dist/interpolate.js.map +1 -0
- package/dist/paths.d.ts +23 -0
- package/dist/paths.d.ts.map +1 -0
- package/dist/paths.js +43 -0
- package/dist/paths.js.map +1 -0
- package/dist/runner-core.d.ts +49 -0
- package/dist/runner-core.d.ts.map +1 -0
- package/dist/runner-core.js +208 -0
- package/dist/runner-core.js.map +1 -0
- package/dist/runtime.d.ts +234 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +2284 -0
- package/dist/runtime.js.map +1 -0
- package/dist/schema.d.ts +257 -0
- package/dist/schema.d.ts.map +1 -0
- package/dist/schema.js +795 -0
- package/dist/schema.js.map +1 -0
- package/dist/stale.d.ts +72 -0
- package/dist/stale.d.ts.map +1 -0
- package/dist/stale.js +179 -0
- package/dist/stale.js.map +1 -0
- package/dist/store.d.ts +237 -0
- package/dist/store.d.ts.map +1 -0
- package/dist/store.js +761 -0
- package/dist/store.js.map +1 -0
- package/dist/typebox-helpers.d.ts +15 -0
- package/dist/typebox-helpers.d.ts.map +1 -0
- package/dist/typebox-helpers.js +19 -0
- package/dist/typebox-helpers.js.map +1 -0
- package/dist/usage.d.ts +21 -0
- package/dist/usage.d.ts.map +1 -0
- package/dist/usage.js +33 -0
- package/dist/usage.js.map +1 -0
- package/dist/verify.d.ts +38 -0
- package/dist/verify.d.ts.map +1 -0
- package/dist/verify.js +324 -0
- package/dist/verify.js.map +1 -0
- package/dist/workspace.d.ts +64 -0
- package/dist/workspace.d.ts.map +1 -0
- package/dist/workspace.js +178 -0
- package/dist/workspace.js.map +1 -0
- package/package.json +40 -0
package/dist/runtime.js
ADDED
|
@@ -0,0 +1,2284 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Taskflow runtime — the orchestration engine.
|
|
3
|
+
*
|
|
4
|
+
* Resolves the phase DAG into topological layers and executes each phase by
|
|
5
|
+
* delegating to isolated subagents. Intermediate phase outputs live here (in
|
|
6
|
+
* RunState) and never enter the host conversation's context window — only the
|
|
7
|
+
* final phase output is returned to the caller.
|
|
8
|
+
*
|
|
9
|
+
* Supports resume: phases whose resolved input hash matches a cached completed
|
|
10
|
+
* result are skipped.
|
|
11
|
+
*/
|
|
12
|
+
import * as path from "node:path";
|
|
13
|
+
import * as fs from "node:fs";
|
|
14
|
+
import { coerceArray, evaluateCondition, interpolate, safeParse, tryEvaluateCondition } from "./interpolate.js";
|
|
15
|
+
import { isFailed, isTransientError, mapWithConcurrencyLimit } from "./runner-core.js";
|
|
16
|
+
/** Default runner used when no host injected one: fail loudly rather than
|
|
17
|
+
* silently spawn anything (core is host-neutral and cannot spawn pi/codex). */
|
|
18
|
+
const noRunnerInjected = async (_cwd, _agents, agentName, task) => ({
|
|
19
|
+
agent: agentName,
|
|
20
|
+
task,
|
|
21
|
+
exitCode: 1,
|
|
22
|
+
output: "",
|
|
23
|
+
stderr: "No subagent runner injected. A host adapter must set RuntimeDeps.runTask (e.g. piSubagentRunner or codexSubagentRunner).",
|
|
24
|
+
usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, cost: 0, contextTokens: 0, turns: 0 },
|
|
25
|
+
errorMessage: "No subagent runner injected",
|
|
26
|
+
stopReason: "error",
|
|
27
|
+
});
|
|
28
|
+
import { aggregateUsage, emptyUsage } from "./usage.js";
|
|
29
|
+
import { dependenciesOf, finalPhase, LOOP_DEFAULT_MAX_ITERATIONS, LOOP_HARD_MAX_ITERATIONS, MAX_DYNAMIC_MAP_ITEMS, MAX_DYNAMIC_NESTING, parseTtlMs, resolveArgs, topoLayers, TOURNAMENT_DEFAULT_VARIANTS, TOURNAMENT_HARD_MAX_VARIANTS, validateTaskflow } from "./schema.js";
|
|
30
|
+
import { verifyTaskflow } from "./verify.js";
|
|
31
|
+
import { hashInput, newRunId, runsDir } from "./store.js";
|
|
32
|
+
import { CacheStore, resolveFingerprint } from "./cache.js";
|
|
33
|
+
import { compileTaskflowToIR, phaseFingerprint } from "./flowir/index.js";
|
|
34
|
+
import { computeStaleFrontier, declaredReadMapOfDef, readMapOf } from "./stale.js";
|
|
35
|
+
import { ctxDirFor, drainPendingSpawns, initCtxDir, registerNode, setNodeStatus } from "./context-store.js";
|
|
36
|
+
import { allocateWorkspace, isWorkspaceKeyword } from "./workspace.js";
|
|
37
|
+
/** Compute the incremental-reuse summary from a run's terminal phase states.
|
|
38
|
+
* Pure, total, never throws. A phase is "reused" iff it carries a `cacheHit`
|
|
39
|
+
* marker (set by `cachedPhase` for both within-run resume and cross-run hits). */
|
|
40
|
+
export function summarizeReuse(state) {
|
|
41
|
+
let executed = 0;
|
|
42
|
+
let reusedRunOnly = 0;
|
|
43
|
+
let reusedCrossRun = 0;
|
|
44
|
+
let savedUSD = 0;
|
|
45
|
+
for (const ps of Object.values(state.phases)) {
|
|
46
|
+
if (ps.status !== "done")
|
|
47
|
+
continue;
|
|
48
|
+
if (ps.cacheHit === "run-only") {
|
|
49
|
+
reusedRunOnly++;
|
|
50
|
+
savedUSD += ps.usage?.cost ?? 0; // within-run resume preserves prior usage
|
|
51
|
+
}
|
|
52
|
+
else if (ps.cacheHit === "cross-run") {
|
|
53
|
+
reusedCrossRun++; // cross-run hits zero their usage — cost not recoverable
|
|
54
|
+
}
|
|
55
|
+
else {
|
|
56
|
+
executed++;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
return {
|
|
60
|
+
executed,
|
|
61
|
+
reusedRunOnly,
|
|
62
|
+
reusedCrossRun,
|
|
63
|
+
done: executed + reusedRunOnly + reusedCrossRun,
|
|
64
|
+
savedUSD,
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
function buildInterpolationContext(state, previousOutput, locals, onRead) {
|
|
68
|
+
const steps = {};
|
|
69
|
+
for (const [id, ps] of Object.entries(state.phases)) {
|
|
70
|
+
// Include both done AND failed phases so downstream phases can see
|
|
71
|
+
// error info. Skipped phases (upstream failure cascade) are excluded.
|
|
72
|
+
if (ps.status === "done" || ps.status === "failed") {
|
|
73
|
+
if (ps.output !== undefined) {
|
|
74
|
+
steps[id] = { output: ps.output, json: ps.json };
|
|
75
|
+
}
|
|
76
|
+
else if (ps.status === "failed") {
|
|
77
|
+
// M-3: Failed phases without output get a placeholder so
|
|
78
|
+
// downstream references like {steps.X.output} resolve to a
|
|
79
|
+
// sensible value instead of leaving the raw placeholder intact.
|
|
80
|
+
steps[id] = { output: "[previous phase failed]", json: undefined };
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
return { args: state.args, steps, previousOutput, locals, onRead };
|
|
85
|
+
}
|
|
86
|
+
function resultToPhaseState(id, r, inputHash, parseJson) {
|
|
87
|
+
const failed = isFailed(r);
|
|
88
|
+
const attempts = attemptsOf(r);
|
|
89
|
+
// For failed phases, embed the error info in the output so downstream
|
|
90
|
+
// phases (and the user) can see what went wrong. The raw r.output is
|
|
91
|
+
// often a useless placeholder like "(upstream error: subagent failed)".
|
|
92
|
+
const output = failed
|
|
93
|
+
? r.errorMessage || r.stderr || r.output
|
|
94
|
+
: r.output;
|
|
95
|
+
return {
|
|
96
|
+
id,
|
|
97
|
+
status: failed ? "failed" : "done",
|
|
98
|
+
output,
|
|
99
|
+
json: parseJson && !failed ? safeParse(r.output) : undefined,
|
|
100
|
+
usage: r.usage,
|
|
101
|
+
model: r.model,
|
|
102
|
+
attempts: attempts > 1 ? attempts : undefined,
|
|
103
|
+
error: failed ? r.errorMessage || r.stderr || r.output : undefined,
|
|
104
|
+
inputHash,
|
|
105
|
+
endedAt: Date.now(),
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* Synthesize a 0-token `RunResult` from a cached per-item `PhaseState` so a
|
|
110
|
+
* cross-run per-item cache hit flows through `mergePhaseState` as a normal
|
|
111
|
+
* successful fan-out item. `stopReason: "cache-hit"` is NOT in `isFailed`'s
|
|
112
|
+
* failure set (only "error"/"aborted"/non-zero exit), so the item counts as
|
|
113
|
+
* success. Usage is `emptyUsage()` — a cached item spent no new tokens this
|
|
114
|
+
* run, so `mergePhaseState`'s `aggregateUsage` charges nothing for it.
|
|
115
|
+
*
|
|
116
|
+
* Used only by the `map` per-item cache path (see `runFanout`). Fail-open by
|
|
117
|
+
* construction: this is only reached AFTER a successful `cachedPhase` lookup,
|
|
118
|
+
* so `ps.output` is always present.
|
|
119
|
+
*/
|
|
120
|
+
function phaseStateToRunResult(ps, it) {
|
|
121
|
+
return {
|
|
122
|
+
agent: it.agent,
|
|
123
|
+
task: it.task,
|
|
124
|
+
exitCode: 0,
|
|
125
|
+
output: ps.output ?? "",
|
|
126
|
+
stderr: "",
|
|
127
|
+
usage: emptyUsage(),
|
|
128
|
+
model: ps.model,
|
|
129
|
+
stopReason: "cache-hit",
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
/** Convert observed read refs (e.g. "steps.scout.output") into a structured
|
|
133
|
+
* readSet keyed by upstream phase id, tagging each with the version
|
|
134
|
+
* (= inputHash) that was current when read. Only `steps.*` refs are upstream
|
|
135
|
+
* phase dependencies; args/item/previous are invocation/loop values. */
|
|
136
|
+
function readRefsToReads(refs, state) {
|
|
137
|
+
const out = [];
|
|
138
|
+
const seen = new Set();
|
|
139
|
+
for (const ref of refs) {
|
|
140
|
+
const m = /^steps\.([A-Za-z0-9_-]+)\b/.exec(ref);
|
|
141
|
+
if (!m)
|
|
142
|
+
continue;
|
|
143
|
+
const stepId = m[1];
|
|
144
|
+
if (seen.has(stepId))
|
|
145
|
+
continue;
|
|
146
|
+
seen.add(stepId);
|
|
147
|
+
out.push({ stepId, version: state.phases[stepId]?.inputHash });
|
|
148
|
+
}
|
|
149
|
+
return out;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Surface unresolved interpolation placeholders (the `missing[]` from
|
|
153
|
+
* `interpolate()`). Without this they are silently left intact in the task —
|
|
154
|
+
* the doc comment in interpolate.ts promises "a recorded warning". We both
|
|
155
|
+
* log to the console and return a string to attach to PhaseState.warnings so
|
|
156
|
+
* the warning is persisted in the run record and visible in `/tf runs`.
|
|
157
|
+
* Returns undefined when nothing is missing.
|
|
158
|
+
*/
|
|
159
|
+
function warnUnresolvedRefs(phaseId, missing) {
|
|
160
|
+
if (!missing.length)
|
|
161
|
+
return undefined;
|
|
162
|
+
const unique = Array.from(new Set(missing));
|
|
163
|
+
const msg = `unresolved refs in task: ${unique.map((m) => `{${m}}`).join(", ")} — left intact (check dependsOn / placeholder spelling)`;
|
|
164
|
+
console.warn(`[taskflow] phase '${phaseId}': ${msg}`);
|
|
165
|
+
return msg;
|
|
166
|
+
}
|
|
167
|
+
/** Attempts recorded by the retry wrapper (defaults to 1). */
|
|
168
|
+
function attemptsOf(r) {
|
|
169
|
+
const a = r.attempts;
|
|
170
|
+
return typeof a === "number" && a > 0 ? a : 1;
|
|
171
|
+
}
|
|
172
|
+
/** Cancellable delay used between retry attempts. */
|
|
173
|
+
function delay(ms, signal) {
|
|
174
|
+
return new Promise((resolve) => {
|
|
175
|
+
if (ms <= 0)
|
|
176
|
+
return resolve();
|
|
177
|
+
let onAbort;
|
|
178
|
+
const t = setTimeout(() => {
|
|
179
|
+
if (signal && onAbort)
|
|
180
|
+
signal.removeEventListener("abort", onAbort);
|
|
181
|
+
resolve();
|
|
182
|
+
}, ms);
|
|
183
|
+
if (signal) {
|
|
184
|
+
if (signal.aborted) {
|
|
185
|
+
clearTimeout(t);
|
|
186
|
+
return resolve();
|
|
187
|
+
}
|
|
188
|
+
onAbort = () => {
|
|
189
|
+
clearTimeout(t);
|
|
190
|
+
resolve();
|
|
191
|
+
};
|
|
192
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
193
|
+
}
|
|
194
|
+
});
|
|
195
|
+
}
|
|
196
|
+
function failPhase(id, error) {
|
|
197
|
+
return { id, status: "failed", error, inputHash: hashInput(id, error), endedAt: Date.now(), usage: emptyUsage() };
|
|
198
|
+
}
|
|
199
|
+
/**
|
|
200
|
+
* Normalize an inline `flow.def` payload into a full Taskflow shape.
|
|
201
|
+
* Accepts: a full Taskflow ({name?,phases:[...]}), a bare phases array, or
|
|
202
|
+
* {phases:[...]}. Returns undefined if the shape is unrecognized. A recognized
|
|
203
|
+
* shape with ZERO phases is returned as-is (caller treats it as a no-op) so the
|
|
204
|
+
* empty-plan case is distinguishable from a malformed one.
|
|
205
|
+
*
|
|
206
|
+
* The payload is deep-cloned so the runtime never shares references with (or
|
|
207
|
+
* mutates) the upstream phase's parsed JSON. Cloning also drops any non-own /
|
|
208
|
+
* prototype-shadowing `__proto__` own-property that a crafted JSON could carry.
|
|
209
|
+
*/
|
|
210
|
+
function normalizeInlineDef(parsed, phaseId) {
|
|
211
|
+
let shaped;
|
|
212
|
+
if (Array.isArray(parsed)) {
|
|
213
|
+
shaped = { name: `${phaseId}-inline`, phases: parsed };
|
|
214
|
+
}
|
|
215
|
+
else if (parsed && typeof parsed === "object") {
|
|
216
|
+
const o = parsed;
|
|
217
|
+
if (Array.isArray(o.phases)) {
|
|
218
|
+
const name = typeof o.name === "string" && o.name.length > 0 ? o.name : `${phaseId}-inline`;
|
|
219
|
+
shaped = { ...o, name, phases: o.phases };
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
if (!shaped)
|
|
223
|
+
return undefined;
|
|
224
|
+
// Deep clone via JSON round-trip: severs shared references with upstream output
|
|
225
|
+
// and drops any own "__proto__" key (JSON.stringify omits it). As belt-and-
|
|
226
|
+
// suspenders, also delete inert `constructor`/`prototype` own-keys a crafted
|
|
227
|
+
// payload could carry, so the returned object is clean of pollution vectors.
|
|
228
|
+
try {
|
|
229
|
+
const clone = JSON.parse(JSON.stringify(shaped));
|
|
230
|
+
for (const k of ["__proto__", "constructor", "prototype"]) {
|
|
231
|
+
if (Object.prototype.hasOwnProperty.call(clone, k))
|
|
232
|
+
delete clone[k];
|
|
233
|
+
}
|
|
234
|
+
return clone;
|
|
235
|
+
}
|
|
236
|
+
catch {
|
|
237
|
+
return undefined;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
/**
|
|
241
|
+
* Clamp a runtime-generated sub-flow's budget so it can only ever be TIGHTER
|
|
242
|
+
* than the parent's, never looser. A generated def cannot raise the spend cap by
|
|
243
|
+
* declaring its own large budget. Each dimension becomes min(child, parent).
|
|
244
|
+
*/
|
|
245
|
+
function clampSubFlowBudget(sub, parentBudget) {
|
|
246
|
+
if (!parentBudget)
|
|
247
|
+
return sub;
|
|
248
|
+
const child = sub.budget;
|
|
249
|
+
const clamped = {
|
|
250
|
+
maxUSD: Math.min(child?.maxUSD ?? Infinity, parentBudget.maxUSD ?? Infinity),
|
|
251
|
+
maxTokens: Math.min(child?.maxTokens ?? Infinity, parentBudget.maxTokens ?? Infinity),
|
|
252
|
+
};
|
|
253
|
+
// Drop Infinity dimensions (no cap on that axis).
|
|
254
|
+
const budget = {};
|
|
255
|
+
if (Number.isFinite(clamped.maxUSD))
|
|
256
|
+
budget.maxUSD = clamped.maxUSD;
|
|
257
|
+
if (Number.isFinite(clamped.maxTokens))
|
|
258
|
+
budget.maxTokens = clamped.maxTokens;
|
|
259
|
+
return { ...sub, budget: budget.maxUSD === undefined && budget.maxTokens === undefined ? undefined : budget };
|
|
260
|
+
}
|
|
261
|
+
/** Aggregate run cost/tokens so far and test against the budget. */
|
|
262
|
+
function overBudget(state) {
|
|
263
|
+
const budget = state.def.budget;
|
|
264
|
+
if (!budget)
|
|
265
|
+
return { over: false, reason: "" };
|
|
266
|
+
const u = aggregateUsage(Object.values(state.phases).map((p) => p.usage ?? emptyUsage()));
|
|
267
|
+
if (budget.maxUSD !== undefined && u.cost > budget.maxUSD) {
|
|
268
|
+
return { over: true, reason: `cost $${u.cost.toFixed(3)} exceeded cap $${budget.maxUSD}` };
|
|
269
|
+
}
|
|
270
|
+
if (budget.maxTokens !== undefined && u.input + u.output > budget.maxTokens) {
|
|
271
|
+
return { over: true, reason: `tokens ${u.input + u.output} exceeded cap ${budget.maxTokens}` };
|
|
272
|
+
}
|
|
273
|
+
return { over: false, reason: "" };
|
|
274
|
+
}
|
|
275
|
+
/** Merge several sub-results into a single PhaseState (for map/parallel). */
|
|
276
|
+
function mergePhaseState(id, results, inputHash, parseJson) {
|
|
277
|
+
const budgetSkips = results.filter((r) => r.stopReason === "budget-skipped");
|
|
278
|
+
const ran = results.filter((r) => r.stopReason !== "budget-skipped");
|
|
279
|
+
const anyFailed = ran.some(isFailed);
|
|
280
|
+
const usage = aggregateUsage(results.map((r) => r.usage));
|
|
281
|
+
// B12: surface the model(s) used in the fan-out so consumers can show
|
|
282
|
+
// which model produced the merged output.
|
|
283
|
+
const model = ran.find((r) => r.model !== undefined)?.model;
|
|
284
|
+
// Combine outputs as a labelled list; also expose a JSON array of outputs.
|
|
285
|
+
// For failed items, use the error message instead of the useless placeholder.
|
|
286
|
+
// Labels are positionally aligned to the ORIGINAL `over` array: we iterate
|
|
287
|
+
// over ALL results (including budget-skipped, which are filtered to null) and
|
|
288
|
+
// use `results.length` as N, so item k's label reads `[k/N]` matching its
|
|
289
|
+
// position in `over` — not its rank among non-skipped items. Per-item cache
|
|
290
|
+
// hits (`stopReason: "cache-hit"`) are not budget-skipped, so they keep their
|
|
291
|
+
// original positional label.
|
|
292
|
+
const combinedText = results
|
|
293
|
+
.map((r, i) => {
|
|
294
|
+
if (r.stopReason === "budget-skipped")
|
|
295
|
+
return null;
|
|
296
|
+
const label = `### [${i + 1}/${results.length}] ${r.agent}${isFailed(r) ? " (failed)" : ""}`;
|
|
297
|
+
const content = isFailed(r) ? (r.errorMessage || r.stderr || r.output) : r.output;
|
|
298
|
+
return `${label}\n\n${content}`;
|
|
299
|
+
})
|
|
300
|
+
.filter((x) => x !== null)
|
|
301
|
+
.join("\n\n---\n\n");
|
|
302
|
+
// Only successful runs feed the parsed JSON array (no error/skip strings).
|
|
303
|
+
const jsonArray = parseJson ? ran.filter((r) => !isFailed(r)).map((r) => safeParse(r.output) ?? r.output) : undefined;
|
|
304
|
+
const failedCount = ran.filter(isFailed).length;
|
|
305
|
+
const attempts = results.reduce((sum, r) => sum + attemptsOf(r), 0);
|
|
306
|
+
const errors = ran.filter(isFailed).map((r) => `${r.agent}: ${r.errorMessage ?? r.stderr}`);
|
|
307
|
+
if (budgetSkips.length)
|
|
308
|
+
errors.push(`${budgetSkips.length} item(s) skipped: budget exceeded`);
|
|
309
|
+
return {
|
|
310
|
+
id,
|
|
311
|
+
status: anyFailed ? "failed" : "done",
|
|
312
|
+
output: combinedText,
|
|
313
|
+
json: jsonArray,
|
|
314
|
+
usage,
|
|
315
|
+
model,
|
|
316
|
+
attempts: attempts > results.length ? attempts : undefined,
|
|
317
|
+
budgetTruncated: budgetSkips.length > 0 || undefined,
|
|
318
|
+
subProgress: { done: ran.length, total: results.length, running: 0, failed: failedCount },
|
|
319
|
+
error: errors.length ? errors.join("; ") : undefined,
|
|
320
|
+
inputHash,
|
|
321
|
+
endedAt: Date.now(),
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* A live-update sink that mirrors a subagent's streaming progress into a single
|
|
326
|
+
* phase's state row, then notifies the TUI. Shared by all single-agent phases.
|
|
327
|
+
*/
|
|
328
|
+
function liveSink(state, phaseId, emitProgress) {
|
|
329
|
+
return (l) => {
|
|
330
|
+
const live = state.phases[phaseId];
|
|
331
|
+
if (live) {
|
|
332
|
+
live.liveText = l.text;
|
|
333
|
+
live.usage = l.usage;
|
|
334
|
+
live.model = l.model;
|
|
335
|
+
}
|
|
336
|
+
emitProgress();
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
/**
|
|
340
|
+
* Pre-read files listed in a phase's `context` field and return them as
|
|
341
|
+
* markdown code blocks. Handles:
|
|
342
|
+
* - literal paths
|
|
343
|
+
* - interpolation refs (e.g. `{steps.scout.json}` resolving to `["a.ts"]`)
|
|
344
|
+
* - per-file truncation via `contextLimit`
|
|
345
|
+
*
|
|
346
|
+
* The result is a single string that should be prepended to the phase task so
|
|
347
|
+
* the subagent never needs to spend turns on file exploration.
|
|
348
|
+
*/
|
|
349
|
+
const CONTEXT_MAX_FILE_BYTES = 10 * 1024 * 1024; // 10 MB
|
|
350
|
+
const MAX_TOTAL_CONTEXT_CHARS = 200_000;
|
|
351
|
+
async function resolvePhaseContext(phase, ctx) {
|
|
352
|
+
const entries = phase.context;
|
|
353
|
+
if (!entries || entries.length === 0)
|
|
354
|
+
return "";
|
|
355
|
+
const limit = phase.contextLimit ?? 8000;
|
|
356
|
+
const paths = [];
|
|
357
|
+
for (const entry of entries) {
|
|
358
|
+
const r = interpolate(entry, ctx);
|
|
359
|
+
if (r.text !== entry) {
|
|
360
|
+
// Resolved — may be a JSON array from {steps.X.json}
|
|
361
|
+
const parsed = safeParse(r.text);
|
|
362
|
+
if (Array.isArray(parsed)) {
|
|
363
|
+
for (const item of parsed) {
|
|
364
|
+
if (typeof item === "string" && item.trim())
|
|
365
|
+
paths.push(item.trim());
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
else if (typeof r.text === "string" && r.text.trim()) {
|
|
369
|
+
paths.push(r.text.trim());
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
else {
|
|
373
|
+
// Unchanged — literal path
|
|
374
|
+
paths.push(entry);
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
const unique = Array.from(new Set(paths));
|
|
378
|
+
// Diagnose JSON blobs masquerading as file paths — common when a context
|
|
379
|
+
// entry like {steps.discover.output} resolves to {"files":[...]} instead
|
|
380
|
+
// of a flat path or JSON array. The author should use {steps.discover.json.files}.
|
|
381
|
+
const jsonBlobs = unique.filter((p) => p.startsWith("{"));
|
|
382
|
+
for (const blob of jsonBlobs) {
|
|
383
|
+
console.warn(`[taskflow] Context entry "${blob.slice(0, 80)}…" looks like a JSON object, not a file path. ` +
|
|
384
|
+
`Use {steps.<id>.json.<field>} to extract a specific field.`);
|
|
385
|
+
}
|
|
386
|
+
const filtered = jsonBlobs.length ? unique.filter((p) => !p.startsWith("{")) : unique;
|
|
387
|
+
const blocks = [];
|
|
388
|
+
for (const p of filtered) {
|
|
389
|
+
try {
|
|
390
|
+
const abs = path.resolve(p);
|
|
391
|
+
const stat = fs.statSync(abs);
|
|
392
|
+
if (!stat.isFile())
|
|
393
|
+
continue;
|
|
394
|
+
if (stat.size > CONTEXT_MAX_FILE_BYTES)
|
|
395
|
+
continue;
|
|
396
|
+
const content = fs.readFileSync(abs, "utf-8");
|
|
397
|
+
const truncated = content.length > limit
|
|
398
|
+
? content.slice(0, limit) + `\n... [truncated ${content.length - limit} chars]`
|
|
399
|
+
: content;
|
|
400
|
+
const ext = path.extname(p).slice(1) || "txt";
|
|
401
|
+
blocks.push(`## File: ${p}\n\n\`\`\`${ext}\n${truncated}\n\`\`\``);
|
|
402
|
+
}
|
|
403
|
+
catch {
|
|
404
|
+
console.warn(`[taskflow] Skipped unreadable context file: ${p}`);
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
// Safety cap: truncate total context when too many files are listed.
|
|
408
|
+
let result = blocks.join("\n\n") + "\n\n";
|
|
409
|
+
if (result.length > MAX_TOTAL_CONTEXT_CHARS) {
|
|
410
|
+
result = result.slice(0, MAX_TOTAL_CONTEXT_CHARS) + `\n\n... [truncated ${result.length - MAX_TOTAL_CONTEXT_CHARS} total chars]`;
|
|
411
|
+
}
|
|
412
|
+
return result;
|
|
413
|
+
}
|
|
414
|
+
/**
|
|
415
|
+
* Run an inline sub-flow queued via `ctx_spawn({subflow})`. Reuses the SAME
|
|
416
|
+
* validation + execution machinery as a `flow{def}` phase (normalizeInlineDef →
|
|
417
|
+
* validateTaskflow(dynamic) → verifyTaskflow → nested executeTaskflow), so a
|
|
418
|
+
* spawned DAG is held to the same safety bar as an author-written one.
|
|
419
|
+
*
|
|
420
|
+
* Crucially it extends `deps._stack` with a `def:spawn-<childNodeId>` frame so
|
|
421
|
+
* the existing inline-nesting guard counts spawn-subflows AND flow{def} on the
|
|
422
|
+
* SAME counter — neither axis can independently reach MAX_DYNAMIC_NESTING and
|
|
423
|
+
* multiply with the other (verdict Issue 1). Failures are fail-open: a bad
|
|
424
|
+
* subflow returns a diagnostic string, never throws.
|
|
425
|
+
*/
|
|
426
|
+
/**
|
|
427
|
+
* The effective working directory for a phase's execution. Honours an allocated
|
|
428
|
+
* workspace override (`_cwdOverride`, set by the executePhase wrapper for
|
|
429
|
+
* isolated `temp`/`dedicated`/`worktree` cwds) and never passes a reserved
|
|
430
|
+
* keyword through to a runner (keywords are resolved upstream into a real dir).
|
|
431
|
+
* Single source of truth — do not inline this formula (divergence here caused
|
|
432
|
+
* two isolation-leak bugs in the 0.0.23 review).
|
|
433
|
+
*/
|
|
434
|
+
function resolveEffCwd(deps, phase) {
|
|
435
|
+
return deps._cwdOverride ?? (isWorkspaceKeyword(phase.cwd) ? deps.cwd : phase.cwd ?? deps.cwd);
|
|
436
|
+
}
|
|
437
|
+
async function runInlineSubflow(subflowSpec, defaultAgent, childNodeId, phase, deps, state) {
|
|
438
|
+
const stack = deps._stack ?? [];
|
|
439
|
+
const inlineDepth = stack.filter((s) => s.startsWith("def:")).length;
|
|
440
|
+
if (inlineDepth >= MAX_DYNAMIC_NESTING) {
|
|
441
|
+
return { output: `(spawned subflow rejected: nesting exceeded MAX_DYNAMIC_NESTING (${MAX_DYNAMIC_NESTING}))`, usage: emptyUsage() };
|
|
442
|
+
}
|
|
443
|
+
const wrapped = normalizeInlineDef(subflowSpec, childNodeId);
|
|
444
|
+
if (!wrapped)
|
|
445
|
+
return { output: "(spawned subflow is not a Taskflow / phases array)", usage: emptyUsage() };
|
|
446
|
+
if (wrapped.phases.length === 0)
|
|
447
|
+
return { output: "(spawned subflow had zero phases — no-op)", usage: emptyUsage() };
|
|
448
|
+
// Inner phases without their own agent inherit the assignment's defaultAgent.
|
|
449
|
+
if (defaultAgent) {
|
|
450
|
+
for (const p of wrapped.phases)
|
|
451
|
+
if (!p.agent)
|
|
452
|
+
p.agent = defaultAgent;
|
|
453
|
+
}
|
|
454
|
+
const spawnCwd = resolveEffCwd(deps, phase);
|
|
455
|
+
const dynCwd = spawnCwd;
|
|
456
|
+
const v = validateTaskflow(wrapped, { dynamic: true, cwd: dynCwd });
|
|
457
|
+
if (!v.ok)
|
|
458
|
+
return { output: `(spawned subflow failed validation: ${v.errors.join("; ")})`, usage: emptyUsage() };
|
|
459
|
+
const ver = verifyTaskflow({ name: wrapped.name, phases: wrapped.phases, budget: wrapped.budget, concurrency: wrapped.concurrency });
|
|
460
|
+
if (!ver.ok) {
|
|
461
|
+
const errs = ver.issues.filter((i) => i.severity === "error").map((i) => i.message);
|
|
462
|
+
return { output: `(spawned subflow failed verification: ${errs.join("; ")})`, usage: emptyUsage() };
|
|
463
|
+
}
|
|
464
|
+
const subDef = clampSubFlowBudget(wrapped, state.def.budget);
|
|
465
|
+
const subState = {
|
|
466
|
+
runId: newRunId(subDef.name),
|
|
467
|
+
flowName: subDef.name,
|
|
468
|
+
def: subDef,
|
|
469
|
+
args: resolveArgs(subDef, {}),
|
|
470
|
+
status: "running",
|
|
471
|
+
phases: {},
|
|
472
|
+
createdAt: Date.now(),
|
|
473
|
+
updatedAt: Date.now(),
|
|
474
|
+
cwd: dynCwd,
|
|
475
|
+
};
|
|
476
|
+
try {
|
|
477
|
+
const subResult = await executeTaskflow(subState, {
|
|
478
|
+
...deps,
|
|
479
|
+
cwd: dynCwd,
|
|
480
|
+
// The parent phase's isolated workspace (if any) applies only to the
|
|
481
|
+
// parent — each spawned sub-phase resolves its own cwd. Clear the
|
|
482
|
+
// override so the whole subflow doesn't inherit the parent's dir
|
|
483
|
+
// (mirrors the `flow` phase handler discipline).
|
|
484
|
+
_cwdOverride: undefined,
|
|
485
|
+
// Don't let spawned sub-phases persist the parent's run state.
|
|
486
|
+
persist: undefined,
|
|
487
|
+
// Unify the nesting counter across both recursion axes (verdict Issue 1).
|
|
488
|
+
_stack: [...stack, state.flowName, `def:spawn-${childNodeId}`],
|
|
489
|
+
_ctxDir: deps._ctxDir,
|
|
490
|
+
onProgress: undefined,
|
|
491
|
+
});
|
|
492
|
+
// Sum every sub-phase's usage so the parent's budget guard sees spawn spend
|
|
493
|
+
// (verdict Issue 2).
|
|
494
|
+
const usage = aggregateUsage(Object.values(subResult.state.phases).map((p) => p.usage ?? emptyUsage()));
|
|
495
|
+
return { output: subResult.finalOutput ?? "", usage };
|
|
496
|
+
}
|
|
497
|
+
catch (e) {
|
|
498
|
+
return { output: `(spawned subflow failed: ${e instanceof Error ? e.message : String(e)})`, usage: emptyUsage() };
|
|
499
|
+
}
|
|
500
|
+
}
|
|
501
|
+
async function runSpawnedChildren(assignments, ctxDir, parentNodeId, phase, deps, state, run) {
|
|
502
|
+
const capped = assignments.slice(0, MAX_DYNAMIC_MAP_ITEMS);
|
|
503
|
+
const lines = [];
|
|
504
|
+
const usages = [];
|
|
505
|
+
// Effective cwd for flat spawned tasks: honour a workspace override and never
|
|
506
|
+
// pass a reserved keyword through to the runner.
|
|
507
|
+
const spawnCwd = resolveEffCwd(deps, phase);
|
|
508
|
+
let idx = 0;
|
|
509
|
+
for (const a of capped) {
|
|
510
|
+
if (deps.signal?.aborted || overBudget(state).over)
|
|
511
|
+
break;
|
|
512
|
+
idx++;
|
|
513
|
+
const childNodeId = `${parentNodeId}--c${idx}`.replace(/[^A-Za-z0-9._-]+/g, "_");
|
|
514
|
+
const isSubflow = a.subflow !== undefined && a.subflow !== null;
|
|
515
|
+
const agentName = isSubflow ? "(subflow)" : resolveAgent(a.agent ?? phase.agent, deps, state);
|
|
516
|
+
registerNode(ctxDir, childNodeId, `${phase.id}:spawn`, parentNodeId, "running");
|
|
517
|
+
let out = "";
|
|
518
|
+
try {
|
|
519
|
+
if (isSubflow) {
|
|
520
|
+
const sub = await runInlineSubflow(a.subflow, a.defaultAgent ?? phase.agent, childNodeId, phase, deps, state);
|
|
521
|
+
out = sub.output;
|
|
522
|
+
usages.push(sub.usage);
|
|
523
|
+
setNodeStatus(ctxDir, childNodeId, "done");
|
|
524
|
+
}
|
|
525
|
+
else {
|
|
526
|
+
const r = await run(spawnCwd, deps.agents, agentName, a.task ?? "", { model: phase.model, thinking: phase.thinking, tools: phase.tools, cwd: spawnCwd, signal: deps.signal, ctxDir, nodeId: childNodeId }, deps.globalThinking);
|
|
527
|
+
out = r.output ?? "";
|
|
528
|
+
if (r.usage)
|
|
529
|
+
usages.push(r.usage);
|
|
530
|
+
setNodeStatus(ctxDir, childNodeId, isFailed(r) ? "failed" : "done");
|
|
531
|
+
// A child may itself have queued spawns — recurse (depth-capped by the tool).
|
|
532
|
+
const grand = drainPendingSpawns(ctxDir, childNodeId);
|
|
533
|
+
if (grand.length > 0 && !deps.signal?.aborted && !overBudget(state).over) {
|
|
534
|
+
const rec = await runSpawnedChildren(grand, ctxDir, childNodeId, phase, deps, state, run);
|
|
535
|
+
if (rec.reports)
|
|
536
|
+
out += rec.reports;
|
|
537
|
+
usages.push(rec.usage);
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
catch (e) {
|
|
542
|
+
setNodeStatus(ctxDir, childNodeId, "failed");
|
|
543
|
+
out = `(spawned child failed: ${e instanceof Error ? e.message : String(e)})`;
|
|
544
|
+
}
|
|
545
|
+
lines.push(`### spawned child ${idx} (${agentName})\n${out}`);
|
|
546
|
+
}
|
|
547
|
+
const usage = aggregateUsage(usages);
|
|
548
|
+
if (lines.length === 0)
|
|
549
|
+
return { reports: undefined, usage };
|
|
550
|
+
return { reports: `\n\n<!-- ctx_spawn: ${lines.length} child report(s) -->\n${lines.join("\n\n")}`, usage };
|
|
551
|
+
}
|
|
552
|
+
async function executePhase(phase, state, deps, prior, emitProgress, _retryDepth = 0, opts) {
|
|
553
|
+
// Non-keyword cwd (or none): no workspace lifecycle — run directly.
|
|
554
|
+
if (!isWorkspaceKeyword(phase.cwd)) {
|
|
555
|
+
return executePhaseInner(phase, state, deps, prior, emitProgress, _retryDepth, opts);
|
|
556
|
+
}
|
|
557
|
+
let ws;
|
|
558
|
+
try {
|
|
559
|
+
ws = allocateWorkspace(phase.cwd, {
|
|
560
|
+
baseCwd: deps.cwd,
|
|
561
|
+
runId: state.runId,
|
|
562
|
+
phaseId: phase.id,
|
|
563
|
+
runsRoot: runsDir(deps.cwd),
|
|
564
|
+
});
|
|
565
|
+
}
|
|
566
|
+
catch {
|
|
567
|
+
ws = undefined; // fail-open: run in the base cwd
|
|
568
|
+
}
|
|
569
|
+
const innerDeps = ws ? { ...deps, _cwdOverride: ws.dir } : deps;
|
|
570
|
+
try {
|
|
571
|
+
const ps = await executePhaseInner(phase, state, innerDeps, prior, emitProgress, _retryDepth, opts);
|
|
572
|
+
if (ws && (ws.kind !== "inherited" || ws.note)) {
|
|
573
|
+
const tag = ws.kind === "inherited" ? "workspace" : `workspace:${ws.kind}`;
|
|
574
|
+
const msg = ws.note ? `${tag} — ${ws.note}` : `${tag} at ${ws.dir}`;
|
|
575
|
+
ps.warnings = [...(ps.warnings ?? []), msg];
|
|
576
|
+
}
|
|
577
|
+
return ps;
|
|
578
|
+
}
|
|
579
|
+
finally {
|
|
580
|
+
try {
|
|
581
|
+
ws?.teardown();
|
|
582
|
+
}
|
|
583
|
+
catch {
|
|
584
|
+
/* fail-open: teardown best-effort */
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
}
|
|
588
|
+
async function executePhaseInner(phase, state, deps, prior, emitProgress, _retryDepth = 0, opts) {
|
|
589
|
+
const type = phase.type ?? "agent";
|
|
590
|
+
const concurrency = phase.concurrency ?? state.def.concurrency ?? 8;
|
|
591
|
+
const previousOutput = lastCompletedOutput(state, phase);
|
|
592
|
+
const run = deps.runTask ?? noRunnerInjected;
|
|
593
|
+
// Effective working directory for THIS phase's execution. When an isolated
|
|
594
|
+
// workspace was allocated (worktree isolation), `_cwdOverride` is its dir and
|
|
595
|
+
// takes precedence; otherwise a literal `phase.cwd` (non-keyword) or the run
|
|
596
|
+
// cwd is used. Keyword cwds are never passed to a runner (they're resolved
|
|
597
|
+
// upstream in the executePhase wrapper).
|
|
598
|
+
const effCwd = resolveEffCwd(deps, phase);
|
|
599
|
+
// Shared Context Tree opt-in (per-phase or flow-wide). When on, the subagent
|
|
600
|
+
// gets ctx_* tools backed by a per-run blackboard directory. nodeId is
|
|
601
|
+
// deterministic per phase so a resume re-uses the same tree node (idempotent
|
|
602
|
+
// upsert in registerNode prevents duplication). Sub-items (map/parallel) get
|
|
603
|
+
// a suffixed nodeId so concurrent siblings write to distinct findings files.
|
|
604
|
+
const sharing = (phase.shareContext ?? state.def.contextSharing) === true;
|
|
605
|
+
let ctxDir;
|
|
606
|
+
if (sharing) {
|
|
607
|
+
try {
|
|
608
|
+
ctxDir = deps._ctxDir ?? initCtxDir(ctxDirFor(runsDir(deps.cwd), state.runId));
|
|
609
|
+
}
|
|
610
|
+
catch {
|
|
611
|
+
ctxDir = undefined; // fail-open: degrade to no sharing
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
const nodeIdFor = (suffix) => `${phase.id}${suffix ? `-${suffix}` : ""}`.replace(/[^A-Za-z0-9._-]+/g, "_");
|
|
615
|
+
// Resolve context pre-read files once, before any type branching.
|
|
616
|
+
// The content is prepended to every task so the subagent never spends
|
|
617
|
+
// turns on file exploration for files the flow author already knows.
|
|
618
|
+
// M3 observed-readSet: collect every upstream ref this phase resolves, so we
|
|
619
|
+
// can record what its result ACTUALLY depended on (not just its declared
|
|
620
|
+
// dependsOn). Shared by every interpolation in this phase (task / when / …).
|
|
621
|
+
const readRefs = [];
|
|
622
|
+
const onRead = (ref) => {
|
|
623
|
+
readRefs.push(ref);
|
|
624
|
+
};
|
|
625
|
+
const ctx = buildInterpolationContext(state, previousOutput, undefined, onRead);
|
|
626
|
+
// M3 observed-readSet: when conditions are part of the phase's real
|
|
627
|
+
// dependencies. Evaluate them inside executePhaseInner so every upstream
|
|
628
|
+
// interpolation is captured by the shared onRead hook, not silently dropped
|
|
629
|
+
// by a separate out-of-band context.
|
|
630
|
+
if (phase.when !== undefined) {
|
|
631
|
+
if (!evaluateCondition(phase.when, ctx)) {
|
|
632
|
+
return {
|
|
633
|
+
id: phase.id,
|
|
634
|
+
status: "skipped",
|
|
635
|
+
error: `Condition not met: ${phase.when}`,
|
|
636
|
+
endedAt: Date.now(),
|
|
637
|
+
usage: emptyUsage(),
|
|
638
|
+
reads: readRefsToReads(readRefs, state),
|
|
639
|
+
};
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
const preRead = await resolvePhaseContext(phase, ctx);
|
|
643
|
+
// Resolve this phase's cache policy once. Default scope is "run-only" (the
|
|
644
|
+
// historical within-run resume behavior). Only "cross-run" phases resolve a
|
|
645
|
+
// fingerprint and consult the persistent store.
|
|
646
|
+
let cacheScope = (phase.cache?.scope ?? deps.cacheScopeDefault ?? "run-only");
|
|
647
|
+
// Defense in depth: gate/approval/loop/tournament must produce a fresh result
|
|
648
|
+
// each run (schema already rejects explicit cross-run, but the default-scope
|
|
649
|
+
// path must also be blocked). If flowDefHash failed, cross-run is unsafe
|
|
650
|
+
// because the key degrades to flowName-only and reopens cross-flow collisions.
|
|
651
|
+
const CROSS_RUN_BLOCKED_TYPES = new Set(["gate", "approval", "loop", "tournament"]);
|
|
652
|
+
if (cacheScope === "cross-run" && CROSS_RUN_BLOCKED_TYPES.has(type)) {
|
|
653
|
+
cacheScope = "run-only";
|
|
654
|
+
}
|
|
655
|
+
if (state.flowDefHash === "failed" && cacheScope === "cross-run") {
|
|
656
|
+
cacheScope = "run-only";
|
|
657
|
+
}
|
|
658
|
+
const cc = {
|
|
659
|
+
scope: cacheScope,
|
|
660
|
+
ttlMs: phase.cache?.ttl ? (parseTtlMs(phase.cache.ttl) ?? undefined) : undefined,
|
|
661
|
+
fingerprint: cacheScope === "cross-run" ? resolveFingerprint(phase.cache?.fingerprint, effCwd) : "",
|
|
662
|
+
store: deps.cacheStore ?? new CacheStore(deps.cwd),
|
|
663
|
+
prior,
|
|
664
|
+
phaseId: phase.id,
|
|
665
|
+
flowName: state.flowName,
|
|
666
|
+
runId: state.runId,
|
|
667
|
+
flowDefHash: state.flowDefHash === "failed" ? undefined : state.flowDefHash,
|
|
668
|
+
phaseFp: state.phaseFingerprints?.[phase.id],
|
|
669
|
+
forceRerun: opts?.forceRerun,
|
|
670
|
+
thinking: phase.thinking,
|
|
671
|
+
tools: phase.tools,
|
|
672
|
+
preRead,
|
|
673
|
+
};
|
|
674
|
+
const baseRun = (agentName, task, onLive, ctxNodeId) => run(effCwd, deps.agents, agentName, task, {
|
|
675
|
+
model: phase.model,
|
|
676
|
+
thinking: phase.thinking,
|
|
677
|
+
tools: phase.tools,
|
|
678
|
+
cwd: effCwd,
|
|
679
|
+
signal: deps.signal,
|
|
680
|
+
onLive,
|
|
681
|
+
ctxDir: ctxDir,
|
|
682
|
+
nodeId: ctxDir ? ctxNodeId : undefined,
|
|
683
|
+
}, deps.globalThinking);
|
|
684
|
+
// Wrap each subagent call in the phase's retry policy. Usage is summed across
|
|
685
|
+
// attempts; the attempt count rides along on the result for the TUI.
|
|
686
|
+
//
|
|
687
|
+
// Even without an explicit `phase.retry`, transient provider errors (rate
|
|
688
|
+
// limits, overload, 5xx, timeouts) are retried with backoff so a momentary
|
|
689
|
+
// 429 is absorbed inside this run instead of bubbling up and provoking the
|
|
690
|
+
// calling agent to re-invoke the whole tool (which stacks duplicate progress
|
|
691
|
+
// blocks in the transcript).
|
|
692
|
+
const retry = phase.retry;
|
|
693
|
+
const DEFAULT_TRANSIENT_RETRIES = 3;
|
|
694
|
+
const DEFAULT_TRANSIENT_BACKOFF_MS = 2000;
|
|
695
|
+
const DEFAULT_TRANSIENT_FACTOR = 2;
|
|
696
|
+
const runOne = async (agentName, task, onLive, ctxNodeId) => {
|
|
697
|
+
const explicitMax = Math.max(1, 1 + Math.max(0, Math.floor(retry?.max ?? 0)));
|
|
698
|
+
// Allow enough attempts to cover whichever policy applies on a given attempt.
|
|
699
|
+
const maxAttempts = Math.max(explicitMax, 1 + DEFAULT_TRANSIENT_RETRIES);
|
|
700
|
+
const usages = [];
|
|
701
|
+
let last;
|
|
702
|
+
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
703
|
+
if (deps.signal?.aborted)
|
|
704
|
+
break;
|
|
705
|
+
last = await baseRun(agentName, task, onLive, ctxNodeId);
|
|
706
|
+
usages.push(last.usage);
|
|
707
|
+
// B6: aggregate and surface cumulative usage before the retry decision,
|
|
708
|
+
// so the TUI / budget guard see the in-flight spend on every attempt.
|
|
709
|
+
const liveRetry = state.phases[phase.id];
|
|
710
|
+
if (liveRetry)
|
|
711
|
+
liveRetry.usage = aggregateUsage(usages);
|
|
712
|
+
if (!isFailed(last))
|
|
713
|
+
break;
|
|
714
|
+
// Stop retrying on abort or once the run is over budget.
|
|
715
|
+
if (deps.signal?.aborted || overBudget(state).over)
|
|
716
|
+
break;
|
|
717
|
+
// Decide whether THIS failure warrants another attempt. Explicit retry
|
|
718
|
+
// policy covers all failures up to its cap; the transient fallback covers
|
|
719
|
+
// only retryable provider errors. A non-transient failure with no explicit
|
|
720
|
+
// policy stops immediately (no point burning attempts on a hard error).
|
|
721
|
+
const withinExplicit = attempt < explicitMax - 1;
|
|
722
|
+
const transient = isTransientError(last);
|
|
723
|
+
const withinTransient = transient && attempt < DEFAULT_TRANSIENT_RETRIES;
|
|
724
|
+
if (!withinExplicit && !withinTransient)
|
|
725
|
+
break;
|
|
726
|
+
// Backoff: prefer the explicit policy's curve when the phase defines one
|
|
727
|
+
// (covers transient retries too, and keeps tests fast with backoffMs:0),
|
|
728
|
+
// otherwise use the transient defaults.
|
|
729
|
+
const baseMs = retry?.backoffMs != null ? retry.backoffMs : DEFAULT_TRANSIENT_BACKOFF_MS;
|
|
730
|
+
// Factor asymmetry is intentional:
|
|
731
|
+
// - Explicit retry: backoffMs * (factor ?? 1) ^ attempt — user's
|
|
732
|
+
// curve, defaults to flat (factor=1 → constant backoff).
|
|
733
|
+
// - Transient fallback: backoffMs * 2 ^ attempt — exponential.
|
|
734
|
+
// This lets users opt into flat retry with retry: {max:3} without
|
|
735
|
+
// specifying factor, while transient errors get proper exponential
|
|
736
|
+
// backoff.
|
|
737
|
+
const factor = retry ? (retry.factor ?? 1) : DEFAULT_TRANSIENT_FACTOR;
|
|
738
|
+
const wait = Math.min(60000, Math.round(baseMs * factor ** attempt));
|
|
739
|
+
if (wait > 0)
|
|
740
|
+
await delay(wait, deps.signal);
|
|
741
|
+
}
|
|
742
|
+
// Aborted before any attempt ran → return a clean aborted result (no crash).
|
|
743
|
+
if (!last) {
|
|
744
|
+
return {
|
|
745
|
+
agent: agentName,
|
|
746
|
+
task,
|
|
747
|
+
exitCode: 1,
|
|
748
|
+
output: "",
|
|
749
|
+
stderr: "Aborted before execution",
|
|
750
|
+
usage: emptyUsage(),
|
|
751
|
+
stopReason: "aborted",
|
|
752
|
+
errorMessage: "Aborted before execution",
|
|
753
|
+
attempts: 0,
|
|
754
|
+
};
|
|
755
|
+
}
|
|
756
|
+
if (usages.length > 1)
|
|
757
|
+
last.usage = aggregateUsage(usages);
|
|
758
|
+
last.attempts = usages.length;
|
|
759
|
+
return last;
|
|
760
|
+
};
|
|
761
|
+
const parseJson = phase.output === "json";
|
|
762
|
+
// Runs a list of sub-tasks with live fan-out progress + aggregate live usage/activity.
|
|
763
|
+
// `perItem` (map only) enables per-item cross-run caching: each item is looked
|
|
764
|
+
// up in the cache before spawning a subagent, and a successful fresh item is
|
|
765
|
+
// recorded so a later run with that item unchanged hits per-item. When
|
|
766
|
+
// `perItem` is undefined (parallel, or non-cacheable maps) the path is inert.
|
|
767
|
+
const runFanout = async (items, perItem) => {
|
|
768
|
+
let done = 0;
|
|
769
|
+
let running = 0;
|
|
770
|
+
let failed = 0;
|
|
771
|
+
const total = items.length;
|
|
772
|
+
const live = state.phases[phase.id];
|
|
773
|
+
const liveUsages = items.map(() => emptyUsage());
|
|
774
|
+
let latestText = "";
|
|
775
|
+
let latestModel;
|
|
776
|
+
const refresh = () => {
|
|
777
|
+
if (live) {
|
|
778
|
+
live.subProgress = { done, total, running, failed };
|
|
779
|
+
live.usage = aggregateUsage(liveUsages);
|
|
780
|
+
live.liveText = latestText;
|
|
781
|
+
live.model = latestModel;
|
|
782
|
+
}
|
|
783
|
+
emitProgress();
|
|
784
|
+
};
|
|
785
|
+
refresh();
|
|
786
|
+
return mapWithConcurrencyLimit(items, concurrency, async (it, idx) => {
|
|
787
|
+
// Budget guard: stop spawning new fan-out items once the run is over budget.
|
|
788
|
+
if (overBudget(state).over) {
|
|
789
|
+
done++;
|
|
790
|
+
refresh();
|
|
791
|
+
return {
|
|
792
|
+
agent: it.agent,
|
|
793
|
+
task: it.task,
|
|
794
|
+
exitCode: 0,
|
|
795
|
+
output: "(skipped: budget exceeded)",
|
|
796
|
+
stderr: "",
|
|
797
|
+
usage: emptyUsage(),
|
|
798
|
+
stopReason: "budget-skipped",
|
|
799
|
+
};
|
|
800
|
+
}
|
|
801
|
+
// Per-item cross-run cache lookup (map only). A hit synthesizes a 0-token
|
|
802
|
+
// RunResult and returns immediately — the item never spawns a subagent and
|
|
803
|
+
// never reaches the ctx_spawn drain below (a cached item can't have queued
|
|
804
|
+
// new spawns). Fail-open: any error in the lookup path degrades to executing.
|
|
805
|
+
if (perItem) {
|
|
806
|
+
try {
|
|
807
|
+
const ckItem = perItem.keyOf(idx);
|
|
808
|
+
if (ckItem) {
|
|
809
|
+
const hit = cachedPhase(perItem.cc, ckItem);
|
|
810
|
+
if (hit) {
|
|
811
|
+
done++;
|
|
812
|
+
const synth = phaseStateToRunResult(hit, it);
|
|
813
|
+
liveUsages[idx] = emptyUsage();
|
|
814
|
+
if (hit.model)
|
|
815
|
+
latestModel = hit.model;
|
|
816
|
+
refresh();
|
|
817
|
+
return synth;
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
catch {
|
|
822
|
+
/* fail-open: a cache read error must never sink the item */
|
|
823
|
+
}
|
|
824
|
+
}
|
|
825
|
+
running++;
|
|
826
|
+
refresh();
|
|
827
|
+
if (ctxDir) {
|
|
828
|
+
try {
|
|
829
|
+
registerNode(ctxDir, nodeIdFor(String(idx)), phase.id, undefined, "running");
|
|
830
|
+
}
|
|
831
|
+
catch { /* fail-open */ }
|
|
832
|
+
}
|
|
833
|
+
const r = await runOne(it.agent, it.task, (l) => {
|
|
834
|
+
liveUsages[idx] = l.usage;
|
|
835
|
+
if (l.text)
|
|
836
|
+
latestText = l.text;
|
|
837
|
+
if (l.model)
|
|
838
|
+
latestModel = l.model;
|
|
839
|
+
refresh();
|
|
840
|
+
}, ctxDir ? nodeIdFor(String(idx)) : undefined);
|
|
841
|
+
running--;
|
|
842
|
+
done++;
|
|
843
|
+
if (isFailed(r))
|
|
844
|
+
failed++;
|
|
845
|
+
liveUsages[idx] = r.usage;
|
|
846
|
+
// Per-item cross-run cache record (map only): persist a successful fresh
|
|
847
|
+
// item so a later run with this item unchanged hits per-item instead of
|
|
848
|
+
// re-running. Failed and budget-skipped items are never cached (a stale
|
|
849
|
+
// failure would be served on the next run). Fail-open: a write error never
|
|
850
|
+
// sinks the item — the fresh `r` is already in hand and flows downstream.
|
|
851
|
+
if (perItem && !isFailed(r) && r.stopReason !== "budget-skipped") {
|
|
852
|
+
try {
|
|
853
|
+
const ckItem = perItem.keyOf(idx);
|
|
854
|
+
if (ckItem) {
|
|
855
|
+
const ccItem = { ...perItem.cc, phaseId: `${phase.id}#item${idx}` };
|
|
856
|
+
const itemPs = resultToPhaseState(`${phase.id}#item${idx}`, r, ckItem.key, parseJson);
|
|
857
|
+
recordCache(ccItem, itemPs);
|
|
858
|
+
}
|
|
859
|
+
}
|
|
860
|
+
catch {
|
|
861
|
+
/* fail-open: cache write must never sink the item */
|
|
862
|
+
}
|
|
863
|
+
}
|
|
864
|
+
if (ctxDir) {
|
|
865
|
+
try {
|
|
866
|
+
const itemNid = nodeIdFor(String(idx));
|
|
867
|
+
setNodeStatus(ctxDir, itemNid, isFailed(r) ? "failed" : "done");
|
|
868
|
+
// A fan-out item may itself ctx_spawn children. Without this drain a
|
|
869
|
+
// map/parallel item's spawn intents are silently orphaned (the
|
|
870
|
+
// post-run drain below only covers single-agent phases).
|
|
871
|
+
const spawned = drainPendingSpawns(ctxDir, itemNid);
|
|
872
|
+
if (spawned.length > 0 && !deps.signal?.aborted && !overBudget(state).over) {
|
|
873
|
+
const child = await runSpawnedChildren(spawned, ctxDir, itemNid, phase, deps, state, run);
|
|
874
|
+
if (child.reports)
|
|
875
|
+
r.output = `${r.output ?? ""}${child.reports}`;
|
|
876
|
+
if (child.usage) {
|
|
877
|
+
r.usage = aggregateUsage([r.usage ?? emptyUsage(), child.usage]);
|
|
878
|
+
liveUsages[idx] = r.usage;
|
|
879
|
+
}
|
|
880
|
+
}
|
|
881
|
+
}
|
|
882
|
+
catch { /* fail-open */ }
|
|
883
|
+
}
|
|
884
|
+
refresh();
|
|
885
|
+
return r;
|
|
886
|
+
});
|
|
887
|
+
};
|
|
888
|
+
// Single-agent phases: agent, gate, and reduce all run one subagent on an
|
|
889
|
+
// interpolated task. gate additionally parses a verdict; reduce simply pulls
|
|
890
|
+
// its inputs from `from` phases (already exposed via interpolation).
|
|
891
|
+
if (type === "agent" || type === "gate" || type === "reduce") {
|
|
892
|
+
// Eval gate: zero-token machine checks before the LLM gate.
|
|
893
|
+
if (type === "gate" && Array.isArray(phase.eval) && phase.eval.length > 0) {
|
|
894
|
+
const evalCtx = buildInterpolationContext(state, previousOutput, undefined, onRead);
|
|
895
|
+
let allPassed = true;
|
|
896
|
+
for (const check of phase.eval) {
|
|
897
|
+
let expr = check;
|
|
898
|
+
// Pre-process `contains` expressions: "{steps.x.output} contains PASS"
|
|
899
|
+
// Convert to: interpolate LHS, check RHS substring inclusion.
|
|
900
|
+
const containsIdx = expr.indexOf(" contains ");
|
|
901
|
+
if (containsIdx > 0) {
|
|
902
|
+
const lhs = expr.slice(0, containsIdx).trim();
|
|
903
|
+
const rhs = expr.slice(containsIdx + " contains ".length).trim();
|
|
904
|
+
const lhsVal = interpolate(lhs, evalCtx);
|
|
905
|
+
const lhsStr = lhsVal.text;
|
|
906
|
+
if (!lhsStr.includes(rhs)) {
|
|
907
|
+
allPassed = false;
|
|
908
|
+
break;
|
|
909
|
+
}
|
|
910
|
+
continue;
|
|
911
|
+
}
|
|
912
|
+
if (!evaluateCondition(expr, evalCtx)) {
|
|
913
|
+
allPassed = false;
|
|
914
|
+
break;
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
if (allPassed) {
|
|
918
|
+
// All evals passed — skip the LLM gate, return an auto-pass.
|
|
919
|
+
const inputHash = cacheKeys(cc, [phase.id, "eval-skip"]).key;
|
|
920
|
+
const ps = {
|
|
921
|
+
id: phase.id,
|
|
922
|
+
status: "done",
|
|
923
|
+
output: "PASS (eval checks passed — no LLM call)",
|
|
924
|
+
gate: { verdict: "pass" },
|
|
925
|
+
usage: emptyUsage(),
|
|
926
|
+
inputHash,
|
|
927
|
+
endedAt: Date.now(),
|
|
928
|
+
};
|
|
929
|
+
if (readRefs.length)
|
|
930
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
931
|
+
recordCache(cc, ps);
|
|
932
|
+
return ps;
|
|
933
|
+
}
|
|
934
|
+
}
|
|
935
|
+
const interp = interpolate(phase.task ?? "", ctx);
|
|
936
|
+
const text = interp.text;
|
|
937
|
+
const refWarning = warnUnresolvedRefs(phase.id, interp.missing);
|
|
938
|
+
const fullTask = preRead + text;
|
|
939
|
+
const agentName = resolveAgent(phase.agent, deps, state);
|
|
940
|
+
const ck = cacheKeys(cc, [phase.id, agentName, phase.model ?? "", fullTask]);
|
|
941
|
+
const inputHash = ck.key;
|
|
942
|
+
const cached = cachedPhase(cc, ck);
|
|
943
|
+
if (cached)
|
|
944
|
+
return cached;
|
|
945
|
+
const r = await runOne(agentName, fullTask, liveSink(state, phase.id, emitProgress), nodeIdFor());
|
|
946
|
+
const ps = resultToPhaseState(phase.id, r, inputHash, parseJson);
|
|
947
|
+
if (readRefs.length)
|
|
948
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
949
|
+
if (refWarning)
|
|
950
|
+
ps.warnings = [...(ps.warnings ?? []), refWarning];
|
|
951
|
+
if (type === "gate" && ps.status === "done")
|
|
952
|
+
ps.gate = parseGateVerdict(r.output);
|
|
953
|
+
// Shared Context Tree: register this node, mark its terminal status, and
|
|
954
|
+
// pick up any ctx_spawn intents the subagent queued. The spawned child
|
|
955
|
+
// tasks run here (supervision loop) and their reports are folded into this
|
|
956
|
+
// phase's output so the parent — and downstream phases — can see them.
|
|
957
|
+
if (ctxDir) {
|
|
958
|
+
try {
|
|
959
|
+
const nid = nodeIdFor();
|
|
960
|
+
registerNode(ctxDir, nid, phase.id, undefined, ps.status === "failed" ? "failed" : "done");
|
|
961
|
+
const spawned = drainPendingSpawns(ctxDir, nid);
|
|
962
|
+
if (spawned.length > 0 && !deps.signal?.aborted && !overBudget(state).over) {
|
|
963
|
+
const child = await runSpawnedChildren(spawned, ctxDir, nid, phase, deps, state, run);
|
|
964
|
+
if (child.reports)
|
|
965
|
+
ps.output = `${ps.output ?? ""}${child.reports}`;
|
|
966
|
+
// Fold spawned spend into this phase's usage so the run-wide budget
|
|
967
|
+
// guard accounts for it (verdict Issue 2).
|
|
968
|
+
ps.usage = aggregateUsage([ps.usage ?? emptyUsage(), child.usage]);
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
catch {
|
|
972
|
+
/* fail-open: context-tree bookkeeping must never sink the phase */
|
|
973
|
+
}
|
|
974
|
+
}
|
|
975
|
+
// onBlock:retry — re-execute upstream + gate until pass or max attempts.
|
|
976
|
+
if (type === "gate" && ps.gate?.verdict === "block") {
|
|
977
|
+
const onBlockV = phase.onBlock ?? "halt";
|
|
978
|
+
const MAX_RETRY_DEPTH = 3;
|
|
979
|
+
let attempt = 0;
|
|
980
|
+
let gatePs = ps;
|
|
981
|
+
while (onBlockV === "retry" && attempt < (phase.retry?.max ?? 1)) {
|
|
982
|
+
// H1: guard against unbounded spend and user abort
|
|
983
|
+
if (deps.signal?.aborted || overBudget(state).over)
|
|
984
|
+
break;
|
|
985
|
+
attempt++;
|
|
986
|
+
// H2: cap nested retry depth to prevent exponential re-execution
|
|
987
|
+
// when a gate's upstream dependency is itself a gate with onBlock:retry
|
|
988
|
+
if (_retryDepth < MAX_RETRY_DEPTH) {
|
|
989
|
+
// Re-executing upstream deps must NOT inherit this gate's isolated
|
|
990
|
+
// workspace — each dep resolves its own cwd. Strip the override.
|
|
991
|
+
// NOTE: we intentionally pass the gate's `prior` (not the dep's own
|
|
992
|
+
// completed state) so the dep does NOT cache-hit and actually
|
|
993
|
+
// RE-RUNS — re-running upstream is the whole point of onBlock:retry.
|
|
994
|
+
const { _cwdOverride: _dropGateWs, ...depsForUpstream } = deps;
|
|
995
|
+
for (const depId of phase.dependsOn ?? []) {
|
|
996
|
+
const d = state.def.phases.find((p) => p.id === depId);
|
|
997
|
+
if (!d)
|
|
998
|
+
continue;
|
|
999
|
+
const dPs = await executePhase(d, state, depsForUpstream, prior, emitProgress, _retryDepth + 1, undefined);
|
|
1000
|
+
state.phases[depId] = dPs;
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
const retryCtx = buildInterpolationContext(state, lastCompletedOutput(state, phase));
|
|
1004
|
+
const retryText = interpolate(phase.task ?? "", retryCtx).text;
|
|
1005
|
+
const retryTask = preRead + retryText;
|
|
1006
|
+
const retryIH = cacheKeys(cc, [phase.id, agentName, phase.model ?? "", retryTask]).key;
|
|
1007
|
+
const retryR = await runOne(agentName, retryTask, liveSink(state, phase.id, emitProgress));
|
|
1008
|
+
gatePs = resultToPhaseState(phase.id, retryR, retryIH, parseJson);
|
|
1009
|
+
if (gatePs.status === "done")
|
|
1010
|
+
gatePs.gate = parseGateVerdict(retryR.output);
|
|
1011
|
+
if (gatePs.gate?.verdict !== "block" || overBudget(state).over)
|
|
1012
|
+
break;
|
|
1013
|
+
}
|
|
1014
|
+
gatePs.attempts = (ps.attempts ?? 0) + attempt;
|
|
1015
|
+
recordCache(cc, gatePs);
|
|
1016
|
+
return gatePs;
|
|
1017
|
+
}
|
|
1018
|
+
recordCache(cc, ps);
|
|
1019
|
+
return ps;
|
|
1020
|
+
}
|
|
1021
|
+
if (type === "parallel") {
|
|
1022
|
+
const branches = (phase.branches ?? []).map((b) => {
|
|
1023
|
+
const r = interpolate(b.task, ctx);
|
|
1024
|
+
return {
|
|
1025
|
+
agent: resolveAgent(b.agent ?? phase.agent, deps, state),
|
|
1026
|
+
task: preRead + r.text,
|
|
1027
|
+
};
|
|
1028
|
+
});
|
|
1029
|
+
const ck = cacheKeys(cc, [phase.id, phase.model ?? "", JSON.stringify(branches)]);
|
|
1030
|
+
const inputHash = ck.key;
|
|
1031
|
+
const cached = cachedPhase(cc, ck);
|
|
1032
|
+
if (cached)
|
|
1033
|
+
return cached;
|
|
1034
|
+
const results = await runFanout(branches);
|
|
1035
|
+
const ps = mergePhaseState(phase.id, results, inputHash, parseJson);
|
|
1036
|
+
if (readRefs.length)
|
|
1037
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1038
|
+
recordCache(cc, ps);
|
|
1039
|
+
return ps;
|
|
1040
|
+
}
|
|
1041
|
+
if (type === "map") {
|
|
1042
|
+
const overResolved = interpolate(phase.over ?? "", ctx).text;
|
|
1043
|
+
// `over` may itself be a placeholder that resolved to a JSON string.
|
|
1044
|
+
let arr = coerceArray(safeParse(overResolved)) ?? coerceArray(directRef(phase.over ?? "", state));
|
|
1045
|
+
// Breadth cap for untrusted dynamic sub-flows: a `def:` frame in the stack
|
|
1046
|
+
// means we are inside a runtime-generated flow. Truncate giant fan-outs to
|
|
1047
|
+
// bound subprocess blast radius (fail-open: keep the first N rather than abort).
|
|
1048
|
+
let mapTruncated = false;
|
|
1049
|
+
if (arr && (deps._stack ?? []).some((s) => s.startsWith("def:")) && arr.length > MAX_DYNAMIC_MAP_ITEMS) {
|
|
1050
|
+
arr = arr.slice(0, MAX_DYNAMIC_MAP_ITEMS);
|
|
1051
|
+
mapTruncated = true;
|
|
1052
|
+
}
|
|
1053
|
+
if (!arr) {
|
|
1054
|
+
return {
|
|
1055
|
+
id: phase.id,
|
|
1056
|
+
status: "failed",
|
|
1057
|
+
error: `map phase '${phase.id}': 'over' (${phase.over}) did not resolve to an array`,
|
|
1058
|
+
inputHash: hashInput(phase.id, "no-array"),
|
|
1059
|
+
endedAt: Date.now(),
|
|
1060
|
+
usage: emptyUsage(),
|
|
1061
|
+
};
|
|
1062
|
+
}
|
|
1063
|
+
const loopVar = phase.as ?? "item";
|
|
1064
|
+
const tasks = arr.map((item) => {
|
|
1065
|
+
const localCtx = buildInterpolationContext(state, previousOutput, { [loopVar]: item }, onRead);
|
|
1066
|
+
return {
|
|
1067
|
+
agent: resolveAgent(phase.agent, deps, state),
|
|
1068
|
+
task: preRead + interpolate(phase.task ?? "", localCtx).text,
|
|
1069
|
+
};
|
|
1070
|
+
});
|
|
1071
|
+
// Per-item caching is sound ONLY when ALL of:
|
|
1072
|
+
// - cross-run scope: run-only has no persistent store, so per-item entries
|
|
1073
|
+
// could never be re-read (no point keying them).
|
|
1074
|
+
// - no Shared Context Tree (`!sharing`): a sharing map item can read sibling
|
|
1075
|
+
// blackboard writes OUTSIDE its declared deps, so the per-item key (which
|
|
1076
|
+
// folds only the item's own task) under-approximates real reads and could
|
|
1077
|
+
// serve a stale result. Fall back to whole-map.
|
|
1078
|
+
// - not inside a runtime-generated sub-flow (`def:` frame in the stack):
|
|
1079
|
+
// such flows are untrusted / possibly non-deterministic, so per-item reuse
|
|
1080
|
+
// is unsafe. Fall back to whole-map (which still applies breadth caps).
|
|
1081
|
+
// `undefined phaseFingerprint` is NOT a blocker for soundness — it is a
|
|
1082
|
+
// DELIBERATE design choice: per-item keys omit BOTH phaseFp and flowDefHash
|
|
1083
|
+
// (via ccPerItem below) so a changing `over` cannot move unchanged items'
|
|
1084
|
+
// keys. See ccPerItem for the full soundness argument.
|
|
1085
|
+
const perItemCacheable = cc.scope === "cross-run" &&
|
|
1086
|
+
!sharing &&
|
|
1087
|
+
!(deps._stack ?? []).some((s) => s.startsWith("def:"));
|
|
1088
|
+
// Per-item cache context: structural fingerprints (phaseFp + flowDefHash)
|
|
1089
|
+
// are OMITTED so a changing `over` cannot move unchanged items' keys. Both
|
|
1090
|
+
// fingerprints hash `over` (the array source); folding either into a
|
|
1091
|
+
// per-item key means editing one item invalidates EVERY per-item key at
|
|
1092
|
+
// once (no partial reuse) — the bug fixed here. A single item's output is
|
|
1093
|
+
// fully specified by `it.task` (template + {item}/{as} value + any
|
|
1094
|
+
// upstream-output refs + args) + `it.agent` + model + thinking/tools/preRead
|
|
1095
|
+
// + the world-state `fingerprint`; `over` only determines WHICH items
|
|
1096
|
+
// exist, not WHAT any item computes. `flowName` is retained for cross-flow
|
|
1097
|
+
// collision prevention. Soundness: docs/internal/cache-migration.md.
|
|
1098
|
+
// NB: perItemCacheable already gates on scope === "cross-run", which is
|
|
1099
|
+
// blocked upstream when flowDefHash === "failed", so ccPerItem is only
|
|
1100
|
+
// built when flowDefHash is a real hash (or already undefined) — setting
|
|
1101
|
+
// it to undefined here is a safe no-op for the failed case.
|
|
1102
|
+
const ccPerItem = { ...cc, phaseFp: undefined, flowDefHash: undefined };
|
|
1103
|
+
// Pre-compute per-item CacheKeys once so the lookup and the record path use
|
|
1104
|
+
// the IDENTICAL key (built from ccPerItem, NOT the whole-phase cc). The
|
|
1105
|
+
// per-item key folds `it.agent` (Arbiter fix): a different agent means
|
|
1106
|
+
// different output, so a per-item key WITHOUT the agent could serve a stale
|
|
1107
|
+
// cross-agent hit when only `phase.agent` changed (the whole-map key would
|
|
1108
|
+
// correctly miss via JSON.stringify(tasks), but per-item keys would not).
|
|
1109
|
+
const perItemKeys = perItemCacheable
|
|
1110
|
+
? tasks.map((it) => cacheKeys(ccPerItem, [phase.id, it.agent, phase.model ?? "", it.task]))
|
|
1111
|
+
: tasks.map(() => null);
|
|
1112
|
+
const perItem = perItemCacheable
|
|
1113
|
+
? { keyOf: (idx) => perItemKeys[idx] ?? null, cc: ccPerItem }
|
|
1114
|
+
: undefined;
|
|
1115
|
+
// Whole-map key keeps the FULL cc (phaseFp + flowDefHash) so its fast path
|
|
1116
|
+
// and any pre-existing whole-map entries are unchanged (backward compat).
|
|
1117
|
+
const ck = cacheKeys(cc, [phase.id, phase.model ?? "", JSON.stringify(tasks)]);
|
|
1118
|
+
const inputHash = ck.key;
|
|
1119
|
+
const cached = cachedPhase(cc, ck);
|
|
1120
|
+
if (cached)
|
|
1121
|
+
return cached;
|
|
1122
|
+
const results = await runFanout(tasks, perItem);
|
|
1123
|
+
const ps = mergePhaseState(phase.id, results, inputHash, parseJson);
|
|
1124
|
+
if (readRefs.length)
|
|
1125
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1126
|
+
if (mapTruncated) {
|
|
1127
|
+
ps.warnings = [...(ps.warnings ?? []), `map fan-out truncated to MAX_DYNAMIC_MAP_ITEMS (${MAX_DYNAMIC_MAP_ITEMS}) inside a dynamic sub-flow`];
|
|
1128
|
+
// NB: do NOT set ps.budgetTruncated — that field drives the run-level
|
|
1129
|
+
// budget-blocked path and would mislabel the run as "budget exceeded".
|
|
1130
|
+
// This is a safety fan-out cap, not a cost overrun; a warning is enough.
|
|
1131
|
+
}
|
|
1132
|
+
recordCache(cc, ps);
|
|
1133
|
+
return ps;
|
|
1134
|
+
}
|
|
1135
|
+
if (type === "approval") {
|
|
1136
|
+
const readRefs = [];
|
|
1137
|
+
const ctx = buildInterpolationContext(state, previousOutput, undefined, (ref) => readRefs.push(ref));
|
|
1138
|
+
const message = interpolate(phase.task ?? "Approve to continue?", ctx).text;
|
|
1139
|
+
const ck = cacheKeys(cc, [phase.id, phase.model ?? "", "approval", message]);
|
|
1140
|
+
const inputHash = ck.key;
|
|
1141
|
+
const cached = cachedPhase(cc, ck);
|
|
1142
|
+
if (cached)
|
|
1143
|
+
return cached;
|
|
1144
|
+
// Non-interactive (headless/CI/detached): auto-REJECT, fail-open, but record it.
|
|
1145
|
+
// Approval gates are safety boundaries — bypassing them silently in CI would
|
|
1146
|
+
// let unreviewed work ship. Detached/CI runs must not bypass approval gates.
|
|
1147
|
+
if (!deps.requestApproval) {
|
|
1148
|
+
return {
|
|
1149
|
+
id: phase.id,
|
|
1150
|
+
status: "done",
|
|
1151
|
+
output: "(auto-rejected: no interactive approver available)",
|
|
1152
|
+
approval: { decision: "reject", auto: true },
|
|
1153
|
+
gate: { verdict: "block", reason: "(auto-rejected: no interactive approver available)" },
|
|
1154
|
+
usage: emptyUsage(),
|
|
1155
|
+
inputHash,
|
|
1156
|
+
reads: readRefsToReads(readRefs, state),
|
|
1157
|
+
endedAt: Date.now(),
|
|
1158
|
+
};
|
|
1159
|
+
}
|
|
1160
|
+
const decision = await deps.requestApproval({ phaseId: phase.id, message, upstream: previousOutput });
|
|
1161
|
+
const note = decision.note?.trim();
|
|
1162
|
+
const ps = {
|
|
1163
|
+
id: phase.id,
|
|
1164
|
+
status: "done",
|
|
1165
|
+
output: note || `(${decision.decision})`,
|
|
1166
|
+
approval: { decision: decision.decision, note },
|
|
1167
|
+
usage: emptyUsage(),
|
|
1168
|
+
inputHash,
|
|
1169
|
+
reads: readRefsToReads(readRefs, state),
|
|
1170
|
+
endedAt: Date.now(),
|
|
1171
|
+
};
|
|
1172
|
+
// A rejection halts the flow via the same mechanism as a blocking gate.
|
|
1173
|
+
if (decision.decision === "reject") {
|
|
1174
|
+
ps.gate = { verdict: "block", reason: note || "Rejected by user" };
|
|
1175
|
+
}
|
|
1176
|
+
return ps;
|
|
1177
|
+
}
|
|
1178
|
+
if (type === "flow") {
|
|
1179
|
+
const readRefs = [];
|
|
1180
|
+
const ctx = buildInterpolationContext(state, previousOutput, undefined, (ref) => readRefs.push(ref));
|
|
1181
|
+
const hasDef = phase.def !== undefined;
|
|
1182
|
+
const stack = deps._stack ?? [];
|
|
1183
|
+
let subDef;
|
|
1184
|
+
let name;
|
|
1185
|
+
let recursionKey; // identity used for cache key + recursion guard
|
|
1186
|
+
if (hasDef) {
|
|
1187
|
+
// --- Inline `def`: resolve at runtime, validate, fail-OPEN on any error. ---
|
|
1188
|
+
// Fail-open contract: a bad def NEVER aborts the run. The phase resolves
|
|
1189
|
+
// as `done` with empty output and a `defError` diagnostic, and the
|
|
1190
|
+
// upstream output is preserved for downstream phases. (Authors who want
|
|
1191
|
+
// a bad plan to be a hard failure can add their own gate downstream.)
|
|
1192
|
+
const defFailOpen = (diag) => ({
|
|
1193
|
+
id: phase.id,
|
|
1194
|
+
status: "done",
|
|
1195
|
+
output: "",
|
|
1196
|
+
json: parseJson ? safeParse("") : undefined,
|
|
1197
|
+
usage: emptyUsage(),
|
|
1198
|
+
inputHash: hashInput(phase.id, `flow-def-error:${diag}`),
|
|
1199
|
+
reads: readRefsToReads(readRefs, state),
|
|
1200
|
+
endedAt: Date.now(),
|
|
1201
|
+
defError: diag,
|
|
1202
|
+
});
|
|
1203
|
+
// Nesting guard: each `flow{def}` adds a frame to _stack; cap inline depth.
|
|
1204
|
+
const inlineDepth = stack.filter((s) => s.startsWith("def:")).length;
|
|
1205
|
+
if (inlineDepth >= MAX_DYNAMIC_NESTING) {
|
|
1206
|
+
return defFailOpen(`inline sub-flow nesting exceeded MAX_DYNAMIC_NESTING (${MAX_DYNAMIC_NESTING}): depth ${inlineDepth}`);
|
|
1207
|
+
}
|
|
1208
|
+
const rawDef = phase.def;
|
|
1209
|
+
// String defs are interpolated then JSON-parsed; objects are used directly.
|
|
1210
|
+
let parsed;
|
|
1211
|
+
if (typeof rawDef === "string") {
|
|
1212
|
+
const resolved = interpolate(rawDef, ctx).text;
|
|
1213
|
+
parsed = safeParse(resolved);
|
|
1214
|
+
if (parsed === undefined) {
|
|
1215
|
+
return defFailOpen("inline def string did not parse as JSON");
|
|
1216
|
+
}
|
|
1217
|
+
}
|
|
1218
|
+
else {
|
|
1219
|
+
parsed = rawDef;
|
|
1220
|
+
}
|
|
1221
|
+
// Accept a full Taskflow, a bare phases array, or {phases:[...]}; wrap the latter two.
|
|
1222
|
+
const wrapped = normalizeInlineDef(parsed, phase.id);
|
|
1223
|
+
if (!wrapped) {
|
|
1224
|
+
return defFailOpen("inline def is not a Taskflow, phases array, or {phases:[...]}");
|
|
1225
|
+
}
|
|
1226
|
+
// Empty plan is a valid no-op (a planner deciding there is nothing to do):
|
|
1227
|
+
// succeed with empty output instead of failing validation on zero phases.
|
|
1228
|
+
if (wrapped.phases.length === 0) {
|
|
1229
|
+
return {
|
|
1230
|
+
id: phase.id,
|
|
1231
|
+
status: "done",
|
|
1232
|
+
output: "",
|
|
1233
|
+
json: parseJson ? safeParse("") : undefined,
|
|
1234
|
+
usage: emptyUsage(),
|
|
1235
|
+
inputHash: hashInput(phase.id, "flow-def-empty"),
|
|
1236
|
+
reads: readRefsToReads(readRefs, state),
|
|
1237
|
+
endedAt: Date.now(),
|
|
1238
|
+
};
|
|
1239
|
+
}
|
|
1240
|
+
// Validate with `dynamic` hardening (breadth caps + cwd containment) since
|
|
1241
|
+
// this content is LLM-authored / untrusted. cwd anchors containment checks.
|
|
1242
|
+
const dynCwd = effCwd;
|
|
1243
|
+
const v = validateTaskflow(wrapped, { dynamic: true, cwd: dynCwd });
|
|
1244
|
+
if (!v.ok) {
|
|
1245
|
+
return defFailOpen(`inline def failed validation: ${v.errors.join("; ")}`);
|
|
1246
|
+
}
|
|
1247
|
+
// Static verification (dead-ends, unreachable, gate-exhaustion, budget,
|
|
1248
|
+
// concurrency). Only error-severity issues block; warnings are advisory.
|
|
1249
|
+
const ver = verifyTaskflow({ name: wrapped.name, phases: wrapped.phases, budget: wrapped.budget, concurrency: wrapped.concurrency });
|
|
1250
|
+
if (!ver.ok) {
|
|
1251
|
+
const errs = ver.issues.filter((i) => i.severity === "error").map((i) => i.message);
|
|
1252
|
+
return defFailOpen(`inline def failed verification: ${errs.join("; ")}`);
|
|
1253
|
+
}
|
|
1254
|
+
// Budget containment: a generated def may not raise the parent's cap. Clamp
|
|
1255
|
+
// each dimension to min(child, parent) so it can only ever be tighter.
|
|
1256
|
+
subDef = clampSubFlowBudget(wrapped, state.def.budget);
|
|
1257
|
+
name = subDef.name;
|
|
1258
|
+
recursionKey = `def:${name}`;
|
|
1259
|
+
}
|
|
1260
|
+
else {
|
|
1261
|
+
// --- Saved flow via `use` (unchanged behavior). ---
|
|
1262
|
+
const useName = phase.use;
|
|
1263
|
+
if (!useName)
|
|
1264
|
+
return failPhase(phase.id, `flow phase '${phase.id}' requires 'use' or 'def'`);
|
|
1265
|
+
if (!deps.loadFlow)
|
|
1266
|
+
return failPhase(phase.id, `flow phase '${phase.id}': no sub-flow loader available`);
|
|
1267
|
+
subDef = deps.loadFlow(useName);
|
|
1268
|
+
if (!subDef)
|
|
1269
|
+
return failPhase(phase.id, `flow phase '${phase.id}': saved flow not found: '${useName}'`);
|
|
1270
|
+
name = useName;
|
|
1271
|
+
recursionKey = useName;
|
|
1272
|
+
}
|
|
1273
|
+
if (recursionKey === state.flowName || stack.includes(recursionKey)) {
|
|
1274
|
+
return failPhase(phase.id, `flow phase '${phase.id}': recursive sub-flow ${[...stack, state.flowName, recursionKey].join(" -> ")}`);
|
|
1275
|
+
}
|
|
1276
|
+
// Resolve sub-flow args (interpolate string values), then apply declared defaults.
|
|
1277
|
+
const provided = {};
|
|
1278
|
+
for (const [k, v] of Object.entries(phase.with ?? {})) {
|
|
1279
|
+
provided[k] = typeof v === "string" ? interpolate(v, ctx).text : v;
|
|
1280
|
+
}
|
|
1281
|
+
const subArgs = resolveArgs(subDef, provided);
|
|
1282
|
+
// For inline defs the cache identity must include the resolved def content so
|
|
1283
|
+
// that a different generated plan yields a different key (and an identical plan
|
|
1284
|
+
// hits cache). For saved flows the name is the identity (historical behavior).
|
|
1285
|
+
const flowIdentity = hasDef ? `def:${JSON.stringify(subDef)}` : `flow:${name}`;
|
|
1286
|
+
const ck = cacheKeys(cc, [phase.id, flowIdentity, preRead, JSON.stringify(subArgs)]);
|
|
1287
|
+
const inputHash = ck.key;
|
|
1288
|
+
const cached = cachedPhase(cc, ck);
|
|
1289
|
+
if (cached)
|
|
1290
|
+
return cached;
|
|
1291
|
+
const live = state.phases[phase.id];
|
|
1292
|
+
// Sub-flows enforce their own budget; if they declare none, inherit the
|
|
1293
|
+
// parent cap as a soft per-flow ceiling (best-effort — spend does not cross
|
|
1294
|
+
// flow boundaries, so the parent's already-spent total is not subtracted).
|
|
1295
|
+
const subDefEffective = subDef.budget || !state.def.budget ? subDef : { ...subDef, budget: state.def.budget };
|
|
1296
|
+
const subState = {
|
|
1297
|
+
runId: newRunId(subDef.name),
|
|
1298
|
+
flowName: subDef.name,
|
|
1299
|
+
def: subDefEffective,
|
|
1300
|
+
args: subArgs,
|
|
1301
|
+
status: "running",
|
|
1302
|
+
phases: {},
|
|
1303
|
+
createdAt: Date.now(),
|
|
1304
|
+
updatedAt: Date.now(),
|
|
1305
|
+
cwd: effCwd,
|
|
1306
|
+
};
|
|
1307
|
+
// B8: pass this flow phase's preRead content to every sub-flow phase by
|
|
1308
|
+
// wrapping runTask — sub-phase preRead still gets prepended on top of it.
|
|
1309
|
+
const baseRunTask = deps.runTask ?? noRunnerInjected;
|
|
1310
|
+
const subRunTask = (cwd, agents, agentName, subTask, opts, globalThinking) => baseRunTask(cwd, agents, agentName, preRead + subTask, opts, globalThinking);
|
|
1311
|
+
const subResult = await executeTaskflow(subState, {
|
|
1312
|
+
...deps,
|
|
1313
|
+
// Override deps.cwd with the flow phase's own cwd so that sub-flow
|
|
1314
|
+
// phases without an explicit cwd derive their subagents from the
|
|
1315
|
+
// flow's cwd (not the caller's cwd).
|
|
1316
|
+
cwd: effCwd,
|
|
1317
|
+
// The workspace override applies only to THIS flow phase, not to the
|
|
1318
|
+
// nested sub-phases (each resolves its own cwd). Clear it so the child
|
|
1319
|
+
// phases don't all inherit this phase's isolated dir as an override.
|
|
1320
|
+
_cwdOverride: undefined,
|
|
1321
|
+
runTask: subRunTask,
|
|
1322
|
+
_stack: hasDef ? [...stack, state.flowName, recursionKey] : [...stack, state.flowName],
|
|
1323
|
+
_ctxDir: ctxDir ?? deps._ctxDir,
|
|
1324
|
+
persist: undefined,
|
|
1325
|
+
onProgress: () => {
|
|
1326
|
+
if (live) {
|
|
1327
|
+
const ph = Object.values(subState.phases);
|
|
1328
|
+
// B-F015: `done` must include both success and failure so the
|
|
1329
|
+
// renderer's `done - failed` shows the true success count.
|
|
1330
|
+
live.subProgress = {
|
|
1331
|
+
done: ph.filter((p) => p.status === "done" || p.status === "failed").length,
|
|
1332
|
+
total: subDef.phases.length,
|
|
1333
|
+
running: ph.filter((p) => p.status === "running").length,
|
|
1334
|
+
failed: ph.filter((p) => p.status === "failed").length,
|
|
1335
|
+
};
|
|
1336
|
+
const cur = ph.find((p) => p.status === "running");
|
|
1337
|
+
if (cur)
|
|
1338
|
+
live.liveText = `↳ ${cur.id}${cur.liveText ? `: ${cur.liveText}` : ""}`;
|
|
1339
|
+
live.usage = aggregateUsage(ph.map((p) => p.usage ?? emptyUsage()));
|
|
1340
|
+
}
|
|
1341
|
+
emitProgress();
|
|
1342
|
+
},
|
|
1343
|
+
});
|
|
1344
|
+
const sp = Object.values(subState.phases);
|
|
1345
|
+
const flowPs = {
|
|
1346
|
+
id: phase.id,
|
|
1347
|
+
status: subResult.ok ? "done" : "failed",
|
|
1348
|
+
output: subResult.finalOutput,
|
|
1349
|
+
json: parseJson ? safeParse(subResult.finalOutput) : undefined,
|
|
1350
|
+
usage: subResult.totalUsage,
|
|
1351
|
+
// B-F015: include failed in `done` so the renderer's
|
|
1352
|
+
// `done - failed` formula gives the success count (matches the
|
|
1353
|
+
// map/parallel runner's overlapping-counter convention).
|
|
1354
|
+
subProgress: {
|
|
1355
|
+
done: sp.filter((p) => p.status === "done" || p.status === "failed").length,
|
|
1356
|
+
total: subDef.phases.length,
|
|
1357
|
+
running: 0,
|
|
1358
|
+
failed: sp.filter((p) => p.status === "failed").length,
|
|
1359
|
+
},
|
|
1360
|
+
error: subResult.ok ? undefined : `sub-flow '${name}' ${subResult.state.status}`,
|
|
1361
|
+
inputHash,
|
|
1362
|
+
reads: readRefsToReads(readRefs, state),
|
|
1363
|
+
endedAt: Date.now(),
|
|
1364
|
+
};
|
|
1365
|
+
recordCache(cc, flowPs);
|
|
1366
|
+
return flowPs;
|
|
1367
|
+
}
|
|
1368
|
+
// loop-until-done: run the body repeatedly until `until` is truthy, the output
|
|
1369
|
+
// converges to a fixed point, or maxIterations is hit (always terminates).
|
|
1370
|
+
if (type === "loop") {
|
|
1371
|
+
const readRefs = [];
|
|
1372
|
+
const agentName = resolveAgent(phase.agent, deps, state);
|
|
1373
|
+
const rawMax = phase.maxIterations ?? LOOP_DEFAULT_MAX_ITERATIONS;
|
|
1374
|
+
const maxIters = Math.max(1, Math.min(LOOP_HARD_MAX_ITERATIONS, Math.floor(rawMax)));
|
|
1375
|
+
const convergence = phase.convergence ?? true;
|
|
1376
|
+
// Canonical first-iteration body for the cache key. It must fold in the
|
|
1377
|
+
// interpolated task/upstream refs so that a changed upstream changes the
|
|
1378
|
+
// key and recompute no longer silently reuses a stale loop (critic finding).
|
|
1379
|
+
const firstBodyCtx = buildInterpolationContext(state, previousOutput, {
|
|
1380
|
+
loop: { iteration: 1, lastOutput: "", maxIterations: maxIters },
|
|
1381
|
+
}, (ref) => readRefs.push(ref));
|
|
1382
|
+
const firstBody = preRead + interpolate(phase.task ?? "", firstBodyCtx).text;
|
|
1383
|
+
const inputHash = hashInput(phase.id, "loop", phase.until ?? "", firstBody, String(maxIters));
|
|
1384
|
+
const usages = [];
|
|
1385
|
+
const loopWarnings = [];
|
|
1386
|
+
let lastOutput = "";
|
|
1387
|
+
let prevOutput;
|
|
1388
|
+
let iterations = 0;
|
|
1389
|
+
let stop = "maxIterations";
|
|
1390
|
+
let failedResult;
|
|
1391
|
+
for (let i = 1; i <= maxIters; i++) {
|
|
1392
|
+
if (deps.signal?.aborted) {
|
|
1393
|
+
stop = "aborted";
|
|
1394
|
+
break;
|
|
1395
|
+
}
|
|
1396
|
+
iterations = i;
|
|
1397
|
+
// The body sees its iteration number and the prior iteration's output.
|
|
1398
|
+
const bodyCtx = buildInterpolationContext(state, previousOutput, {
|
|
1399
|
+
loop: { iteration: i, lastOutput, maxIterations: maxIters },
|
|
1400
|
+
}, (ref) => readRefs.push(ref));
|
|
1401
|
+
const body = preRead + interpolate(phase.task ?? "", bodyCtx).text;
|
|
1402
|
+
const r = await runOne(agentName, body, liveSink(state, phase.id, emitProgress));
|
|
1403
|
+
usages.push(r.usage);
|
|
1404
|
+
if (isFailed(r)) {
|
|
1405
|
+
failedResult = r;
|
|
1406
|
+
stop = "failed";
|
|
1407
|
+
break;
|
|
1408
|
+
}
|
|
1409
|
+
prevOutput = lastOutput;
|
|
1410
|
+
lastOutput = r.output;
|
|
1411
|
+
// Expose this iteration's output as {steps.<thisId>.output|json} so the
|
|
1412
|
+
// `until` condition can inspect it (e.g. "{steps.refine.json.done}==true").
|
|
1413
|
+
// Loop locals ({loop.iteration} etc.) are available to the condition too.
|
|
1414
|
+
const untilCtx = buildInterpolationContext(state, previousOutput, {
|
|
1415
|
+
loop: { iteration: i, lastOutput, maxIterations: maxIters },
|
|
1416
|
+
}, (ref) => readRefs.push(ref));
|
|
1417
|
+
untilCtx.steps[phase.id] = { output: lastOutput, json: safeParse(lastOutput) };
|
|
1418
|
+
const { value: done, error: condErr } = tryEvaluateCondition(phase.until ?? "", untilCtx);
|
|
1419
|
+
// A malformed condition must not spin forever: stop and surface a warning
|
|
1420
|
+
// so the author learns the `until` never actually evaluated.
|
|
1421
|
+
if (condErr) {
|
|
1422
|
+
loopWarnings.push(`loop 'until' could not be evaluated (stopped early): ${condErr}`);
|
|
1423
|
+
stop = "until";
|
|
1424
|
+
break;
|
|
1425
|
+
}
|
|
1426
|
+
if (done) {
|
|
1427
|
+
stop = "until";
|
|
1428
|
+
break;
|
|
1429
|
+
}
|
|
1430
|
+
// Fixed-point convergence: identical consecutive output ⇒ further work is wasted.
|
|
1431
|
+
if (convergence && prevOutput !== undefined && prevOutput === lastOutput) {
|
|
1432
|
+
stop = "converged";
|
|
1433
|
+
break;
|
|
1434
|
+
}
|
|
1435
|
+
}
|
|
1436
|
+
const aggUsage = usages.length ? aggregateUsage(usages) : emptyUsage();
|
|
1437
|
+
if (failedResult || stop === "failed" || stop === "aborted") {
|
|
1438
|
+
return {
|
|
1439
|
+
id: phase.id,
|
|
1440
|
+
status: "failed",
|
|
1441
|
+
output: lastOutput || undefined,
|
|
1442
|
+
usage: aggUsage,
|
|
1443
|
+
error: failedResult?.errorMessage || failedResult?.stderr || (stop === "aborted" ? "Aborted" : `loop '${phase.id}' iteration ${iterations} failed`),
|
|
1444
|
+
loop: { iterations, stop },
|
|
1445
|
+
warnings: loopWarnings.length ? loopWarnings : undefined,
|
|
1446
|
+
inputHash,
|
|
1447
|
+
reads: readRefsToReads(readRefs, state),
|
|
1448
|
+
endedAt: Date.now(),
|
|
1449
|
+
};
|
|
1450
|
+
}
|
|
1451
|
+
return {
|
|
1452
|
+
id: phase.id,
|
|
1453
|
+
status: "done",
|
|
1454
|
+
output: lastOutput,
|
|
1455
|
+
json: parseJson ? safeParse(lastOutput) : undefined,
|
|
1456
|
+
usage: aggUsage,
|
|
1457
|
+
loop: { iterations, stop },
|
|
1458
|
+
warnings: loopWarnings.length ? loopWarnings : undefined,
|
|
1459
|
+
inputHash,
|
|
1460
|
+
reads: readRefsToReads(readRefs, state),
|
|
1461
|
+
endedAt: Date.now(),
|
|
1462
|
+
};
|
|
1463
|
+
}
|
|
1464
|
+
// tournament: spawn N competing variants, then a judge picks the best (or
|
|
1465
|
+
// synthesizes an aggregate). Combines the parallel fan-out with a gate-style
|
|
1466
|
+
// verdict, expressed as a single declarative phase.
|
|
1467
|
+
if (type === "tournament") {
|
|
1468
|
+
const mode = (phase.mode ?? "best");
|
|
1469
|
+
// Competitors: explicit `branches` win; otherwise N copies of `task`.
|
|
1470
|
+
let competitors;
|
|
1471
|
+
if (phase.branches && phase.branches.length > 0) {
|
|
1472
|
+
competitors = phase.branches.map((b) => ({
|
|
1473
|
+
agent: resolveAgent(b.agent ?? phase.agent, deps, state),
|
|
1474
|
+
task: preRead + interpolate(b.task, ctx).text,
|
|
1475
|
+
}));
|
|
1476
|
+
}
|
|
1477
|
+
else {
|
|
1478
|
+
const n = Math.max(2, Math.min(TOURNAMENT_HARD_MAX_VARIANTS, Math.floor(phase.variants ?? TOURNAMENT_DEFAULT_VARIANTS)));
|
|
1479
|
+
const body = preRead + interpolate(phase.task ?? "", ctx).text;
|
|
1480
|
+
competitors = Array.from({ length: n }, () => ({ agent: resolveAgent(phase.agent, deps, state), task: body }));
|
|
1481
|
+
}
|
|
1482
|
+
// The inputHash must fold in the resolved competitors (which embed the
|
|
1483
|
+
// interpolated task/upstream refs) and the judge rubric, otherwise a changed
|
|
1484
|
+
// upstream produces the same key and recompute silently reuses a stale
|
|
1485
|
+
// tournament (critic finding: unsound for cross-run/recompute).
|
|
1486
|
+
const rubric = interpolate(phase.judge ?? "", ctx).text.trim();
|
|
1487
|
+
const inputHash = hashInput(phase.id, "tournament", mode, String(competitors.length), JSON.stringify(competitors.map((c) => ({ agent: c.agent, task: c.task }))), rubric);
|
|
1488
|
+
const results = await runFanout(competitors);
|
|
1489
|
+
const ran = results.filter((r) => r.stopReason !== "budget-skipped");
|
|
1490
|
+
const ok = ran.filter((r) => !isFailed(r));
|
|
1491
|
+
const variantUsage = aggregateUsage(results.map((r) => r.usage));
|
|
1492
|
+
// Winner numbers are 1-based over `ran` (exactly what the judge is shown).
|
|
1493
|
+
// Using indexOf on the stable `ran` array is reference-based and correct even
|
|
1494
|
+
// when two variants produce byte-identical output.
|
|
1495
|
+
const ranIdx = (r) => ran.indexOf(r) + 1;
|
|
1496
|
+
const budgetSkipCount = results.filter((r) => r.stopReason === "budget-skipped").length;
|
|
1497
|
+
// All competitors failed → the tournament fails (nothing to judge).
|
|
1498
|
+
if (ok.length === 0) {
|
|
1499
|
+
return {
|
|
1500
|
+
id: phase.id,
|
|
1501
|
+
status: "failed",
|
|
1502
|
+
usage: variantUsage,
|
|
1503
|
+
error: `tournament '${phase.id}': all ${competitors.length} variants failed`,
|
|
1504
|
+
budgetTruncated: budgetSkipCount > 0 || undefined,
|
|
1505
|
+
tournament: { variants: competitors.length, winner: 0, mode },
|
|
1506
|
+
inputHash,
|
|
1507
|
+
reads: readRefsToReads(readRefs, state),
|
|
1508
|
+
endedAt: Date.now(),
|
|
1509
|
+
};
|
|
1510
|
+
}
|
|
1511
|
+
// Only one competitor survived → no contest; it wins by default (skip judge).
|
|
1512
|
+
if (ok.length === 1) {
|
|
1513
|
+
return {
|
|
1514
|
+
id: phase.id,
|
|
1515
|
+
status: "done",
|
|
1516
|
+
output: ok[0].output,
|
|
1517
|
+
json: parseJson ? safeParse(ok[0].output) : undefined,
|
|
1518
|
+
usage: variantUsage,
|
|
1519
|
+
model: ok[0].model,
|
|
1520
|
+
budgetTruncated: budgetSkipCount > 0 || undefined,
|
|
1521
|
+
tournament: { variants: competitors.length, winner: ranIdx(ok[0]), mode, reason: "only surviving variant" },
|
|
1522
|
+
inputHash,
|
|
1523
|
+
reads: readRefsToReads(readRefs, state),
|
|
1524
|
+
endedAt: Date.now(),
|
|
1525
|
+
};
|
|
1526
|
+
}
|
|
1527
|
+
// Guard: skip the judge if the run is over budget or aborted.
|
|
1528
|
+
if (deps.signal?.aborted || overBudget(state).over) {
|
|
1529
|
+
return {
|
|
1530
|
+
id: phase.id,
|
|
1531
|
+
status: "done",
|
|
1532
|
+
output: ok[0].output,
|
|
1533
|
+
json: parseJson ? safeParse(ok[0].output) : undefined,
|
|
1534
|
+
usage: variantUsage,
|
|
1535
|
+
model: ok[0].model,
|
|
1536
|
+
budgetTruncated: budgetSkipCount > 0 || undefined,
|
|
1537
|
+
warnings: ["judge skipped: run aborted or budget exceeded"],
|
|
1538
|
+
tournament: { variants: competitors.length, winner: ranIdx(ok[0]), mode, reason: "judge skipped" },
|
|
1539
|
+
inputHash,
|
|
1540
|
+
reads: readRefsToReads(readRefs, state),
|
|
1541
|
+
endedAt: Date.now(),
|
|
1542
|
+
};
|
|
1543
|
+
}
|
|
1544
|
+
// Build the judge prompt: label every variant output, then the rubric.
|
|
1545
|
+
const labelled = ran
|
|
1546
|
+
.map((r, i) => `### Variant ${i + 1}${isFailed(r) ? " (failed — ineligible)" : ""}\n\n${r.output}`)
|
|
1547
|
+
.join("\n\n---\n\n");
|
|
1548
|
+
const finalRubric = rubric ||
|
|
1549
|
+
"You are judging competing answers to the same task. Pick the single best variant on correctness, completeness, and clarity.";
|
|
1550
|
+
const directive = mode === "best"
|
|
1551
|
+
? `End your reply with a line exactly: WINNER: <number> (1–${ran.length}), choosing the strongest eligible variant.`
|
|
1552
|
+
: `Synthesize the strongest possible answer by combining the best parts of the eligible variants. Then end with a line: WINNER: <number> indicating which variant contributed most.`;
|
|
1553
|
+
const judgeTask = `${finalRubric}\n\nThe candidate variants:\n\n${labelled}\n\n${directive}`;
|
|
1554
|
+
const judgeAgent = resolveAgent(phase.judgeAgent ?? phase.agent, deps, state);
|
|
1555
|
+
const judgeRes = await runOne(judgeAgent, judgeTask, liveSink(state, phase.id, emitProgress));
|
|
1556
|
+
const judgeUsage = aggregateUsage([variantUsage, judgeRes.usage]);
|
|
1557
|
+
if (isFailed(judgeRes)) {
|
|
1558
|
+
// Judge failed: fall back to the first eligible variant (fail-open, never
|
|
1559
|
+
// lose the work). Report the variant we actually used, not a hardcoded 1.
|
|
1560
|
+
return {
|
|
1561
|
+
id: phase.id,
|
|
1562
|
+
status: "done",
|
|
1563
|
+
output: ok[0].output,
|
|
1564
|
+
json: parseJson ? safeParse(ok[0].output) : undefined,
|
|
1565
|
+
usage: judgeUsage,
|
|
1566
|
+
model: ok[0].model,
|
|
1567
|
+
budgetTruncated: budgetSkipCount > 0 || undefined,
|
|
1568
|
+
warnings: [`judge failed (${judgeRes.errorMessage ?? "error"}); used variant ${ranIdx(ok[0])}`],
|
|
1569
|
+
tournament: { variants: competitors.length, winner: ranIdx(ok[0]), mode, reason: "judge failed" },
|
|
1570
|
+
inputHash,
|
|
1571
|
+
reads: readRefsToReads(readRefs, state),
|
|
1572
|
+
endedAt: Date.now(),
|
|
1573
|
+
};
|
|
1574
|
+
}
|
|
1575
|
+
const { winner, reason } = parseTournamentWinner(judgeRes.output, ran.length);
|
|
1576
|
+
const winnerResult = ran[winner - 1];
|
|
1577
|
+
const winnerIneligible = !winnerResult || isFailed(winnerResult);
|
|
1578
|
+
// In 'best' mode the output is the winning variant verbatim; in 'aggregate'
|
|
1579
|
+
// mode it is the judge's synthesized answer.
|
|
1580
|
+
const chosen = winnerIneligible ? ok[0] : winnerResult;
|
|
1581
|
+
const winnerIdx = ranIdx(chosen);
|
|
1582
|
+
const output = mode === "aggregate" ? judgeRes.output : chosen.output;
|
|
1583
|
+
return {
|
|
1584
|
+
id: phase.id,
|
|
1585
|
+
status: "done",
|
|
1586
|
+
output,
|
|
1587
|
+
json: parseJson ? safeParse(output) : undefined,
|
|
1588
|
+
usage: judgeUsage,
|
|
1589
|
+
model: mode === "aggregate" ? judgeRes.model : chosen.model,
|
|
1590
|
+
budgetTruncated: budgetSkipCount > 0 || undefined,
|
|
1591
|
+
warnings: winnerIneligible ? [`judge picked an ineligible variant; used variant ${winnerIdx}`] : undefined,
|
|
1592
|
+
tournament: { variants: competitors.length, winner: winnerIdx, mode, reason },
|
|
1593
|
+
inputHash,
|
|
1594
|
+
reads: readRefsToReads(readRefs, state),
|
|
1595
|
+
endedAt: Date.now(),
|
|
1596
|
+
};
|
|
1597
|
+
}
|
|
1598
|
+
return {
|
|
1599
|
+
id: phase.id,
|
|
1600
|
+
status: "failed",
|
|
1601
|
+
error: `Unknown phase type: ${type}`,
|
|
1602
|
+
endedAt: Date.now(),
|
|
1603
|
+
usage: emptyUsage(),
|
|
1604
|
+
};
|
|
1605
|
+
}
|
|
1606
|
+
/** Resolve a `{steps.x.json}`-style ref directly to its parsed value (bypassing stringify). */
|
|
1607
|
+
function directRef(over, state) {
|
|
1608
|
+
const m = over.match(/^\{steps\.([a-zA-Z0-9_-]+)\.(output|json)(?:\.([a-zA-Z0-9_-]+(?:\.[a-zA-Z0-9_-]+)*))?\}$/);
|
|
1609
|
+
if (!m)
|
|
1610
|
+
return undefined;
|
|
1611
|
+
const step = state.phases[m[1]];
|
|
1612
|
+
if (!step || step.status !== "done")
|
|
1613
|
+
return undefined;
|
|
1614
|
+
let value;
|
|
1615
|
+
if (m[2] === "json")
|
|
1616
|
+
value = step.json ?? safeParse(step.output ?? "");
|
|
1617
|
+
else
|
|
1618
|
+
value = safeParse(step.output ?? "");
|
|
1619
|
+
if (m[3]) {
|
|
1620
|
+
for (const key of m[3].split(".")) {
|
|
1621
|
+
if (value == null || typeof value !== "object")
|
|
1622
|
+
return undefined;
|
|
1623
|
+
value = value[key];
|
|
1624
|
+
}
|
|
1625
|
+
}
|
|
1626
|
+
return value;
|
|
1627
|
+
}
|
|
1628
|
+
function lastCompletedOutput(state, phase) {
|
|
1629
|
+
const deps = dependenciesOf(phase);
|
|
1630
|
+
for (let i = deps.length - 1; i >= 0; i--) {
|
|
1631
|
+
const ps = state.phases[deps[i]];
|
|
1632
|
+
if (ps?.status === "done")
|
|
1633
|
+
return ps.output;
|
|
1634
|
+
}
|
|
1635
|
+
return undefined;
|
|
1636
|
+
}
|
|
1637
|
+
/** Fold the phase fingerprint into the base hash parts to form the cache keys.
|
|
1638
|
+
*
|
|
1639
|
+
* Four keys are produced for backward compatibility (see
|
|
1640
|
+
* docs/internal/cache-migration.md):
|
|
1641
|
+
* - `key` : `v3:phasefp:<subfp>` — the current write key (per-phase
|
|
1642
|
+
* structural sub-fingerprint; falls back to the whole-flow hash when
|
|
1643
|
+
* `cc.phaseFp` is absent).
|
|
1644
|
+
* - `v2Key` : `v2:flowdef:<flowDefHash>` — pre-M6 whole-flow key.
|
|
1645
|
+
* - `bareKey` : bare `flowdef:<flowDefHash>` (unversioned) — pre-H1 entries.
|
|
1646
|
+
* - `legacyKey`: the flowdef line omitted — pre-flowDefHash entries.
|
|
1647
|
+
* `cachedPhase` consults all four READ-ONLY on a miss; `recordCache` writes
|
|
1648
|
+
* only `key`. This means an upgrade never produces a miss-storm: existing
|
|
1649
|
+
* entries (whichever shape) still hit, and new writes converge on `key`. */
|
|
1650
|
+
export function cacheKeys(cc, baseParts) {
|
|
1651
|
+
// Fold the full cache identity into the hash: flow name (prevents collisions
|
|
1652
|
+
// across different flows that share a phase.id + task + model), the per-phase
|
|
1653
|
+
// thinking/tools config (changing either changes the subagent's output), the
|
|
1654
|
+
// resolved context pre-read content, and the world-state fingerprint.
|
|
1655
|
+
const tail = [
|
|
1656
|
+
...baseParts,
|
|
1657
|
+
`think:${cc.thinking ?? ""}`,
|
|
1658
|
+
`tools:${JSON.stringify(cc.tools ?? [])}`,
|
|
1659
|
+
`ctx:${cc.preRead ?? ""}`,
|
|
1660
|
+
];
|
|
1661
|
+
const fold = (parts) => cc.fingerprint ? hashInput(...parts, cc.fingerprint) : hashInput(...parts);
|
|
1662
|
+
// Per-phase sub-fingerprint; falls back to the whole-flow hash when absent
|
|
1663
|
+
// (sub-flow inner states, or soundness fallback) — preserving pre-M6 behavior.
|
|
1664
|
+
const fp = cc.phaseFp ?? cc.flowDefHash ?? "";
|
|
1665
|
+
const fdh = cc.flowDefHash ?? "";
|
|
1666
|
+
return {
|
|
1667
|
+
key: fold([`flow:${cc.flowName}`, `v3:phasefp:${fp}`, ...tail]),
|
|
1668
|
+
v2Key: fold([`flow:${cc.flowName}`, `v2:flowdef:${fdh}`, ...tail]),
|
|
1669
|
+
bareKey: fold([`flow:${cc.flowName}`, `flowdef:${fdh}`, ...tail]),
|
|
1670
|
+
legacyKey: fold([`flow:${cc.flowName}`, ...tail]),
|
|
1671
|
+
};
|
|
1672
|
+
}
|
|
1673
|
+
/**
|
|
1674
|
+
* Resume/memoization lookup. Honors scope:
|
|
1675
|
+
* - "off": never reuse (even within-run).
|
|
1676
|
+
* - "run-only": within-run resume only (historical behavior).
|
|
1677
|
+
* - "cross-run": within-run first, then the persistent cross-run store.
|
|
1678
|
+
* On a cross-run hit, usage is zeroed and `cacheHit` records the source.
|
|
1679
|
+
*
|
|
1680
|
+
* The cross-run read is FOUR-TIER and READ-ONLY for fallback keys: it tries
|
|
1681
|
+
* `keys.key` (current `v3:phasefp:` shape) first, then `keys.v2Key` (pre-M6
|
|
1682
|
+
* `v2:flowdef:`), then `keys.bareKey` (pre-H1 bare `flowdef:`), then
|
|
1683
|
+
* `keys.legacyKey` (pre-flowDefHash, no flowdef line).
|
|
1684
|
+
* A hit on ANY tier is restored as a cache hit; we do NOT write-through (no
|
|
1685
|
+
* re-store under the new key) so the cache size stays stable and the legacy
|
|
1686
|
+
* entry ages out naturally. See docs/internal/cache-migration.md.
|
|
1687
|
+
*/
|
|
1688
|
+
function cachedPhase(cc, keys) {
|
|
1689
|
+
if (cc.scope === "off")
|
|
1690
|
+
return null;
|
|
1691
|
+
if (cc.forceRerun)
|
|
1692
|
+
return null;
|
|
1693
|
+
// 1. within-run resume (fastest; always allowed unless scope is off). Flag
|
|
1694
|
+
// it as a `run-only` cache hit so the run summary can count it as reused
|
|
1695
|
+
// work (it spent no new tokens). The prior usage is preserved verbatim so
|
|
1696
|
+
// the summary can report what the reuse would otherwise have cost.
|
|
1697
|
+
if (cc.prior && cc.prior.status === "done" && cc.prior.inputHash === keys.key) {
|
|
1698
|
+
return { ...cc.prior, status: "done", cacheHit: "run-only" };
|
|
1699
|
+
}
|
|
1700
|
+
// 2. cross-run memoization (opt-in) — four-tier read-only fallback.
|
|
1701
|
+
if (cc.scope === "cross-run") {
|
|
1702
|
+
for (const k of [keys.key, keys.v2Key, keys.bareKey, keys.legacyKey]) {
|
|
1703
|
+
const e = cc.store.get(k, cc.ttlMs);
|
|
1704
|
+
if (!e)
|
|
1705
|
+
continue;
|
|
1706
|
+
// If we stored the full PhaseState, restore it (preserving gate,
|
|
1707
|
+
// approval, reads, loop/tournament metadata, warnings) and just mark
|
|
1708
|
+
// the cache hit + zero usage. Fallback to the legacy trimmed surface
|
|
1709
|
+
// for entries written before this change.
|
|
1710
|
+
if (e.state) {
|
|
1711
|
+
return { ...e.state, inputHash: keys.key, usage: emptyUsage(), cacheHit: "cross-run", endedAt: Date.now() };
|
|
1712
|
+
}
|
|
1713
|
+
return {
|
|
1714
|
+
id: cc.phaseId,
|
|
1715
|
+
status: "done",
|
|
1716
|
+
inputHash: keys.key,
|
|
1717
|
+
output: e.output,
|
|
1718
|
+
json: e.json,
|
|
1719
|
+
model: e.model,
|
|
1720
|
+
usage: emptyUsage(),
|
|
1721
|
+
cacheHit: "cross-run",
|
|
1722
|
+
endedAt: Date.now(),
|
|
1723
|
+
};
|
|
1724
|
+
}
|
|
1725
|
+
}
|
|
1726
|
+
return null;
|
|
1727
|
+
}
|
|
1728
|
+
/** Persist a freshly-computed phase result to the cross-run store (best-effort). */
|
|
1729
|
+
function recordCache(cc, ps) {
|
|
1730
|
+
if (cc.scope !== "cross-run")
|
|
1731
|
+
return;
|
|
1732
|
+
if (ps.status !== "done" || !ps.inputHash)
|
|
1733
|
+
return;
|
|
1734
|
+
if (ps.cacheHit)
|
|
1735
|
+
return; // don't re-store a value we just read from cache
|
|
1736
|
+
cc.store.put({
|
|
1737
|
+
key: ps.inputHash,
|
|
1738
|
+
createdAt: Date.now(),
|
|
1739
|
+
output: ps.output,
|
|
1740
|
+
json: ps.json,
|
|
1741
|
+
model: ps.model,
|
|
1742
|
+
state: ps,
|
|
1743
|
+
flowName: cc.flowName,
|
|
1744
|
+
phaseId: cc.phaseId,
|
|
1745
|
+
runId: cc.runId,
|
|
1746
|
+
});
|
|
1747
|
+
}
|
|
1748
|
+
/**
|
|
1749
|
+
* Resolve an agent name against available agents. Falls back to the default
|
|
1750
|
+
* agent if the requested agent isn't found, logging a warning via safeEmit.
|
|
1751
|
+
*/
|
|
1752
|
+
function resolveAgent(name, deps, state) {
|
|
1753
|
+
const resolved = name ?? defaultAgent(deps);
|
|
1754
|
+
if (name && !deps.agents.some((a) => a.name === name)) {
|
|
1755
|
+
const fallback = defaultAgent(deps);
|
|
1756
|
+
// Log only once per run to avoid noise.
|
|
1757
|
+
if (!state.__unknownAgentWarned) {
|
|
1758
|
+
state.__unknownAgentWarned = new Set();
|
|
1759
|
+
}
|
|
1760
|
+
if (!state.__unknownAgentWarned.has(name)) {
|
|
1761
|
+
state.__unknownAgentWarned.add(name);
|
|
1762
|
+
console.warn(`[taskflow] Unknown agent "${name}", falling back to "${fallback}". Use action=agents to list available agents.`);
|
|
1763
|
+
}
|
|
1764
|
+
return fallback;
|
|
1765
|
+
}
|
|
1766
|
+
return resolved;
|
|
1767
|
+
}
|
|
1768
|
+
function defaultAgent(deps) {
|
|
1769
|
+
return deps.agents[0]?.name ?? "default";
|
|
1770
|
+
}
|
|
1771
|
+
/**
|
|
1772
|
+
* Parse a gate phase's output into a verdict. Blocks the flow only on an
|
|
1773
|
+
* explicit negative signal; ambiguous output passes (fail-open).
|
|
1774
|
+
* Accepts JSON ({continue|pass: bool} or {verdict: "..."}) or a text marker
|
|
1775
|
+
* `VERDICT: PASS|BLOCK|FAIL|STOP|OK|REJECT|HALT` (last occurrence wins).
|
|
1776
|
+
*/
|
|
1777
|
+
export function parseGateVerdict(output) {
|
|
1778
|
+
const json = safeParse(output);
|
|
1779
|
+
if (json && typeof json === "object") {
|
|
1780
|
+
const o = json;
|
|
1781
|
+
if (typeof o.continue === "boolean")
|
|
1782
|
+
return { verdict: o.continue ? "pass" : "block", reason: asReason(o.reason) };
|
|
1783
|
+
if (typeof o.pass === "boolean")
|
|
1784
|
+
return { verdict: o.pass ? "pass" : "block", reason: asReason(o.reason) };
|
|
1785
|
+
if (typeof o.verdict === "string") {
|
|
1786
|
+
// Note: do NOT include standalone "no" — natural-language verdicts like
|
|
1787
|
+
// "No issues found" / "no errors" would otherwise be false-positive BLOCK.
|
|
1788
|
+
// Fail-open covers any ambiguous text.
|
|
1789
|
+
const block = /block|fail|stop|reject|halt/i.test(o.verdict);
|
|
1790
|
+
return { verdict: block ? "block" : "pass", reason: asReason(o.reason) };
|
|
1791
|
+
}
|
|
1792
|
+
}
|
|
1793
|
+
const matches = [...output.matchAll(/VERDICT\s*[:=]\s*(PASS|BLOCK|FAIL|STOP|OK|REJECT|HALT)/gi)];
|
|
1794
|
+
if (matches.length) {
|
|
1795
|
+
const v = matches[matches.length - 1][1].toUpperCase();
|
|
1796
|
+
const pass = v === "PASS" || v === "OK";
|
|
1797
|
+
return { verdict: pass ? "pass" : "block" };
|
|
1798
|
+
}
|
|
1799
|
+
return { verdict: "pass" };
|
|
1800
|
+
}
|
|
1801
|
+
function asReason(v) {
|
|
1802
|
+
return typeof v === "string" && v.trim() ? v.trim() : undefined;
|
|
1803
|
+
}
|
|
1804
|
+
/**
|
|
1805
|
+
* Parse a judge's pick of the winning variant. Accepts JSON ({"winner":n} or
|
|
1806
|
+
* {"best":n}) or a `WINNER: n` line (last match wins). Clamps to [1, count].
|
|
1807
|
+
* Fail-open: an unreadable verdict defaults to variant 1 so the work is never
|
|
1808
|
+
* lost. Returns the 1-based index plus an optional reason.
|
|
1809
|
+
*/
|
|
1810
|
+
export function parseTournamentWinner(output, count) {
|
|
1811
|
+
const clamp = (n) => Math.min(Math.max(1, Math.floor(n)), Math.max(1, count));
|
|
1812
|
+
const json = safeParse(output);
|
|
1813
|
+
if (json && typeof json === "object") {
|
|
1814
|
+
const o = json;
|
|
1815
|
+
const raw = o.winner ?? o.best ?? o.choice;
|
|
1816
|
+
const n = typeof raw === "number" ? raw : typeof raw === "string" ? Number(raw) : NaN;
|
|
1817
|
+
if (Number.isFinite(n))
|
|
1818
|
+
return { winner: clamp(n), reason: asReason(o.reason) };
|
|
1819
|
+
}
|
|
1820
|
+
const matches = [...output.matchAll(/WINNER\s*[:=]\s*#?\s*(\d+)/gi)];
|
|
1821
|
+
if (matches.length) {
|
|
1822
|
+
const n = Number(matches[matches.length - 1][1]);
|
|
1823
|
+
if (Number.isFinite(n))
|
|
1824
|
+
return { winner: clamp(n) };
|
|
1825
|
+
}
|
|
1826
|
+
return { winner: 1, reason: "no parseable winner; defaulted to variant 1" };
|
|
1827
|
+
}
|
|
1828
|
+
/**
|
|
1829
|
+
* Best-effort invocation of the user-provided `persist` + `onProgress` callbacks.
|
|
1830
|
+
*
|
|
1831
|
+
* A throw from a host-supplied callback must NEVER replace the runtime's
|
|
1832
|
+
* outcome — neither the original crash message in `executeTaskflow`'s catch
|
|
1833
|
+
* block, nor the final output of a successful run. Callbacks are observability
|
|
1834
|
+
* hooks; the run survives their failure.
|
|
1835
|
+
*
|
|
1836
|
+
* Used at every "checkpoint" call site (phase start, phase end, terminal state).
|
|
1837
|
+
* For high-frequency live updates inside a phase, see `safeProgress` below.
|
|
1838
|
+
*/
|
|
1839
|
+
function safeEmit(deps, state) {
|
|
1840
|
+
try {
|
|
1841
|
+
deps.persist?.(state);
|
|
1842
|
+
}
|
|
1843
|
+
catch {
|
|
1844
|
+
// user callback — must not break the run
|
|
1845
|
+
}
|
|
1846
|
+
try {
|
|
1847
|
+
deps.onProgress?.(state);
|
|
1848
|
+
}
|
|
1849
|
+
catch {
|
|
1850
|
+
// user callback — must not break the run
|
|
1851
|
+
}
|
|
1852
|
+
}
|
|
1853
|
+
/**
|
|
1854
|
+
* Like `safeEmit` but for the high-frequency live-update channel only.
|
|
1855
|
+
* Skips `persist` (which is intentionally checkpoint-only) and swallows any
|
|
1856
|
+
* throw from the user-supplied `onProgress` so a misbehaving TUI sink cannot
|
|
1857
|
+
* disrupt an in-flight phase.
|
|
1858
|
+
*/
|
|
1859
|
+
function safeProgress(deps, state) {
|
|
1860
|
+
try {
|
|
1861
|
+
deps.onProgress?.(state);
|
|
1862
|
+
}
|
|
1863
|
+
catch {
|
|
1864
|
+
// user callback — must not break the run
|
|
1865
|
+
}
|
|
1866
|
+
}
|
|
1867
|
+
/** Scan a flow for dependencies that cannot be observed through the readSet.
|
|
1868
|
+
* These include Shared Context Tree, sub-flows, context: file pre-reads, and
|
|
1869
|
+
* interpolation placeholders that do not resolve through `steps.*` (previous,
|
|
1870
|
+
* args, item). Recomputing flows with such deps with dryRun:false risks
|
|
1871
|
+
* silently reusing stale upstream state. */
|
|
1872
|
+
function hasUnobservedDependencies(state) {
|
|
1873
|
+
const scan = (text) => /\{(previous\.output|args\.|item\b|item\.)/.test(text);
|
|
1874
|
+
for (const p of state.def.phases) {
|
|
1875
|
+
if (p.shareContext === true)
|
|
1876
|
+
return true;
|
|
1877
|
+
if (state.def.contextSharing === true)
|
|
1878
|
+
return true;
|
|
1879
|
+
if (p.type === "flow")
|
|
1880
|
+
return true;
|
|
1881
|
+
if (p.context && p.context.length > 0)
|
|
1882
|
+
return true;
|
|
1883
|
+
if (scan(p.task ?? ""))
|
|
1884
|
+
return true;
|
|
1885
|
+
if (p.when && scan(p.when))
|
|
1886
|
+
return true;
|
|
1887
|
+
if (p.until && scan(p.until))
|
|
1888
|
+
return true;
|
|
1889
|
+
if (Array.isArray(p.eval) && p.eval.some(scan))
|
|
1890
|
+
return true;
|
|
1891
|
+
}
|
|
1892
|
+
return false;
|
|
1893
|
+
}
|
|
1894
|
+
/** Recompute a completed run minimally: force-rerun the `seeds`, then walk
|
|
1895
|
+
* their stale frontier in topological order. The cache provides early cutoff
|
|
1896
|
+
* for free — a downstream whose inputHash didn't move (because the seed's new
|
|
1897
|
+
* output happened to equal the old) hits its prior and is reused rather than
|
|
1898
|
+
* re-executed. `dryRun` computes the worst-case frontier without spending a
|
|
1899
|
+
* token. Returns a fresh state + a report. Throws only when dryRun:false is
|
|
1900
|
+
* requested for a flow with unobserved dependencies; callers should surface
|
|
1901
|
+
* that as a user-facing error. */
|
|
1902
|
+
export async function recomputeTaskflow(state, deps, seeds,
|
|
1903
|
+
// Fail-safe default: a real recompute overwrites the run and spends tokens.
|
|
1904
|
+
// The tool/command wrappers can explicitly opt into dryRun:false.
|
|
1905
|
+
opts = { dryRun: true }) {
|
|
1906
|
+
// Never mutate the caller's RunState in-place. Recompute is a speculative
|
|
1907
|
+
// replay; only the caller decides whether to persist the new state.
|
|
1908
|
+
const newState = structuredClone(state);
|
|
1909
|
+
const reads = readMapOf(newState.phases);
|
|
1910
|
+
// M2: derive the declared read-map fresh from the def so the frontier uses
|
|
1911
|
+
// the UNION (observed ∪ declared). Derived here (not read from the persisted
|
|
1912
|
+
// `RunState.declaredDeps`) so old runs — pre-H1, no persisted declaredDeps —
|
|
1913
|
+
// also get union semantics. The persisted field is audit/provenance only.
|
|
1914
|
+
const declared = declaredReadMapOfDef(newState.def);
|
|
1915
|
+
const frontier = computeStaleFrontier(reads, seeds, declared);
|
|
1916
|
+
const allIds = Object.keys(newState.phases);
|
|
1917
|
+
if (opts.dryRun) {
|
|
1918
|
+
// Explain each phase WITHOUT executing: a frontier phase "may rerun"
|
|
1919
|
+
// because it (transitively) reads a changed seed; everything else is
|
|
1920
|
+
// reused as unreachable. We name the in-frontier upstream(s) as the cause.
|
|
1921
|
+
const seedSet0 = new Set(seeds);
|
|
1922
|
+
const upstreamsOf = (id) => {
|
|
1923
|
+
const observed = (newState.phases[id]?.reads ?? []).map((r) => r.stepId).filter((u) => u !== id);
|
|
1924
|
+
const decl = (declared.get(id) ?? []).filter((u) => u !== id);
|
|
1925
|
+
return [...new Set([...observed, ...decl])];
|
|
1926
|
+
};
|
|
1927
|
+
const decisions = allIds.map((id) => {
|
|
1928
|
+
if (!frontier.has(id)) {
|
|
1929
|
+
return { phaseId: id, outcome: "reused", reason: "not reachable from any changed seed" };
|
|
1930
|
+
}
|
|
1931
|
+
if (seedSet0.has(id)) {
|
|
1932
|
+
return { phaseId: id, outcome: "rerun", reason: "forced by recompute request (seed)" };
|
|
1933
|
+
}
|
|
1934
|
+
const causes = upstreamsOf(id).filter((u) => frontier.has(u));
|
|
1935
|
+
return {
|
|
1936
|
+
phaseId: id,
|
|
1937
|
+
outcome: "rerun",
|
|
1938
|
+
reason: "reads a phase in the stale frontier; may re-run if that upstream's output moves",
|
|
1939
|
+
causedBy: causes.length ? causes : undefined,
|
|
1940
|
+
};
|
|
1941
|
+
});
|
|
1942
|
+
return {
|
|
1943
|
+
report: {
|
|
1944
|
+
dryRun: true,
|
|
1945
|
+
aborted: false,
|
|
1946
|
+
seeds,
|
|
1947
|
+
rerun: [...frontier],
|
|
1948
|
+
reused: allIds.filter((id) => !frontier.has(id)),
|
|
1949
|
+
cutoff: [],
|
|
1950
|
+
decisions,
|
|
1951
|
+
},
|
|
1952
|
+
state: newState,
|
|
1953
|
+
};
|
|
1954
|
+
}
|
|
1955
|
+
// Guard: observed readSet only tracks `{steps.X.*}` interpolation refs. It is
|
|
1956
|
+
// blind to Shared Context Tree (ctx_read/ctx_write), sub-flow internals,
|
|
1957
|
+
// context: file pre-reads, {previous.output}, and loop locals ({args.*},
|
|
1958
|
+
// {item.*}). Recomputing such a run with dryRun:false could silently skip
|
|
1959
|
+
// phases whose deps changed outside the observed frontier and then persist a
|
|
1960
|
+
// corrupted run over the original.
|
|
1961
|
+
if (hasUnobservedDependencies(newState)) {
|
|
1962
|
+
throw new Error("recompute dryRun:false is unsafe for this run: it contains dependencies " +
|
|
1963
|
+
"(shareContext, flow/ctx_spawn, context: files, {previous.output}, {args.*}, or {item.*}) " +
|
|
1964
|
+
"that are not tracked by the observed readSet. Use dryRun:true to inspect " +
|
|
1965
|
+
"the frontier, or change the upstream phase and re-run the whole flow.");
|
|
1966
|
+
}
|
|
1967
|
+
// Real recompute: topological order over the frontier so a downstream always
|
|
1968
|
+
// sees its (already-refreshed) upstreams when it re-evaluates its cache key.
|
|
1969
|
+
// The order must respect declared dependsOn, observed reads, AND declared
|
|
1970
|
+
// reads (M2 union): pi-taskflow allows interpolation refs without an
|
|
1971
|
+
// explicit dependsOn edge, and a declared-but-unobserved edge (e.g. a `when`
|
|
1972
|
+
// ref that never fired) must still order the reader after its upstream so
|
|
1973
|
+
// the reader evaluates its cache key against the refreshed upstream (no
|
|
1974
|
+
// false early-cutoff).
|
|
1975
|
+
const seedSet = new Set(seeds);
|
|
1976
|
+
function depsFor(phaseId) {
|
|
1977
|
+
// A phase reading its own prior output (e.g. a loop `until` checking
|
|
1978
|
+
// `{steps.thisId.output}`) must not create a self-edge in the scheduling
|
|
1979
|
+
// graph — otherwise topoLayers would deadlock on the self-loop.
|
|
1980
|
+
const observed = (newState.phases[phaseId]?.reads ?? [])
|
|
1981
|
+
.map((r) => r.stepId)
|
|
1982
|
+
.filter((id) => id !== phaseId);
|
|
1983
|
+
const declared_ = (declared.get(phaseId) ?? []).filter((id) => id !== phaseId);
|
|
1984
|
+
return [...new Set([...observed, ...declared_])];
|
|
1985
|
+
}
|
|
1986
|
+
const augmentedPhases = newState.def.phases.map((p) => ({
|
|
1987
|
+
...p,
|
|
1988
|
+
dependsOn: [...new Set([...(p.dependsOn ?? []), ...depsFor(p.id)])],
|
|
1989
|
+
}));
|
|
1990
|
+
const order = topoLayers(augmentedPhases)
|
|
1991
|
+
.flat()
|
|
1992
|
+
.map((p) => p.id)
|
|
1993
|
+
.filter((id) => frontier.has(id));
|
|
1994
|
+
const rerun = [];
|
|
1995
|
+
const cutoff = [];
|
|
1996
|
+
const decisions = [];
|
|
1997
|
+
// Phases whose OUTPUT actually moved this recompute (seed forced, or result
|
|
1998
|
+
// changed). Used to attribute a downstream rerun to the specific upstream(s)
|
|
1999
|
+
// that changed — the "why" of the decision trace.
|
|
2000
|
+
const outputMoved = new Set();
|
|
2001
|
+
const noop = () => { };
|
|
2002
|
+
let aborted = false;
|
|
2003
|
+
for (const id of order) {
|
|
2004
|
+
// A partial recompute must NOT be persisted over the original run — the
|
|
2005
|
+
// caller discards `state` when `aborted` is set.
|
|
2006
|
+
if (deps.signal?.aborted) {
|
|
2007
|
+
aborted = true;
|
|
2008
|
+
break;
|
|
2009
|
+
}
|
|
2010
|
+
const phase = newState.def.phases.find((p) => p.id === id);
|
|
2011
|
+
if (!phase)
|
|
2012
|
+
continue;
|
|
2013
|
+
const before = newState.phases[id]?.inputHash;
|
|
2014
|
+
const isSeed = seedSet.has(id);
|
|
2015
|
+
const execOpts = isSeed ? { forceRerun: true } : undefined;
|
|
2016
|
+
// The upstream(s) of this phase whose output moved — the cause of a rerun.
|
|
2017
|
+
const changedUpstreams = depsFor(id).filter((u) => outputMoved.has(u));
|
|
2018
|
+
try {
|
|
2019
|
+
const ps = await executePhase(phase, newState, deps, newState.phases[id], noop, 0, execOpts);
|
|
2020
|
+
newState.phases[id] = ps;
|
|
2021
|
+
// A phase counts as "rerun" if it was a forced seed OR its result moved;
|
|
2022
|
+
// otherwise it hit its cache (inputHash unchanged) → early cutoff.
|
|
2023
|
+
if (isSeed || ps.inputHash !== before) {
|
|
2024
|
+
rerun.push(id);
|
|
2025
|
+
outputMoved.add(id);
|
|
2026
|
+
decisions.push(isSeed
|
|
2027
|
+
? { phaseId: id, outcome: "rerun", reason: "forced by recompute request (seed)" }
|
|
2028
|
+
: {
|
|
2029
|
+
phaseId: id,
|
|
2030
|
+
outcome: "rerun",
|
|
2031
|
+
reason: "input changed — an upstream's output moved",
|
|
2032
|
+
causedBy: changedUpstreams.length ? changedUpstreams : undefined,
|
|
2033
|
+
});
|
|
2034
|
+
}
|
|
2035
|
+
else {
|
|
2036
|
+
cutoff.push(id);
|
|
2037
|
+
decisions.push({
|
|
2038
|
+
phaseId: id,
|
|
2039
|
+
outcome: "cutoff",
|
|
2040
|
+
reason: "input unchanged — upstream(s) re-ran but produced identical output (early cutoff)",
|
|
2041
|
+
causedBy: depsFor(id).filter((u) => frontier.has(u)).length
|
|
2042
|
+
? depsFor(id).filter((u) => frontier.has(u))
|
|
2043
|
+
: undefined,
|
|
2044
|
+
});
|
|
2045
|
+
}
|
|
2046
|
+
}
|
|
2047
|
+
catch {
|
|
2048
|
+
// A failing recompute phase is recorded as rerun (it was attempted).
|
|
2049
|
+
rerun.push(id);
|
|
2050
|
+
outputMoved.add(id);
|
|
2051
|
+
decisions.push({ phaseId: id, outcome: "failed", reason: "re-execution attempted but the phase failed" });
|
|
2052
|
+
}
|
|
2053
|
+
}
|
|
2054
|
+
// Frontier-external phases were never touched — record them as reused.
|
|
2055
|
+
for (const id of allIds) {
|
|
2056
|
+
if (!frontier.has(id)) {
|
|
2057
|
+
decisions.push({ phaseId: id, outcome: "reused", reason: "not reachable from any changed seed" });
|
|
2058
|
+
}
|
|
2059
|
+
}
|
|
2060
|
+
return {
|
|
2061
|
+
report: {
|
|
2062
|
+
dryRun: false,
|
|
2063
|
+
aborted,
|
|
2064
|
+
seeds,
|
|
2065
|
+
rerun,
|
|
2066
|
+
reused: allIds.filter((id) => !frontier.has(id)),
|
|
2067
|
+
cutoff,
|
|
2068
|
+
decisions,
|
|
2069
|
+
},
|
|
2070
|
+
state: newState,
|
|
2071
|
+
};
|
|
2072
|
+
}
|
|
2073
|
+
export async function executeTaskflow(state, deps) {
|
|
2074
|
+
const def = state.def;
|
|
2075
|
+
try {
|
|
2076
|
+
return await runTaskflowLayers(state, deps);
|
|
2077
|
+
}
|
|
2078
|
+
catch (e) {
|
|
2079
|
+
// A thrown phase must not leave the run wedged in "running" (which breaks
|
|
2080
|
+
// resume). Mark any in-flight phase + the run as failed, persist, and return.
|
|
2081
|
+
const message = e instanceof Error ? e.message : String(e);
|
|
2082
|
+
for (const p of Object.values(state.phases)) {
|
|
2083
|
+
if (p.status === "running") {
|
|
2084
|
+
p.status = "failed";
|
|
2085
|
+
p.error = p.error ?? message;
|
|
2086
|
+
p.endedAt = Date.now();
|
|
2087
|
+
}
|
|
2088
|
+
}
|
|
2089
|
+
state.status = "failed";
|
|
2090
|
+
safeEmit(deps, state);
|
|
2091
|
+
const totalUsage = aggregateUsage(Object.values(state.phases).map((p) => p.usage ?? emptyUsage()));
|
|
2092
|
+
return { state, finalOutput: `Taskflow '${def.name}' crashed: ${message}`, ok: false, totalUsage };
|
|
2093
|
+
}
|
|
2094
|
+
}
|
|
2095
|
+
async function runTaskflowLayers(state, deps) {
|
|
2096
|
+
const def = state.def;
|
|
2097
|
+
const layers = topoLayers(def.phases);
|
|
2098
|
+
// Content-fingerprint the desugared definition ONCE per run and fold it into
|
|
2099
|
+
// every phase's cache key (overstory hash algorithm; see ./flowir/hash.ts).
|
|
2100
|
+
// Reused by every phase, persisted on the RunState for audit/resume.
|
|
2101
|
+
// Never throws into the run — a hash failure leaves the field unset and the
|
|
2102
|
+
// cache key degrades to the legacy flowName-only shape.
|
|
2103
|
+
//
|
|
2104
|
+
// Routed through the FlowIR compile seam (M1): `compileTaskflowToIR`
|
|
2105
|
+
// produces the content-addressed IR whose `hash` (== flowDefHash in the
|
|
2106
|
+
// stub) folds into the cache key, and whose `meta.declaredDeps` (M2 declared
|
|
2107
|
+
// plane) is persisted for audit/provenance. The declared plane is also
|
|
2108
|
+
// derived fresh from `def` in recompute (so old runs get union semantics
|
|
2109
|
+
// too); the persisted copy is for display.
|
|
2110
|
+
if (state.flowDefHash === undefined) {
|
|
2111
|
+
try {
|
|
2112
|
+
const ir = await compileTaskflowToIR(def);
|
|
2113
|
+
state.flowDefHash = ir.hash ?? "failed";
|
|
2114
|
+
state.declaredDeps = ir.meta.declaredDeps;
|
|
2115
|
+
if (ir.errors.length) {
|
|
2116
|
+
console.warn(`[taskflow] IR compile errors for '${def.name}': ${ir.errors.map((e) => e.message).join("; ")}`);
|
|
2117
|
+
}
|
|
2118
|
+
}
|
|
2119
|
+
catch (e) {
|
|
2120
|
+
// Fail-safe: warn loudly rather than silently degrading to the legacy
|
|
2121
|
+
// flowName-only key, which would reopen the cross-flow collision hole.
|
|
2122
|
+
console.warn(`[taskflow] flowDefHash failed for '${def.name}': ${e instanceof Error ? e.message : String(e)}. ` +
|
|
2123
|
+
"Cross-run cache is disabled for this run to prevent stale cross-flow hits.");
|
|
2124
|
+
state.flowDefHash = "failed";
|
|
2125
|
+
}
|
|
2126
|
+
}
|
|
2127
|
+
// M6: per-phase structural sub-fingerprints. Computed once per run (when
|
|
2128
|
+
// cross-run is potentially active) so editing phase B invalidates only B +
|
|
2129
|
+
// its transitive dependents, not independent siblings. Each value is either
|
|
2130
|
+
// a precise per-phase hash or the whole-flow `flowDefHash` (soundness
|
|
2131
|
+
// fallback for shareContext / `flow` phases). Skipped entirely when
|
|
2132
|
+
// `flowDefHash === "failed"` (cross-run is disabled for the run anyway).
|
|
2133
|
+
// Never throws into the run — a per-phase error degrades that phase to the
|
|
2134
|
+
// whole-flow hash (safe, = pre-M6 behavior).
|
|
2135
|
+
if (state.flowDefHash !== "failed" && state.phaseFingerprints === undefined) {
|
|
2136
|
+
const whole = state.flowDefHash ?? "";
|
|
2137
|
+
const map = {};
|
|
2138
|
+
for (const p of def.phases) {
|
|
2139
|
+
try {
|
|
2140
|
+
map[p.id] = (await phaseFingerprint(def, p.id)) ?? whole;
|
|
2141
|
+
}
|
|
2142
|
+
catch {
|
|
2143
|
+
map[p.id] = whole; // fail-open → whole-flow scope
|
|
2144
|
+
}
|
|
2145
|
+
}
|
|
2146
|
+
state.phaseFingerprints = map;
|
|
2147
|
+
}
|
|
2148
|
+
state.status = "running";
|
|
2149
|
+
safeEmit(deps, state);
|
|
2150
|
+
let aborted = false;
|
|
2151
|
+
let gateBlocked = false;
|
|
2152
|
+
let gateReason = "";
|
|
2153
|
+
let gateOutput = "";
|
|
2154
|
+
// `budgetBlocked` gates the skipping of remaining phases once the cap is hit
|
|
2155
|
+
// and also drives the terminal "blocked" status — a maxUSD ceiling must never
|
|
2156
|
+
// silently do nothing.
|
|
2157
|
+
let budgetBlocked = false;
|
|
2158
|
+
let budgetReason = "";
|
|
2159
|
+
const byId = new Map(def.phases.map((p) => [p.id, p]));
|
|
2160
|
+
for (const layer of layers) {
|
|
2161
|
+
if (deps.signal?.aborted) {
|
|
2162
|
+
aborted = true;
|
|
2163
|
+
break;
|
|
2164
|
+
}
|
|
2165
|
+
// Phases within a layer have no inter-dependencies → run concurrently.
|
|
2166
|
+
const layerConcurrency = Math.max(1, def.concurrency ?? 8);
|
|
2167
|
+
await mapWithConcurrencyLimit(layer, layerConcurrency, async (phase) => {
|
|
2168
|
+
// Snapshot prior state BEFORE marking running, so resume cache checks work.
|
|
2169
|
+
const prior = state.phases[phase.id];
|
|
2170
|
+
// Determine whether this phase should run, or be skipped (and why).
|
|
2171
|
+
const deps_ = dependenciesOf(phase);
|
|
2172
|
+
const join = phase.join ?? "all";
|
|
2173
|
+
// An `optional` dependency that failed still counts as satisfied.
|
|
2174
|
+
const depOk = (d) => {
|
|
2175
|
+
const s = state.phases[d]?.status;
|
|
2176
|
+
if (s === "done")
|
|
2177
|
+
return true;
|
|
2178
|
+
if (s === "failed" && byId.get(d)?.optional)
|
|
2179
|
+
return true;
|
|
2180
|
+
return false;
|
|
2181
|
+
};
|
|
2182
|
+
const depsSatisfied = deps_.length === 0 ? true : join === "any" ? deps_.some(depOk) : deps_.every(depOk);
|
|
2183
|
+
let skipReason;
|
|
2184
|
+
if (gateBlocked)
|
|
2185
|
+
skipReason = `Gate blocked${gateReason ? `: ${gateReason}` : ""}`;
|
|
2186
|
+
else if (budgetBlocked)
|
|
2187
|
+
skipReason = `Budget exceeded${budgetReason ? `: ${budgetReason}` : ""}`;
|
|
2188
|
+
else if (!depsSatisfied)
|
|
2189
|
+
skipReason = join === "any" ? "All dependencies failed or were skipped" : "Upstream dependency not satisfied";
|
|
2190
|
+
if (skipReason) {
|
|
2191
|
+
if (skipReason.startsWith("Budget exceeded"))
|
|
2192
|
+
budgetBlocked = true;
|
|
2193
|
+
state.phases[phase.id] = {
|
|
2194
|
+
id: phase.id,
|
|
2195
|
+
status: "skipped",
|
|
2196
|
+
error: skipReason,
|
|
2197
|
+
endedAt: Date.now(),
|
|
2198
|
+
usage: emptyUsage(),
|
|
2199
|
+
};
|
|
2200
|
+
safeEmit(deps, state);
|
|
2201
|
+
return;
|
|
2202
|
+
}
|
|
2203
|
+
const startedAt = Date.now();
|
|
2204
|
+
// Re-running a phase (resume after a previous failed/done attempt) must
|
|
2205
|
+
// start from a clean "running" state. Spreading the prior PhaseState
|
|
2206
|
+
// would carry over its terminal `endedAt` (and `error`/`gate`/`output`),
|
|
2207
|
+
// leaving a running phase with an old endedAt < new startedAt — which
|
|
2208
|
+
// renders as a frozen NEGATIVE elapsed time in the TUI. Keep only the
|
|
2209
|
+
// fields that are still meaningful across attempts (model, attempts).
|
|
2210
|
+
const priorPs = state.phases[phase.id];
|
|
2211
|
+
state.phases[phase.id] = {
|
|
2212
|
+
id: phase.id,
|
|
2213
|
+
status: "running",
|
|
2214
|
+
startedAt,
|
|
2215
|
+
...(priorPs?.model ? { model: priorPs.model } : {}),
|
|
2216
|
+
...(priorPs?.attempts ? { attempts: priorPs.attempts } : {}),
|
|
2217
|
+
};
|
|
2218
|
+
safeProgress(deps, state);
|
|
2219
|
+
const ps = await executePhase(phase, state, deps, prior, () => safeProgress(deps, state));
|
|
2220
|
+
// Preserve the phase start time: executePhase returns a fresh PhaseState
|
|
2221
|
+
// that omits startedAt (cached/resumed results carry their own).
|
|
2222
|
+
state.phases[phase.id] = ps.startedAt ? ps : { ...ps, startedAt };
|
|
2223
|
+
// A blocking verdict (gate phase OR a rejected approval) halts the flow.
|
|
2224
|
+
const ptype = phase.type ?? "agent";
|
|
2225
|
+
if (ps.gate?.verdict === "block" && (ptype === "gate" || ptype === "approval")) {
|
|
2226
|
+
gateBlocked = true;
|
|
2227
|
+
gateReason = ps.gate.reason ?? "";
|
|
2228
|
+
gateOutput = ps.output ?? "";
|
|
2229
|
+
}
|
|
2230
|
+
// A fan-out cut short by the cap is itself a budget skip.
|
|
2231
|
+
if (ps.budgetTruncated) {
|
|
2232
|
+
budgetBlocked = true;
|
|
2233
|
+
if (!budgetReason)
|
|
2234
|
+
budgetReason = "fan-out truncated by budget";
|
|
2235
|
+
}
|
|
2236
|
+
// Budget ceiling: once exceeded, remaining phases are skipped.
|
|
2237
|
+
// For concurrent same-layer phases, the check runs after each phase
|
|
2238
|
+
// completes, so at most (concurrency - 1) extra phases may run before
|
|
2239
|
+
// the budget is detected as exceeded. This bounded overshoot is
|
|
2240
|
+
// acceptable: budgetBlocked prevents cascading into subsequent layers.
|
|
2241
|
+
const ob = overBudget(state);
|
|
2242
|
+
if (ob.over) {
|
|
2243
|
+
budgetBlocked = true;
|
|
2244
|
+
budgetReason = ob.reason;
|
|
2245
|
+
}
|
|
2246
|
+
safeEmit(deps, state);
|
|
2247
|
+
});
|
|
2248
|
+
}
|
|
2249
|
+
const fp = finalPhase(def.phases);
|
|
2250
|
+
let finalState = state.phases[fp.id];
|
|
2251
|
+
// If the designated final phase produced no output (skipped/blocked), fall
|
|
2252
|
+
// back to the last phase (in definition order) that actually completed.
|
|
2253
|
+
if (!finalState || finalState.status !== "done") {
|
|
2254
|
+
const doneInOrder = def.phases.map((p) => state.phases[p.id]).filter((p) => p?.status === "done");
|
|
2255
|
+
if (doneInOrder.length)
|
|
2256
|
+
finalState = doneInOrder[doneInOrder.length - 1];
|
|
2257
|
+
}
|
|
2258
|
+
// A failed non-optional phase fails the run; optional failures are tolerated.
|
|
2259
|
+
const anyFailed = Object.entries(state.phases).some(([id, p]) => p.status === "failed" && !byId.get(id)?.optional);
|
|
2260
|
+
state.status = aborted
|
|
2261
|
+
? "paused"
|
|
2262
|
+
: gateBlocked || budgetBlocked
|
|
2263
|
+
? "blocked"
|
|
2264
|
+
: anyFailed
|
|
2265
|
+
? "failed"
|
|
2266
|
+
: "completed";
|
|
2267
|
+
safeEmit(deps, state);
|
|
2268
|
+
let finalOutput = finalState?.output ?? "(no output)";
|
|
2269
|
+
if (gateBlocked) {
|
|
2270
|
+
finalOutput = `Gate blocked the workflow.${gateReason ? `\nReason: ${gateReason}` : ""}${gateOutput ? `\n\n${gateOutput}` : ""}`;
|
|
2271
|
+
}
|
|
2272
|
+
else if (budgetBlocked) {
|
|
2273
|
+
finalOutput = `Budget exceeded — run halted.${budgetReason ? `\nReason: ${budgetReason}` : ""}${finalState?.output ? `\n\n${finalState.output}` : ""}`;
|
|
2274
|
+
}
|
|
2275
|
+
const totalUsage = aggregateUsage(Object.values(state.phases).map((p) => p.usage ?? emptyUsage()));
|
|
2276
|
+
return {
|
|
2277
|
+
state,
|
|
2278
|
+
finalOutput,
|
|
2279
|
+
ok: state.status === "completed",
|
|
2280
|
+
totalUsage,
|
|
2281
|
+
reuse: summarizeReuse(state),
|
|
2282
|
+
};
|
|
2283
|
+
}
|
|
2284
|
+
//# sourceMappingURL=runtime.js.map
|