@fyeeme/pi-dynamic-workflows 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +373 -0
- package/README.zh-CN.md +359 -0
- package/index.ts +650 -0
- package/package.json +58 -0
- package/sessions/spawn.ts +15 -0
- package/src/agent/dispatch.ts +76 -0
- package/src/budget/caps.ts +42 -0
- package/src/budget/index.ts +8 -0
- package/src/budget/pool.ts +118 -0
- package/src/cache/index.ts +7 -0
- package/src/cache/journal.ts +184 -0
- package/src/cache/key.ts +97 -0
- package/src/determinism/ast-guard.ts +196 -0
- package/src/errors.ts +55 -0
- package/src/format.ts +27 -0
- package/src/index.ts +28 -0
- package/src/inspect.ts +237 -0
- package/src/lifecycle.ts +75 -0
- package/src/loader.ts +50 -0
- package/src/outcomes.ts +113 -0
- package/src/planner.ts +66 -0
- package/src/runner/index.ts +188 -0
- package/src/runner/stage-executor.ts +1078 -0
- package/src/state/index.ts +1 -0
- package/src/state/names.ts +33 -0
- package/src/types.ts +332 -0
- package/src/ui-groups.ts +43 -0
|
@@ -0,0 +1,1078 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Stage executor — dispatches one step by `type`, wiring the five CC-fusion
|
|
3
|
+
* modules plus the three composite patterns:
|
|
4
|
+
* - cache/journal → computeCacheKey + Journal lookup/append (cache-resume)
|
|
5
|
+
* - budget → BudgetPool guardBatch/guardSpawn + caps (runaway + budget-exceeded, incl. maxTokens)
|
|
6
|
+
* - spawn → AgentDispatch (real spawnAgent or injected fake) + registry
|
|
7
|
+
* - lifecycle → listeners threaded through (agent start/end/skip/retry)
|
|
8
|
+
* - abort → per-call AbortController via the registry + run signal
|
|
9
|
+
*
|
|
10
|
+
* All seven step types are implemented. Every agent call — including those
|
|
11
|
+
* inside composites — goes through dispatchAgentCall (cache + budget + spawn +
|
|
12
|
+
* journal + lifecycle + an early signal-abort short-circuit), so resume/budget/
|
|
13
|
+
* abort apply uniformly. Composite verdicts are coerced (LLMs return "true"/"0"
|
|
14
|
+
* as strings) and JSON is parsed via the shared string-aware parseFirstJson.
|
|
15
|
+
*
|
|
16
|
+
* Timing note: `Date.now()` is used here for per-step duration stats only. This
|
|
17
|
+
* is engine code, NOT a workflow `.ts` body, so the Task 3 ast-guard does not
|
|
18
|
+
* apply; run identity (`now`) is still supplied deterministically by the caller.
|
|
19
|
+
* (maxDurationMs uses a live clock via Date.now() at the guard points — engine
|
|
20
|
+
* code is not AST-guarded — so wall-clock duration is enforced; maxTokens is
|
|
21
|
+
* enforced via BudgetPool.isExhausted on each spawn.)
|
|
22
|
+
*/
|
|
23
|
+
import {
|
|
24
|
+
assertBatchSize,
|
|
25
|
+
assertLifetimeAgents,
|
|
26
|
+
BudgetExceededError,
|
|
27
|
+
BudgetPool,
|
|
28
|
+
} from "../budget/index.ts";
|
|
29
|
+
import { computeCacheKey, type Journal } from "../cache/index.ts";
|
|
30
|
+
import { RETRYABLE_CATEGORIES, WorkflowError, type ErrorCategory } from "../errors.ts";
|
|
31
|
+
import { type AgentLifecycleListeners, notifyCacheHit, notifyLog, notifyUpdate } from "../lifecycle.ts";
|
|
32
|
+
import { parseFirstJson } from "../outcomes.ts";
|
|
33
|
+
import type {
|
|
34
|
+
AdversarialStep,
|
|
35
|
+
AgentCallSpec,
|
|
36
|
+
AgentOpts,
|
|
37
|
+
AgentStep,
|
|
38
|
+
ClassifyRouteStep,
|
|
39
|
+
CodeStep,
|
|
40
|
+
LoopUntilDryStep,
|
|
41
|
+
LogStep,
|
|
42
|
+
SubWorkflowStep,
|
|
43
|
+
FanOutStep,
|
|
44
|
+
LoopUntilStep,
|
|
45
|
+
RunStatus,
|
|
46
|
+
StepContext,
|
|
47
|
+
StepDefinition,
|
|
48
|
+
StepResult,
|
|
49
|
+
StepStats,
|
|
50
|
+
TournamentStep,
|
|
51
|
+
} from "../types.ts";
|
|
52
|
+
import type { AgentSpawnOptions, AgentSpawnRegistry, AgentSpawnResult } from "../agent/dispatch.ts";
|
|
53
|
+
import { mapWithConcurrencyLimit } from "../agent/dispatch.ts";
|
|
54
|
+
import { stepIdOf } from "../format.ts";
|
|
55
|
+
|
|
56
|
+
/** Injectable agent dispatch — same shape as spawnAgent. Default = real spawnAgent. */
|
|
57
|
+
export type AgentDispatch = (
|
|
58
|
+
registry: AgentSpawnRegistry,
|
|
59
|
+
opts: AgentSpawnOptions,
|
|
60
|
+
) => Promise<AgentSpawnResult>;
|
|
61
|
+
|
|
62
|
+
/** Shared, mutable execution state threaded through every step of a run. */
|
|
63
|
+
export interface StepExecContext {
|
|
64
|
+
readonly workflowName: string;
|
|
65
|
+
readonly dispatch: AgentDispatch;
|
|
66
|
+
readonly registry: AgentSpawnRegistry;
|
|
67
|
+
readonly journal: Journal;
|
|
68
|
+
readonly pool: BudgetPool;
|
|
69
|
+
readonly signal?: AbortSignal;
|
|
70
|
+
readonly listeners?: AgentLifecycleListeners;
|
|
71
|
+
/** Run inception time (ms), supplied deterministically by the caller — used
|
|
72
|
+
* for journal timestamps and as the child BudgetPool originMs. NOT used for
|
|
73
|
+
* budget duration checks (those read Date.now() at the guard points). */
|
|
74
|
+
readonly now: number;
|
|
75
|
+
/** Cumulative agents spawned so far this run (lifetime-cap counter). */
|
|
76
|
+
spawned: number;
|
|
77
|
+
/** Current classify_route nesting depth (cycle guard). */
|
|
78
|
+
depth: number;
|
|
79
|
+
/** A3: per-step budget-exhaustion policy. Set by runStepSequence from
|
|
80
|
+
* step.onBudgetExhaust before each step (default "throw"). */
|
|
81
|
+
budgetPolicy: "throw" | "null";
|
|
82
|
+
/** A6: max resolved-prompt byte size; oversize throws a size-limit error. */
|
|
83
|
+
maxPromptBytes: number;
|
|
84
|
+
/** A3: step ids that degraded to null under the "null" policy this run. */
|
|
85
|
+
degradedStepIds: Set<string>;
|
|
86
|
+
/** Recursion opt-in propagated to every spawned workflow sub-agent: when
|
|
87
|
+
* false (default) children load WITHOUT the subagent/fan-out tools.
|
|
88
|
+
* Opt-in re-enables them, bounded by any maxSpawnDepth cap. */
|
|
89
|
+
allowChildRecursion: boolean;
|
|
90
|
+
/** Non-cached dispatch attempts this run (null-degraded and dispatch-throw
|
|
91
|
+
* paths included) — the denominator for resume cache-hit accounting. */
|
|
92
|
+
dispatched: number;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Outcome of a dispatched agent call (shared by all step kinds that call agents). */
|
|
96
|
+
interface AgentCallOutcome {
|
|
97
|
+
readonly value: string | null;
|
|
98
|
+
readonly ok: boolean;
|
|
99
|
+
readonly aborted: boolean;
|
|
100
|
+
readonly cached: boolean;
|
|
101
|
+
readonly stats: StepStats;
|
|
102
|
+
/** A5: error category of this call's failure. dispatch-error marks a
|
|
103
|
+
* retryable agent failure (dispatch rejection OR a failed subprocess);
|
|
104
|
+
* absent for aborts and successes. */
|
|
105
|
+
readonly errorCategory?: ErrorCategory;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const MAX_ROUTE_DEPTH = 8;
|
|
109
|
+
|
|
110
|
+
/** Default per-prompt byte-size cap (A6). Overridable via RunWorkflowOptions. */
|
|
111
|
+
export const DEFAULT_MAX_PROMPT_BYTES = 256 * 1024;
|
|
112
|
+
|
|
113
|
+
/** Non-printable C0 controls + DEL, excluding the common whitespace (\t \n \r).
|
|
114
|
+
* Presence triggers a control-chars rejection (A6) — these almost always
|
|
115
|
+
* indicate corrupted/serialized binary data leaking into a prompt. */
|
|
116
|
+
// A6: reject prompts containing non-printable / format control characters before
|
|
117
|
+
// any spawn. Covers C0 (minus tab/LF/CR) + DEL + C1 (U+0080–U+009F) + zero-width
|
|
118
|
+
// and bidi format chars (U+200B–U+200F, U+202A–U+202E, U+2060–U+206F, U+FEFF) —
|
|
119
|
+
// the known prompt-injection vectors (bidi override, ZWJ/ZWNJ smuggling, BOM).
|
|
120
|
+
// The message below says "non-printable control characters" and the regex means
|
|
121
|
+
// it: previously only C0+DEL were blocked while bidi/zero-width sailed through.
|
|
122
|
+
const CONTROL_CHARS = /[\x00-\x08\x0B\x0C\x0E-\x1F\x7F\u0080-\u009F\u200B-\u200F\u202A-\u202E\u2060-\u206F\uFEFF]/;
|
|
123
|
+
|
|
124
|
+
/** Base workflow-subagent discipline prompt (B1+B2). Prepended to every agent
|
|
125
|
+
* dispatch's systemPrompt so the model knows it is a workflow subagent and
|
|
126
|
+
* returns the literal result without confirmations / fences / prose. Injected
|
|
127
|
+
* BEFORE cache-key computation (design D1) so it participates in the key. */
|
|
128
|
+
const BASE_WORKFLOW_SUBAGENT_PROMPT = [
|
|
129
|
+
"You are a subagent spawned by a deterministic workflow orchestration script.",
|
|
130
|
+
"Your final text response is returned verbatim as a string to the calling script \u2014 it is your return value, not a message to a human.",
|
|
131
|
+
"- Output the literal result (data, JSON, or text). Do NOT output confirmations like \"Done.\" or \"Sent.\"",
|
|
132
|
+
"- If asked for JSON, return ONLY the raw JSON \u2014 no code fences, no prose, no markdown.",
|
|
133
|
+
"- Be concise. The script will parse your output.",
|
|
134
|
+
].join("\n");
|
|
135
|
+
|
|
136
|
+
// ---------------------------------------------------------------------------
|
|
137
|
+
// Dispatch
|
|
138
|
+
// ---------------------------------------------------------------------------
|
|
139
|
+
|
|
140
|
+
export async function executeStep(
|
|
141
|
+
step: StepDefinition,
|
|
142
|
+
ctx: StepContext,
|
|
143
|
+
exec: StepExecContext,
|
|
144
|
+
): Promise<StepResult> {
|
|
145
|
+
switch (step.type) {
|
|
146
|
+
case "agent":
|
|
147
|
+
return execAgent(step, ctx, exec);
|
|
148
|
+
case "code":
|
|
149
|
+
return execCode(step, ctx);
|
|
150
|
+
case "log":
|
|
151
|
+
return execLog(step, ctx, exec);
|
|
152
|
+
case "fan_out":
|
|
153
|
+
return execFanOut(step, ctx, exec);
|
|
154
|
+
case "loop_until":
|
|
155
|
+
return execLoopUntil(step, ctx, exec);
|
|
156
|
+
case "adversarial":
|
|
157
|
+
return execAdversarial(step, ctx, exec);
|
|
158
|
+
case "tournament":
|
|
159
|
+
return execTournament(step, ctx, exec);
|
|
160
|
+
case "classify_route":
|
|
161
|
+
return execClassifyRoute(step, ctx, exec);
|
|
162
|
+
case "sub_workflow":
|
|
163
|
+
return execSubWorkflow(step, ctx, exec);
|
|
164
|
+
case "loop_until_dry":
|
|
165
|
+
return execLoopUntilDry(step, ctx, exec);
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// ---------------------------------------------------------------------------
|
|
170
|
+
// Shared agent dispatch — cache-resume + budget + spawn + journal + lifecycle
|
|
171
|
+
// ---------------------------------------------------------------------------
|
|
172
|
+
|
|
173
|
+
async function dispatchAgentCall(
|
|
174
|
+
callId: string,
|
|
175
|
+
prompt: string,
|
|
176
|
+
signature: AgentOpts,
|
|
177
|
+
exec: StepExecContext,
|
|
178
|
+
): Promise<AgentCallOutcome> {
|
|
179
|
+
// If the run is already aborted, do not spawn (avoids spawning processes that
|
|
180
|
+
// are killed immediately by the signal listener).
|
|
181
|
+
if (exec.signal?.aborted) {
|
|
182
|
+
return { value: "", ok: false, aborted: true, cached: false, stats: zeroStats };
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// D1: prepend the base workflow-subagent discipline prompt to the signature's
|
|
186
|
+
// systemPrompt BEFORE computing the cache key, so it participates in the key
|
|
187
|
+
// (injecting after the key would let runs under different base-prompt versions
|
|
188
|
+
// collide). A step override appends after the base (spec B1+B2).
|
|
189
|
+
const effectiveSignature: AgentOpts = signature.systemPrompt
|
|
190
|
+
? { ...signature, systemPrompt: `${BASE_WORKFLOW_SUBAGENT_PROMPT}\n\n${signature.systemPrompt}` }
|
|
191
|
+
: { ...signature, systemPrompt: BASE_WORKFLOW_SUBAGENT_PROMPT };
|
|
192
|
+
|
|
193
|
+
const key = computeCacheKey({ workflowName: exec.workflowName, prompt, signature: effectiveSignature });
|
|
194
|
+
|
|
195
|
+
const cached = exec.journal.lookup(key);
|
|
196
|
+
if (cached?.type === "result" && cached.ok) {
|
|
197
|
+
notifyCacheHit(exec.listeners, callId);
|
|
198
|
+
return { value: cached.value as string, ok: true, aborted: false, cached: true, stats: zeroStats };
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// A6: reject oversized / control-character payloads before any spawn. These
|
|
202
|
+
// run AFTER the cache-hit return so a config change (e.g. lowering
|
|
203
|
+
// maxPromptBytes between runs) cannot kill resume replay of a previously
|
|
204
|
+
// cached call — a replay spawns nothing. The checks gate NEW spawns and
|
|
205
|
+
// cover BOTH the task prompt and the effective systemPrompt (which always
|
|
206
|
+
// contains the base discipline prompt), so oversized/binary data cannot
|
|
207
|
+
// dodge the guard by being placed in the system prompt. These throw terminal
|
|
208
|
+
// WorkflowErrors (size-limit / control-chars) which propagate past
|
|
209
|
+
// runWithRetry's retry loop — only dispatch-error is retryable (A5:
|
|
210
|
+
// runWithRetry gates on RETRYABLE_CATEGORIES, not bare status). A rejection
|
|
211
|
+
// also aborts the step's in-flight siblings so a concurrent batch fails fast
|
|
212
|
+
// instead of the allSettled wait blocking on a stalled sibling.
|
|
213
|
+
assertPromptAllowed(exec, callId, prompt, effectiveSignature.systemPrompt);
|
|
214
|
+
|
|
215
|
+
const releaseSpawn = guardSpawn(exec, callId, 1);
|
|
216
|
+
if (releaseSpawn === null) {
|
|
217
|
+
// A3: budget exhausted under the "null" policy — degrade this call to null
|
|
218
|
+
// (guardSpawn already attributed the step to degradedStepIds). No spawn,
|
|
219
|
+
// no slot consumed, nothing journaled. Counted as a dispatch attempt so
|
|
220
|
+
// resume's cachedTotal covers the degraded path.
|
|
221
|
+
exec.dispatched++;
|
|
222
|
+
return { value: null, ok: true, aborted: false, cached: false, stats: zeroStats };
|
|
223
|
+
}
|
|
224
|
+
// Every non-cached call that passes the guard is a dispatch attempt (the
|
|
225
|
+
// dispatch-throw path increments here too — the start was announced even
|
|
226
|
+
// though no agent launched, so the total must not undercount it).
|
|
227
|
+
exec.dispatched++;
|
|
228
|
+
notifyStart(exec, callId);
|
|
229
|
+
await exec.journal.append({ type: "started", key, at: exec.now });
|
|
230
|
+
let res: AgentSpawnResult;
|
|
231
|
+
let durationMs: number;
|
|
232
|
+
try {
|
|
233
|
+
({ res, durationMs } = await timed(() =>
|
|
234
|
+
exec.dispatch(exec.registry, dispatchOpts(callId, prompt, effectiveSignature, exec.signal, exec.listeners, exec.allowChildRecursion)),
|
|
235
|
+
));
|
|
236
|
+
} catch (e) {
|
|
237
|
+
releaseSpawn(); // dispatch never started — return the reserved slot
|
|
238
|
+
// dispatch rejected — write a terminal result so resume sees closure (not
|
|
239
|
+
// an orphan `started`), per CC's "result always written" invariant. Returns
|
|
240
|
+
// a failed outcome; runWithRetry treats status 'failed' as retryable.
|
|
241
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
242
|
+
await exec.journal.append({ type: "result", key, at: exec.now, ok: false, value: `dispatch error: ${msg}` });
|
|
243
|
+
notifyEnd(exec, callId, false, zeroStats);
|
|
244
|
+
// Fail-fast for concurrent batches: a dispatch error is a real failure —
|
|
245
|
+
// terminate the step's in-flight siblings (SIGTERM via their per-call
|
|
246
|
+
// controllers) so mapWithConcurrencyLimit's allSettled wait does not hang
|
|
247
|
+
// forever on a stalled sibling subprocess.
|
|
248
|
+
abortStepCalls(exec, stepIdOf(callId));
|
|
249
|
+
return {
|
|
250
|
+
value: `dispatch error: ${msg}`,
|
|
251
|
+
ok: false,
|
|
252
|
+
aborted: false,
|
|
253
|
+
cached: false,
|
|
254
|
+
stats: zeroStats,
|
|
255
|
+
errorCategory: "dispatch-error",
|
|
256
|
+
};
|
|
257
|
+
}
|
|
258
|
+
applyOutcome(exec, res);
|
|
259
|
+
const value = finalText(res);
|
|
260
|
+
const ok = !res.aborted && res.exitCode === 0 && res.stopReason !== "error" && res.stopReason !== "aborted";
|
|
261
|
+
const stats = usageStats(res, durationMs, ok);
|
|
262
|
+
// On failure with no assistant output, surface the subprocess stderr /
|
|
263
|
+
// errorMessage / exitCode so the caller can diagnose WHY it failed
|
|
264
|
+
// (unknown model, missing provider, bad PATH, …). Without this the
|
|
265
|
+
// step result is just "" and the error is invisible.
|
|
266
|
+
const diag = ok || value ? value : diagnoseFailure(res);
|
|
267
|
+
await exec.journal.append({ type: "result", key, at: exec.now, ok, value: diag });
|
|
268
|
+
notifyEnd(exec, callId, ok, stats, res.model, diag);
|
|
269
|
+
// Fail-fast for concurrent batches: a settled-but-failed call (non-abort) is a
|
|
270
|
+
// real failure — terminate the step's in-flight siblings so the batch's
|
|
271
|
+
// allSettled wait cannot block forever on a stalled sibling subprocess. A
|
|
272
|
+
// degraded (budget-null) or aborted call is NOT a failure and must not abort
|
|
273
|
+
// its siblings.
|
|
274
|
+
if (!ok && !res.aborted) abortStepCalls(exec, stepIdOf(callId));
|
|
275
|
+
// A failed-but-settled subprocess (exitCode≠0 / stopReason "error" / killed)
|
|
276
|
+
// is a retryable dispatch-error — typically a transient provider/model
|
|
277
|
+
// failure. Aborts (external cancel, skip, maxTurns) carry no category: they
|
|
278
|
+
// settle as skipped and must not auto-retry.
|
|
279
|
+
return {
|
|
280
|
+
value: diag,
|
|
281
|
+
ok,
|
|
282
|
+
aborted: res.aborted,
|
|
283
|
+
cached: false,
|
|
284
|
+
stats,
|
|
285
|
+
errorCategory: ok || res.aborted ? undefined : "dispatch-error",
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/** Build a human-readable failure reason from a subprocess result that
|
|
290
|
+
* produced no assistant text. Surfaces stderr / errorMessage / exitCode. */
|
|
291
|
+
function diagnoseFailure(res: AgentSpawnResult): string {
|
|
292
|
+
const parts: string[] = ["[agent failed"];
|
|
293
|
+
if (res.exitCode !== 0) parts.push(`exit ${res.exitCode}`);
|
|
294
|
+
if (res.stopReason) parts.push(`stop:${res.stopReason}`);
|
|
295
|
+
if (res.errorMessage) parts.push(res.errorMessage);
|
|
296
|
+
// Last few non-empty stderr lines are usually the real cause.
|
|
297
|
+
const stderrTail = res.stderr.trim().split("\n").filter(Boolean).slice(-4).join(" | ");
|
|
298
|
+
if (stderrTail) parts.push(stderrTail);
|
|
299
|
+
parts.push("]");
|
|
300
|
+
return parts.join(" ");
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
// ---------------------------------------------------------------------------
|
|
304
|
+
// agent
|
|
305
|
+
// ---------------------------------------------------------------------------
|
|
306
|
+
|
|
307
|
+
async function execAgent(step: AgentStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
308
|
+
const prompt = typeof step.prompt === "function" ? await step.prompt(ctx) : step.prompt;
|
|
309
|
+
const callId = `${step.id}#${exec.spawned + 1}`;
|
|
310
|
+
const outcome = await dispatchAgentCall(callId, prompt, step, exec);
|
|
311
|
+
const status = outcome.ok ? "done" : outcome.aborted ? "skipped" : "failed";
|
|
312
|
+
return stepResult(step.id, "agent", status, outcome.value, outcome.stats, undefined, outcome.errorCategory);
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
// ---------------------------------------------------------------------------
|
|
316
|
+
// code (pure transform — no dispatch, no cache, not budgeted)
|
|
317
|
+
// ---------------------------------------------------------------------------
|
|
318
|
+
|
|
319
|
+
async function execCode(step: CodeStep, ctx: StepContext): Promise<StepResult> {
|
|
320
|
+
const start = Date.now();
|
|
321
|
+
try {
|
|
322
|
+
const value = await step.transform(ctx);
|
|
323
|
+
return stepResult(step.id, "code", "done", value, {
|
|
324
|
+
tokens: 0,
|
|
325
|
+
cost: 0,
|
|
326
|
+
durationMs: Date.now() - start,
|
|
327
|
+
agents: 0,
|
|
328
|
+
failures: 0,
|
|
329
|
+
});
|
|
330
|
+
} catch (e) {
|
|
331
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
332
|
+
// Surface the error message (runStepSequence only appends sr.results when
|
|
333
|
+
// it's a non-empty string); a bare catch left the failure cause invisible.
|
|
334
|
+
return stepResult(step.id, "code", "failed", `code error: ${msg}`, {
|
|
335
|
+
tokens: 0,
|
|
336
|
+
cost: 0,
|
|
337
|
+
durationMs: Date.now() - start,
|
|
338
|
+
agents: 0,
|
|
339
|
+
failures: 1,
|
|
340
|
+
});
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
// log (C2): emit a narrative line via the onLog listener. Pure string, zero
|
|
345
|
+
// dispatch/tokens, not cached — same profile as `code` but fires onLog and the
|
|
346
|
+
// widget renders it as a distinct narrative line.
|
|
347
|
+
async function execLog(step: LogStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
348
|
+
const start = Date.now();
|
|
349
|
+
// A `log` step is non-critical narrative — a throwing message function (e.g.
|
|
350
|
+
// referencing a missing upstream field) must NOT crash the whole run. Fall
|
|
351
|
+
// back to an inline error marker and keep status "done" so the run continues.
|
|
352
|
+
let message: string;
|
|
353
|
+
try {
|
|
354
|
+
message = typeof step.message === "function" ? await step.message(ctx) : step.message;
|
|
355
|
+
} catch (e) {
|
|
356
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
357
|
+
message = `[log error: ${msg}]`;
|
|
358
|
+
}
|
|
359
|
+
notifyLog(exec.listeners, step.id, message);
|
|
360
|
+
return stepResult(step.id, "log", "done", message, {
|
|
361
|
+
tokens: 0,
|
|
362
|
+
cost: 0,
|
|
363
|
+
durationMs: Date.now() - start,
|
|
364
|
+
agents: 0,
|
|
365
|
+
failures: 0,
|
|
366
|
+
});
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
// ---------------------------------------------------------------------------
|
|
370
|
+
// fan_out (parallel agents over a list)
|
|
371
|
+
// ---------------------------------------------------------------------------
|
|
372
|
+
|
|
373
|
+
async function execFanOut(step: FanOutStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
374
|
+
const items = [...step.over(ctx)];
|
|
375
|
+
const n = items.length;
|
|
376
|
+
guardBatch(exec, n, `fan_out "${step.id}"`);
|
|
377
|
+
|
|
378
|
+
const parallelism = step.parallelism ?? Math.min(n, 8);
|
|
379
|
+
const itemSpecs = items.map((item, i) => step.agent(item, i, ctx));
|
|
380
|
+
const start = Date.now();
|
|
381
|
+
|
|
382
|
+
const outcomes = await mapWithConcurrencyLimit(itemSpecs, parallelism, (spec, i) =>
|
|
383
|
+
dispatchAgentCall(`${step.id}#${i + 1}`, spec.prompt, spec, exec),
|
|
384
|
+
);
|
|
385
|
+
|
|
386
|
+
const durationMs = Date.now() - start;
|
|
387
|
+
const merged = step.merge ? await step.merge(outcomes.map((o) => o.value), ctx) : outcomes.map((o) => o.value);
|
|
388
|
+
const stats = aggregateStats(outcomes.map((o) => o.stats), durationMs);
|
|
389
|
+
const status = outcomesStatus(outcomes);
|
|
390
|
+
// A5: a batch whose failures are all retryable dispatch-errors carries the
|
|
391
|
+
// category so runWithRetry can retry the whole fan_out (successful items
|
|
392
|
+
// replay as cache hits and are not re-charged). Aborted-only batches settle
|
|
393
|
+
// as skipped and carry none.
|
|
394
|
+
const errorCategory = status === "failed" && outcomes.some((o) => o.errorCategory === "dispatch-error") ? "dispatch-error" : undefined;
|
|
395
|
+
return stepResult(step.id, "fan_out", status, merged, stats, undefined, errorCategory);
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
// ---------------------------------------------------------------------------
|
|
399
|
+
// loop_until (iterate a body agent until / maxIterations / budget)
|
|
400
|
+
// ---------------------------------------------------------------------------
|
|
401
|
+
|
|
402
|
+
async function execLoopUntil(step: LoopUntilStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
403
|
+
const maxIter = step.maxIterations ?? 10;
|
|
404
|
+
const start = Date.now();
|
|
405
|
+
const values: unknown[] = [];
|
|
406
|
+
let stats = zeroStats;
|
|
407
|
+
let status: StepResult["status"] = "done";
|
|
408
|
+
let iter = 0;
|
|
409
|
+
|
|
410
|
+
while (iter < maxIter) {
|
|
411
|
+
if (exec.signal?.aborted) {
|
|
412
|
+
status = "skipped";
|
|
413
|
+
break;
|
|
414
|
+
}
|
|
415
|
+
const prompt = await step.prompt(ctx, iter);
|
|
416
|
+
const outcome = await dispatchAgentCall(`${step.id}#${iter + 1}`, prompt, step, exec);
|
|
417
|
+
stats = addStats(stats, outcome.stats);
|
|
418
|
+
values.push(outcome.value);
|
|
419
|
+
if (!outcome.ok) {
|
|
420
|
+
status = outcome.aborted ? "skipped" : "failed";
|
|
421
|
+
break;
|
|
422
|
+
}
|
|
423
|
+
iter++;
|
|
424
|
+
if (step.until(ctx, iter)) break;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
return stepResult(step.id, "loop_until", status, values, withDuration(stats, start), iter);
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
// ---------------------------------------------------------------------------
|
|
431
|
+
// adversarial (produce one candidate, N judges grade it, tally)
|
|
432
|
+
// ---------------------------------------------------------------------------
|
|
433
|
+
|
|
434
|
+
/** Judge call opts inherit the produce opts (model/tools/systemPrompt) and are
|
|
435
|
+
* overridden per-field by an explicit `step.judge`. Without this, a step-level
|
|
436
|
+
* `model` applied only to the produce call and judges silently fell back to the
|
|
437
|
+
* default model (review m14). */
|
|
438
|
+
function judgeOpts(produce: AgentOpts, judge?: AgentOpts): AgentOpts {
|
|
439
|
+
return {
|
|
440
|
+
model: judge?.model ?? produce.model,
|
|
441
|
+
tools: judge?.tools ?? produce.tools,
|
|
442
|
+
systemPrompt: judge?.systemPrompt ?? produce.systemPrompt,
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
async function execAdversarial(step: AdversarialStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
447
|
+
const judgeCount = step.judges ?? 3;
|
|
448
|
+
if (judgeCount < 1) throw new Error(`adversarial "${step.id}" requires judges >= 1`);
|
|
449
|
+
const minPass = step.minPass ?? Math.ceil(judgeCount / 2);
|
|
450
|
+
guardBatch(exec, 1 + judgeCount, `adversarial "${step.id}"`);
|
|
451
|
+
const start = Date.now();
|
|
452
|
+
let stats = zeroStats;
|
|
453
|
+
|
|
454
|
+
const producePrompt = await resolvePrompt(step.produce, ctx);
|
|
455
|
+
const candidate = await dispatchAgentCall(`${step.id}#produce`, producePrompt, step.produce, exec);
|
|
456
|
+
stats = addStats(stats, candidate.stats);
|
|
457
|
+
|
|
458
|
+
// A3: a degraded produce (budget exhausted under the "null" policy) returns
|
|
459
|
+
// value null — the step degrades to a null result per the README contract,
|
|
460
|
+
// NOT a fabricated "done" with an empty candidate. No judges are dispatched:
|
|
461
|
+
// the budget is exhausted, they would only degrade too.
|
|
462
|
+
if (candidate.value === null) {
|
|
463
|
+
return stepResult(step.id, "adversarial", "done", null, withDuration(stats, start), 1);
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
const outcomes: AgentCallOutcome[] = [candidate];
|
|
467
|
+
const judges: { pass: boolean; reason: string }[] = [];
|
|
468
|
+
if (candidate.ok) {
|
|
469
|
+
const judgeSpecs = Array.from({ length: judgeCount }, (_, i) => judgePrompt(i, candidate.value ?? "", step.rubric));
|
|
470
|
+
const judgeOutcomes = await mapWithConcurrencyLimit(judgeSpecs, Math.min(judgeCount, 8), (prompt, i) =>
|
|
471
|
+
dispatchAgentCall(`${step.id}#judge${i + 1}`, prompt, judgeOpts(step.produce, step.judge), exec),
|
|
472
|
+
);
|
|
473
|
+
for (const o of judgeOutcomes) {
|
|
474
|
+
stats = addStats(stats, o.stats);
|
|
475
|
+
outcomes.push(o);
|
|
476
|
+
const parsed = parseFirstJson(o.value ?? "") as { pass?: unknown; reason?: unknown } | undefined;
|
|
477
|
+
judges.push({ pass: parsePassBool(parsed?.pass), reason: parseReason(parsed?.reason) });
|
|
478
|
+
}
|
|
479
|
+
} else {
|
|
480
|
+
// populate judge slots so the result shape is stable on produce failure
|
|
481
|
+
for (let i = 0; i < judgeCount; i++) judges.push({ pass: false, reason: "produce failed" });
|
|
482
|
+
}
|
|
483
|
+
const passCount = judges.filter((j) => j.pass).length;
|
|
484
|
+
|
|
485
|
+
return stepResult(
|
|
486
|
+
step.id,
|
|
487
|
+
"adversarial",
|
|
488
|
+
outcomesStatus(outcomes),
|
|
489
|
+
{ candidate: candidate.value ?? "", passed: passCount >= minPass, passCount, minPass, judges },
|
|
490
|
+
withDuration(stats, start),
|
|
491
|
+
1 + judgeCount,
|
|
492
|
+
);
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
// ---------------------------------------------------------------------------
|
|
496
|
+
// tournament (N distinct candidates, M judges rank, pick winner)
|
|
497
|
+
// ---------------------------------------------------------------------------
|
|
498
|
+
|
|
499
|
+
async function execTournament(step: TournamentStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
500
|
+
if (step.candidates < 1) throw new Error(`tournament "${step.id}" requires candidates >= 1`);
|
|
501
|
+
if (step.judges < 1) throw new Error(`tournament "${step.id}" requires judges >= 1`);
|
|
502
|
+
guardBatch(exec, step.candidates + step.judges, `tournament "${step.id}"`);
|
|
503
|
+
const start = Date.now();
|
|
504
|
+
let stats = zeroStats;
|
|
505
|
+
const outcomes: AgentCallOutcome[] = [];
|
|
506
|
+
|
|
507
|
+
const producePrompt = await resolvePrompt(step.produce, ctx);
|
|
508
|
+
const candSpecs = Array.from({ length: step.candidates }, (_, i) => ({
|
|
509
|
+
prompt: `${producePrompt}\n\nAttempt ${i + 1}: take a distinct approach.`,
|
|
510
|
+
}));
|
|
511
|
+
const candOutcomes = await mapWithConcurrencyLimit(candSpecs, Math.min(step.candidates, 8), (spec, i) =>
|
|
512
|
+
dispatchAgentCall(`${step.id}#cand${i + 1}`, spec.prompt, step.produce, exec),
|
|
513
|
+
);
|
|
514
|
+
const candidates = candOutcomes.map((o) => {
|
|
515
|
+
stats = addStats(stats, o.stats);
|
|
516
|
+
outcomes.push(o);
|
|
517
|
+
return o.value ?? "";
|
|
518
|
+
});
|
|
519
|
+
|
|
520
|
+
const judgeSpecs = Array.from({ length: step.judges }, (_, j) => rankPrompt(j, candidates));
|
|
521
|
+
const judgeOutcomes = await mapWithConcurrencyLimit(judgeSpecs, Math.min(step.judges, 8), (prompt, j) =>
|
|
522
|
+
dispatchAgentCall(`${step.id}#judge${j + 1}`, prompt, judgeOpts(step.produce, step.judge), exec),
|
|
523
|
+
);
|
|
524
|
+
const judgePicks = judgeOutcomes.map((o) => {
|
|
525
|
+
stats = addStats(stats, o.stats);
|
|
526
|
+
outcomes.push(o);
|
|
527
|
+
const parsed = parseFirstJson(o.value ?? "") as { winner?: unknown; reason?: unknown } | undefined;
|
|
528
|
+
return { winner: parseWinnerNum(parsed?.winner), reason: parseReason(parsed?.reason) };
|
|
529
|
+
});
|
|
530
|
+
const winner = tallyWinner(
|
|
531
|
+
judgePicks.map((j) => j.winner),
|
|
532
|
+
step.candidates,
|
|
533
|
+
);
|
|
534
|
+
|
|
535
|
+
// A5: like fan_out, a tournament whose failures are retryable dispatch-errors
|
|
536
|
+
// carries the category so runWithRetry can retry the whole composite.
|
|
537
|
+
const status = outcomesStatus(outcomes);
|
|
538
|
+
const errorCategory = status === "failed" && outcomes.some((o) => o.errorCategory === "dispatch-error") ? "dispatch-error" : undefined;
|
|
539
|
+
|
|
540
|
+
return stepResult(
|
|
541
|
+
step.id,
|
|
542
|
+
"tournament",
|
|
543
|
+
status,
|
|
544
|
+
{ candidates, winner, judges: judgePicks },
|
|
545
|
+
withDuration(stats, start),
|
|
546
|
+
step.candidates + step.judges,
|
|
547
|
+
errorCategory,
|
|
548
|
+
);
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
// ---------------------------------------------------------------------------
|
|
552
|
+
// classify_route (classify → run the matching route's sub-steps)
|
|
553
|
+
// ---------------------------------------------------------------------------
|
|
554
|
+
|
|
555
|
+
async function execClassifyRoute(step: ClassifyRouteStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
556
|
+
const start = Date.now();
|
|
557
|
+
const classifyPrompt = await resolvePrompt(step.classifier, ctx);
|
|
558
|
+
const classify = await dispatchAgentCall(`${step.id}#classify`, classifyPrompt, step.classifier, exec);
|
|
559
|
+
if (!classify.ok) {
|
|
560
|
+
return stepResult(
|
|
561
|
+
step.id,
|
|
562
|
+
"classify_route",
|
|
563
|
+
classify.aborted ? "skipped" : "failed",
|
|
564
|
+
undefined,
|
|
565
|
+
withDuration(classify.stats, start),
|
|
566
|
+
undefined,
|
|
567
|
+
classify.aborted ? undefined : "dispatch-error",
|
|
568
|
+
);
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
// A3: a degraded classifier (budget exhausted under the "null" policy)
|
|
572
|
+
// returns value null — the step degrades to a null result per the README
|
|
573
|
+
// contract; do not fabricate a route run from an empty category.
|
|
574
|
+
if (classify.value === null) {
|
|
575
|
+
return stepResult(step.id, "classify_route", "done", null, withDuration(classify.stats, start));
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
const parsed = parseFirstJson(classify.value ?? "") as { category?: unknown } | undefined;
|
|
579
|
+
const category = parseCategoryStr(parsed?.category);
|
|
580
|
+
const routeSteps = step.routes[category] ?? step.fallback ?? [];
|
|
581
|
+
|
|
582
|
+
exec.depth++;
|
|
583
|
+
let sub: SequenceOutcome;
|
|
584
|
+
try {
|
|
585
|
+
sub = await runStepSequence(routeSteps, ctx.input, exec);
|
|
586
|
+
} finally {
|
|
587
|
+
exec.depth--;
|
|
588
|
+
}
|
|
589
|
+
const status: StepResult["status"] = sub.status === "completed" ? "done" : sub.status === "aborted" ? "skipped" : "failed";
|
|
590
|
+
|
|
591
|
+
return stepResult(
|
|
592
|
+
step.id,
|
|
593
|
+
"classify_route",
|
|
594
|
+
status,
|
|
595
|
+
{ category, matched: category in step.routes, route: sub.steps, routeStatus: sub.status },
|
|
596
|
+
withDuration(addStats(classify.stats, aggregateStats(sub.steps.map((s) => s.stats), 0)), start),
|
|
597
|
+
undefined,
|
|
598
|
+
sub.errorCategory,
|
|
599
|
+
);
|
|
600
|
+
}
|
|
601
|
+
|
|
602
|
+
// ---------------------------------------------------------------------------
|
|
603
|
+
// sub_workflow (nested child workflow — CC's workflow() pattern)
|
|
604
|
+
// ---------------------------------------------------------------------------
|
|
605
|
+
|
|
606
|
+
async function execSubWorkflow(step: SubWorkflowStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
607
|
+
const start = Date.now();
|
|
608
|
+
// Resolve the child's input: static value or function of parent ctx. The
|
|
609
|
+
// function form is awaited (and typed to allow a Promise), matching the
|
|
610
|
+
// async convention of CodeStep.transform / LogStep.message — an un-awaited
|
|
611
|
+
// Promise would leak into the child's ctx.input as "[object Promise]" with
|
|
612
|
+
// its rejection silently dropped.
|
|
613
|
+
const childInput = typeof step.input === "function"
|
|
614
|
+
? await (step.input as (c: StepContext) => unknown | Promise<unknown>)(ctx)
|
|
615
|
+
: step.input ?? ctx.input;
|
|
616
|
+
|
|
617
|
+
// Child shares parent's journal, pool, registry, and signal by default.
|
|
618
|
+
// inheritBudget: false means the child uses its own BudgetPool (isolated caps).
|
|
619
|
+
// Override workflowName with the CHILD's name so cache keys are scoped to the
|
|
620
|
+
// child workflow — otherwise two sibling sub_workflows whose agents share a
|
|
621
|
+
// prompt+signature collide on the parent's name and replay each other's cache.
|
|
622
|
+
const childExec: StepExecContext = step.inheritBudget === false
|
|
623
|
+
? { ...exec, workflowName: step.workflow.name, depth: exec.depth + 1, pool: new BudgetPool(step.workflow.budget ?? {}, Date.now()) }
|
|
624
|
+
: { ...exec, workflowName: step.workflow.name, depth: exec.depth + 1 };
|
|
625
|
+
|
|
626
|
+
let sub: SequenceOutcome;
|
|
627
|
+
if (childExec.depth > MAX_ROUTE_DEPTH) {
|
|
628
|
+
sub = { steps: [], status: "failed", error: `sub_workflow nesting exceeded depth ${MAX_ROUTE_DEPTH}` };
|
|
629
|
+
} else {
|
|
630
|
+
sub = await runStepSequence(step.workflow.steps, childInput, childExec);
|
|
631
|
+
}
|
|
632
|
+
// Propagate the child's lifetime counter back to the parent: childExec is a
|
|
633
|
+
// spread copy, so without this sync the parent's `spawned` undercounts every
|
|
634
|
+
// agent the child spawned (callId numbering + the assertLifetimeAgents backstop
|
|
635
|
+
// in guardSpawn both depend on this being run-cumulative, per its docstring).
|
|
636
|
+
exec.spawned = childExec.spawned;
|
|
637
|
+
|
|
638
|
+
const status: StepResult["status"] = sub.status === "completed" ? "done"
|
|
639
|
+
: sub.status === "aborted" ? "skipped"
|
|
640
|
+
: "failed";
|
|
641
|
+
|
|
642
|
+
return stepResult(
|
|
643
|
+
step.id,
|
|
644
|
+
"sub_workflow",
|
|
645
|
+
status,
|
|
646
|
+
{ steps: sub.steps, status: sub.status, workflowName: step.workflow.name, error: sub.error },
|
|
647
|
+
withDuration(aggregateStats(sub.steps.map((s) => s.stats), 0), start),
|
|
648
|
+
undefined,
|
|
649
|
+
sub.errorCategory,
|
|
650
|
+
);
|
|
651
|
+
}
|
|
652
|
+
|
|
653
|
+
// ---------------------------------------------------------------------------
|
|
654
|
+
// loop_until_dry (keep discovering until K rounds return nothing new)
|
|
655
|
+
// ---------------------------------------------------------------------------
|
|
656
|
+
|
|
657
|
+
async function execLoopUntilDry(step: LoopUntilDryStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
658
|
+
const maxRounds = step.maxRounds ?? 10;
|
|
659
|
+
const dryThreshold = step.dryThreshold ?? 2;
|
|
660
|
+
const keyOf = step.keyOf;
|
|
661
|
+
const merge = step.merge ?? ((known: unknown[], fresh: unknown[]) => known.concat(fresh));
|
|
662
|
+
const start = Date.now();
|
|
663
|
+
let known: unknown[] = [];
|
|
664
|
+
let stats = zeroStats;
|
|
665
|
+
let dry = 0;
|
|
666
|
+
let round = 0;
|
|
667
|
+
let status: StepResult["status"] = "done";
|
|
668
|
+
|
|
669
|
+
while (round < maxRounds) {
|
|
670
|
+
if (exec.signal?.aborted) { status = "skipped"; break; }
|
|
671
|
+
const prompt = await step.prompt(ctx, known);
|
|
672
|
+
const outcome = await dispatchAgentCall(`${step.id}#r${round + 1}`, prompt, {}, exec);
|
|
673
|
+
stats = addStats(stats, outcome.stats);
|
|
674
|
+
if (!outcome.ok) {
|
|
675
|
+
status = outcome.aborted ? "skipped" : "failed";
|
|
676
|
+
break;
|
|
677
|
+
}
|
|
678
|
+
const parsed = parseFirstJson(outcome.value ?? "") as unknown;
|
|
679
|
+
const freshItems: unknown[] = Array.isArray(parsed) ? parsed : parsed !== undefined && parsed !== null ? [parsed] : [];
|
|
680
|
+
const seen = new Set(known.map(keyOf));
|
|
681
|
+
const novel = freshItems.filter((item) => !seen.has(keyOf(item)));
|
|
682
|
+
if (novel.length === 0) {
|
|
683
|
+
dry++;
|
|
684
|
+
if (dry >= dryThreshold) {
|
|
685
|
+
// Completeness critic: ask "what's missing?" one last time.
|
|
686
|
+
if (step.critic) {
|
|
687
|
+
const criticPrompt = await step.critic.prompt(ctx, known);
|
|
688
|
+
const criticOutcome = await dispatchAgentCall(`${step.id}#critic`, criticPrompt, {}, exec);
|
|
689
|
+
stats = addStats(stats, criticOutcome.stats);
|
|
690
|
+
if (criticOutcome.ok) {
|
|
691
|
+
const criticParsed = parseFirstJson(criticOutcome.value ?? "") as unknown;
|
|
692
|
+
const criticItems: unknown[] = Array.isArray(criticParsed) ? criticParsed : [];
|
|
693
|
+
const criticNovel = criticItems.filter((item) => !seen.has(keyOf(item)));
|
|
694
|
+
if (criticNovel.length > 0) {
|
|
695
|
+
known = merge(known, criticNovel);
|
|
696
|
+
dry = 0; // reset dry counter and keep going
|
|
697
|
+
round++;
|
|
698
|
+
continue;
|
|
699
|
+
}
|
|
700
|
+
}
|
|
701
|
+
}
|
|
702
|
+
break;
|
|
703
|
+
}
|
|
704
|
+
} else {
|
|
705
|
+
dry = 0;
|
|
706
|
+
known = merge(known, novel);
|
|
707
|
+
}
|
|
708
|
+
round++;
|
|
709
|
+
}
|
|
710
|
+
|
|
711
|
+
return stepResult(step.id, "loop_until_dry", status, known, withDuration(stats, start), round);
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
// ---------------------------------------------------------------------------
|
|
715
|
+
// runStepSequence — run a list of steps (top-level workflow OR a classify route)
|
|
716
|
+
// ---------------------------------------------------------------------------
|
|
717
|
+
|
|
718
|
+
export interface SequenceOutcome {
|
|
719
|
+
readonly steps: readonly StepResult[];
|
|
720
|
+
readonly status: RunStatus;
|
|
721
|
+
readonly error?: string;
|
|
722
|
+
/** A5: category of the terminal error that failed the sequence (from a
|
|
723
|
+
* thrown WorkflowError), surfaced on RunResult.errorCategory. */
|
|
724
|
+
readonly errorCategory?: ErrorCategory;
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
export async function runStepSequence(
|
|
728
|
+
steps: readonly StepDefinition[],
|
|
729
|
+
input: unknown,
|
|
730
|
+
exec: StepExecContext,
|
|
731
|
+
): Promise<SequenceOutcome> {
|
|
732
|
+
if (exec.depth > MAX_ROUTE_DEPTH) {
|
|
733
|
+
return { steps: [], status: "failed", error: `classify_route nesting exceeded depth ${MAX_ROUTE_DEPTH} (cycle?)` };
|
|
734
|
+
}
|
|
735
|
+
const prior = new Map<string, { results: unknown; stats: StepStats }>();
|
|
736
|
+
const out: StepResult[] = [];
|
|
737
|
+
const ctx: StepContext = {
|
|
738
|
+
input,
|
|
739
|
+
step(id: string) {
|
|
740
|
+
const p = prior.get(id);
|
|
741
|
+
if (!p) throw new Error(`step "${id}" has not executed yet (or does not exist)`);
|
|
742
|
+
return p;
|
|
743
|
+
},
|
|
744
|
+
};
|
|
745
|
+
|
|
746
|
+
for (const step of steps) {
|
|
747
|
+
if (exec.signal?.aborted) return { steps: out, status: "aborted", error: "aborted by signal" };
|
|
748
|
+
// A3: apply the step's budget-exhaustion policy for the duration of this step.
|
|
749
|
+
exec.budgetPolicy = step.onBudgetExhaust ?? "throw";
|
|
750
|
+
let sr: StepResult;
|
|
751
|
+
try {
|
|
752
|
+
sr = await runWithRetry(step, ctx, exec);
|
|
753
|
+
} catch (e) {
|
|
754
|
+
const wfe = e instanceof WorkflowError ? e : undefined;
|
|
755
|
+
return {
|
|
756
|
+
steps: out,
|
|
757
|
+
status: "failed",
|
|
758
|
+
error: e instanceof Error ? e.message : String(e),
|
|
759
|
+
errorCategory: wfe?.category,
|
|
760
|
+
};
|
|
761
|
+
}
|
|
762
|
+
prior.set(step.id, { results: sr.results, stats: sr.stats });
|
|
763
|
+
out.push(sr);
|
|
764
|
+
// Post-step signal check: a run aborted mid-step (e.g. a fan_out whose
|
|
765
|
+
// items all aborted) must never report "completed".
|
|
766
|
+
if (exec.signal?.aborted) return { steps: out, status: "aborted", error: "aborted by signal" };
|
|
767
|
+
if (sr.status === "failed") {
|
|
768
|
+
const why = typeof sr.results === "string" && sr.results ? `: ${sr.results}` : "";
|
|
769
|
+
return { steps: out, status: "failed", error: `step "${step.id}" failed${why}`, errorCategory: sr.errorCategory };
|
|
770
|
+
}
|
|
771
|
+
if (sr.status === "skipped") {
|
|
772
|
+
const aborted = !!exec.signal?.aborted;
|
|
773
|
+
return { steps: out, status: aborted ? "aborted" : "failed", error: aborted ? "aborted by signal" : `step "${step.id}" skipped` };
|
|
774
|
+
}
|
|
775
|
+
}
|
|
776
|
+
return { steps: out, status: "completed" };
|
|
777
|
+
}
|
|
778
|
+
|
|
779
|
+
async function runWithRetry(step: StepDefinition, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
|
|
780
|
+
let sr = await executeStep(step, ctx, exec);
|
|
781
|
+
const max = step.retry?.maxRetries ?? 0;
|
|
782
|
+
let attempt = 0;
|
|
783
|
+
let stats = sr.stats; // accumulate every attempt's stats (the pool is charged per dispatch)
|
|
784
|
+
// A5: retry is gated by error category, not by bare status. Only retryable
|
|
785
|
+
// categories (dispatch-error / unexpected-state) auto-retry — a code
|
|
786
|
+
// transform failure or a terminal category (size-limit, determinism, …) must
|
|
787
|
+
// not burn budget on a deterministic re-run.
|
|
788
|
+
while (sr.status === "failed" && attempt < max && sr.errorCategory !== undefined && RETRYABLE_CATEGORIES.includes(sr.errorCategory)) {
|
|
789
|
+
if (exec.signal?.aborted) break;
|
|
790
|
+
attempt++;
|
|
791
|
+
sr = await executeStep(step, ctx, exec);
|
|
792
|
+
stats = addStats(stats, sr.stats);
|
|
793
|
+
}
|
|
794
|
+
return stats === sr.stats ? sr : { ...sr, stats };
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
// ---------------------------------------------------------------------------
|
|
798
|
+
// prompt / verdict coercion helpers
|
|
799
|
+
// ---------------------------------------------------------------------------
|
|
800
|
+
|
|
801
|
+
async function resolvePrompt(spec: AgentCallSpec, ctx: StepContext): Promise<string> {
|
|
802
|
+
return typeof spec.prompt === "function" ? spec.prompt(ctx) : spec.prompt;
|
|
803
|
+
}
|
|
804
|
+
|
|
805
|
+
function judgePrompt(index: number, candidate: string, rubric: readonly string[]): string {
|
|
806
|
+
const criteria = rubric.map((r, i) => `${i + 1}. ${r}`).join("\n");
|
|
807
|
+
// The base workflow-subagent systemPrompt already carries the verbatim / raw-JSON
|
|
808
|
+
// discipline, so this task prompt states only the task + the JSON schema it wants.
|
|
809
|
+
return `You are judge ${index + 1}. Evaluate this candidate:\n\n${candidate}\n\nAgainst these criteria:\n${criteria}\n\nReturn JSON matching: {"pass": true|false, "reason": "..."}`;
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
function rankPrompt(index: number, candidates: readonly string[]): string {
|
|
813
|
+
const listing = candidates.map((c, i) => `[${i}] ${c}`).join("\n\n");
|
|
814
|
+
// Base systemPrompt carries the verbatim / raw-JSON discipline; task prompt
|
|
815
|
+
// states only the ranking task + the JSON schema it wants.
|
|
816
|
+
return `You are judge ${index + 1}. Rank these candidates:\n\n${listing}\n\nReturn JSON matching: {"winner": <index>, "reason": "..."}`;
|
|
817
|
+
}
|
|
818
|
+
|
|
819
|
+
/** LLMs sometimes stringify booleans/numbers ("true", "0"); coerce leniently. */
|
|
820
|
+
function parsePassBool(v: unknown): boolean {
|
|
821
|
+
if (typeof v === "string") {
|
|
822
|
+
const s = v.trim().toLowerCase();
|
|
823
|
+
return s === "true" || s === "yes" || s === "y" || s === "1";
|
|
824
|
+
}
|
|
825
|
+
return v === true || v === 1;
|
|
826
|
+
}
|
|
827
|
+
function parseWinnerNum(v: unknown): number | undefined {
|
|
828
|
+
if (typeof v === "number") return Number.isFinite(v) ? Math.trunc(v) : undefined;
|
|
829
|
+
// Number("") / Number(null) === 0 would record a spurious vote for candidate 0.
|
|
830
|
+
if (typeof v !== "string" || !v.trim()) return undefined;
|
|
831
|
+
const n = Number(v);
|
|
832
|
+
return Number.isFinite(n) ? Math.trunc(n) : undefined;
|
|
833
|
+
}
|
|
834
|
+
function parseCategoryStr(v: unknown): string {
|
|
835
|
+
// Trim: LLMs often emit {"category": " bug"} with stray whitespace, which
|
|
836
|
+
// would silently miss the route key and fall through to fallback.
|
|
837
|
+
return typeof v === "string" ? v.trim() : v === undefined || v === null ? "" : String(v).trim();
|
|
838
|
+
}
|
|
839
|
+
function parseReason(v: unknown): string {
|
|
840
|
+
return typeof v === "string" ? v : "";
|
|
841
|
+
}
|
|
842
|
+
|
|
843
|
+
function tallyWinner(picks: readonly (number | undefined)[], n: number): number {
|
|
844
|
+
if (n <= 0) return -1;
|
|
845
|
+
const counts = new Array(n).fill(0);
|
|
846
|
+
for (const p of picks) if (typeof p === "number" && p >= 0 && p < n) counts[p]!++;
|
|
847
|
+
let best = 0;
|
|
848
|
+
let voted = false;
|
|
849
|
+
for (let i = 0; i < n; i++) {
|
|
850
|
+
if (counts[i]! > 0) voted = true;
|
|
851
|
+
if (counts[i]! > counts[best]!) best = i;
|
|
852
|
+
}
|
|
853
|
+
return voted ? best : -1;
|
|
854
|
+
}
|
|
855
|
+
|
|
856
|
+
function outcomesStatus(outcomes: readonly { ok: boolean; aborted: boolean }[]): StepResult["status"] {
|
|
857
|
+
// A real failure (non-abort) fails the step. An aborted item (per-call skip or
|
|
858
|
+
// run signal) does NOT fail individually — but if EVERY item aborted (e.g. all
|
|
859
|
+
// per-call skipped), the step did zero real work → 'skipped' so the run stops
|
|
860
|
+
// instead of reporting 'done' with empty results.
|
|
861
|
+
if (outcomes.length > 0 && outcomes.every((o) => o.aborted)) return "skipped";
|
|
862
|
+
if (outcomes.some((o) => !o.ok && !o.aborted)) return "failed";
|
|
863
|
+
return "done";
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
/** Terminate every in-flight call belonging to `stepId` (via its per-call
|
|
867
|
+
* controller → SIGTERM in spawnAgent). Called from dispatchAgentCall on any
|
|
868
|
+
* failure of that call (settle failure, dispatch throw, A6 rejection): the
|
|
869
|
+
* hung siblings are killed so the batch's allSettled wait settles instead of
|
|
870
|
+
* blocking the run forever on a stalled subprocess. Aborting a healthy
|
|
871
|
+
* in-flight item is fine — the step is already failing, its results are
|
|
872
|
+
* discarded. A degraded (budget-null) or aborted call is NOT a failure and
|
|
873
|
+
* never triggers this. */
|
|
874
|
+
function abortStepCalls(exec: StepExecContext, stepId: string): void {
|
|
875
|
+
for (const [callId, controller] of exec.registry.controllers) {
|
|
876
|
+
if (stepIdOf(callId) === stepId) controller.abort();
|
|
877
|
+
}
|
|
878
|
+
}
|
|
879
|
+
|
|
880
|
+
/** A6: reject oversized / control-character payloads before any spawn. On
|
|
881
|
+
* rejection the step's in-flight siblings are aborted first (fail-fast for
|
|
882
|
+
* concurrent batches), then a terminal WorkflowError is thrown. */
|
|
883
|
+
function assertPromptAllowed(exec: StepExecContext, callId: string, prompt: string, systemPrompt: string | undefined): void {
|
|
884
|
+
const promptBytes = Buffer.byteLength(prompt, "utf8");
|
|
885
|
+
if (promptBytes > exec.maxPromptBytes) {
|
|
886
|
+
abortStepCalls(exec, stepIdOf(callId));
|
|
887
|
+
throw new WorkflowError(
|
|
888
|
+
`prompt size ${promptBytes} bytes exceeds limit ${exec.maxPromptBytes} bytes`,
|
|
889
|
+
{ category: "size-limit", detail: { bytes: promptBytes, limit: exec.maxPromptBytes } },
|
|
890
|
+
);
|
|
891
|
+
}
|
|
892
|
+
if (CONTROL_CHARS.test(prompt)) {
|
|
893
|
+
abortStepCalls(exec, stepIdOf(callId));
|
|
894
|
+
throw new WorkflowError("prompt contains non-printable control characters", { category: "control-chars" });
|
|
895
|
+
}
|
|
896
|
+
if (systemPrompt) {
|
|
897
|
+
const sysBytes = Buffer.byteLength(systemPrompt, "utf8");
|
|
898
|
+
if (sysBytes > exec.maxPromptBytes) {
|
|
899
|
+
abortStepCalls(exec, stepIdOf(callId));
|
|
900
|
+
throw new WorkflowError(
|
|
901
|
+
`systemPrompt size ${sysBytes} bytes exceeds limit ${exec.maxPromptBytes} bytes`,
|
|
902
|
+
{ category: "size-limit", detail: { bytes: sysBytes, limit: exec.maxPromptBytes } },
|
|
903
|
+
);
|
|
904
|
+
}
|
|
905
|
+
if (CONTROL_CHARS.test(systemPrompt)) {
|
|
906
|
+
abortStepCalls(exec, stepIdOf(callId));
|
|
907
|
+
throw new WorkflowError("systemPrompt contains non-printable control characters", { category: "control-chars" });
|
|
908
|
+
}
|
|
909
|
+
}
|
|
910
|
+
}
|
|
911
|
+
|
|
912
|
+
// ---------------------------------------------------------------------------
|
|
913
|
+
// budget / spawn / stats helpers
|
|
914
|
+
// ---------------------------------------------------------------------------
|
|
915
|
+
|
|
916
|
+
/** Pre-check a whole batch (fan_out items, or a composite's total agents) fits the budget + MAX_BATCH cap.
|
|
917
|
+
* Does NOT reserve — individual dispatchAgentCall → guardSpawn → pool.reserve(1) atomically reserves
|
|
918
|
+
* per-agent. This avoids double-counting: a prior guardBatch reserve(total) + per-call guardSpawn
|
|
919
|
+
* reserve(1) charged 2x the agent slots. Serial runStepSequence has no cross-step TOCTOU to exploit
|
|
920
|
+
* the pre-check gap; within-step concurrency is guarded by reserve(1)'s atomic increment. */
|
|
921
|
+
function guardBatch(exec: StepExecContext, total: number, label: string): void {
|
|
922
|
+
assertBatchSize(total);
|
|
923
|
+
// Under the "null" policy (A3), skip the canSpawn pre-check: per-item
|
|
924
|
+
// guardSpawn degrades excess items to null instead. The hard MAX_BATCH cap
|
|
925
|
+
// above still throws regardless of policy.
|
|
926
|
+
if (exec.budgetPolicy !== "null" && !exec.pool.canSpawn(total, Date.now())) {
|
|
927
|
+
throw new BudgetExceededError(`${label} needs ${total} agents but the budget is exhausted`);
|
|
928
|
+
}
|
|
929
|
+
}
|
|
930
|
+
|
|
931
|
+
/** Reserve one agent slot and check token budget. Returns a release handle —
|
|
932
|
+
* call it ONLY if the dispatch never started (spawn threw).
|
|
933
|
+
*
|
|
934
|
+
* Under the "null" budget policy (A3), returns `null` instead of throwing
|
|
935
|
+
* when the budget is exhausted: the caller degrades that call to a null
|
|
936
|
+
* outcome. This is the single atomic chokepoint (sync section, no await gap),
|
|
937
|
+
* so concurrent fan_out workers each see an accurate count — closing the
|
|
938
|
+
* TOCTOU that a pre-check before reserve would reintroduce.
|
|
939
|
+
*
|
|
940
|
+
* Lifetime accounting: `exec.spawned` is incremented SYNCHRONOUSLY here ... */
|
|
941
|
+
function guardSpawn(exec: StepExecContext, callId: string, n: number): (() => void) | null {
|
|
942
|
+
exec.spawned += n;
|
|
943
|
+
assertLifetimeAgents(exec.spawned);
|
|
944
|
+
// isExhausted enforces maxTokens (and maxAgents) before committing.
|
|
945
|
+
if (exec.pool.isExhausted(Date.now())) {
|
|
946
|
+
exec.spawned -= n;
|
|
947
|
+
if (exec.budgetPolicy === "null") {
|
|
948
|
+
exec.degradedStepIds.add(stepIdOf(callId));
|
|
949
|
+
return null;
|
|
950
|
+
}
|
|
951
|
+
throw new BudgetExceededError("agent spawn refused — budget exhausted");
|
|
952
|
+
}
|
|
953
|
+
const releasePool = exec.pool.reserve(n);
|
|
954
|
+
return () => {
|
|
955
|
+
exec.spawned -= n;
|
|
956
|
+
releasePool();
|
|
957
|
+
};
|
|
958
|
+
}
|
|
959
|
+
|
|
960
|
+
function applyOutcome(exec: StepExecContext, res: AgentSpawnResult): void {
|
|
961
|
+
// `spawned` was incremented synchronously in guardSpawn; a settled agent keeps
|
|
962
|
+
// its slot, so don't touch it here — only record token spend.
|
|
963
|
+
exec.pool.track({ tokens: res.usage.input + res.usage.output });
|
|
964
|
+
}
|
|
965
|
+
|
|
966
|
+
async function timed(fn: () => Promise<AgentSpawnResult>): Promise<{ res: AgentSpawnResult; durationMs: number }> {
|
|
967
|
+
const start = Date.now();
|
|
968
|
+
const res = await fn();
|
|
969
|
+
return { res, durationMs: Date.now() - start };
|
|
970
|
+
}
|
|
971
|
+
|
|
972
|
+
function dispatchOpts(
|
|
973
|
+
callId: string,
|
|
974
|
+
prompt: string,
|
|
975
|
+
spec: AgentOpts,
|
|
976
|
+
signal?: AbortSignal,
|
|
977
|
+
listeners?: AgentLifecycleListeners,
|
|
978
|
+
allowChildRecursion = false,
|
|
979
|
+
): AgentSpawnOptions {
|
|
980
|
+
return {
|
|
981
|
+
callId,
|
|
982
|
+
task: prompt,
|
|
983
|
+
model: spec.model,
|
|
984
|
+
tools: spec.tools ? [...spec.tools] : undefined,
|
|
985
|
+
systemPrompt: spec.systemPrompt,
|
|
986
|
+
signal,
|
|
987
|
+
allowChildRecursion,
|
|
988
|
+
// C3: bridge the spawn's streamed deltas to the lifecycle onUpdate listener,
|
|
989
|
+
// attributed to this callId. When no listener is registered, the subprocess
|
|
990
|
+
// drops the deltas (its onUpdate stays undefined — same as before).
|
|
991
|
+
onUpdate: listeners?.onUpdate ? (delta) => notifyUpdate(listeners, callId, delta) : undefined,
|
|
992
|
+
};
|
|
993
|
+
}
|
|
994
|
+
|
|
995
|
+
function finalText(res: AgentSpawnResult): string {
|
|
996
|
+
for (let i = res.messages.length - 1; i >= 0; i--) {
|
|
997
|
+
const msg = res.messages[i];
|
|
998
|
+
if (msg?.role === "assistant") {
|
|
999
|
+
for (const part of msg.content) {
|
|
1000
|
+
if (part.type === "text") return part.text;
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
}
|
|
1004
|
+
return "";
|
|
1005
|
+
}
|
|
1006
|
+
|
|
1007
|
+
const zeroStats: StepStats = { tokens: 0, cost: 0, durationMs: 0, agents: 0, failures: 0 };
|
|
1008
|
+
|
|
1009
|
+
function usageStats(res: AgentSpawnResult, durationMs: number, ok: boolean): StepStats {
|
|
1010
|
+
return {
|
|
1011
|
+
tokens: res.usage.input + res.usage.output,
|
|
1012
|
+
cost: res.usage.cost,
|
|
1013
|
+
durationMs,
|
|
1014
|
+
agents: 1,
|
|
1015
|
+
failures: ok ? 0 : 1,
|
|
1016
|
+
};
|
|
1017
|
+
}
|
|
1018
|
+
|
|
1019
|
+
function addStats(a: StepStats, b: StepStats): StepStats {
|
|
1020
|
+
return {
|
|
1021
|
+
tokens: a.tokens + b.tokens,
|
|
1022
|
+
cost: a.cost + b.cost,
|
|
1023
|
+
durationMs: a.durationMs + b.durationMs,
|
|
1024
|
+
agents: a.agents + b.agents,
|
|
1025
|
+
failures: a.failures + b.failures,
|
|
1026
|
+
};
|
|
1027
|
+
}
|
|
1028
|
+
|
|
1029
|
+
/** Sum a list of StepStats. `durationMs` is the wall-clock override (0 to keep the per-call sum). */
|
|
1030
|
+
export function aggregateStats(stats: readonly StepStats[], durationMs: number): StepStats {
|
|
1031
|
+
let tokens = 0;
|
|
1032
|
+
let cost = 0;
|
|
1033
|
+
let agents = 0;
|
|
1034
|
+
let failures = 0;
|
|
1035
|
+
let dur = 0;
|
|
1036
|
+
for (const s of stats) {
|
|
1037
|
+
tokens += s.tokens;
|
|
1038
|
+
cost += s.cost;
|
|
1039
|
+
agents += s.agents;
|
|
1040
|
+
failures += s.failures;
|
|
1041
|
+
dur += s.durationMs;
|
|
1042
|
+
}
|
|
1043
|
+
return { tokens, cost, durationMs: durationMs > 0 ? durationMs : dur, agents, failures };
|
|
1044
|
+
}
|
|
1045
|
+
|
|
1046
|
+
function withDuration(stats: StepStats, start: number): StepStats {
|
|
1047
|
+
return { ...stats, durationMs: Date.now() - start };
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
function stepResult(
|
|
1051
|
+
id: string,
|
|
1052
|
+
type: StepResult["type"],
|
|
1053
|
+
status: StepResult["status"],
|
|
1054
|
+
results: unknown,
|
|
1055
|
+
stats: StepStats,
|
|
1056
|
+
iterations?: number,
|
|
1057
|
+
errorCategory?: ErrorCategory,
|
|
1058
|
+
): StepResult {
|
|
1059
|
+
const base: StepResult = { id, type, status, results, stats };
|
|
1060
|
+
if (iterations !== undefined) return { ...base, iterations, errorCategory };
|
|
1061
|
+
if (errorCategory !== undefined) return { ...base, errorCategory };
|
|
1062
|
+
return base;
|
|
1063
|
+
}
|
|
1064
|
+
|
|
1065
|
+
function notifyStart(exec: StepExecContext, callId: string): void {
|
|
1066
|
+
try {
|
|
1067
|
+
exec.listeners?.onAgentStart?.(callId);
|
|
1068
|
+
} catch {
|
|
1069
|
+
/* listener robustness — a throwing listener never blocks dispatch */
|
|
1070
|
+
}
|
|
1071
|
+
}
|
|
1072
|
+
function notifyEnd(exec: StepExecContext, callId: string, ok: boolean, stats: StepStats, model?: string, output?: string): void {
|
|
1073
|
+
try {
|
|
1074
|
+
exec.listeners?.onAgentEnd?.(callId, ok, stats, model, output);
|
|
1075
|
+
} catch {
|
|
1076
|
+
/* listener robustness */
|
|
1077
|
+
}
|
|
1078
|
+
}
|