@fyeeme/pi-dynamic-workflows 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1078 @@
1
+ /**
2
+ * Stage executor — dispatches one step by `type`, wiring the five CC-fusion
3
+ * modules plus the three composite patterns:
4
+ * - cache/journal → computeCacheKey + Journal lookup/append (cache-resume)
5
+ * - budget → BudgetPool guardBatch/guardSpawn + caps (runaway + budget-exceeded, incl. maxTokens)
6
+ * - spawn → AgentDispatch (real spawnAgent or injected fake) + registry
7
+ * - lifecycle → listeners threaded through (agent start/end/skip/retry)
8
+ * - abort → per-call AbortController via the registry + run signal
9
+ *
10
+ * All seven step types are implemented. Every agent call — including those
11
+ * inside composites — goes through dispatchAgentCall (cache + budget + spawn +
12
+ * journal + lifecycle + an early signal-abort short-circuit), so resume/budget/
13
+ * abort apply uniformly. Composite verdicts are coerced (LLMs return "true"/"0"
14
+ * as strings) and JSON is parsed via the shared string-aware parseFirstJson.
15
+ *
16
+ * Timing note: `Date.now()` is used here for per-step duration stats only. This
17
+ * is engine code, NOT a workflow `.ts` body, so the Task 3 ast-guard does not
18
+ * apply; run identity (`now`) is still supplied deterministically by the caller.
19
+ * (maxDurationMs uses a live clock via Date.now() at the guard points — engine
20
+ * code is not AST-guarded — so wall-clock duration is enforced; maxTokens is
21
+ * enforced via BudgetPool.isExhausted on each spawn.)
22
+ */
23
+ import {
24
+ assertBatchSize,
25
+ assertLifetimeAgents,
26
+ BudgetExceededError,
27
+ BudgetPool,
28
+ } from "../budget/index.ts";
29
+ import { computeCacheKey, type Journal } from "../cache/index.ts";
30
+ import { RETRYABLE_CATEGORIES, WorkflowError, type ErrorCategory } from "../errors.ts";
31
+ import { type AgentLifecycleListeners, notifyCacheHit, notifyLog, notifyUpdate } from "../lifecycle.ts";
32
+ import { parseFirstJson } from "../outcomes.ts";
33
+ import type {
34
+ AdversarialStep,
35
+ AgentCallSpec,
36
+ AgentOpts,
37
+ AgentStep,
38
+ ClassifyRouteStep,
39
+ CodeStep,
40
+ LoopUntilDryStep,
41
+ LogStep,
42
+ SubWorkflowStep,
43
+ FanOutStep,
44
+ LoopUntilStep,
45
+ RunStatus,
46
+ StepContext,
47
+ StepDefinition,
48
+ StepResult,
49
+ StepStats,
50
+ TournamentStep,
51
+ } from "../types.ts";
52
+ import type { AgentSpawnOptions, AgentSpawnRegistry, AgentSpawnResult } from "../agent/dispatch.ts";
53
+ import { mapWithConcurrencyLimit } from "../agent/dispatch.ts";
54
+ import { stepIdOf } from "../format.ts";
55
+
56
+ /** Injectable agent dispatch — same shape as spawnAgent. Default = real spawnAgent. */
57
+ export type AgentDispatch = (
58
+ registry: AgentSpawnRegistry,
59
+ opts: AgentSpawnOptions,
60
+ ) => Promise<AgentSpawnResult>;
61
+
62
+ /** Shared, mutable execution state threaded through every step of a run. */
63
+ export interface StepExecContext {
64
+ readonly workflowName: string;
65
+ readonly dispatch: AgentDispatch;
66
+ readonly registry: AgentSpawnRegistry;
67
+ readonly journal: Journal;
68
+ readonly pool: BudgetPool;
69
+ readonly signal?: AbortSignal;
70
+ readonly listeners?: AgentLifecycleListeners;
71
+ /** Run inception time (ms), supplied deterministically by the caller — used
72
+ * for journal timestamps and as the child BudgetPool originMs. NOT used for
73
+ * budget duration checks (those read Date.now() at the guard points). */
74
+ readonly now: number;
75
+ /** Cumulative agents spawned so far this run (lifetime-cap counter). */
76
+ spawned: number;
77
+ /** Current classify_route nesting depth (cycle guard). */
78
+ depth: number;
79
+ /** A3: per-step budget-exhaustion policy. Set by runStepSequence from
80
+ * step.onBudgetExhaust before each step (default "throw"). */
81
+ budgetPolicy: "throw" | "null";
82
+ /** A6: max resolved-prompt byte size; oversize throws a size-limit error. */
83
+ maxPromptBytes: number;
84
+ /** A3: step ids that degraded to null under the "null" policy this run. */
85
+ degradedStepIds: Set<string>;
86
+ /** Recursion opt-in propagated to every spawned workflow sub-agent: when
87
+ * false (default) children load WITHOUT the subagent/fan-out tools.
88
+ * Opt-in re-enables them, bounded by any maxSpawnDepth cap. */
89
+ allowChildRecursion: boolean;
90
+ /** Non-cached dispatch attempts this run (null-degraded and dispatch-throw
91
+ * paths included) — the denominator for resume cache-hit accounting. */
92
+ dispatched: number;
93
+ }
94
+
95
+ /** Outcome of a dispatched agent call (shared by all step kinds that call agents). */
96
+ interface AgentCallOutcome {
97
+ readonly value: string | null;
98
+ readonly ok: boolean;
99
+ readonly aborted: boolean;
100
+ readonly cached: boolean;
101
+ readonly stats: StepStats;
102
+ /** A5: error category of this call's failure. dispatch-error marks a
103
+ * retryable agent failure (dispatch rejection OR a failed subprocess);
104
+ * absent for aborts and successes. */
105
+ readonly errorCategory?: ErrorCategory;
106
+ }
107
+
108
+ const MAX_ROUTE_DEPTH = 8;
109
+
110
+ /** Default per-prompt byte-size cap (A6). Overridable via RunWorkflowOptions. */
111
+ export const DEFAULT_MAX_PROMPT_BYTES = 256 * 1024;
112
+
113
+ /** Non-printable C0 controls + DEL, excluding the common whitespace (\t \n \r).
114
+ * Presence triggers a control-chars rejection (A6) — these almost always
115
+ * indicate corrupted/serialized binary data leaking into a prompt. */
116
+ // A6: reject prompts containing non-printable / format control characters before
117
+ // any spawn. Covers C0 (minus tab/LF/CR) + DEL + C1 (U+0080–U+009F) + zero-width
118
+ // and bidi format chars (U+200B–U+200F, U+202A–U+202E, U+2060–U+206F, U+FEFF) —
119
+ // the known prompt-injection vectors (bidi override, ZWJ/ZWNJ smuggling, BOM).
120
+ // The message below says "non-printable control characters" and the regex means
121
+ // it: previously only C0+DEL were blocked while bidi/zero-width sailed through.
122
+ const CONTROL_CHARS = /[\x00-\x08\x0B\x0C\x0E-\x1F\x7F\u0080-\u009F\u200B-\u200F\u202A-\u202E\u2060-\u206F\uFEFF]/;
123
+
124
+ /** Base workflow-subagent discipline prompt (B1+B2). Prepended to every agent
125
+ * dispatch's systemPrompt so the model knows it is a workflow subagent and
126
+ * returns the literal result without confirmations / fences / prose. Injected
127
+ * BEFORE cache-key computation (design D1) so it participates in the key. */
128
+ const BASE_WORKFLOW_SUBAGENT_PROMPT = [
129
+ "You are a subagent spawned by a deterministic workflow orchestration script.",
130
+ "Your final text response is returned verbatim as a string to the calling script \u2014 it is your return value, not a message to a human.",
131
+ "- Output the literal result (data, JSON, or text). Do NOT output confirmations like \"Done.\" or \"Sent.\"",
132
+ "- If asked for JSON, return ONLY the raw JSON \u2014 no code fences, no prose, no markdown.",
133
+ "- Be concise. The script will parse your output.",
134
+ ].join("\n");
135
+
136
+ // ---------------------------------------------------------------------------
137
+ // Dispatch
138
+ // ---------------------------------------------------------------------------
139
+
140
+ export async function executeStep(
141
+ step: StepDefinition,
142
+ ctx: StepContext,
143
+ exec: StepExecContext,
144
+ ): Promise<StepResult> {
145
+ switch (step.type) {
146
+ case "agent":
147
+ return execAgent(step, ctx, exec);
148
+ case "code":
149
+ return execCode(step, ctx);
150
+ case "log":
151
+ return execLog(step, ctx, exec);
152
+ case "fan_out":
153
+ return execFanOut(step, ctx, exec);
154
+ case "loop_until":
155
+ return execLoopUntil(step, ctx, exec);
156
+ case "adversarial":
157
+ return execAdversarial(step, ctx, exec);
158
+ case "tournament":
159
+ return execTournament(step, ctx, exec);
160
+ case "classify_route":
161
+ return execClassifyRoute(step, ctx, exec);
162
+ case "sub_workflow":
163
+ return execSubWorkflow(step, ctx, exec);
164
+ case "loop_until_dry":
165
+ return execLoopUntilDry(step, ctx, exec);
166
+ }
167
+ }
168
+
169
+ // ---------------------------------------------------------------------------
170
+ // Shared agent dispatch — cache-resume + budget + spawn + journal + lifecycle
171
+ // ---------------------------------------------------------------------------
172
+
173
+ async function dispatchAgentCall(
174
+ callId: string,
175
+ prompt: string,
176
+ signature: AgentOpts,
177
+ exec: StepExecContext,
178
+ ): Promise<AgentCallOutcome> {
179
+ // If the run is already aborted, do not spawn (avoids spawning processes that
180
+ // are killed immediately by the signal listener).
181
+ if (exec.signal?.aborted) {
182
+ return { value: "", ok: false, aborted: true, cached: false, stats: zeroStats };
183
+ }
184
+
185
+ // D1: prepend the base workflow-subagent discipline prompt to the signature's
186
+ // systemPrompt BEFORE computing the cache key, so it participates in the key
187
+ // (injecting after the key would let runs under different base-prompt versions
188
+ // collide). A step override appends after the base (spec B1+B2).
189
+ const effectiveSignature: AgentOpts = signature.systemPrompt
190
+ ? { ...signature, systemPrompt: `${BASE_WORKFLOW_SUBAGENT_PROMPT}\n\n${signature.systemPrompt}` }
191
+ : { ...signature, systemPrompt: BASE_WORKFLOW_SUBAGENT_PROMPT };
192
+
193
+ const key = computeCacheKey({ workflowName: exec.workflowName, prompt, signature: effectiveSignature });
194
+
195
+ const cached = exec.journal.lookup(key);
196
+ if (cached?.type === "result" && cached.ok) {
197
+ notifyCacheHit(exec.listeners, callId);
198
+ return { value: cached.value as string, ok: true, aborted: false, cached: true, stats: zeroStats };
199
+ }
200
+
201
+ // A6: reject oversized / control-character payloads before any spawn. These
202
+ // run AFTER the cache-hit return so a config change (e.g. lowering
203
+ // maxPromptBytes between runs) cannot kill resume replay of a previously
204
+ // cached call — a replay spawns nothing. The checks gate NEW spawns and
205
+ // cover BOTH the task prompt and the effective systemPrompt (which always
206
+ // contains the base discipline prompt), so oversized/binary data cannot
207
+ // dodge the guard by being placed in the system prompt. These throw terminal
208
+ // WorkflowErrors (size-limit / control-chars) which propagate past
209
+ // runWithRetry's retry loop — only dispatch-error is retryable (A5:
210
+ // runWithRetry gates on RETRYABLE_CATEGORIES, not bare status). A rejection
211
+ // also aborts the step's in-flight siblings so a concurrent batch fails fast
212
+ // instead of the allSettled wait blocking on a stalled sibling.
213
+ assertPromptAllowed(exec, callId, prompt, effectiveSignature.systemPrompt);
214
+
215
+ const releaseSpawn = guardSpawn(exec, callId, 1);
216
+ if (releaseSpawn === null) {
217
+ // A3: budget exhausted under the "null" policy — degrade this call to null
218
+ // (guardSpawn already attributed the step to degradedStepIds). No spawn,
219
+ // no slot consumed, nothing journaled. Counted as a dispatch attempt so
220
+ // resume's cachedTotal covers the degraded path.
221
+ exec.dispatched++;
222
+ return { value: null, ok: true, aborted: false, cached: false, stats: zeroStats };
223
+ }
224
+ // Every non-cached call that passes the guard is a dispatch attempt (the
225
+ // dispatch-throw path increments here too — the start was announced even
226
+ // though no agent launched, so the total must not undercount it).
227
+ exec.dispatched++;
228
+ notifyStart(exec, callId);
229
+ await exec.journal.append({ type: "started", key, at: exec.now });
230
+ let res: AgentSpawnResult;
231
+ let durationMs: number;
232
+ try {
233
+ ({ res, durationMs } = await timed(() =>
234
+ exec.dispatch(exec.registry, dispatchOpts(callId, prompt, effectiveSignature, exec.signal, exec.listeners, exec.allowChildRecursion)),
235
+ ));
236
+ } catch (e) {
237
+ releaseSpawn(); // dispatch never started — return the reserved slot
238
+ // dispatch rejected — write a terminal result so resume sees closure (not
239
+ // an orphan `started`), per CC's "result always written" invariant. Returns
240
+ // a failed outcome; runWithRetry treats status 'failed' as retryable.
241
+ const msg = e instanceof Error ? e.message : String(e);
242
+ await exec.journal.append({ type: "result", key, at: exec.now, ok: false, value: `dispatch error: ${msg}` });
243
+ notifyEnd(exec, callId, false, zeroStats);
244
+ // Fail-fast for concurrent batches: a dispatch error is a real failure —
245
+ // terminate the step's in-flight siblings (SIGTERM via their per-call
246
+ // controllers) so mapWithConcurrencyLimit's allSettled wait does not hang
247
+ // forever on a stalled sibling subprocess.
248
+ abortStepCalls(exec, stepIdOf(callId));
249
+ return {
250
+ value: `dispatch error: ${msg}`,
251
+ ok: false,
252
+ aborted: false,
253
+ cached: false,
254
+ stats: zeroStats,
255
+ errorCategory: "dispatch-error",
256
+ };
257
+ }
258
+ applyOutcome(exec, res);
259
+ const value = finalText(res);
260
+ const ok = !res.aborted && res.exitCode === 0 && res.stopReason !== "error" && res.stopReason !== "aborted";
261
+ const stats = usageStats(res, durationMs, ok);
262
+ // On failure with no assistant output, surface the subprocess stderr /
263
+ // errorMessage / exitCode so the caller can diagnose WHY it failed
264
+ // (unknown model, missing provider, bad PATH, …). Without this the
265
+ // step result is just "" and the error is invisible.
266
+ const diag = ok || value ? value : diagnoseFailure(res);
267
+ await exec.journal.append({ type: "result", key, at: exec.now, ok, value: diag });
268
+ notifyEnd(exec, callId, ok, stats, res.model, diag);
269
+ // Fail-fast for concurrent batches: a settled-but-failed call (non-abort) is a
270
+ // real failure — terminate the step's in-flight siblings so the batch's
271
+ // allSettled wait cannot block forever on a stalled sibling subprocess. A
272
+ // degraded (budget-null) or aborted call is NOT a failure and must not abort
273
+ // its siblings.
274
+ if (!ok && !res.aborted) abortStepCalls(exec, stepIdOf(callId));
275
+ // A failed-but-settled subprocess (exitCode≠0 / stopReason "error" / killed)
276
+ // is a retryable dispatch-error — typically a transient provider/model
277
+ // failure. Aborts (external cancel, skip, maxTurns) carry no category: they
278
+ // settle as skipped and must not auto-retry.
279
+ return {
280
+ value: diag,
281
+ ok,
282
+ aborted: res.aborted,
283
+ cached: false,
284
+ stats,
285
+ errorCategory: ok || res.aborted ? undefined : "dispatch-error",
286
+ };
287
+ }
288
+
289
+ /** Build a human-readable failure reason from a subprocess result that
290
+ * produced no assistant text. Surfaces stderr / errorMessage / exitCode. */
291
+ function diagnoseFailure(res: AgentSpawnResult): string {
292
+ const parts: string[] = ["[agent failed"];
293
+ if (res.exitCode !== 0) parts.push(`exit ${res.exitCode}`);
294
+ if (res.stopReason) parts.push(`stop:${res.stopReason}`);
295
+ if (res.errorMessage) parts.push(res.errorMessage);
296
+ // Last few non-empty stderr lines are usually the real cause.
297
+ const stderrTail = res.stderr.trim().split("\n").filter(Boolean).slice(-4).join(" | ");
298
+ if (stderrTail) parts.push(stderrTail);
299
+ parts.push("]");
300
+ return parts.join(" ");
301
+ }
302
+
303
+ // ---------------------------------------------------------------------------
304
+ // agent
305
+ // ---------------------------------------------------------------------------
306
+
307
+ async function execAgent(step: AgentStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
308
+ const prompt = typeof step.prompt === "function" ? await step.prompt(ctx) : step.prompt;
309
+ const callId = `${step.id}#${exec.spawned + 1}`;
310
+ const outcome = await dispatchAgentCall(callId, prompt, step, exec);
311
+ const status = outcome.ok ? "done" : outcome.aborted ? "skipped" : "failed";
312
+ return stepResult(step.id, "agent", status, outcome.value, outcome.stats, undefined, outcome.errorCategory);
313
+ }
314
+
315
+ // ---------------------------------------------------------------------------
316
+ // code (pure transform — no dispatch, no cache, not budgeted)
317
+ // ---------------------------------------------------------------------------
318
+
319
+ async function execCode(step: CodeStep, ctx: StepContext): Promise<StepResult> {
320
+ const start = Date.now();
321
+ try {
322
+ const value = await step.transform(ctx);
323
+ return stepResult(step.id, "code", "done", value, {
324
+ tokens: 0,
325
+ cost: 0,
326
+ durationMs: Date.now() - start,
327
+ agents: 0,
328
+ failures: 0,
329
+ });
330
+ } catch (e) {
331
+ const msg = e instanceof Error ? e.message : String(e);
332
+ // Surface the error message (runStepSequence only appends sr.results when
333
+ // it's a non-empty string); a bare catch left the failure cause invisible.
334
+ return stepResult(step.id, "code", "failed", `code error: ${msg}`, {
335
+ tokens: 0,
336
+ cost: 0,
337
+ durationMs: Date.now() - start,
338
+ agents: 0,
339
+ failures: 1,
340
+ });
341
+ }
342
+ }
343
+
344
+ // log (C2): emit a narrative line via the onLog listener. Pure string, zero
345
+ // dispatch/tokens, not cached — same profile as `code` but fires onLog and the
346
+ // widget renders it as a distinct narrative line.
347
+ async function execLog(step: LogStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
348
+ const start = Date.now();
349
+ // A `log` step is non-critical narrative — a throwing message function (e.g.
350
+ // referencing a missing upstream field) must NOT crash the whole run. Fall
351
+ // back to an inline error marker and keep status "done" so the run continues.
352
+ let message: string;
353
+ try {
354
+ message = typeof step.message === "function" ? await step.message(ctx) : step.message;
355
+ } catch (e) {
356
+ const msg = e instanceof Error ? e.message : String(e);
357
+ message = `[log error: ${msg}]`;
358
+ }
359
+ notifyLog(exec.listeners, step.id, message);
360
+ return stepResult(step.id, "log", "done", message, {
361
+ tokens: 0,
362
+ cost: 0,
363
+ durationMs: Date.now() - start,
364
+ agents: 0,
365
+ failures: 0,
366
+ });
367
+ }
368
+
369
+ // ---------------------------------------------------------------------------
370
+ // fan_out (parallel agents over a list)
371
+ // ---------------------------------------------------------------------------
372
+
373
+ async function execFanOut(step: FanOutStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
374
+ const items = [...step.over(ctx)];
375
+ const n = items.length;
376
+ guardBatch(exec, n, `fan_out "${step.id}"`);
377
+
378
+ const parallelism = step.parallelism ?? Math.min(n, 8);
379
+ const itemSpecs = items.map((item, i) => step.agent(item, i, ctx));
380
+ const start = Date.now();
381
+
382
+ const outcomes = await mapWithConcurrencyLimit(itemSpecs, parallelism, (spec, i) =>
383
+ dispatchAgentCall(`${step.id}#${i + 1}`, spec.prompt, spec, exec),
384
+ );
385
+
386
+ const durationMs = Date.now() - start;
387
+ const merged = step.merge ? await step.merge(outcomes.map((o) => o.value), ctx) : outcomes.map((o) => o.value);
388
+ const stats = aggregateStats(outcomes.map((o) => o.stats), durationMs);
389
+ const status = outcomesStatus(outcomes);
390
+ // A5: a batch whose failures are all retryable dispatch-errors carries the
391
+ // category so runWithRetry can retry the whole fan_out (successful items
392
+ // replay as cache hits and are not re-charged). Aborted-only batches settle
393
+ // as skipped and carry none.
394
+ const errorCategory = status === "failed" && outcomes.some((o) => o.errorCategory === "dispatch-error") ? "dispatch-error" : undefined;
395
+ return stepResult(step.id, "fan_out", status, merged, stats, undefined, errorCategory);
396
+ }
397
+
398
+ // ---------------------------------------------------------------------------
399
+ // loop_until (iterate a body agent until / maxIterations / budget)
400
+ // ---------------------------------------------------------------------------
401
+
402
+ async function execLoopUntil(step: LoopUntilStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
403
+ const maxIter = step.maxIterations ?? 10;
404
+ const start = Date.now();
405
+ const values: unknown[] = [];
406
+ let stats = zeroStats;
407
+ let status: StepResult["status"] = "done";
408
+ let iter = 0;
409
+
410
+ while (iter < maxIter) {
411
+ if (exec.signal?.aborted) {
412
+ status = "skipped";
413
+ break;
414
+ }
415
+ const prompt = await step.prompt(ctx, iter);
416
+ const outcome = await dispatchAgentCall(`${step.id}#${iter + 1}`, prompt, step, exec);
417
+ stats = addStats(stats, outcome.stats);
418
+ values.push(outcome.value);
419
+ if (!outcome.ok) {
420
+ status = outcome.aborted ? "skipped" : "failed";
421
+ break;
422
+ }
423
+ iter++;
424
+ if (step.until(ctx, iter)) break;
425
+ }
426
+
427
+ return stepResult(step.id, "loop_until", status, values, withDuration(stats, start), iter);
428
+ }
429
+
430
+ // ---------------------------------------------------------------------------
431
+ // adversarial (produce one candidate, N judges grade it, tally)
432
+ // ---------------------------------------------------------------------------
433
+
434
+ /** Judge call opts inherit the produce opts (model/tools/systemPrompt) and are
435
+ * overridden per-field by an explicit `step.judge`. Without this, a step-level
436
+ * `model` applied only to the produce call and judges silently fell back to the
437
+ * default model (review m14). */
438
+ function judgeOpts(produce: AgentOpts, judge?: AgentOpts): AgentOpts {
439
+ return {
440
+ model: judge?.model ?? produce.model,
441
+ tools: judge?.tools ?? produce.tools,
442
+ systemPrompt: judge?.systemPrompt ?? produce.systemPrompt,
443
+ };
444
+ }
445
+
446
+ async function execAdversarial(step: AdversarialStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
447
+ const judgeCount = step.judges ?? 3;
448
+ if (judgeCount < 1) throw new Error(`adversarial "${step.id}" requires judges >= 1`);
449
+ const minPass = step.minPass ?? Math.ceil(judgeCount / 2);
450
+ guardBatch(exec, 1 + judgeCount, `adversarial "${step.id}"`);
451
+ const start = Date.now();
452
+ let stats = zeroStats;
453
+
454
+ const producePrompt = await resolvePrompt(step.produce, ctx);
455
+ const candidate = await dispatchAgentCall(`${step.id}#produce`, producePrompt, step.produce, exec);
456
+ stats = addStats(stats, candidate.stats);
457
+
458
+ // A3: a degraded produce (budget exhausted under the "null" policy) returns
459
+ // value null — the step degrades to a null result per the README contract,
460
+ // NOT a fabricated "done" with an empty candidate. No judges are dispatched:
461
+ // the budget is exhausted, they would only degrade too.
462
+ if (candidate.value === null) {
463
+ return stepResult(step.id, "adversarial", "done", null, withDuration(stats, start), 1);
464
+ }
465
+
466
+ const outcomes: AgentCallOutcome[] = [candidate];
467
+ const judges: { pass: boolean; reason: string }[] = [];
468
+ if (candidate.ok) {
469
+ const judgeSpecs = Array.from({ length: judgeCount }, (_, i) => judgePrompt(i, candidate.value ?? "", step.rubric));
470
+ const judgeOutcomes = await mapWithConcurrencyLimit(judgeSpecs, Math.min(judgeCount, 8), (prompt, i) =>
471
+ dispatchAgentCall(`${step.id}#judge${i + 1}`, prompt, judgeOpts(step.produce, step.judge), exec),
472
+ );
473
+ for (const o of judgeOutcomes) {
474
+ stats = addStats(stats, o.stats);
475
+ outcomes.push(o);
476
+ const parsed = parseFirstJson(o.value ?? "") as { pass?: unknown; reason?: unknown } | undefined;
477
+ judges.push({ pass: parsePassBool(parsed?.pass), reason: parseReason(parsed?.reason) });
478
+ }
479
+ } else {
480
+ // populate judge slots so the result shape is stable on produce failure
481
+ for (let i = 0; i < judgeCount; i++) judges.push({ pass: false, reason: "produce failed" });
482
+ }
483
+ const passCount = judges.filter((j) => j.pass).length;
484
+
485
+ return stepResult(
486
+ step.id,
487
+ "adversarial",
488
+ outcomesStatus(outcomes),
489
+ { candidate: candidate.value ?? "", passed: passCount >= minPass, passCount, minPass, judges },
490
+ withDuration(stats, start),
491
+ 1 + judgeCount,
492
+ );
493
+ }
494
+
495
+ // ---------------------------------------------------------------------------
496
+ // tournament (N distinct candidates, M judges rank, pick winner)
497
+ // ---------------------------------------------------------------------------
498
+
499
+ async function execTournament(step: TournamentStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
500
+ if (step.candidates < 1) throw new Error(`tournament "${step.id}" requires candidates >= 1`);
501
+ if (step.judges < 1) throw new Error(`tournament "${step.id}" requires judges >= 1`);
502
+ guardBatch(exec, step.candidates + step.judges, `tournament "${step.id}"`);
503
+ const start = Date.now();
504
+ let stats = zeroStats;
505
+ const outcomes: AgentCallOutcome[] = [];
506
+
507
+ const producePrompt = await resolvePrompt(step.produce, ctx);
508
+ const candSpecs = Array.from({ length: step.candidates }, (_, i) => ({
509
+ prompt: `${producePrompt}\n\nAttempt ${i + 1}: take a distinct approach.`,
510
+ }));
511
+ const candOutcomes = await mapWithConcurrencyLimit(candSpecs, Math.min(step.candidates, 8), (spec, i) =>
512
+ dispatchAgentCall(`${step.id}#cand${i + 1}`, spec.prompt, step.produce, exec),
513
+ );
514
+ const candidates = candOutcomes.map((o) => {
515
+ stats = addStats(stats, o.stats);
516
+ outcomes.push(o);
517
+ return o.value ?? "";
518
+ });
519
+
520
+ const judgeSpecs = Array.from({ length: step.judges }, (_, j) => rankPrompt(j, candidates));
521
+ const judgeOutcomes = await mapWithConcurrencyLimit(judgeSpecs, Math.min(step.judges, 8), (prompt, j) =>
522
+ dispatchAgentCall(`${step.id}#judge${j + 1}`, prompt, judgeOpts(step.produce, step.judge), exec),
523
+ );
524
+ const judgePicks = judgeOutcomes.map((o) => {
525
+ stats = addStats(stats, o.stats);
526
+ outcomes.push(o);
527
+ const parsed = parseFirstJson(o.value ?? "") as { winner?: unknown; reason?: unknown } | undefined;
528
+ return { winner: parseWinnerNum(parsed?.winner), reason: parseReason(parsed?.reason) };
529
+ });
530
+ const winner = tallyWinner(
531
+ judgePicks.map((j) => j.winner),
532
+ step.candidates,
533
+ );
534
+
535
+ // A5: like fan_out, a tournament whose failures are retryable dispatch-errors
536
+ // carries the category so runWithRetry can retry the whole composite.
537
+ const status = outcomesStatus(outcomes);
538
+ const errorCategory = status === "failed" && outcomes.some((o) => o.errorCategory === "dispatch-error") ? "dispatch-error" : undefined;
539
+
540
+ return stepResult(
541
+ step.id,
542
+ "tournament",
543
+ status,
544
+ { candidates, winner, judges: judgePicks },
545
+ withDuration(stats, start),
546
+ step.candidates + step.judges,
547
+ errorCategory,
548
+ );
549
+ }
550
+
551
+ // ---------------------------------------------------------------------------
552
+ // classify_route (classify → run the matching route's sub-steps)
553
+ // ---------------------------------------------------------------------------
554
+
555
+ async function execClassifyRoute(step: ClassifyRouteStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
556
+ const start = Date.now();
557
+ const classifyPrompt = await resolvePrompt(step.classifier, ctx);
558
+ const classify = await dispatchAgentCall(`${step.id}#classify`, classifyPrompt, step.classifier, exec);
559
+ if (!classify.ok) {
560
+ return stepResult(
561
+ step.id,
562
+ "classify_route",
563
+ classify.aborted ? "skipped" : "failed",
564
+ undefined,
565
+ withDuration(classify.stats, start),
566
+ undefined,
567
+ classify.aborted ? undefined : "dispatch-error",
568
+ );
569
+ }
570
+
571
+ // A3: a degraded classifier (budget exhausted under the "null" policy)
572
+ // returns value null — the step degrades to a null result per the README
573
+ // contract; do not fabricate a route run from an empty category.
574
+ if (classify.value === null) {
575
+ return stepResult(step.id, "classify_route", "done", null, withDuration(classify.stats, start));
576
+ }
577
+
578
+ const parsed = parseFirstJson(classify.value ?? "") as { category?: unknown } | undefined;
579
+ const category = parseCategoryStr(parsed?.category);
580
+ const routeSteps = step.routes[category] ?? step.fallback ?? [];
581
+
582
+ exec.depth++;
583
+ let sub: SequenceOutcome;
584
+ try {
585
+ sub = await runStepSequence(routeSteps, ctx.input, exec);
586
+ } finally {
587
+ exec.depth--;
588
+ }
589
+ const status: StepResult["status"] = sub.status === "completed" ? "done" : sub.status === "aborted" ? "skipped" : "failed";
590
+
591
+ return stepResult(
592
+ step.id,
593
+ "classify_route",
594
+ status,
595
+ { category, matched: category in step.routes, route: sub.steps, routeStatus: sub.status },
596
+ withDuration(addStats(classify.stats, aggregateStats(sub.steps.map((s) => s.stats), 0)), start),
597
+ undefined,
598
+ sub.errorCategory,
599
+ );
600
+ }
601
+
602
+ // ---------------------------------------------------------------------------
603
+ // sub_workflow (nested child workflow — CC's workflow() pattern)
604
+ // ---------------------------------------------------------------------------
605
+
606
+ async function execSubWorkflow(step: SubWorkflowStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
607
+ const start = Date.now();
608
+ // Resolve the child's input: static value or function of parent ctx. The
609
+ // function form is awaited (and typed to allow a Promise), matching the
610
+ // async convention of CodeStep.transform / LogStep.message — an un-awaited
611
+ // Promise would leak into the child's ctx.input as "[object Promise]" with
612
+ // its rejection silently dropped.
613
+ const childInput = typeof step.input === "function"
614
+ ? await (step.input as (c: StepContext) => unknown | Promise<unknown>)(ctx)
615
+ : step.input ?? ctx.input;
616
+
617
+ // Child shares parent's journal, pool, registry, and signal by default.
618
+ // inheritBudget: false means the child uses its own BudgetPool (isolated caps).
619
+ // Override workflowName with the CHILD's name so cache keys are scoped to the
620
+ // child workflow — otherwise two sibling sub_workflows whose agents share a
621
+ // prompt+signature collide on the parent's name and replay each other's cache.
622
+ const childExec: StepExecContext = step.inheritBudget === false
623
+ ? { ...exec, workflowName: step.workflow.name, depth: exec.depth + 1, pool: new BudgetPool(step.workflow.budget ?? {}, Date.now()) }
624
+ : { ...exec, workflowName: step.workflow.name, depth: exec.depth + 1 };
625
+
626
+ let sub: SequenceOutcome;
627
+ if (childExec.depth > MAX_ROUTE_DEPTH) {
628
+ sub = { steps: [], status: "failed", error: `sub_workflow nesting exceeded depth ${MAX_ROUTE_DEPTH}` };
629
+ } else {
630
+ sub = await runStepSequence(step.workflow.steps, childInput, childExec);
631
+ }
632
+ // Propagate the child's lifetime counter back to the parent: childExec is a
633
+ // spread copy, so without this sync the parent's `spawned` undercounts every
634
+ // agent the child spawned (callId numbering + the assertLifetimeAgents backstop
635
+ // in guardSpawn both depend on this being run-cumulative, per its docstring).
636
+ exec.spawned = childExec.spawned;
637
+
638
+ const status: StepResult["status"] = sub.status === "completed" ? "done"
639
+ : sub.status === "aborted" ? "skipped"
640
+ : "failed";
641
+
642
+ return stepResult(
643
+ step.id,
644
+ "sub_workflow",
645
+ status,
646
+ { steps: sub.steps, status: sub.status, workflowName: step.workflow.name, error: sub.error },
647
+ withDuration(aggregateStats(sub.steps.map((s) => s.stats), 0), start),
648
+ undefined,
649
+ sub.errorCategory,
650
+ );
651
+ }
652
+
653
+ // ---------------------------------------------------------------------------
654
+ // loop_until_dry (keep discovering until K rounds return nothing new)
655
+ // ---------------------------------------------------------------------------
656
+
657
+ async function execLoopUntilDry(step: LoopUntilDryStep, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
658
+ const maxRounds = step.maxRounds ?? 10;
659
+ const dryThreshold = step.dryThreshold ?? 2;
660
+ const keyOf = step.keyOf;
661
+ const merge = step.merge ?? ((known: unknown[], fresh: unknown[]) => known.concat(fresh));
662
+ const start = Date.now();
663
+ let known: unknown[] = [];
664
+ let stats = zeroStats;
665
+ let dry = 0;
666
+ let round = 0;
667
+ let status: StepResult["status"] = "done";
668
+
669
+ while (round < maxRounds) {
670
+ if (exec.signal?.aborted) { status = "skipped"; break; }
671
+ const prompt = await step.prompt(ctx, known);
672
+ const outcome = await dispatchAgentCall(`${step.id}#r${round + 1}`, prompt, {}, exec);
673
+ stats = addStats(stats, outcome.stats);
674
+ if (!outcome.ok) {
675
+ status = outcome.aborted ? "skipped" : "failed";
676
+ break;
677
+ }
678
+ const parsed = parseFirstJson(outcome.value ?? "") as unknown;
679
+ const freshItems: unknown[] = Array.isArray(parsed) ? parsed : parsed !== undefined && parsed !== null ? [parsed] : [];
680
+ const seen = new Set(known.map(keyOf));
681
+ const novel = freshItems.filter((item) => !seen.has(keyOf(item)));
682
+ if (novel.length === 0) {
683
+ dry++;
684
+ if (dry >= dryThreshold) {
685
+ // Completeness critic: ask "what's missing?" one last time.
686
+ if (step.critic) {
687
+ const criticPrompt = await step.critic.prompt(ctx, known);
688
+ const criticOutcome = await dispatchAgentCall(`${step.id}#critic`, criticPrompt, {}, exec);
689
+ stats = addStats(stats, criticOutcome.stats);
690
+ if (criticOutcome.ok) {
691
+ const criticParsed = parseFirstJson(criticOutcome.value ?? "") as unknown;
692
+ const criticItems: unknown[] = Array.isArray(criticParsed) ? criticParsed : [];
693
+ const criticNovel = criticItems.filter((item) => !seen.has(keyOf(item)));
694
+ if (criticNovel.length > 0) {
695
+ known = merge(known, criticNovel);
696
+ dry = 0; // reset dry counter and keep going
697
+ round++;
698
+ continue;
699
+ }
700
+ }
701
+ }
702
+ break;
703
+ }
704
+ } else {
705
+ dry = 0;
706
+ known = merge(known, novel);
707
+ }
708
+ round++;
709
+ }
710
+
711
+ return stepResult(step.id, "loop_until_dry", status, known, withDuration(stats, start), round);
712
+ }
713
+
714
+ // ---------------------------------------------------------------------------
715
+ // runStepSequence — run a list of steps (top-level workflow OR a classify route)
716
+ // ---------------------------------------------------------------------------
717
+
718
+ export interface SequenceOutcome {
719
+ readonly steps: readonly StepResult[];
720
+ readonly status: RunStatus;
721
+ readonly error?: string;
722
+ /** A5: category of the terminal error that failed the sequence (from a
723
+ * thrown WorkflowError), surfaced on RunResult.errorCategory. */
724
+ readonly errorCategory?: ErrorCategory;
725
+ }
726
+
727
+ export async function runStepSequence(
728
+ steps: readonly StepDefinition[],
729
+ input: unknown,
730
+ exec: StepExecContext,
731
+ ): Promise<SequenceOutcome> {
732
+ if (exec.depth > MAX_ROUTE_DEPTH) {
733
+ return { steps: [], status: "failed", error: `classify_route nesting exceeded depth ${MAX_ROUTE_DEPTH} (cycle?)` };
734
+ }
735
+ const prior = new Map<string, { results: unknown; stats: StepStats }>();
736
+ const out: StepResult[] = [];
737
+ const ctx: StepContext = {
738
+ input,
739
+ step(id: string) {
740
+ const p = prior.get(id);
741
+ if (!p) throw new Error(`step "${id}" has not executed yet (or does not exist)`);
742
+ return p;
743
+ },
744
+ };
745
+
746
+ for (const step of steps) {
747
+ if (exec.signal?.aborted) return { steps: out, status: "aborted", error: "aborted by signal" };
748
+ // A3: apply the step's budget-exhaustion policy for the duration of this step.
749
+ exec.budgetPolicy = step.onBudgetExhaust ?? "throw";
750
+ let sr: StepResult;
751
+ try {
752
+ sr = await runWithRetry(step, ctx, exec);
753
+ } catch (e) {
754
+ const wfe = e instanceof WorkflowError ? e : undefined;
755
+ return {
756
+ steps: out,
757
+ status: "failed",
758
+ error: e instanceof Error ? e.message : String(e),
759
+ errorCategory: wfe?.category,
760
+ };
761
+ }
762
+ prior.set(step.id, { results: sr.results, stats: sr.stats });
763
+ out.push(sr);
764
+ // Post-step signal check: a run aborted mid-step (e.g. a fan_out whose
765
+ // items all aborted) must never report "completed".
766
+ if (exec.signal?.aborted) return { steps: out, status: "aborted", error: "aborted by signal" };
767
+ if (sr.status === "failed") {
768
+ const why = typeof sr.results === "string" && sr.results ? `: ${sr.results}` : "";
769
+ return { steps: out, status: "failed", error: `step "${step.id}" failed${why}`, errorCategory: sr.errorCategory };
770
+ }
771
+ if (sr.status === "skipped") {
772
+ const aborted = !!exec.signal?.aborted;
773
+ return { steps: out, status: aborted ? "aborted" : "failed", error: aborted ? "aborted by signal" : `step "${step.id}" skipped` };
774
+ }
775
+ }
776
+ return { steps: out, status: "completed" };
777
+ }
778
+
779
+ async function runWithRetry(step: StepDefinition, ctx: StepContext, exec: StepExecContext): Promise<StepResult> {
780
+ let sr = await executeStep(step, ctx, exec);
781
+ const max = step.retry?.maxRetries ?? 0;
782
+ let attempt = 0;
783
+ let stats = sr.stats; // accumulate every attempt's stats (the pool is charged per dispatch)
784
+ // A5: retry is gated by error category, not by bare status. Only retryable
785
+ // categories (dispatch-error / unexpected-state) auto-retry — a code
786
+ // transform failure or a terminal category (size-limit, determinism, …) must
787
+ // not burn budget on a deterministic re-run.
788
+ while (sr.status === "failed" && attempt < max && sr.errorCategory !== undefined && RETRYABLE_CATEGORIES.includes(sr.errorCategory)) {
789
+ if (exec.signal?.aborted) break;
790
+ attempt++;
791
+ sr = await executeStep(step, ctx, exec);
792
+ stats = addStats(stats, sr.stats);
793
+ }
794
+ return stats === sr.stats ? sr : { ...sr, stats };
795
+ }
796
+
797
+ // ---------------------------------------------------------------------------
798
+ // prompt / verdict coercion helpers
799
+ // ---------------------------------------------------------------------------
800
+
801
+ async function resolvePrompt(spec: AgentCallSpec, ctx: StepContext): Promise<string> {
802
+ return typeof spec.prompt === "function" ? spec.prompt(ctx) : spec.prompt;
803
+ }
804
+
805
+ function judgePrompt(index: number, candidate: string, rubric: readonly string[]): string {
806
+ const criteria = rubric.map((r, i) => `${i + 1}. ${r}`).join("\n");
807
+ // The base workflow-subagent systemPrompt already carries the verbatim / raw-JSON
808
+ // discipline, so this task prompt states only the task + the JSON schema it wants.
809
+ return `You are judge ${index + 1}. Evaluate this candidate:\n\n${candidate}\n\nAgainst these criteria:\n${criteria}\n\nReturn JSON matching: {"pass": true|false, "reason": "..."}`;
810
+ }
811
+
812
+ function rankPrompt(index: number, candidates: readonly string[]): string {
813
+ const listing = candidates.map((c, i) => `[${i}] ${c}`).join("\n\n");
814
+ // Base systemPrompt carries the verbatim / raw-JSON discipline; task prompt
815
+ // states only the ranking task + the JSON schema it wants.
816
+ return `You are judge ${index + 1}. Rank these candidates:\n\n${listing}\n\nReturn JSON matching: {"winner": <index>, "reason": "..."}`;
817
+ }
818
+
819
+ /** LLMs sometimes stringify booleans/numbers ("true", "0"); coerce leniently. */
820
+ function parsePassBool(v: unknown): boolean {
821
+ if (typeof v === "string") {
822
+ const s = v.trim().toLowerCase();
823
+ return s === "true" || s === "yes" || s === "y" || s === "1";
824
+ }
825
+ return v === true || v === 1;
826
+ }
827
+ function parseWinnerNum(v: unknown): number | undefined {
828
+ if (typeof v === "number") return Number.isFinite(v) ? Math.trunc(v) : undefined;
829
+ // Number("") / Number(null) === 0 would record a spurious vote for candidate 0.
830
+ if (typeof v !== "string" || !v.trim()) return undefined;
831
+ const n = Number(v);
832
+ return Number.isFinite(n) ? Math.trunc(n) : undefined;
833
+ }
834
+ function parseCategoryStr(v: unknown): string {
835
+ // Trim: LLMs often emit {"category": " bug"} with stray whitespace, which
836
+ // would silently miss the route key and fall through to fallback.
837
+ return typeof v === "string" ? v.trim() : v === undefined || v === null ? "" : String(v).trim();
838
+ }
839
+ function parseReason(v: unknown): string {
840
+ return typeof v === "string" ? v : "";
841
+ }
842
+
843
+ function tallyWinner(picks: readonly (number | undefined)[], n: number): number {
844
+ if (n <= 0) return -1;
845
+ const counts = new Array(n).fill(0);
846
+ for (const p of picks) if (typeof p === "number" && p >= 0 && p < n) counts[p]!++;
847
+ let best = 0;
848
+ let voted = false;
849
+ for (let i = 0; i < n; i++) {
850
+ if (counts[i]! > 0) voted = true;
851
+ if (counts[i]! > counts[best]!) best = i;
852
+ }
853
+ return voted ? best : -1;
854
+ }
855
+
856
+ function outcomesStatus(outcomes: readonly { ok: boolean; aborted: boolean }[]): StepResult["status"] {
857
+ // A real failure (non-abort) fails the step. An aborted item (per-call skip or
858
+ // run signal) does NOT fail individually — but if EVERY item aborted (e.g. all
859
+ // per-call skipped), the step did zero real work → 'skipped' so the run stops
860
+ // instead of reporting 'done' with empty results.
861
+ if (outcomes.length > 0 && outcomes.every((o) => o.aborted)) return "skipped";
862
+ if (outcomes.some((o) => !o.ok && !o.aborted)) return "failed";
863
+ return "done";
864
+ }
865
+
866
+ /** Terminate every in-flight call belonging to `stepId` (via its per-call
867
+ * controller → SIGTERM in spawnAgent). Called from dispatchAgentCall on any
868
+ * failure of that call (settle failure, dispatch throw, A6 rejection): the
869
+ * hung siblings are killed so the batch's allSettled wait settles instead of
870
+ * blocking the run forever on a stalled subprocess. Aborting a healthy
871
+ * in-flight item is fine — the step is already failing, its results are
872
+ * discarded. A degraded (budget-null) or aborted call is NOT a failure and
873
+ * never triggers this. */
874
+ function abortStepCalls(exec: StepExecContext, stepId: string): void {
875
+ for (const [callId, controller] of exec.registry.controllers) {
876
+ if (stepIdOf(callId) === stepId) controller.abort();
877
+ }
878
+ }
879
+
880
+ /** A6: reject oversized / control-character payloads before any spawn. On
881
+ * rejection the step's in-flight siblings are aborted first (fail-fast for
882
+ * concurrent batches), then a terminal WorkflowError is thrown. */
883
+ function assertPromptAllowed(exec: StepExecContext, callId: string, prompt: string, systemPrompt: string | undefined): void {
884
+ const promptBytes = Buffer.byteLength(prompt, "utf8");
885
+ if (promptBytes > exec.maxPromptBytes) {
886
+ abortStepCalls(exec, stepIdOf(callId));
887
+ throw new WorkflowError(
888
+ `prompt size ${promptBytes} bytes exceeds limit ${exec.maxPromptBytes} bytes`,
889
+ { category: "size-limit", detail: { bytes: promptBytes, limit: exec.maxPromptBytes } },
890
+ );
891
+ }
892
+ if (CONTROL_CHARS.test(prompt)) {
893
+ abortStepCalls(exec, stepIdOf(callId));
894
+ throw new WorkflowError("prompt contains non-printable control characters", { category: "control-chars" });
895
+ }
896
+ if (systemPrompt) {
897
+ const sysBytes = Buffer.byteLength(systemPrompt, "utf8");
898
+ if (sysBytes > exec.maxPromptBytes) {
899
+ abortStepCalls(exec, stepIdOf(callId));
900
+ throw new WorkflowError(
901
+ `systemPrompt size ${sysBytes} bytes exceeds limit ${exec.maxPromptBytes} bytes`,
902
+ { category: "size-limit", detail: { bytes: sysBytes, limit: exec.maxPromptBytes } },
903
+ );
904
+ }
905
+ if (CONTROL_CHARS.test(systemPrompt)) {
906
+ abortStepCalls(exec, stepIdOf(callId));
907
+ throw new WorkflowError("systemPrompt contains non-printable control characters", { category: "control-chars" });
908
+ }
909
+ }
910
+ }
911
+
912
+ // ---------------------------------------------------------------------------
913
+ // budget / spawn / stats helpers
914
+ // ---------------------------------------------------------------------------
915
+
916
+ /** Pre-check a whole batch (fan_out items, or a composite's total agents) fits the budget + MAX_BATCH cap.
917
+ * Does NOT reserve — individual dispatchAgentCall → guardSpawn → pool.reserve(1) atomically reserves
918
+ * per-agent. This avoids double-counting: a prior guardBatch reserve(total) + per-call guardSpawn
919
+ * reserve(1) charged 2x the agent slots. Serial runStepSequence has no cross-step TOCTOU to exploit
920
+ * the pre-check gap; within-step concurrency is guarded by reserve(1)'s atomic increment. */
921
+ function guardBatch(exec: StepExecContext, total: number, label: string): void {
922
+ assertBatchSize(total);
923
+ // Under the "null" policy (A3), skip the canSpawn pre-check: per-item
924
+ // guardSpawn degrades excess items to null instead. The hard MAX_BATCH cap
925
+ // above still throws regardless of policy.
926
+ if (exec.budgetPolicy !== "null" && !exec.pool.canSpawn(total, Date.now())) {
927
+ throw new BudgetExceededError(`${label} needs ${total} agents but the budget is exhausted`);
928
+ }
929
+ }
930
+
931
+ /** Reserve one agent slot and check token budget. Returns a release handle —
932
+ * call it ONLY if the dispatch never started (spawn threw).
933
+ *
934
+ * Under the "null" budget policy (A3), returns `null` instead of throwing
935
+ * when the budget is exhausted: the caller degrades that call to a null
936
+ * outcome. This is the single atomic chokepoint (sync section, no await gap),
937
+ * so concurrent fan_out workers each see an accurate count — closing the
938
+ * TOCTOU that a pre-check before reserve would reintroduce.
939
+ *
940
+ * Lifetime accounting: `exec.spawned` is incremented SYNCHRONOUSLY here ... */
941
+ function guardSpawn(exec: StepExecContext, callId: string, n: number): (() => void) | null {
942
+ exec.spawned += n;
943
+ assertLifetimeAgents(exec.spawned);
944
+ // isExhausted enforces maxTokens (and maxAgents) before committing.
945
+ if (exec.pool.isExhausted(Date.now())) {
946
+ exec.spawned -= n;
947
+ if (exec.budgetPolicy === "null") {
948
+ exec.degradedStepIds.add(stepIdOf(callId));
949
+ return null;
950
+ }
951
+ throw new BudgetExceededError("agent spawn refused — budget exhausted");
952
+ }
953
+ const releasePool = exec.pool.reserve(n);
954
+ return () => {
955
+ exec.spawned -= n;
956
+ releasePool();
957
+ };
958
+ }
959
+
960
+ function applyOutcome(exec: StepExecContext, res: AgentSpawnResult): void {
961
+ // `spawned` was incremented synchronously in guardSpawn; a settled agent keeps
962
+ // its slot, so don't touch it here — only record token spend.
963
+ exec.pool.track({ tokens: res.usage.input + res.usage.output });
964
+ }
965
+
966
+ async function timed(fn: () => Promise<AgentSpawnResult>): Promise<{ res: AgentSpawnResult; durationMs: number }> {
967
+ const start = Date.now();
968
+ const res = await fn();
969
+ return { res, durationMs: Date.now() - start };
970
+ }
971
+
972
+ function dispatchOpts(
973
+ callId: string,
974
+ prompt: string,
975
+ spec: AgentOpts,
976
+ signal?: AbortSignal,
977
+ listeners?: AgentLifecycleListeners,
978
+ allowChildRecursion = false,
979
+ ): AgentSpawnOptions {
980
+ return {
981
+ callId,
982
+ task: prompt,
983
+ model: spec.model,
984
+ tools: spec.tools ? [...spec.tools] : undefined,
985
+ systemPrompt: spec.systemPrompt,
986
+ signal,
987
+ allowChildRecursion,
988
+ // C3: bridge the spawn's streamed deltas to the lifecycle onUpdate listener,
989
+ // attributed to this callId. When no listener is registered, the subprocess
990
+ // drops the deltas (its onUpdate stays undefined — same as before).
991
+ onUpdate: listeners?.onUpdate ? (delta) => notifyUpdate(listeners, callId, delta) : undefined,
992
+ };
993
+ }
994
+
995
+ function finalText(res: AgentSpawnResult): string {
996
+ for (let i = res.messages.length - 1; i >= 0; i--) {
997
+ const msg = res.messages[i];
998
+ if (msg?.role === "assistant") {
999
+ for (const part of msg.content) {
1000
+ if (part.type === "text") return part.text;
1001
+ }
1002
+ }
1003
+ }
1004
+ return "";
1005
+ }
1006
+
1007
+ const zeroStats: StepStats = { tokens: 0, cost: 0, durationMs: 0, agents: 0, failures: 0 };
1008
+
1009
+ function usageStats(res: AgentSpawnResult, durationMs: number, ok: boolean): StepStats {
1010
+ return {
1011
+ tokens: res.usage.input + res.usage.output,
1012
+ cost: res.usage.cost,
1013
+ durationMs,
1014
+ agents: 1,
1015
+ failures: ok ? 0 : 1,
1016
+ };
1017
+ }
1018
+
1019
+ function addStats(a: StepStats, b: StepStats): StepStats {
1020
+ return {
1021
+ tokens: a.tokens + b.tokens,
1022
+ cost: a.cost + b.cost,
1023
+ durationMs: a.durationMs + b.durationMs,
1024
+ agents: a.agents + b.agents,
1025
+ failures: a.failures + b.failures,
1026
+ };
1027
+ }
1028
+
1029
+ /** Sum a list of StepStats. `durationMs` is the wall-clock override (0 to keep the per-call sum). */
1030
+ export function aggregateStats(stats: readonly StepStats[], durationMs: number): StepStats {
1031
+ let tokens = 0;
1032
+ let cost = 0;
1033
+ let agents = 0;
1034
+ let failures = 0;
1035
+ let dur = 0;
1036
+ for (const s of stats) {
1037
+ tokens += s.tokens;
1038
+ cost += s.cost;
1039
+ agents += s.agents;
1040
+ failures += s.failures;
1041
+ dur += s.durationMs;
1042
+ }
1043
+ return { tokens, cost, durationMs: durationMs > 0 ? durationMs : dur, agents, failures };
1044
+ }
1045
+
1046
+ function withDuration(stats: StepStats, start: number): StepStats {
1047
+ return { ...stats, durationMs: Date.now() - start };
1048
+ }
1049
+
1050
+ function stepResult(
1051
+ id: string,
1052
+ type: StepResult["type"],
1053
+ status: StepResult["status"],
1054
+ results: unknown,
1055
+ stats: StepStats,
1056
+ iterations?: number,
1057
+ errorCategory?: ErrorCategory,
1058
+ ): StepResult {
1059
+ const base: StepResult = { id, type, status, results, stats };
1060
+ if (iterations !== undefined) return { ...base, iterations, errorCategory };
1061
+ if (errorCategory !== undefined) return { ...base, errorCategory };
1062
+ return base;
1063
+ }
1064
+
1065
+ function notifyStart(exec: StepExecContext, callId: string): void {
1066
+ try {
1067
+ exec.listeners?.onAgentStart?.(callId);
1068
+ } catch {
1069
+ /* listener robustness — a throwing listener never blocks dispatch */
1070
+ }
1071
+ }
1072
+ function notifyEnd(exec: StepExecContext, callId: string, ok: boolean, stats: StepStats, model?: string, output?: string): void {
1073
+ try {
1074
+ exec.listeners?.onAgentEnd?.(callId, ok, stats, model, output);
1075
+ } catch {
1076
+ /* listener robustness */
1077
+ }
1078
+ }