@esso0428/pi-subagents 0.17.6 → 0.17.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/CONTRIBUTING.md +4 -0
  3. package/dist/abortable.d.ts +13 -0
  4. package/dist/abortable.d.ts.map +1 -0
  5. package/dist/abortable.js +43 -0
  6. package/dist/abortable.js.map +1 -0
  7. package/dist/agent-color.d.ts +36 -0
  8. package/dist/agent-color.d.ts.map +1 -0
  9. package/dist/agent-color.js +124 -0
  10. package/dist/agent-color.js.map +1 -0
  11. package/dist/agent-file-toggle.d.ts +126 -0
  12. package/dist/agent-file-toggle.d.ts.map +1 -0
  13. package/dist/agent-file-toggle.js +259 -0
  14. package/dist/agent-file-toggle.js.map +1 -0
  15. package/dist/agent-history.d.ts +4 -0
  16. package/dist/agent-history.d.ts.map +1 -1
  17. package/dist/agent-history.js +47 -1
  18. package/dist/agent-history.js.map +1 -1
  19. package/dist/agent-manager.d.ts +370 -56
  20. package/dist/agent-manager.d.ts.map +1 -1
  21. package/dist/agent-manager.js +1123 -409
  22. package/dist/agent-manager.js.map +1 -1
  23. package/dist/agent-runner.d.ts +100 -10
  24. package/dist/agent-runner.d.ts.map +1 -1
  25. package/dist/agent-runner.js +166 -21
  26. package/dist/agent-runner.js.map +1 -1
  27. package/dist/agent-types.d.ts +57 -5
  28. package/dist/agent-types.d.ts.map +1 -1
  29. package/dist/agent-types.js +164 -32
  30. package/dist/agent-types.js.map +1 -1
  31. package/dist/child-context.d.ts +3 -0
  32. package/dist/child-context.d.ts.map +1 -0
  33. package/dist/child-context.js +13 -0
  34. package/dist/child-context.js.map +1 -0
  35. package/dist/cross-extension-rpc.d.ts +23 -3
  36. package/dist/cross-extension-rpc.d.ts.map +1 -1
  37. package/dist/cross-extension-rpc.js +79 -17
  38. package/dist/cross-extension-rpc.js.map +1 -1
  39. package/dist/custom-agents.d.ts +38 -1
  40. package/dist/custom-agents.d.ts.map +1 -1
  41. package/dist/custom-agents.js +164 -12
  42. package/dist/custom-agents.js.map +1 -1
  43. package/dist/index.d.ts +34 -0
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +1912 -492
  46. package/dist/index.js.map +1 -1
  47. package/dist/invocation-config.d.ts +87 -2
  48. package/dist/invocation-config.d.ts.map +1 -1
  49. package/dist/invocation-config.js +71 -3
  50. package/dist/invocation-config.js.map +1 -1
  51. package/dist/mention-clone.d.ts +88 -0
  52. package/dist/mention-clone.d.ts.map +1 -0
  53. package/dist/mention-clone.js +154 -0
  54. package/dist/mention-clone.js.map +1 -0
  55. package/dist/mention.d.ts +82 -0
  56. package/dist/mention.d.ts.map +1 -0
  57. package/dist/mention.js +132 -0
  58. package/dist/mention.js.map +1 -0
  59. package/dist/model-resolver.d.ts +17 -0
  60. package/dist/model-resolver.d.ts.map +1 -1
  61. package/dist/model-resolver.js +15 -0
  62. package/dist/model-resolver.js.map +1 -1
  63. package/dist/model-scope.d.ts +50 -0
  64. package/dist/model-scope.d.ts.map +1 -0
  65. package/dist/model-scope.js +49 -0
  66. package/dist/model-scope.js.map +1 -0
  67. package/dist/nested-tools.d.ts +57 -0
  68. package/dist/nested-tools.d.ts.map +1 -0
  69. package/dist/nested-tools.js +301 -0
  70. package/dist/nested-tools.js.map +1 -0
  71. package/dist/output-file.d.ts +22 -3
  72. package/dist/output-file.d.ts.map +1 -1
  73. package/dist/output-file.js +58 -7
  74. package/dist/output-file.js.map +1 -1
  75. package/dist/prompts.d.ts +23 -0
  76. package/dist/prompts.d.ts.map +1 -1
  77. package/dist/prompts.js +20 -2
  78. package/dist/prompts.js.map +1 -1
  79. package/dist/schedule.d.ts.map +1 -1
  80. package/dist/schedule.js +36 -15
  81. package/dist/schedule.js.map +1 -1
  82. package/dist/settings.d.ts +228 -2
  83. package/dist/settings.d.ts.map +1 -1
  84. package/dist/settings.js +94 -0
  85. package/dist/settings.js.map +1 -1
  86. package/dist/status-note.d.ts +49 -1
  87. package/dist/status-note.d.ts.map +1 -1
  88. package/dist/status-note.js +62 -1
  89. package/dist/status-note.js.map +1 -1
  90. package/dist/structured-output.d.ts +62 -0
  91. package/dist/structured-output.d.ts.map +1 -0
  92. package/dist/structured-output.js +113 -0
  93. package/dist/structured-output.js.map +1 -0
  94. package/dist/types.d.ts +176 -10
  95. package/dist/types.d.ts.map +1 -1
  96. package/dist/ui/agent-mention.d.ts +83 -0
  97. package/dist/ui/agent-mention.d.ts.map +1 -0
  98. package/dist/ui/agent-mention.js +188 -0
  99. package/dist/ui/agent-mention.js.map +1 -0
  100. package/dist/ui/agent-widget.d.ts +97 -75
  101. package/dist/ui/agent-widget.d.ts.map +1 -1
  102. package/dist/ui/agent-widget.js +398 -420
  103. package/dist/ui/agent-widget.js.map +1 -1
  104. package/dist/ui/conversation-blocks.d.ts.map +1 -1
  105. package/dist/ui/conversation-blocks.js +6 -0
  106. package/dist/ui/conversation-blocks.js.map +1 -1
  107. package/dist/ui/conversation-timeline.d.ts +10 -2
  108. package/dist/ui/conversation-timeline.d.ts.map +1 -1
  109. package/dist/ui/conversation-timeline.js +130 -23
  110. package/dist/ui/conversation-timeline.js.map +1 -1
  111. package/dist/ui/conversation-viewer.d.ts +15 -5
  112. package/dist/ui/conversation-viewer.d.ts.map +1 -1
  113. package/dist/ui/conversation-viewer.js +202 -50
  114. package/dist/ui/conversation-viewer.js.map +1 -1
  115. package/dist/ui/fleet-list.d.ts +198 -0
  116. package/dist/ui/fleet-list.d.ts.map +1 -0
  117. package/dist/ui/fleet-list.js +487 -0
  118. package/dist/ui/fleet-list.js.map +1 -0
  119. package/dist/ui/schedule-menu.d.ts.map +1 -1
  120. package/dist/ui/schedule-menu.js +6 -7
  121. package/dist/ui/schedule-menu.js.map +1 -1
  122. package/dist/ui/select-item.d.ts +28 -0
  123. package/dist/ui/select-item.d.ts.map +1 -0
  124. package/dist/ui/select-item.js +35 -0
  125. package/dist/ui/select-item.js.map +1 -0
  126. package/dist/ui/workflow-card.d.ts +176 -0
  127. package/dist/ui/workflow-card.d.ts.map +1 -0
  128. package/dist/ui/workflow-card.js +333 -0
  129. package/dist/ui/workflow-card.js.map +1 -0
  130. package/dist/ui/workflow-dialog.d.ts +306 -0
  131. package/dist/ui/workflow-dialog.d.ts.map +1 -0
  132. package/dist/ui/workflow-dialog.js +844 -0
  133. package/dist/ui/workflow-dialog.js.map +1 -0
  134. package/dist/ui/workflow-menu.d.ts +61 -0
  135. package/dist/ui/workflow-menu.d.ts.map +1 -0
  136. package/dist/ui/workflow-menu.js +148 -0
  137. package/dist/ui/workflow-menu.js.map +1 -0
  138. package/dist/usage.d.ts +86 -1
  139. package/dist/usage.d.ts.map +1 -1
  140. package/dist/usage.js +72 -1
  141. package/dist/usage.js.map +1 -1
  142. package/dist/workflow/collisions.d.ts +96 -0
  143. package/dist/workflow/collisions.d.ts.map +1 -0
  144. package/dist/workflow/collisions.js +89 -0
  145. package/dist/workflow/collisions.js.map +1 -0
  146. package/dist/workflow/entry.d.ts +33 -0
  147. package/dist/workflow/entry.d.ts.map +1 -0
  148. package/dist/workflow/entry.js +30 -0
  149. package/dist/workflow/entry.js.map +1 -0
  150. package/dist/workflow/host.d.ts +63 -0
  151. package/dist/workflow/host.d.ts.map +1 -0
  152. package/dist/workflow/host.js +363 -0
  153. package/dist/workflow/host.js.map +1 -0
  154. package/dist/workflow/journal.d.ts +98 -0
  155. package/dist/workflow/journal.d.ts.map +1 -0
  156. package/dist/workflow/journal.js +121 -0
  157. package/dist/workflow/journal.js.map +1 -0
  158. package/dist/workflow/json-schema.d.ts +52 -0
  159. package/dist/workflow/json-schema.d.ts.map +1 -0
  160. package/dist/workflow/json-schema.js +112 -0
  161. package/dist/workflow/json-schema.js.map +1 -0
  162. package/dist/workflow/meta.d.ts +68 -0
  163. package/dist/workflow/meta.d.ts.map +1 -0
  164. package/dist/workflow/meta.js +318 -0
  165. package/dist/workflow/meta.js.map +1 -0
  166. package/dist/workflow/progress.d.ts +225 -0
  167. package/dist/workflow/progress.d.ts.map +1 -0
  168. package/dist/workflow/progress.js +362 -0
  169. package/dist/workflow/progress.js.map +1 -0
  170. package/dist/workflow/runtime.d.ts +335 -0
  171. package/dist/workflow/runtime.d.ts.map +1 -0
  172. package/dist/workflow/runtime.js +831 -0
  173. package/dist/workflow/runtime.js.map +1 -0
  174. package/dist/workflow/saved.d.ts +91 -0
  175. package/dist/workflow/saved.d.ts.map +1 -0
  176. package/dist/workflow/saved.js +204 -0
  177. package/dist/workflow/saved.js.map +1 -0
  178. package/dist/workflow/task.d.ts +137 -0
  179. package/dist/workflow/task.d.ts.map +1 -0
  180. package/dist/workflow/task.js +208 -0
  181. package/dist/workflow/task.js.map +1 -0
  182. package/dist/workflow/tool-description.d.ts +39 -0
  183. package/dist/workflow/tool-description.d.ts.map +1 -0
  184. package/dist/workflow/tool-description.js +200 -0
  185. package/dist/workflow/tool-description.js.map +1 -0
  186. package/dist/workflow/worker-source.d.ts +48 -0
  187. package/dist/workflow/worker-source.d.ts.map +1 -0
  188. package/dist/workflow/worker-source.js +779 -0
  189. package/dist/workflow/worker-source.js.map +1 -0
  190. package/dist/worktree.d.ts +10 -3
  191. package/dist/worktree.d.ts.map +1 -1
  192. package/dist/worktree.js +58 -54
  193. package/dist/worktree.js.map +1 -1
  194. package/dist/xml.d.ts +11 -0
  195. package/dist/xml.d.ts.map +1 -0
  196. package/dist/xml.js +13 -0
  197. package/dist/xml.js.map +1 -0
  198. package/docs/rpc.md +183 -0
  199. package/docs/superpowers/plans/2026-09-30-upstream-event-workflow-partial-history.md +195 -0
  200. package/docs/superpowers/specs/2026-09-30-upstream-event-workflow-partial-history-design.md +49 -0
  201. package/docs/workflows.md +437 -0
  202. package/examples/agent-tool-description.md +7 -7
  203. package/examples/workflows/compose.js +51 -0
  204. package/examples/workflows/fan-out-audit.js +47 -0
  205. package/examples/workflows/gated-fix.js +60 -0
  206. package/examples/workflows/lib/count-child.js +27 -0
  207. package/examples/workflows/review-panel.js +63 -0
  208. package/examples/workflows/structured-findings.js +78 -0
  209. package/package.json +1 -1
  210. package/src/abortable.ts +43 -0
  211. package/src/agent-color.ts +161 -0
  212. package/src/agent-file-toggle.ts +269 -0
  213. package/src/agent-history.ts +54 -2
  214. package/src/agent-manager.ts +1263 -402
  215. package/src/agent-runner.ts +251 -27
  216. package/src/agent-types.ts +188 -32
  217. package/src/child-context.ts +15 -0
  218. package/src/cross-extension-rpc.ts +96 -20
  219. package/src/custom-agents.ts +170 -13
  220. package/src/index.ts +2029 -536
  221. package/src/invocation-config.ts +118 -3
  222. package/src/mention-clone.ts +196 -0
  223. package/src/mention.ts +141 -0
  224. package/src/model-resolver.ts +18 -0
  225. package/src/model-scope.ts +70 -0
  226. package/src/nested-tools.ts +424 -0
  227. package/src/output-file.ts +61 -6
  228. package/src/prompts.ts +45 -2
  229. package/src/schedule.ts +35 -14
  230. package/src/settings.ts +312 -2
  231. package/src/status-note.ts +66 -1
  232. package/src/structured-output.ts +130 -0
  233. package/src/types.ts +177 -10
  234. package/src/ui/agent-mention.ts +216 -0
  235. package/src/ui/agent-widget.ts +393 -441
  236. package/src/ui/conversation-blocks.ts +6 -0
  237. package/src/ui/conversation-timeline.ts +139 -25
  238. package/src/ui/conversation-viewer.ts +212 -48
  239. package/src/ui/fleet-list.ts +558 -0
  240. package/src/ui/schedule-menu.ts +9 -8
  241. package/src/ui/select-item.ts +45 -0
  242. package/src/ui/workflow-card.ts +470 -0
  243. package/src/ui/workflow-dialog.ts +1115 -0
  244. package/src/ui/workflow-menu.ts +193 -0
  245. package/src/usage.ts +109 -2
  246. package/src/workflow/collisions.ts +123 -0
  247. package/src/workflow/entry.ts +47 -0
  248. package/src/workflow/host.ts +403 -0
  249. package/src/workflow/journal.ts +164 -0
  250. package/src/workflow/json-schema.ts +128 -0
  251. package/src/workflow/meta.ts +325 -0
  252. package/src/workflow/progress.ts +550 -0
  253. package/src/workflow/runtime.ts +1219 -0
  254. package/src/workflow/saved.ts +217 -0
  255. package/src/workflow/task.ts +302 -0
  256. package/src/workflow/tool-description.ts +200 -0
  257. package/src/workflow/worker-source.ts +781 -0
  258. package/src/worktree.ts +69 -55
  259. package/src/xml.ts +13 -0
  260. package/vitest.config.ts +0 -18
@@ -0,0 +1,1219 @@
1
+ /**
2
+ * runtime.ts — the host half of a workflow run.
3
+ *
4
+ * Owns the worker lifecycle, the RPC bridge, the concurrency semaphore, the
5
+ * per-run caps, and the progress log. The script's only route to an agent is a
6
+ * `call` message landing here, which is what makes the caps and the abort story
7
+ * enforceable at all: a script cannot go around them because it has nothing to
8
+ * go around them *with*.
9
+ *
10
+ * Spawning is injected rather than imported. `AgentManager` is a large, stateful
11
+ * dependency and wiring it in directly would make every test here an integration
12
+ * test; a {@link WorkflowHost} stub is a dozen lines. The adapter that binds this
13
+ * to the real manager lives at the call site.
14
+ */
15
+
16
+ import { cpus } from "node:os";
17
+ import { Worker } from "node:worker_threads";
18
+ import { type JournalKeyInput, journalKey, type WorkflowJournalEntry } from "./journal.js";
19
+ import { type CompiledSchema, compileJsonSchema } from "./json-schema.js";
20
+ import { extractMeta, type WorkflowMeta } from "./meta.js";
21
+ import type { WorkflowAgentEntry, WorkflowEntry } from "./progress.js";
22
+ import { WORKER_SOURCE } from "./worker-source.js";
23
+
24
+ /** Matches the `script` field's `maxLength` in the tool schema. */
25
+ export const MAX_SCRIPT_LENGTH = 524_288;
26
+
27
+ /** Agents one run may schedule, in total. */
28
+ export const WORKFLOW_AGENT_CAP = 1000;
29
+
30
+ /** Items one `parallel()` or `pipeline()` call may take. */
31
+ export const WORKFLOW_ITEM_CAP = 4096;
32
+
33
+ /** Nested `workflow()` invocations allowed per run. */
34
+ export const WORKFLOW_NESTED_CAP = 256;
35
+
36
+ /** How much of a prompt or result is kept for the UI. */
37
+ const PREVIEW_LENGTH = 200;
38
+
39
+ export class WorkflowRuntimeError extends Error {}
40
+
41
+ /**
42
+ * Concurrent agents allowed, leaving two cores for the host and the TUI.
43
+ *
44
+ * `Math.max(1, …)` is not decoration: the raw `min(16, cpus - 2)` is 0 on a one-
45
+ * or two-core machine, and a semaphore with zero permits never hands out a slot,
46
+ * so the run would hang before its first agent rather than fail.
47
+ */
48
+ export function workflowConcurrency(cpuCount: number = cpus().length): number {
49
+ return Math.max(1, Math.min(16, cpuCount - 2));
50
+ }
51
+
52
+ /** One agent the script asked for. `agentId` is the handle for {@link WorkflowHost.abortAgent}. */
53
+ export interface WorkflowSpawnRequest {
54
+ agentId: string;
55
+ /** Position in the run, and the progress entry's stable identity. */
56
+ index: number;
57
+ prompt: string;
58
+ label: string;
59
+ agentType: string;
60
+ model?: string;
61
+ /**
62
+ * Reasoning effort for this child, as one of pi's thinking levels.
63
+ *
64
+ * Typed as a plain string because this interface is the host boundary and
65
+ * deliberately knows nothing about pi — `host.ts` is where it becomes a
66
+ * `ThinkingLevel`. The worker has already rejected anything off the list.
67
+ */
68
+ effort?: string;
69
+ isolation?: "worktree";
70
+ /**
71
+ * Called by the host once the child's EFFECTIVE configuration is known —
72
+ * which is when its session exists, not when the spawn resolves.
73
+ *
74
+ * Without it a row could only ever show what the script asked for: a fuzzy
75
+ * `model: "haiku"` stays `haiku` instead of the id it resolved to, an
76
+ * `agent()` that named no model shows nothing at all, and a level pi clamped
77
+ * is presented as the level that was requested (#168, #182).
78
+ *
79
+ * Plain strings, like `effort` above: this interface is the host boundary and
80
+ * deliberately knows nothing about pi's `AgentInvocation`. Optional, so a host
81
+ * that cannot report any of this simply does not, and the row keeps the
82
+ * requested values it started with.
83
+ */
84
+ onResolved?(info: {
85
+ /**
86
+ * The host's own id for the child — the manager's `AgentRecord` id here.
87
+ *
88
+ * Reported as soon as the host has one, which is earlier than the rest of
89
+ * this: the model is knowable only once a session exists, but the id is
90
+ * what lets a reader open that child's conversation, and a child that
91
+ * never got a model is exactly the one worth opening.
92
+ */
93
+ recordId?: string;
94
+ modelName?: string;
95
+ modelId?: string;
96
+ thinking?: string;
97
+ requestedThinking?: string;
98
+ requestedModel?: string;
99
+ }): void;
100
+ /**
101
+ * Compiled from the script's `agent({ schema })`.
102
+ *
103
+ * The host must give the child a `StructuredOutput` tool built from it and
104
+ * return the validated payload as JSON text. Compiled rather than raw so the
105
+ * runtime can re-check the answer without re-parsing the schema per call.
106
+ */
107
+ schema?: CompiledSchema;
108
+ phaseIndex?: number;
109
+ phaseTitle?: string;
110
+ /**
111
+ * The `gate` command this agent is being spawned under, when it has one.
112
+ *
113
+ * Passed down rather than run purely from here because an isolated child's
114
+ * worktree is destroyed as part of its own settle: a host that can reach
115
+ * inside that settle runs the gate there, against the tree the child wrote,
116
+ * and reports the outcome back as {@link WorkflowSpawnResult.gate}. A host
117
+ * that ignores this leaves the gate to {@link applyGate}, which then runs it
118
+ * itself — so exactly one execution either way.
119
+ */
120
+ gate?: string;
121
+ }
122
+
123
+ export interface WorkflowSpawnResult {
124
+ ok: boolean;
125
+ /** The agent's answer. Present when `ok`. */
126
+ text?: string;
127
+ /** Why it failed. Present when not `ok`. */
128
+ error?: string;
129
+ /** The user dismissed it rather than it failing; renders as skipped. */
130
+ skipped?: boolean;
131
+ tokens?: number;
132
+ /**
133
+ * Output tokens only, for the script's `budget.spent()`.
134
+ *
135
+ * Separate from {@link tokens}, which is the lifetime total. Claude Code's
136
+ * budget counts output, and a fan-out's re-sent input would swamp it.
137
+ */
138
+ outputTokens?: number;
139
+ /** Whether the child needed an extra prompt to produce its structured answer. */
140
+ structuredRetried?: boolean;
141
+ toolCalls?: number;
142
+ /**
143
+ * Where the child actually ran.
144
+ *
145
+ * Only meaningful for `isolation: "worktree"`, and the whole reason it exists:
146
+ * a gate has to run against the tree the child edited, not the main one, or it
147
+ * verifies the wrong working copy. Left unset, a gate runs wherever the host
148
+ * runs commands by default.
149
+ *
150
+ * Usually unset for a worktree child even so: the copy is removed during the
151
+ * child's own settle, so it no longer exists by the time this is read. That
152
+ * is what {@link gate} is for.
153
+ */
154
+ cwd?: string;
155
+ /**
156
+ * The outcome of this agent's `gate`, when the host already ran it.
157
+ *
158
+ * Set only by a host that ran the command itself — inside the child's
159
+ * worktree, while that directory still existed. Its presence is what tells
160
+ * {@link applyGate} the command has already been executed; the pass/fail
161
+ * decision and the error shaping still happen there, in one place.
162
+ */
163
+ gate?: WorkflowGateResult;
164
+ }
165
+
166
+ /** Outcome of a `gate` command. `output` is what the user is shown when it fails. */
167
+ export interface WorkflowGateResult {
168
+ ok: boolean;
169
+ /** Combined stdout/stderr, or whatever the host wants surfaced as the failure. */
170
+ output: string;
171
+ }
172
+
173
+ /** The one seam between a workflow and the rest of the extension. */
174
+ /** How a script names another workflow: a saved name, or a path to a file. */
175
+ export interface WorkflowScriptRef {
176
+ name?: string;
177
+ scriptPath?: string;
178
+ }
179
+
180
+ export type WorkflowScriptSource =
181
+ | { ok: true; script: string; path?: string }
182
+ | { ok: false; message: string };
183
+
184
+ export interface WorkflowHost {
185
+ spawnAgent(request: WorkflowSpawnRequest): Promise<WorkflowSpawnResult>;
186
+ /** Called for every in-flight agent when the run aborts. */
187
+ abortAgent(agentId: string): void;
188
+ /**
189
+ * Continue a child that already ran in this run, keeping its context.
190
+ *
191
+ * `agentId` is one previously handed out in a {@link WorkflowSpawnRequest};
192
+ * the child keeps the agent type, model and tool contract it started with, so
193
+ * only the follow-up prompt crosses.
194
+ *
195
+ * Optional: a host without it rejects `resume` rather than quietly starting a
196
+ * fresh child that has none of the context the script is counting on.
197
+ */
198
+ resumeAgent?(
199
+ agentId: string,
200
+ prompt: string,
201
+ /**
202
+ * Same reporter {@link WorkflowSpawnRequest.onResolved} carries, for the
203
+ * same reason: a resumed row is rebuilt from scratch, so without it the
204
+ * continuation of a child would show the model the script *asked* for while
205
+ * the row above it shows the one that ran.
206
+ */
207
+ onResolved?: WorkflowSpawnRequest["onResolved"],
208
+ ): Promise<WorkflowSpawnResult>;
209
+ /**
210
+ * Run a `gate` command and report whether it passed.
211
+ *
212
+ * `cwd` is the child's worktree when it had one. Optional for the same reason
213
+ * as {@link resumeAgent}, and more sharply: a gate that silently does not run
214
+ * would mark unverified work as verified, so the runtime fails the call
215
+ * instead of skipping it.
216
+ */
217
+ runGate?(command: string, options: { agentId: string; cwd?: string }): Promise<WorkflowGateResult>;
218
+ /**
219
+ * Resolve a nested `workflow()` reference to source.
220
+ *
221
+ * The runtime knows nothing about the filesystem or about pi, so it asks. It
222
+ * still decides whether what comes back *is* a workflow — see
223
+ * {@link validateScript} — because those rules belong with the runtime that
224
+ * enforces them everywhere else.
225
+ *
226
+ * Optional for the same reason as {@link resumeAgent}: a host without it
227
+ * rejects `workflow()` outright rather than silently running nothing.
228
+ */
229
+ loadWorkflow?(ref: WorkflowScriptRef): Promise<WorkflowScriptSource> | WorkflowScriptSource;
230
+ }
231
+
232
+ /**
233
+ * What a run can be told to do while it is going, from the workflows dialog.
234
+ *
235
+ * Every method is best-effort and idempotent: the dialog renders off a progress
236
+ * log that lags the runtime slightly, so it will sometimes ask for something
237
+ * that has just stopped being possible. `false` means "there was nothing to do
238
+ * that to" — a caller can say so, but it is never an error.
239
+ */
240
+ export interface WorkflowControl {
241
+ /**
242
+ * Stop *starting* agents. Ones already running are left to finish, because
243
+ * killing model work mid-turn throws away everything it has spent and there
244
+ * is no way to hand it back its context.
245
+ */
246
+ pause(): void;
247
+ resume(): void;
248
+ isPaused(): boolean;
249
+ /**
250
+ * Give up on the agent at `index`: its `agent()` call returns `null`, exactly
251
+ * as a terminal failure does, and the row renders skipped.
252
+ *
253
+ * Immediate for a running agent and for one held at a pause. An agent parked
254
+ * behind the concurrency limit takes its skip when it reaches the front —
255
+ * the alternative is a cancellable semaphore for a case that resolves itself
256
+ * as soon as any sibling finishes.
257
+ */
258
+ skip(index: number): boolean;
259
+ /**
260
+ * Start the agent at `index` over: the child is stopped and the same call is
261
+ * re-run, so the script's `agent()` promise is still the one waiting and it
262
+ * gets the new answer.
263
+ *
264
+ * Only while it is running — that is the whole window. Once the call has
265
+ * settled its value is already the script's, and re-running would produce a
266
+ * result with nowhere to go.
267
+ */
268
+ retry(index: number): boolean;
269
+ }
270
+
271
+ export interface RunWorkflowOptions {
272
+ /** Full script source, starting with `export const meta = { … }`. */
273
+ script: string;
274
+ args?: unknown;
275
+ host: WorkflowHost;
276
+ signal?: AbortSignal;
277
+ /** Fired per batch, not per entry — see the worker's progress batching. */
278
+ onProgress?(entries: readonly WorkflowEntry[]): void;
279
+ concurrency?: number;
280
+ agentCap?: number;
281
+ itemCap?: number;
282
+ /**
283
+ * Hands the caller the run's control surface, once per run.
284
+ *
285
+ * A callback rather than a return value because `runWorkflow` resolves when
286
+ * the run is *over*, which is the one moment there is nothing left to
287
+ * control. Fired before the first agent starts.
288
+ */
289
+ onControl?(control: WorkflowControl): void;
290
+ /**
291
+ * How many nested `workflow()` invocations one run may make in total.
292
+ *
293
+ * Each costs a compile and a scope rather than a thread, so the ceiling is
294
+ * generous — but unbounded is worse than capped, on the same reasoning as
295
+ * {@link agentCap}.
296
+ */
297
+ nestedCap?: number;
298
+ /**
299
+ * Replay and record, for `resumeFromRunId`.
300
+ *
301
+ * The runtime does no file IO — `entries` come in already read and `append`
302
+ * goes back out — so its tests stay free of a filesystem, the same reason
303
+ * spawning is behind {@link WorkflowHost}.
304
+ */
305
+ journal?: {
306
+ /** A previous run's settled calls, in position order. Empty replays nothing. */
307
+ entries?: readonly WorkflowJournalEntry[];
308
+ /** Called as each call of *this* run settles, so it can be resumed in turn. */
309
+ append?(entry: WorkflowJournalEntry): void;
310
+ };
311
+ }
312
+
313
+ export interface WorkflowRunResult {
314
+ status: "completed" | "failed" | "killed";
315
+ meta: WorkflowMeta;
316
+ /** The script's return value, JSON-checked at the boundary. */
317
+ value?: unknown;
318
+ error?: string;
319
+ /** The append-only log, in emission order. */
320
+ progress: WorkflowEntry[];
321
+ /** Agents scheduled, including those that failed. */
322
+ agentCount: number;
323
+ /** How many of those came back from the journal instead of being spawned. */
324
+ replayedCount: number;
325
+ }
326
+
327
+ /* ------------------------------------------------------------------------- *
328
+ * JSON boundary — host side
329
+ * ------------------------------------------------------------------------- */
330
+
331
+ function boundaryError(what: string, path: string): WorkflowRuntimeError {
332
+ return new WorkflowRuntimeError(
333
+ `Cannot pass ${what} across the workflow VM boundary (at ${path}).`,
334
+ );
335
+ }
336
+
337
+ function walk(value: unknown, path: string, seen: Set<object>): void {
338
+ if (value === null) return;
339
+ const kind = typeof value;
340
+ if (kind === "string" || kind === "boolean") return;
341
+ if (kind === "number") {
342
+ if (!Number.isFinite(value)) throw boundaryError("a non-finite number", path);
343
+ return;
344
+ }
345
+ if (kind === "undefined") {
346
+ if (path === "args") return;
347
+ throw boundaryError("undefined", path);
348
+ }
349
+ if (kind === "bigint") throw boundaryError("a BigInt", path);
350
+ if (kind === "symbol") throw boundaryError("a symbol", path);
351
+ if (kind === "function") throw boundaryError("a function", path);
352
+ if (kind !== "object") throw boundaryError(`a ${kind}`, path);
353
+
354
+ const object = value as object;
355
+ if (seen.has(object)) throw boundaryError("a circular structure", path);
356
+ seen.add(object);
357
+
358
+ if (Object.getOwnPropertySymbols(object).length > 0) {
359
+ throw boundaryError("an object with symbol keys", path);
360
+ }
361
+
362
+ if (Array.isArray(object)) {
363
+ for (let i = 0; i < object.length; i++) {
364
+ if (!Object.hasOwn(object, i)) throw boundaryError("a sparse array", `${path}[${i}]`);
365
+ walk(object[i], `${path}[${i}]`, seen);
366
+ }
367
+ seen.delete(object);
368
+ return;
369
+ }
370
+
371
+ const prototype = Object.getPrototypeOf(object);
372
+ if (prototype !== null && prototype !== Object.prototype) {
373
+ throw boundaryError("a non-plain object", path);
374
+ }
375
+ for (const [key, entry] of Object.entries(object)) {
376
+ walk(entry, `${path}.${key}`, seen);
377
+ }
378
+ seen.delete(object);
379
+ }
380
+
381
+ /**
382
+ * Reject anything that cannot survive the round trip to the worker and into a
383
+ * resume journal. Structured clone would happily carry a `Map` or a cycle that
384
+ * the journal then cannot represent, so the check is stricter than the transport.
385
+ */
386
+ export function assertBoundarySafe(value: unknown, path: string): void {
387
+ walk(value, path, new Set());
388
+ }
389
+
390
+ /* ------------------------------------------------------------------------- *
391
+ * Semaphore
392
+ * ------------------------------------------------------------------------- */
393
+
394
+ class Semaphore {
395
+ private active = 0;
396
+ private readonly waiters: (() => void)[] = [];
397
+
398
+ constructor(private readonly limit: number) {}
399
+
400
+ acquire(): Promise<void> {
401
+ if (this.active < this.limit) {
402
+ this.active++;
403
+ return Promise.resolve();
404
+ }
405
+ return new Promise<void>(resolve => {
406
+ this.waiters.push(resolve);
407
+ });
408
+ }
409
+
410
+ release(): void {
411
+ const next = this.waiters.shift();
412
+ // Hand the permit straight over rather than decrementing and re-acquiring;
413
+ // otherwise a burst of releases can let more than `limit` through.
414
+ if (next) next();
415
+ else this.active--;
416
+ }
417
+
418
+ /** Wake everyone so aborted callers can observe the abort and bail. */
419
+ drain(): void {
420
+ while (this.waiters.length > 0) {
421
+ const next = this.waiters.shift();
422
+ next?.();
423
+ }
424
+ }
425
+ }
426
+
427
+ /* ------------------------------------------------------------------------- *
428
+ * Messages
429
+ * ------------------------------------------------------------------------- */
430
+
431
+ interface AgentCallPayload {
432
+ prompt: string;
433
+ label?: string;
434
+ model?: string;
435
+ agentType?: string;
436
+ isolation?: "worktree";
437
+ phaseIndex?: number;
438
+ phaseTitle?: string;
439
+ /** Shell command that has to pass before the agent counts as done. */
440
+ gate?: string;
441
+ /** Label of an earlier child in this run to continue instead of starting one. */
442
+ resume?: string;
443
+ /** Reasoning effort, already validated against pi's thinking levels worker-side. */
444
+ effort?: string;
445
+ /** Raw JSON Schema from `agent({ schema })`, compiled before anything spawns. */
446
+ schema?: unknown;
447
+ }
448
+
449
+ type WorkerMessage =
450
+ | { type: "call"; callId: number; method: string; payload: AgentCallPayload }
451
+ | { type: "progress"; entries: WorkflowEntry[] }
452
+ | { type: "complete"; resultJson?: string }
453
+ | { type: "error"; message: string; stack?: string };
454
+
455
+ /** Everything below 0x20 except tab, newline and carriage return, plus DEL. */
456
+ const CONTROL_CHARACTERS = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F]/;
457
+
458
+ const preview = (text: string) =>
459
+ text.length <= PREVIEW_LENGTH ? text : `${text.slice(0, PREVIEW_LENGTH - 1)}…`;
460
+
461
+ /** First line of the prompt, trimmed — the fallback display name for an agent. */
462
+ function derivedLabel(prompt: string): string {
463
+ const line = prompt.split("\n", 1)[0].trim();
464
+ return line.length <= 60 ? line || "agent" : `${line.slice(0, 59)}…`;
465
+ }
466
+
467
+ /**
468
+ * A child `resume` can revive, remembered under its label.
469
+ *
470
+ * The spawn options travel with it because `resume` deliberately takes none: the
471
+ * revived child keeps the agent type, model and isolation it was started with,
472
+ * and the progress entry has to show the same thing the first entry showed.
473
+ */
474
+ interface CompletedChild {
475
+ agentId: string;
476
+ label: string;
477
+ agentType: string;
478
+ model?: string;
479
+ isolation?: "worktree";
480
+ }
481
+
482
+ /**
483
+ * Turn a failing gate into a failing agent.
484
+ *
485
+ * Deliberately no new state, no new entry type: a gated agent whose command
486
+ * fails is *a failed agent*, so the card, the dialog and `agent()`'s `null`
487
+ * return all handle it with the code they already have. The command output
488
+ * becomes the error, because that is the thing worth reading.
489
+ *
490
+ * The single place that decides whether a gate passed. The command may have
491
+ * been run by the host instead (inside a worktree that no longer exists by
492
+ * now), but only ever by one of the two: a host that ran it says so with
493
+ * `result.gate`, and this then shapes that outcome rather than running it
494
+ * again.
495
+ */
496
+ /**
497
+ * Hold a schema'd result to its schema, host-side.
498
+ *
499
+ * The child's own tool already validated whatever it passed, so this normally
500
+ * agrees. It exists for the cases where nothing did: a host that ignores
501
+ * `schema` entirely, a replayed journal entry from before the schema changed,
502
+ * or a payload that reached us some other way. The script asked for a shape;
503
+ * exactly one place should be able to promise it.
504
+ */
505
+ function applySchema(result: WorkflowSpawnResult, compiled: CompiledSchema): WorkflowSpawnResult {
506
+ let parsed: unknown;
507
+ try {
508
+ parsed = JSON.parse(result.text ?? "");
509
+ } catch {
510
+ return {
511
+ ...result,
512
+ ok: false,
513
+ error: "The agent did not return structured output: its answer was not JSON.",
514
+ };
515
+ }
516
+ const verdict = compiled.check(parsed);
517
+ if (verdict === true) return result;
518
+ return {
519
+ ...result,
520
+ ok: false,
521
+ error: `The agent's answer did not match the requested schema: ${verdict}`,
522
+ };
523
+ }
524
+
525
+ async function applyGate(
526
+ result: WorkflowSpawnResult,
527
+ command: string,
528
+ agentId: string,
529
+ runGate: NonNullable<WorkflowHost["runGate"]>,
530
+ ): Promise<WorkflowSpawnResult> {
531
+ const outcome =
532
+ result.gate ??
533
+ (await runGate(command, {
534
+ agentId,
535
+ // Where the child worked, when it had a worktree of its own. Gating the
536
+ // main tree instead would verify code the child never touched.
537
+ ...(result.cwd !== undefined ? { cwd: result.cwd } : {}),
538
+ }));
539
+ const { gate: _ran, ...kept } = result;
540
+ if (outcome.ok) return kept;
541
+ const { text: _discarded, ...rest } = kept;
542
+ const output = outcome.output.trim();
543
+ return { ...rest, ok: false, error: output === "" ? `Gate command failed: ${command}` : output };
544
+ }
545
+
546
+ /**
547
+ * Nico's wording, kept verbatim — this is the one borrowed check whose message a
548
+ * user is likely to search for.
549
+ */
550
+ function unawaitedLaunchMessage(labels: readonly string[]): string {
551
+ const list = labels.map(label => `'${label}'`).join(", ");
552
+ return `workflow script completed with unawaited agent launch(es): ${list}. Await or return each launch.`;
553
+ }
554
+
555
+ /**
556
+ * Run one workflow script to completion.
557
+ *
558
+ * Rejects before starting for a script that cannot run at all (bad `meta`, over
559
+ * the size limit, control characters, non-JSON `args`). Everything after the
560
+ * worker is live resolves instead, carrying the failure in `status` — by then
561
+ * there is a progress log worth handing back.
562
+ */
563
+ /**
564
+ * Everything a script must satisfy before it is compiled.
565
+ *
566
+ * Extracted so a nested `workflow()` is held to exactly the same standard as a
567
+ * top-level run: same size limit, same character rules, same `meta` contract.
568
+ * The host resolves a reference to source; deciding whether that source is a
569
+ * workflow stays here, where the rules live.
570
+ */
571
+ export function validateScript(script: string): { meta: WorkflowMeta; body: string } {
572
+ if (script.length > MAX_SCRIPT_LENGTH) {
573
+ throw new WorkflowRuntimeError(
574
+ `Workflow script is ${script.length} characters, over the limit of ${MAX_SCRIPT_LENGTH}.`,
575
+ );
576
+ }
577
+ if (CONTROL_CHARACTERS.test(script)) {
578
+ throw new WorkflowRuntimeError(
579
+ "Workflow script contains control characters. Only tab, carriage return and newline are allowed.",
580
+ );
581
+ }
582
+ return extractMeta(script);
583
+ }
584
+
585
+ export async function runWorkflow(options: RunWorkflowOptions): Promise<WorkflowRunResult> {
586
+ const { script, host } = options;
587
+
588
+ assertBoundarySafe(options.args, "args");
589
+
590
+ const { meta, body } = validateScript(script);
591
+ const agentCap = options.agentCap ?? WORKFLOW_AGENT_CAP;
592
+ const itemCap = options.itemCap ?? WORKFLOW_ITEM_CAP;
593
+ const semaphore = new Semaphore(options.concurrency ?? workflowConcurrency());
594
+
595
+ const progress: WorkflowEntry[] = [];
596
+ const inflight = new Set<string>();
597
+ /** Label → the child that ran under it, last one wins. The `resume` handle. */
598
+ const completedByLabel = new Map<string, CompletedChild>();
599
+ /**
600
+ * Launches the host has accepted and not yet answered, in call order.
601
+ *
602
+ * This is the whole unawaited-launch mechanism: a script that drops an
603
+ * `agent()` promise still gets its call answered eventually, but it returns
604
+ * first — so anything left here when `complete` arrives is a result nobody is
605
+ * waiting for. Tracking it host-side avoids proxying `Promise` inside the
606
+ * realm, which §2.4 rules out, and reading stack traces, which is brittle.
607
+ */
608
+ const openLaunches = new Map<number, string>();
609
+ let agentCount = 0;
610
+ let aborted = false;
611
+ let settled = false;
612
+
613
+ /* --- resume state ---------------------------------------------------- */
614
+
615
+ const journalEntries = options.journal?.entries ?? [];
616
+ const recordJournal = options.journal?.append;
617
+ /**
618
+ * Whether the replayable prefix is still intact.
619
+ *
620
+ * Once a position misses — different key, a journaled failure, or nothing
621
+ * recorded there — every later call runs live, however well it matches.
622
+ * See the header of journal.ts for why this is a prefix and not a lookup.
623
+ */
624
+ // A journal from a run that used `agent({ resume })` is declined whole: see
625
+ // journal.ts on why a replayed agent leaves nothing for a later resume to
626
+ // continue. Declining up front beats stranding the first `resume` call
627
+ // partway through a run that has already spent its cheap half.
628
+ const journalResumes = journalEntries.some(entry => entry.resumed);
629
+ let prefixIntact = journalEntries.length > 0 && !journalResumes;
630
+ let replayedCount = 0;
631
+
632
+ /* --- live control ---------------------------------------------------- */
633
+
634
+ /**
635
+ * Agents that still have an unanswered `agent()` call, by index.
636
+ *
637
+ * The window in which skip and retry mean anything: before the entry appears
638
+ * there is nothing to act on, and after it is gone the script already has its
639
+ * value. `started` is what separates the two — a retry needs a child to stop.
640
+ */
641
+ interface LiveAgent {
642
+ agentId: string;
643
+ started: boolean;
644
+ intent?: "skip" | "retry";
645
+ /** Wakes it out of a pause hold, so a skip does not wait for a resume. */
646
+ wake?: () => void;
647
+ }
648
+ const liveAgents = new Map<number, LiveAgent>();
649
+
650
+ /**
651
+ * Output tokens this run has spent, mirrored to the script as
652
+ * `budget.spent()`.
653
+ *
654
+ * The host owns the number and every response carries it, rather than the
655
+ * worker accumulating its own: two counters would drift, and there is nothing
656
+ * to gain from the second one. Nor is there observable staleness — tokens
657
+ * only accrue through agents, and the script only learns anything through
658
+ * agent responses.
659
+ */
660
+ let spentOutputTokens = 0;
661
+
662
+ let paused = false;
663
+ /** Read through a call for the same reason `intent()` is — see below. */
664
+ const isPaused = () => paused;
665
+ const pauseWaiters = new Set<() => void>();
666
+ /** Release everyone held at a pause — on resume, and on the way out. */
667
+ function releasePause(): void {
668
+ for (const wake of [...pauseWaiters]) wake();
669
+ pauseWaiters.clear();
670
+ }
671
+ /** Park here while the run is paused, so no new agent is started. */
672
+ function pauseGate(live: LiveAgent): Promise<void> {
673
+ if (!paused || aborted || settled) return Promise.resolve();
674
+ return new Promise<void>(resolve => {
675
+ const wake = () => {
676
+ pauseWaiters.delete(wake);
677
+ live.wake = undefined;
678
+ resolve();
679
+ };
680
+ live.wake = wake;
681
+ pauseWaiters.add(wake);
682
+ });
683
+ }
684
+
685
+ options.onControl?.({
686
+ pause: () => { paused = true; },
687
+ resume: () => { paused = false; releasePause(); },
688
+ isPaused: () => paused,
689
+ skip: index => {
690
+ const live = liveAgents.get(index);
691
+ if (live === undefined || live.intent !== undefined) return false;
692
+ live.intent = "skip";
693
+ // A running child is stopped, which comes back as a skipped result; a
694
+ // held one is woken so it can bail at the gate it is parked on.
695
+ if (live.started) host.abortAgent(live.agentId);
696
+ else live.wake?.();
697
+ return true;
698
+ },
699
+ retry: index => {
700
+ const live = liveAgents.get(index);
701
+ if (live === undefined || !live.started || live.intent !== undefined) return false;
702
+ live.intent = "retry";
703
+ host.abortAgent(live.agentId);
704
+ return true;
705
+ },
706
+ });
707
+
708
+ /** The journal entry to reuse at `index`, or undefined to run it live. */
709
+ function replayAt(index: number, key: string): WorkflowJournalEntry | undefined {
710
+ if (!prefixIntact) return undefined;
711
+ const entry = journalEntries[index];
712
+ if (entry === undefined || entry.index !== index || entry.key !== key || !entry.ok) {
713
+ prefixIntact = false;
714
+ return undefined;
715
+ }
716
+ return entry;
717
+ }
718
+
719
+ const worker = new Worker(WORKER_SOURCE, {
720
+ eval: true,
721
+ workerData: {
722
+ body,
723
+ metaJson: JSON.stringify(meta),
724
+ argsJson: options.args === undefined ? undefined : JSON.stringify(options.args),
725
+ itemCap,
726
+ nestedCap: options.nestedCap ?? WORKFLOW_NESTED_CAP,
727
+ },
728
+ });
729
+
730
+ return await new Promise<WorkflowRunResult>(resolve => {
731
+ const emit = (entries: WorkflowEntry[]) => {
732
+ if (entries.length === 0) return;
733
+ progress.push(...entries);
734
+ options.onProgress?.(entries);
735
+ };
736
+
737
+ const respond = (callId: number, ok: boolean, value?: unknown, error?: string, fatal?: boolean) => {
738
+ // Cleared before the settled check: a launch answered by a run that is
739
+ // already finishing is not an unawaited launch either.
740
+ openLaunches.delete(callId);
741
+ if (settled) return;
742
+ // `spent` rides on every response, so the worker's `budget.spent()` is a
743
+ // mirror of this number rather than a second tally of its own.
744
+ worker.postMessage({ type: "response", callId, ok, value, error, fatal, spent: spentOutputTokens });
745
+ };
746
+
747
+ const finish = (result: Omit<WorkflowRunResult, "meta" | "progress" | "agentCount" | "replayedCount">) => {
748
+ if (settled) return;
749
+ settled = true;
750
+ options.signal?.removeEventListener("abort", onAbort);
751
+ // Symmetric with `semaphore.drain()` below: everything parked is woken so
752
+ // it observes the settle and unwinds. Nothing depends on it — the run's
753
+ // promise resolves either way — it just does not leave live-agent
754
+ // bookkeeping behind for a run that is over.
755
+ releasePause();
756
+ for (const agentId of inflight) host.abortAgent(agentId);
757
+ inflight.clear();
758
+ semaphore.drain();
759
+ // Resolve only once the thread is actually down, so a caller that awaits
760
+ // runWorkflow() is guaranteed not to be leaking one.
761
+ const settle = () => resolve({ ...result, meta, progress, agentCount, replayedCount });
762
+ void worker.terminate().then(settle, settle);
763
+ };
764
+
765
+ function onAbort() {
766
+ aborted = true;
767
+ // terminate() is why this runs in a worker at all: it stops a script that
768
+ // is spinning or wedged mid-await, which an in-process vm cannot do.
769
+ finish({ status: "killed", error: "Workflow aborted." });
770
+ }
771
+
772
+ if (options.signal) {
773
+ if (options.signal.aborted) {
774
+ onAbort();
775
+ return;
776
+ }
777
+ options.signal.addEventListener("abort", onAbort, { once: true });
778
+ }
779
+
780
+ async function handleAgent(callId: number, payload: AgentCallPayload): Promise<void> {
781
+ // Bound now: the optional methods are checked once, up front, so a
782
+ // capability the host lacks fails before an agent is spawned rather than
783
+ // after — a gate that never ran must not be mistaken for a gate that
784
+ // passed.
785
+ const runGate = host.runGate?.bind(host);
786
+ const resumeAgent = host.resumeAgent?.bind(host);
787
+ if (payload.gate !== undefined && runGate === undefined) {
788
+ respond(callId, false, undefined, "This workflow host cannot run gate commands.", true);
789
+ return;
790
+ }
791
+ if (payload.resume !== undefined && resumeAgent === undefined) {
792
+ respond(callId, false, undefined, "This workflow host cannot resume agents.", true);
793
+ return;
794
+ }
795
+
796
+ let resumed: CompletedChild | undefined;
797
+ if (payload.resume !== undefined) {
798
+ resumed = completedByLabel.get(payload.resume);
799
+ if (resumed === undefined) {
800
+ const known = [...completedByLabel.keys()];
801
+ // Fatal: a typo'd label is a script bug, and folding it into a null
802
+ // would show up as an agent that mysteriously returned nothing.
803
+ //
804
+ // Unless agents were replayed, in which case it is not a script bug
805
+ // at all — the label's child came back from the journal and has no
806
+ // conversation here to continue. Saying "no agent has completed"
807
+ // would send the reader hunting for a typo that is not there.
808
+ respond(
809
+ callId,
810
+ false,
811
+ undefined,
812
+ replayedCount > 0 ?
813
+ `agent() opts.resume: "${payload.resume}" was replayed from the resume journal, not run, so there is ` +
814
+ "no conversation in this run to continue. Re-run without resumeFromRunId."
815
+ : `agent() opts.resume: no agent has completed under the label "${payload.resume}" in this run. ${
816
+ known.length === 0
817
+ ? "No agent has completed yet."
818
+ : `Known labels: ${known.map(label => `"${label}"`).join(", ")}.`
819
+ }`,
820
+ true,
821
+ );
822
+ return;
823
+ }
824
+ }
825
+
826
+ // Compiled before anything is scheduled. A schema the runtime cannot use
827
+ // is a script bug, so it is fatal like a typo'd resume label — folding it
828
+ // into a null would surface as an agent that mysteriously returned
829
+ // nothing, and it costs no model call to say so here.
830
+ let compiledSchema: CompiledSchema | undefined;
831
+ if (payload.schema !== undefined) {
832
+ const compilation = compileJsonSchema(payload.schema);
833
+ if (!compilation.ok) {
834
+ respond(callId, false, undefined, compilation.message, true);
835
+ return;
836
+ }
837
+ compiledSchema = compilation.compiled;
838
+ }
839
+
840
+ if (agentCount >= agentCap) {
841
+ // Fatal, so parallel()/pipeline() rethrow instead of folding it into a
842
+ // null. A cap that silently drops work is worse than no cap.
843
+ respond(callId, false, undefined, `Workflow exceeded its cap of ${agentCap} agents.`, true);
844
+ return;
845
+ }
846
+ const index = agentCount++;
847
+ // A resumed call is the same child again: it keeps the agent id, so an
848
+ // abort still reaches it, and it keeps its spawn contract, so the row
849
+ // reads the same as the row it continues.
850
+ const agentId = resumed?.agentId ?? `wf-agent-${index}`;
851
+ const label = payload.label ?? resumed?.label ?? derivedLabel(payload.prompt);
852
+ const agentType = resumed?.agentType ?? payload.agentType ?? "general-purpose";
853
+ const model = resumed !== undefined ? resumed.model : payload.model;
854
+ const isolation = resumed !== undefined ? resumed.isolation : payload.isolation;
855
+ openLaunches.set(callId, label);
856
+
857
+ const base: WorkflowAgentEntry = {
858
+ type: "workflow_agent",
859
+ index,
860
+ label,
861
+ state: "start",
862
+ agentId,
863
+ agentType,
864
+ promptPreview: preview(payload.prompt),
865
+ ...(model !== undefined ? { model } : {}),
866
+ ...(isolation !== undefined ? { isolation } : {}),
867
+ ...(payload.phaseIndex !== undefined ? { phaseIndex: payload.phaseIndex } : {}),
868
+ ...(payload.phaseTitle !== undefined ? { phaseTitle: payload.phaseTitle } : {}),
869
+ };
870
+
871
+ const queuedAt = Date.now();
872
+ emit([{ ...base, queuedAt }]);
873
+
874
+ // Replay before the semaphore, not after: a cached answer is not model
875
+ // running, so it must not hold a concurrency slot that a live agent
876
+ // could use. The row still appears in the tree — the run reads as the
877
+ // same shape it had the first time, just faster.
878
+ // The payload's `schema` is the raw object; the key wants it serialized,
879
+ // so the spread is narrowed rather than passed through.
880
+ const keyInput: JournalKeyInput = {
881
+ ...payload,
882
+ schema: payload.schema !== undefined ? JSON.stringify(payload.schema) : undefined,
883
+ };
884
+ let replayed = replayAt(index, journalKey(keyInput));
885
+ // A replayed answer still has to satisfy the schema. The key covers a
886
+ // schema that *changed*, but not a journal that was hand-edited, and not
887
+ // the empty text a torn entry leaves behind — either would hand the
888
+ // script a null from an entry the journal claims succeeded.
889
+ if (replayed !== undefined && compiledSchema !== undefined) {
890
+ const recheck = applySchema({ ok: true, text: replayed.text ?? "" }, compiledSchema);
891
+ if (!recheck.ok) {
892
+ prefixIntact = false;
893
+ replayed = undefined;
894
+ }
895
+ }
896
+ if (replayed !== undefined) {
897
+ replayedCount++;
898
+ const replayedText = replayed.text ?? "";
899
+ const at = Date.now();
900
+ emit([
901
+ {
902
+ ...base,
903
+ queuedAt,
904
+ startedAt: at,
905
+ lastProgressAt: at,
906
+ durationMs: 0,
907
+ state: "done",
908
+ // The row reads as done, because it is — `cached` is what tells the
909
+ // dialog to annotate it "from resume journal" rather than letting a
910
+ // 0ms agent look like one that did the work impossibly fast.
911
+ cached: true,
912
+ resultPreview: preview(replayedText),
913
+ },
914
+ ]);
915
+ openLaunches.delete(callId);
916
+ // Re-recorded so this run's journal is complete on its own terms: a
917
+ // resume of a resume must not have to walk back through a chain of
918
+ // earlier files to find the prefix.
919
+ recordJournal?.({ index, key: replayed.key, ok: true, text: replayedText });
920
+ respond(callId, true, replayedText);
921
+ return;
922
+ }
923
+
924
+ const key = journalKey(keyInput);
925
+ const resumeMark = payload.resume !== undefined ? ({ resumed: true } as const) : {};
926
+
927
+ /** A skip the user asked for, before the child ever started. */
928
+ const settleSkipped = (extra: Partial<WorkflowAgentEntry>) => {
929
+ recordJournal?.({ index, key, ok: false, ...resumeMark });
930
+ emit([{ ...base, queuedAt, ...extra, state: "error", skipped: true, error: "Skipped by user." }]);
931
+ // `null`, exactly as a terminal failure gives — a skipped agent is one
932
+ // the script's `.filter(Boolean)` was already written to survive.
933
+ respond(callId, true, null);
934
+ };
935
+
936
+ // Registered for exactly as long as the call is unanswered, which is the
937
+ // window in which skip and retry mean anything.
938
+ const live: LiveAgent = { agentId, started: false };
939
+ liveAgents.set(index, live);
940
+ // Read through a call, not off the field: `intent` is set from outside
941
+ // this function while it is suspended at an await, so control-flow
942
+ // narrowing across the awaits would be reasoning about a value that has
943
+ // since changed.
944
+ const intent = (): LiveAgent["intent"] => live.intent;
945
+ let attempt = 1;
946
+ try {
947
+ for (;;) {
948
+ // Held before the slot, not after: a paused run must not sit on
949
+ // concurrency it is not using while its running agents drain.
950
+ await pauseGate(live);
951
+ if (intent() === "skip") return settleSkipped({});
952
+
953
+ // A resumed agent waits its turn like any other: it is the same amount of
954
+ // model running at once.
955
+ await semaphore.acquire();
956
+ if (aborted || settled) {
957
+ semaphore.release();
958
+ respond(callId, false, undefined, "Workflow aborted.", true);
959
+ return;
960
+ }
961
+ // Paused while parked behind the limit: this agent was waiting for a
962
+ // permit when the pause landed, so it never passed the gate above.
963
+ // Hand the permit back and go wait at the gate like everything else,
964
+ // or a pause would leak exactly as many agents as were queued.
965
+ if (isPaused() && !aborted && !settled) {
966
+ semaphore.release();
967
+ continue;
968
+ }
969
+ // Skipped while parked behind the limit: the permit arrived, and the
970
+ // only thing left to do with it is give it back.
971
+ if (intent() === "skip") {
972
+ semaphore.release();
973
+ return settleSkipped({});
974
+ }
975
+
976
+ // Carried on every emit from here on, so a retried row keeps saying
977
+ // why it is on its second attempt instead of losing it to the next
978
+ // progress update.
979
+ const attemptMark =
980
+ attempt > 1 ? { attempt, lastAttemptReason: "user-retry" as const } : {};
981
+
982
+ const startedAt = Date.now();
983
+ emit([{ ...base, queuedAt, startedAt, ...attemptMark }]);
984
+
985
+ // Mutates `base` rather than emitting a standalone patch: every later
986
+ // emit spreads it, so the settle path carries the effective values
987
+ // without knowing they were ever corrected. Re-emitting under the
988
+ // same `index` is what the append-only, last-write-wins progress log
989
+ // is for — the row updates in place while the agent is still running.
990
+ const onResolved = (info: {
991
+ recordId?: string;
992
+ modelName?: string;
993
+ modelId?: string;
994
+ thinking?: string;
995
+ requestedThinking?: string;
996
+ requestedModel?: string;
997
+ }) => {
998
+ if (info.recordId !== undefined) base.recordId = info.recordId;
999
+ if (info.modelName !== undefined) base.model = info.modelName;
1000
+ if (info.modelId !== undefined) base.modelId = info.modelId;
1001
+ if (info.thinking !== undefined) base.thinking = info.thinking;
1002
+ if (info.requestedThinking !== undefined) base.requestedThinking = info.requestedThinking;
1003
+ if (info.requestedModel !== undefined) base.requestedModel = info.requestedModel;
1004
+ // `base.state` is still "start", so emitting after the row reached a
1005
+ // terminal state would revert it to running under last-write-wins.
1006
+ // Not reachable from this repo's host, which reports during startup
1007
+ // — but this is the host boundary, and every other promise it makes
1008
+ // is checked rather than trusted.
1009
+ if (!inflight.has(agentId)) return;
1010
+ emit([{ ...base, queuedAt, startedAt, ...attemptMark, lastProgressAt: Date.now() }]);
1011
+ };
1012
+ live.started = true;
1013
+ inflight.add(agentId);
1014
+
1015
+ let result: WorkflowSpawnResult;
1016
+ try {
1017
+ result =
1018
+ resumed !== undefined && resumeAgent !== undefined
1019
+ ? await resumeAgent(resumed.agentId, payload.prompt, onResolved)
1020
+ : await host.spawnAgent({
1021
+ agentId,
1022
+ index,
1023
+ prompt: payload.prompt,
1024
+ label,
1025
+ agentType,
1026
+ ...(model !== undefined ? { model } : {}),
1027
+ ...(payload.effort !== undefined ? { effort: payload.effort } : {}),
1028
+ ...(compiledSchema !== undefined ? { schema: compiledSchema } : {}),
1029
+ ...(isolation !== undefined ? { isolation } : {}),
1030
+ ...(payload.phaseIndex !== undefined ? { phaseIndex: payload.phaseIndex } : {}),
1031
+ ...(payload.phaseTitle !== undefined ? { phaseTitle: payload.phaseTitle } : {}),
1032
+ // Offered, not delegated: a host that can run it inside the
1033
+ // child's worktree does, and hands back `result.gate`.
1034
+ ...(payload.gate !== undefined ? { gate: payload.gate } : {}),
1035
+ onResolved,
1036
+ });
1037
+ if (result.ok) {
1038
+ // Recorded before the gate runs: the child itself finished, so it is
1039
+ // resumable even when its gate rejects the work — "here is what the
1040
+ // gate said, fix it" is the loop this exists for.
1041
+ completedByLabel.set(label, {
1042
+ agentId,
1043
+ label,
1044
+ agentType,
1045
+ ...(model !== undefined ? { model } : {}),
1046
+ ...(isolation !== undefined ? { isolation } : {}),
1047
+ });
1048
+ // Re-checked here, not just in the child's tool: this is the one
1049
+ // place that decides the script's value matches the schema it
1050
+ // asked for, so a host that ignored `schema` fails loudly instead
1051
+ // of handing the script prose. Before the gate, because a gate
1052
+ // verifies work and there is no work to verify if the shape is
1053
+ // wrong — and the reader should see the schema error, not a gate
1054
+ // error standing in front of it.
1055
+ if (compiledSchema !== undefined && result.ok) {
1056
+ result = applySchema(result, compiledSchema);
1057
+ }
1058
+ if (result.ok && payload.gate !== undefined && runGate !== undefined) {
1059
+ result = await applyGate(result, payload.gate, agentId, runGate);
1060
+ }
1061
+ }
1062
+ } catch (error) {
1063
+ result = { ok: false, error: error instanceof Error ? error.message : String(error) };
1064
+ } finally {
1065
+ inflight.delete(agentId);
1066
+ live.started = false;
1067
+ semaphore.release();
1068
+ }
1069
+
1070
+ if (settled) return;
1071
+
1072
+ // The stop that produced this result was ours, so run the same call
1073
+ // again rather than reporting it. The script is still awaiting this
1074
+ // `agent()`, which is the only reason a retry can mean anything.
1075
+ if (intent() === "retry" && !aborted) {
1076
+ live.intent = undefined;
1077
+ attempt++;
1078
+ emit([{ ...base, queuedAt, attempt, lastAttemptReason: "user-retry" }]);
1079
+ continue;
1080
+ }
1081
+
1082
+ // Counted before the response is sent, so the very call that spent
1083
+ // them already sees them in `budget.spent()`. Failed and skipped
1084
+ // agents count too — they burned the tokens either way.
1085
+ spentOutputTokens += result.outputTokens ?? 0;
1086
+
1087
+ const finishedAt = Date.now();
1088
+ const common = {
1089
+ ...base,
1090
+ queuedAt,
1091
+ startedAt,
1092
+ ...attemptMark,
1093
+ lastProgressAt: finishedAt,
1094
+ durationMs: finishedAt - startedAt,
1095
+ ...(result.tokens !== undefined ? { tokens: result.tokens } : {}),
1096
+ ...(result.toolCalls !== undefined ? { toolCalls: result.toolCalls } : {}),
1097
+ };
1098
+
1099
+ if (result.ok) {
1100
+ const text = result.text ?? "";
1101
+ emit([{ ...common, state: "done", resultPreview: preview(text) }]);
1102
+ recordJournal?.({ index, key, ok: true, text, ...resumeMark });
1103
+ respond(callId, true, text);
1104
+ return;
1105
+ }
1106
+ // Recorded as a failure rather than left out: a gap would be read as an
1107
+ // unchanged prefix on the next resume, silently skipping the retry this
1108
+ // whole mechanism exists to make cheap.
1109
+ recordJournal?.({ index, key, ok: false, ...resumeMark });
1110
+ // A dead agent is a null in the script, not a thrown error: Claude Code
1111
+ // scripts .filter(Boolean) rather than try/catch around every call.
1112
+ emit([
1113
+ {
1114
+ ...common,
1115
+ state: "error",
1116
+ // A user skip reaches here as a stopped child, which the host
1117
+ // already reports as skipped — the flag is taken from the result
1118
+ // rather than from the intent so an abort mid-skip still reads
1119
+ // as whatever actually happened to the child.
1120
+ error: result.error ?? "Agent failed.",
1121
+ ...(result.skipped ? { skipped: true } : {}),
1122
+ },
1123
+ ]);
1124
+ respond(callId, true, null);
1125
+ return;
1126
+ }
1127
+ } finally {
1128
+ liveAgents.delete(index);
1129
+ }
1130
+ }
1131
+
1132
+ /**
1133
+ * Resolve one `workflow(ref)` and hand the child's source back compiled.
1134
+ *
1135
+ * Resolution failures are non-fatal — Claude Code documents `workflow()` as
1136
+ * throwing on an unknown name so a script can catch it and carry on. A host
1137
+ * with no `loadWorkflow` at all is fatal, matching how a missing `runGate`
1138
+ * or `resumeAgent` is treated: a capability the script asked for and this
1139
+ * host cannot provide is a wiring error, not a runtime condition.
1140
+ */
1141
+ async function handleLoadWorkflow(callId: number, ref: WorkflowScriptRef): Promise<void> {
1142
+ const loadWorkflow = host.loadWorkflow?.bind(host);
1143
+ if (loadWorkflow === undefined) {
1144
+ respond(callId, false, undefined, "This workflow host cannot run nested workflows.", true);
1145
+ return;
1146
+ }
1147
+ let source: WorkflowScriptSource;
1148
+ try {
1149
+ source = await loadWorkflow(ref);
1150
+ } catch (error) {
1151
+ respond(callId, false, undefined, error instanceof Error ? error.message : String(error));
1152
+ return;
1153
+ }
1154
+ if (!source.ok) {
1155
+ respond(callId, false, undefined, source.message);
1156
+ return;
1157
+ }
1158
+ try {
1159
+ const child = validateScript(source.script);
1160
+ respond(callId, true, {
1161
+ name: child.meta.name,
1162
+ metaJson: JSON.stringify(child.meta),
1163
+ body: child.body,
1164
+ });
1165
+ } catch (error) {
1166
+ respond(callId, false, undefined, error instanceof Error ? error.message : String(error));
1167
+ }
1168
+ }
1169
+
1170
+ worker.on("message", (message: WorkerMessage) => {
1171
+ if (settled) return;
1172
+ switch (message.type) {
1173
+ case "progress":
1174
+ emit(message.entries);
1175
+ break;
1176
+ case "call":
1177
+ if (message.method === "workflow") {
1178
+ void handleLoadWorkflow(message.callId, message.payload as WorkflowScriptRef);
1179
+ break;
1180
+ }
1181
+ if (message.method !== "agent") {
1182
+ respond(message.callId, false, undefined, `Unknown workflow host method "${message.method}".`, true);
1183
+ break;
1184
+ }
1185
+ void handleAgent(message.callId, message.payload as AgentCallPayload);
1186
+ break;
1187
+ case "complete": {
1188
+ // The script is done, so every launch it made should have been
1189
+ // answered by now — a response is sent before the worker can post
1190
+ // this, so anything still open was never awaited. finish() aborts
1191
+ // those children on the way out.
1192
+ const unawaited = [...openLaunches.values()];
1193
+ if (unawaited.length > 0) {
1194
+ finish({ status: "failed", error: unawaitedLaunchMessage(unawaited) });
1195
+ break;
1196
+ }
1197
+ finish({
1198
+ status: "completed",
1199
+ ...(message.resultJson === undefined ? {} : { value: JSON.parse(message.resultJson) }),
1200
+ });
1201
+ break;
1202
+ }
1203
+ case "error":
1204
+ finish({ status: "failed", error: message.message });
1205
+ break;
1206
+ }
1207
+ });
1208
+
1209
+ worker.on("error", error => {
1210
+ finish({ status: "failed", error: error instanceof Error ? error.message : String(error) });
1211
+ });
1212
+
1213
+ worker.on("exit", () => {
1214
+ // Only reachable when the worker dies without reporting — a terminate()
1215
+ // we did not initiate, or a hard crash.
1216
+ finish({ status: "failed", error: "Workflow worker exited before completing." });
1217
+ });
1218
+ });
1219
+ }