@esso0428/pi-subagents 0.17.6 → 0.17.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/CONTRIBUTING.md +4 -0
  3. package/dist/abortable.d.ts +13 -0
  4. package/dist/abortable.d.ts.map +1 -0
  5. package/dist/abortable.js +43 -0
  6. package/dist/abortable.js.map +1 -0
  7. package/dist/agent-color.d.ts +36 -0
  8. package/dist/agent-color.d.ts.map +1 -0
  9. package/dist/agent-color.js +124 -0
  10. package/dist/agent-color.js.map +1 -0
  11. package/dist/agent-file-toggle.d.ts +126 -0
  12. package/dist/agent-file-toggle.d.ts.map +1 -0
  13. package/dist/agent-file-toggle.js +259 -0
  14. package/dist/agent-file-toggle.js.map +1 -0
  15. package/dist/agent-history.d.ts +4 -0
  16. package/dist/agent-history.d.ts.map +1 -1
  17. package/dist/agent-history.js +47 -1
  18. package/dist/agent-history.js.map +1 -1
  19. package/dist/agent-manager.d.ts +370 -56
  20. package/dist/agent-manager.d.ts.map +1 -1
  21. package/dist/agent-manager.js +1123 -409
  22. package/dist/agent-manager.js.map +1 -1
  23. package/dist/agent-runner.d.ts +100 -10
  24. package/dist/agent-runner.d.ts.map +1 -1
  25. package/dist/agent-runner.js +166 -21
  26. package/dist/agent-runner.js.map +1 -1
  27. package/dist/agent-types.d.ts +57 -5
  28. package/dist/agent-types.d.ts.map +1 -1
  29. package/dist/agent-types.js +164 -32
  30. package/dist/agent-types.js.map +1 -1
  31. package/dist/child-context.d.ts +3 -0
  32. package/dist/child-context.d.ts.map +1 -0
  33. package/dist/child-context.js +13 -0
  34. package/dist/child-context.js.map +1 -0
  35. package/dist/cross-extension-rpc.d.ts +23 -3
  36. package/dist/cross-extension-rpc.d.ts.map +1 -1
  37. package/dist/cross-extension-rpc.js +79 -17
  38. package/dist/cross-extension-rpc.js.map +1 -1
  39. package/dist/custom-agents.d.ts +38 -1
  40. package/dist/custom-agents.d.ts.map +1 -1
  41. package/dist/custom-agents.js +164 -12
  42. package/dist/custom-agents.js.map +1 -1
  43. package/dist/index.d.ts +34 -0
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +1912 -492
  46. package/dist/index.js.map +1 -1
  47. package/dist/invocation-config.d.ts +87 -2
  48. package/dist/invocation-config.d.ts.map +1 -1
  49. package/dist/invocation-config.js +71 -3
  50. package/dist/invocation-config.js.map +1 -1
  51. package/dist/mention-clone.d.ts +88 -0
  52. package/dist/mention-clone.d.ts.map +1 -0
  53. package/dist/mention-clone.js +154 -0
  54. package/dist/mention-clone.js.map +1 -0
  55. package/dist/mention.d.ts +82 -0
  56. package/dist/mention.d.ts.map +1 -0
  57. package/dist/mention.js +132 -0
  58. package/dist/mention.js.map +1 -0
  59. package/dist/model-resolver.d.ts +17 -0
  60. package/dist/model-resolver.d.ts.map +1 -1
  61. package/dist/model-resolver.js +15 -0
  62. package/dist/model-resolver.js.map +1 -1
  63. package/dist/model-scope.d.ts +50 -0
  64. package/dist/model-scope.d.ts.map +1 -0
  65. package/dist/model-scope.js +49 -0
  66. package/dist/model-scope.js.map +1 -0
  67. package/dist/nested-tools.d.ts +57 -0
  68. package/dist/nested-tools.d.ts.map +1 -0
  69. package/dist/nested-tools.js +301 -0
  70. package/dist/nested-tools.js.map +1 -0
  71. package/dist/output-file.d.ts +22 -3
  72. package/dist/output-file.d.ts.map +1 -1
  73. package/dist/output-file.js +58 -7
  74. package/dist/output-file.js.map +1 -1
  75. package/dist/prompts.d.ts +23 -0
  76. package/dist/prompts.d.ts.map +1 -1
  77. package/dist/prompts.js +20 -2
  78. package/dist/prompts.js.map +1 -1
  79. package/dist/schedule.d.ts.map +1 -1
  80. package/dist/schedule.js +36 -15
  81. package/dist/schedule.js.map +1 -1
  82. package/dist/settings.d.ts +228 -2
  83. package/dist/settings.d.ts.map +1 -1
  84. package/dist/settings.js +94 -0
  85. package/dist/settings.js.map +1 -1
  86. package/dist/status-note.d.ts +49 -1
  87. package/dist/status-note.d.ts.map +1 -1
  88. package/dist/status-note.js +62 -1
  89. package/dist/status-note.js.map +1 -1
  90. package/dist/structured-output.d.ts +62 -0
  91. package/dist/structured-output.d.ts.map +1 -0
  92. package/dist/structured-output.js +113 -0
  93. package/dist/structured-output.js.map +1 -0
  94. package/dist/types.d.ts +176 -10
  95. package/dist/types.d.ts.map +1 -1
  96. package/dist/ui/agent-mention.d.ts +83 -0
  97. package/dist/ui/agent-mention.d.ts.map +1 -0
  98. package/dist/ui/agent-mention.js +188 -0
  99. package/dist/ui/agent-mention.js.map +1 -0
  100. package/dist/ui/agent-widget.d.ts +97 -75
  101. package/dist/ui/agent-widget.d.ts.map +1 -1
  102. package/dist/ui/agent-widget.js +398 -420
  103. package/dist/ui/agent-widget.js.map +1 -1
  104. package/dist/ui/conversation-blocks.d.ts.map +1 -1
  105. package/dist/ui/conversation-blocks.js +6 -0
  106. package/dist/ui/conversation-blocks.js.map +1 -1
  107. package/dist/ui/conversation-timeline.d.ts +10 -2
  108. package/dist/ui/conversation-timeline.d.ts.map +1 -1
  109. package/dist/ui/conversation-timeline.js +130 -23
  110. package/dist/ui/conversation-timeline.js.map +1 -1
  111. package/dist/ui/conversation-viewer.d.ts +15 -5
  112. package/dist/ui/conversation-viewer.d.ts.map +1 -1
  113. package/dist/ui/conversation-viewer.js +202 -50
  114. package/dist/ui/conversation-viewer.js.map +1 -1
  115. package/dist/ui/fleet-list.d.ts +198 -0
  116. package/dist/ui/fleet-list.d.ts.map +1 -0
  117. package/dist/ui/fleet-list.js +487 -0
  118. package/dist/ui/fleet-list.js.map +1 -0
  119. package/dist/ui/schedule-menu.d.ts.map +1 -1
  120. package/dist/ui/schedule-menu.js +6 -7
  121. package/dist/ui/schedule-menu.js.map +1 -1
  122. package/dist/ui/select-item.d.ts +28 -0
  123. package/dist/ui/select-item.d.ts.map +1 -0
  124. package/dist/ui/select-item.js +35 -0
  125. package/dist/ui/select-item.js.map +1 -0
  126. package/dist/ui/workflow-card.d.ts +176 -0
  127. package/dist/ui/workflow-card.d.ts.map +1 -0
  128. package/dist/ui/workflow-card.js +333 -0
  129. package/dist/ui/workflow-card.js.map +1 -0
  130. package/dist/ui/workflow-dialog.d.ts +306 -0
  131. package/dist/ui/workflow-dialog.d.ts.map +1 -0
  132. package/dist/ui/workflow-dialog.js +844 -0
  133. package/dist/ui/workflow-dialog.js.map +1 -0
  134. package/dist/ui/workflow-menu.d.ts +61 -0
  135. package/dist/ui/workflow-menu.d.ts.map +1 -0
  136. package/dist/ui/workflow-menu.js +148 -0
  137. package/dist/ui/workflow-menu.js.map +1 -0
  138. package/dist/usage.d.ts +86 -1
  139. package/dist/usage.d.ts.map +1 -1
  140. package/dist/usage.js +72 -1
  141. package/dist/usage.js.map +1 -1
  142. package/dist/workflow/collisions.d.ts +96 -0
  143. package/dist/workflow/collisions.d.ts.map +1 -0
  144. package/dist/workflow/collisions.js +89 -0
  145. package/dist/workflow/collisions.js.map +1 -0
  146. package/dist/workflow/entry.d.ts +33 -0
  147. package/dist/workflow/entry.d.ts.map +1 -0
  148. package/dist/workflow/entry.js +30 -0
  149. package/dist/workflow/entry.js.map +1 -0
  150. package/dist/workflow/host.d.ts +63 -0
  151. package/dist/workflow/host.d.ts.map +1 -0
  152. package/dist/workflow/host.js +363 -0
  153. package/dist/workflow/host.js.map +1 -0
  154. package/dist/workflow/journal.d.ts +98 -0
  155. package/dist/workflow/journal.d.ts.map +1 -0
  156. package/dist/workflow/journal.js +121 -0
  157. package/dist/workflow/journal.js.map +1 -0
  158. package/dist/workflow/json-schema.d.ts +52 -0
  159. package/dist/workflow/json-schema.d.ts.map +1 -0
  160. package/dist/workflow/json-schema.js +112 -0
  161. package/dist/workflow/json-schema.js.map +1 -0
  162. package/dist/workflow/meta.d.ts +68 -0
  163. package/dist/workflow/meta.d.ts.map +1 -0
  164. package/dist/workflow/meta.js +318 -0
  165. package/dist/workflow/meta.js.map +1 -0
  166. package/dist/workflow/progress.d.ts +225 -0
  167. package/dist/workflow/progress.d.ts.map +1 -0
  168. package/dist/workflow/progress.js +362 -0
  169. package/dist/workflow/progress.js.map +1 -0
  170. package/dist/workflow/runtime.d.ts +335 -0
  171. package/dist/workflow/runtime.d.ts.map +1 -0
  172. package/dist/workflow/runtime.js +831 -0
  173. package/dist/workflow/runtime.js.map +1 -0
  174. package/dist/workflow/saved.d.ts +91 -0
  175. package/dist/workflow/saved.d.ts.map +1 -0
  176. package/dist/workflow/saved.js +204 -0
  177. package/dist/workflow/saved.js.map +1 -0
  178. package/dist/workflow/task.d.ts +137 -0
  179. package/dist/workflow/task.d.ts.map +1 -0
  180. package/dist/workflow/task.js +208 -0
  181. package/dist/workflow/task.js.map +1 -0
  182. package/dist/workflow/tool-description.d.ts +39 -0
  183. package/dist/workflow/tool-description.d.ts.map +1 -0
  184. package/dist/workflow/tool-description.js +200 -0
  185. package/dist/workflow/tool-description.js.map +1 -0
  186. package/dist/workflow/worker-source.d.ts +48 -0
  187. package/dist/workflow/worker-source.d.ts.map +1 -0
  188. package/dist/workflow/worker-source.js +779 -0
  189. package/dist/workflow/worker-source.js.map +1 -0
  190. package/dist/worktree.d.ts +10 -3
  191. package/dist/worktree.d.ts.map +1 -1
  192. package/dist/worktree.js +58 -54
  193. package/dist/worktree.js.map +1 -1
  194. package/dist/xml.d.ts +11 -0
  195. package/dist/xml.d.ts.map +1 -0
  196. package/dist/xml.js +13 -0
  197. package/dist/xml.js.map +1 -0
  198. package/docs/rpc.md +183 -0
  199. package/docs/superpowers/plans/2026-09-30-upstream-event-workflow-partial-history.md +195 -0
  200. package/docs/superpowers/specs/2026-09-30-upstream-event-workflow-partial-history-design.md +49 -0
  201. package/docs/workflows.md +437 -0
  202. package/examples/agent-tool-description.md +7 -7
  203. package/examples/workflows/compose.js +51 -0
  204. package/examples/workflows/fan-out-audit.js +47 -0
  205. package/examples/workflows/gated-fix.js +60 -0
  206. package/examples/workflows/lib/count-child.js +27 -0
  207. package/examples/workflows/review-panel.js +63 -0
  208. package/examples/workflows/structured-findings.js +78 -0
  209. package/package.json +1 -1
  210. package/src/abortable.ts +43 -0
  211. package/src/agent-color.ts +161 -0
  212. package/src/agent-file-toggle.ts +269 -0
  213. package/src/agent-history.ts +54 -2
  214. package/src/agent-manager.ts +1263 -402
  215. package/src/agent-runner.ts +251 -27
  216. package/src/agent-types.ts +188 -32
  217. package/src/child-context.ts +15 -0
  218. package/src/cross-extension-rpc.ts +96 -20
  219. package/src/custom-agents.ts +170 -13
  220. package/src/index.ts +2029 -536
  221. package/src/invocation-config.ts +118 -3
  222. package/src/mention-clone.ts +196 -0
  223. package/src/mention.ts +141 -0
  224. package/src/model-resolver.ts +18 -0
  225. package/src/model-scope.ts +70 -0
  226. package/src/nested-tools.ts +424 -0
  227. package/src/output-file.ts +61 -6
  228. package/src/prompts.ts +45 -2
  229. package/src/schedule.ts +35 -14
  230. package/src/settings.ts +312 -2
  231. package/src/status-note.ts +66 -1
  232. package/src/structured-output.ts +130 -0
  233. package/src/types.ts +177 -10
  234. package/src/ui/agent-mention.ts +216 -0
  235. package/src/ui/agent-widget.ts +393 -441
  236. package/src/ui/conversation-blocks.ts +6 -0
  237. package/src/ui/conversation-timeline.ts +139 -25
  238. package/src/ui/conversation-viewer.ts +212 -48
  239. package/src/ui/fleet-list.ts +558 -0
  240. package/src/ui/schedule-menu.ts +9 -8
  241. package/src/ui/select-item.ts +45 -0
  242. package/src/ui/workflow-card.ts +470 -0
  243. package/src/ui/workflow-dialog.ts +1115 -0
  244. package/src/ui/workflow-menu.ts +193 -0
  245. package/src/usage.ts +109 -2
  246. package/src/workflow/collisions.ts +123 -0
  247. package/src/workflow/entry.ts +47 -0
  248. package/src/workflow/host.ts +403 -0
  249. package/src/workflow/journal.ts +164 -0
  250. package/src/workflow/json-schema.ts +128 -0
  251. package/src/workflow/meta.ts +325 -0
  252. package/src/workflow/progress.ts +550 -0
  253. package/src/workflow/runtime.ts +1219 -0
  254. package/src/workflow/saved.ts +217 -0
  255. package/src/workflow/task.ts +302 -0
  256. package/src/workflow/tool-description.ts +200 -0
  257. package/src/workflow/worker-source.ts +781 -0
  258. package/src/worktree.ts +69 -55
  259. package/src/xml.ts +13 -0
  260. package/vitest.config.ts +0 -18
@@ -0,0 +1,208 @@
1
+ /**
2
+ * task.ts — the background record one workflow run lives in.
3
+ *
4
+ * A `SubagentWorkflow` tool call returns a task id immediately and the run continues
5
+ * without it, so the run's state cannot live in the tool call's closure: the
6
+ * inline card, the completion notification and (later) the `/agents → Workflows` dialog
7
+ * all read it after `execute` has returned. This is that record, shaped after
8
+ * Claude Code's `local_workflow` task so the fields line up with what the
9
+ * renderers already expect.
10
+ *
11
+ * The progress log is append-only and collapses by index (see `progress.ts`),
12
+ * so every derived counter here is recomputed from the log rather than
13
+ * incremented as entries arrive — a re-emitted agent entry replaces its
14
+ * predecessor, and adding its tokens on top would double-count them.
15
+ */
16
+ import { randomUUID } from "node:crypto";
17
+ import { escapeXml } from "../xml.js";
18
+ import { collapse, elapsedMs, stats } from "./progress.js";
19
+ /** `wf_` + hex, matching Claude Code's `^wf_[a-z0-9-]{6,}$` run ids. */
20
+ export function workflowRunId() {
21
+ return `wf_${randomUUID().replace(/-/g, "").slice(0, 12)}`;
22
+ }
23
+ export function createWorkflowTask(init) {
24
+ return {
25
+ type: "local_workflow",
26
+ id: init.id,
27
+ status: "running",
28
+ script: init.script,
29
+ scriptPath: init.scriptPath,
30
+ args: init.args,
31
+ meta: init.meta,
32
+ workflowName: init.meta?.name,
33
+ toolCallId: init.toolCallId,
34
+ journalPath: init.journalPath,
35
+ replay: init.replay,
36
+ resumedFrom: init.resumedFrom,
37
+ replayedCount: 0,
38
+ workflowProgress: [],
39
+ progressVersion: 0,
40
+ agentCount: 0,
41
+ doneCount: 0,
42
+ totalTokens: 0,
43
+ totalToolCalls: 0,
44
+ logs: [],
45
+ abortController: new AbortController(),
46
+ startTime: init.startTime ?? Date.now(),
47
+ totalPausedMs: 0,
48
+ };
49
+ }
50
+ /**
51
+ * Apply one batch of progress entries.
52
+ *
53
+ * Batched rather than per-entry because that is how the worker emits them, and
54
+ * because every counter below is an O(log) recompute — doing it once per fan-out
55
+ * frame instead of once per agent is the difference that keeps a 200-agent run
56
+ * cheap to render.
57
+ */
58
+ export function updateWorkflowProgressBatch(task, entries) {
59
+ if (entries.length === 0)
60
+ return;
61
+ task.workflowProgress.push(...entries);
62
+ task.progressVersion++;
63
+ const { agents, logs } = collapse(task.workflowProgress);
64
+ task.logs = logs;
65
+ // `agentCount` is what the runtime has scheduled, which can lead what the log
66
+ // has seen — never let a recompute walk it backwards.
67
+ task.agentCount = Math.max(task.agentCount, agents.length);
68
+ let totalTokens = 0;
69
+ let totalToolCalls = 0;
70
+ let done = 0;
71
+ for (const agent of agents) {
72
+ totalTokens += agent.tokens ?? 0;
73
+ totalToolCalls += agent.toolCalls ?? 0;
74
+ // Counted off the collapsed agents, so a re-emitted row counts once.
75
+ if (agent.state === "done")
76
+ done++;
77
+ }
78
+ task.totalTokens = totalTokens;
79
+ task.totalToolCalls = totalToolCalls;
80
+ task.doneCount = done;
81
+ }
82
+ /**
83
+ * Hold the run, and stop its clock.
84
+ *
85
+ * The elapsed figure every surface shows subtracts `totalPausedMs`, so a run
86
+ * left paused overnight does not come back reading as a twelve-hour run.
87
+ */
88
+ export function pauseWorkflowTask(task, now = Date.now()) {
89
+ if (task.status !== "running" || task.control === undefined)
90
+ return false;
91
+ task.control.pause();
92
+ task.status = "paused";
93
+ task.pausedAt = now;
94
+ return true;
95
+ }
96
+ /** Let it go again, banking however long it was held. */
97
+ export function resumeWorkflowTask(task, now = Date.now()) {
98
+ if (task.status !== "paused" || task.control === undefined)
99
+ return false;
100
+ task.control.resume();
101
+ task.status = "running";
102
+ task.totalPausedMs = (task.totalPausedMs ?? 0) + Math.max(0, now - (task.pausedAt ?? now));
103
+ task.pausedAt = undefined;
104
+ return true;
105
+ }
106
+ /** Settle a task from the run's own result. */
107
+ export function completeWorkflowTask(task, result) {
108
+ // Banked before the status moves off "paused": a run that finished while held
109
+ // still spent that time held, and the elapsed figure has to say so.
110
+ if (task.pausedAt !== undefined) {
111
+ task.totalPausedMs = (task.totalPausedMs ?? 0) + Math.max(0, Date.now() - task.pausedAt);
112
+ task.pausedAt = undefined;
113
+ }
114
+ // Nothing left to control, and holding the handle would let the dialog offer
115
+ // pause on a run that has already stopped.
116
+ task.control = undefined;
117
+ task.status = result.status;
118
+ task.meta ??= result.meta;
119
+ task.workflowName ??= result.meta.name;
120
+ task.agentCount = Math.max(task.agentCount, result.agentCount);
121
+ task.replayedCount = result.replayedCount;
122
+ task.value = result.value;
123
+ task.error = result.error;
124
+ task.endTime = Date.now();
125
+ }
126
+ /**
127
+ * Settle a task that never produced a result — a script rejected before the
128
+ * worker started (bad `meta`, oversized source, non-JSON `args`).
129
+ */
130
+ export function failWorkflowTask(task, error) {
131
+ task.control = undefined;
132
+ task.pausedAt = undefined;
133
+ task.status = "failed";
134
+ task.error = error;
135
+ task.endTime = Date.now();
136
+ }
137
+ /** The run's outcome as text, for the notification and the LLM-facing result. */
138
+ export function workflowResultText(task) {
139
+ if (task.error !== undefined)
140
+ return task.error;
141
+ if (task.value === undefined)
142
+ return "No output.";
143
+ if (typeof task.value === "string")
144
+ return task.value;
145
+ return JSON.stringify(task.value, null, 2);
146
+ }
147
+ /**
148
+ * Resolve a `resumeFromRunId` against the runs this session has seen.
149
+ *
150
+ * Same-session only, and deliberately so: the journal lives beside the
151
+ * session's task files, and a run id from another session would silently find
152
+ * nothing to replay — reporting that as "resumed" would be a lie the caller
153
+ * could not see through. An unknown id is an error rather than a cold start,
154
+ * because a caller that asked to resume is expecting not to pay.
155
+ */
156
+ export function resolveResumeTarget(runId, tasks) {
157
+ const id = runId?.trim();
158
+ if (id === undefined || id === "")
159
+ return undefined;
160
+ const prior = tasks.get(id);
161
+ if (prior === undefined) {
162
+ const known = [...tasks.keys()];
163
+ return {
164
+ ok: false,
165
+ message: `No workflow run "${id}" in this session. ` +
166
+ (known.length > 0
167
+ ? `Runs this session: ${known.join(", ")}.`
168
+ : "Nothing has run yet — call this without `resumeFromRunId`."),
169
+ };
170
+ }
171
+ if (prior.status === "running") {
172
+ return {
173
+ ok: false,
174
+ message: `Workflow "${id}" is still running. Stop it from /agents → Workflows before resuming it.`,
175
+ };
176
+ }
177
+ if (prior.journalPath === undefined) {
178
+ return { ok: false, message: `Workflow "${id}" has no journal to resume from.` };
179
+ }
180
+ return {
181
+ ok: true,
182
+ runId: id,
183
+ journalPath: prior.journalPath,
184
+ // The persisted copy, which is what `scriptPath` holds when the call had
185
+ // no file of its own.
186
+ scriptPath: prior.scriptPath ?? "",
187
+ };
188
+ }
189
+ /** `<task-notification>`, in the same shape a finished background agent sends. */
190
+ export function formatWorkflowNotification(task, now = Date.now()) {
191
+ const totals = stats(task.workflowProgress, task.agentCount);
192
+ const status = task.status === "completed" ? "Done"
193
+ : task.status === "killed" ? "Stopped"
194
+ : `Error: ${task.error ?? "unknown"}`;
195
+ const result = workflowResultText(task);
196
+ return [
197
+ `<task-notification>`,
198
+ `<task-id>${task.id}</task-id>`,
199
+ task.toolCallId ? `<tool-use-id>${escapeXml(task.toolCallId)}</tool-use-id>` : null,
200
+ task.scriptPath ? `<script>${escapeXml(task.scriptPath)}</script>` : null,
201
+ `<status>${escapeXml(status)}</status>`,
202
+ `<summary>Workflow "${escapeXml(task.workflowName ?? task.id)}" ${task.status} — ${totals.done}/${totals.total} agents${task.replayedCount > 0 ? `, ${task.replayedCount} replayed from ${escapeXml(task.resumedFrom ?? "an earlier run")}` : ""}</summary>`,
203
+ `<result>${escapeXml(result.length > 4000 ? `${result.slice(0, 4000)}\n...(truncated)` : result)}</result>`,
204
+ `<usage><total_tokens>${task.totalTokens}</total_tokens><tool_uses>${task.totalToolCalls}</tool_uses><duration_ms>${elapsedMs(task, now)}</duration_ms></usage>`,
205
+ `</task-notification>`,
206
+ ].filter(Boolean).join("\n");
207
+ }
208
+ //# sourceMappingURL=task.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"task.js","sourceRoot":"","sources":["../../src/workflow/task.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AACzC,OAAO,EAAE,SAAS,EAAE,MAAM,WAAW,CAAC;AAGtC,OAAO,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,EAA8C,MAAM,eAAe,CAAC;AAGvG,wEAAwE;AACxE,MAAM,UAAU,aAAa;IAC3B,OAAO,MAAM,UAAU,EAAE,CAAC,OAAO,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,EAAE,CAAC;AAC7D,CAAC;AAgED,MAAM,UAAU,kBAAkB,CAAC,IAWlC;IACC,OAAO;QACL,IAAI,EAAE,gBAAgB;QACtB,EAAE,EAAE,IAAI,CAAC,EAAE;QACX,MAAM,EAAE,SAAS;QACjB,MAAM,EAAE,IAAI,CAAC,MAAM;QACnB,UAAU,EAAE,IAAI,CAAC,UAAU;QAC3B,IAAI,EAAE,IAAI,CAAC,IAAI;QACf,IAAI,EAAE,IAAI,CAAC,IAAI;QACf,YAAY,EAAE,IAAI,CAAC,IAAI,EAAE,IAAI;QAC7B,UAAU,EAAE,IAAI,CAAC,UAAU;QAC3B,WAAW,EAAE,IAAI,CAAC,WAAW;QAC7B,MAAM,EAAE,IAAI,CAAC,MAAM;QACnB,WAAW,EAAE,IAAI,CAAC,WAAW;QAC7B,aAAa,EAAE,CAAC;QAChB,gBAAgB,EAAE,EAAE;QACpB,eAAe,EAAE,CAAC;QAClB,UAAU,EAAE,CAAC;QACb,SAAS,EAAE,CAAC;QACZ,WAAW,EAAE,CAAC;QACd,cAAc,EAAE,CAAC;QACjB,IAAI,EAAE,EAAE;QACR,eAAe,EAAE,IAAI,eAAe,EAAE;QACtC,SAAS,EAAE,IAAI,CAAC,SAAS,IAAI,IAAI,CAAC,GAAG,EAAE;QACvC,aAAa,EAAE,CAAC;KACjB,CAAC;AACJ,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,2BAA2B,CACzC,IAAkB,EAClB,OAAiC;IAEjC,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO;IACjC,IAAI,CAAC,gBAAgB,CAAC,IAAI,CAAC,GAAG,OAAO,CAAC,CAAC;IACvC,IAAI,CAAC,eAAe,EAAE,CAAC;IAEvB,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,QAAQ,CAAC,IAAI,CAAC,gBAAgB,CAAC,CAAC;IACzD,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC;IACjB,8EAA8E;IAC9E,sDAAsD;IACtD,IAAI,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,UAAU,EAAE,MAAM,CAAC,MAAM,CAAC,CAAC;IAE3D,IAAI,WAAW,GAAG,CAAC,CAAC;IACpB,IAAI,cAAc,GAAG,CAAC,CAAC;IACvB,IAAI,IAAI,GAAG,CAAC,CAAC;IACb,KAAK,MAAM,KAAK,IAAI,MAAM,EAAE,CAAC;QAC3B,WAAW,IAAI,KAAK,CAAC,MAAM,IAAI,CAAC,CAAC;QACjC,cAAc,IAAI,KAAK,CAAC,SAAS,IAAI,CAAC,CAAC;QACvC,qEAAqE;QACrE,IAAI,KAAK,CAAC,KAAK,KAAK,MAAM;YAAE,IAAI,EAAE,CAAC;IACrC,CAAC;IACD,IAAI,CAAC,WAAW,GAAG,WAAW,CAAC;IAC/B,IAAI,CAAC,cAAc,GAAG,cAAc,CAAC;IACrC,IAAI,CAAC,SAAS,GAAG,IAAI,CAAC;AACxB,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,iBAAiB,CAAC,IAAkB,EAAE,GAAG,GAAG,IAAI,CAAC,GAAG,EAAE;IACpE,IAAI,IAAI,CAAC,MAAM,KAAK,SAAS,IAAI,IAAI,CAAC,OAAO,KAAK,SAAS;QAAE,OAAO,KAAK,CAAC;IAC1E,IAAI,CAAC,OAAO,CAAC,KAAK,EAAE,CAAC;IACrB,IAAI,CAAC,MAAM,GAAG,QAAQ,CAAC;IACvB,IAAI,CAAC,QAAQ,GAAG,GAAG,CAAC;IACpB,OAAO,IAAI,CAAC;AACd,CAAC;AAED,yDAAyD;AACzD,MAAM,UAAU,kBAAkB,CAAC,IAAkB,EAAE,GAAG,GAAG,IAAI,CAAC,GAAG,EAAE;IACrE,IAAI,IAAI,CAAC,MAAM,KAAK,QAAQ,IAAI,IAAI,CAAC,OAAO,KAAK,SAAS;QAAE,OAAO,KAAK,CAAC;IACzE,IAAI,CAAC,OAAO,CAAC,MAAM,EAAE,CAAC;IACtB,IAAI,CAAC,MAAM,GAAG,SAAS,CAAC;IACxB,IAAI,CAAC,aAAa,GAAG,CAAC,IAAI,CAAC,aAAa,IAAI,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,GAAG,GAAG,CAAC,IAAI,CAAC,QAAQ,IAAI,GAAG,CAAC,CAAC,CAAC;IAC3F,IAAI,CAAC,QAAQ,GAAG,SAAS,CAAC;IAC1B,OAAO,IAAI,CAAC;AACd,CAAC;AAED,+CAA+C;AAC/C,MAAM,UAAU,oBAAoB,CAAC,IAAkB,EAAE,MAAyB;IAChF,8EAA8E;IAC9E,oEAAoE;IACpE,IAAI,IAAI,CAAC,QAAQ,KAAK,SAAS,EAAE,CAAC;QAChC,IAAI,CAAC,aAAa,GAAG,CAAC,IAAI,CAAC,aAAa,IAAI,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,QAAQ,CAAC,CAAC;QACzF,IAAI,CAAC,QAAQ,GAAG,SAAS,CAAC;IAC5B,CAAC;IACD,6EAA6E;IAC7E,2CAA2C;IAC3C,IAAI,CAAC,OAAO,GAAG,SAAS,CAAC;IACzB,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC,MAAM,CAAC;IAC5B,IAAI,CAAC,IAAI,KAAK,MAAM,CAAC,IAAI,CAAC;IAC1B,IAAI,CAAC,YAAY,KAAK,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC;IACvC,IAAI,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,UAAU,EAAE,MAAM,CAAC,UAAU,CAAC,CAAC;IAC/D,IAAI,CAAC,aAAa,GAAG,MAAM,CAAC,aAAa,CAAC;IAC1C,IAAI,CAAC,KAAK,GAAG,MAAM,CAAC,KAAK,CAAC;IAC1B,IAAI,CAAC,KAAK,GAAG,MAAM,CAAC,KAAK,CAAC;IAC1B,IAAI,CAAC,OAAO,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC;AAC5B,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,gBAAgB,CAAC,IAAkB,EAAE,KAAa;IAChE,IAAI,CAAC,OAAO,GAAG,SAAS,CAAC;IACzB,IAAI,CAAC,QAAQ,GAAG,SAAS,CAAC;IAC1B,IAAI,CAAC,MAAM,GAAG,QAAQ,CAAC;IACvB,IAAI,CAAC,KAAK,GAAG,KAAK,CAAC;IACnB,IAAI,CAAC,OAAO,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC;AAC5B,CAAC;AAED,iFAAiF;AACjF,MAAM,UAAU,kBAAkB,CAAC,IAAkB;IACnD,IAAI,IAAI,CAAC,KAAK,KAAK,SAAS;QAAE,OAAO,IAAI,CAAC,KAAK,CAAC;IAChD,IAAI,IAAI,CAAC,KAAK,KAAK,SAAS;QAAE,OAAO,YAAY,CAAC;IAClD,IAAI,OAAO,IAAI,CAAC,KAAK,KAAK,QAAQ;QAAE,OAAO,IAAI,CAAC,KAAK,CAAC;IACtD,OAAO,IAAI,CAAC,SAAS,CAAC,IAAI,CAAC,KAAK,EAAE,IAAI,EAAE,CAAC,CAAC,CAAC;AAC7C,CAAC;AAED;;;;;;;;GAQG;AACH,MAAM,UAAU,mBAAmB,CACjC,KAAyB,EACzB,KAAwC;IAKxC,MAAM,EAAE,GAAG,KAAK,EAAE,IAAI,EAAE,CAAC;IACzB,IAAI,EAAE,KAAK,SAAS,IAAI,EAAE,KAAK,EAAE;QAAE,OAAO,SAAS,CAAC;IAEpD,MAAM,KAAK,GAAG,KAAK,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;IAC5B,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;QACxB,MAAM,KAAK,GAAG,CAAC,GAAG,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QAChC,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EACL,oBAAoB,EAAE,qBAAqB;gBAC3C,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC;oBACf,CAAC,CAAC,sBAAsB,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,GAAG;oBAC3C,CAAC,CAAC,4DAA4D,CAAC;SACpE,CAAC;IACJ,CAAC;IACD,IAAI,KAAK,CAAC,MAAM,KAAK,SAAS,EAAE,CAAC;QAC/B,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,aAAa,EAAE,0EAA0E;SACnG,CAAC;IACJ,CAAC;IACD,IAAI,KAAK,CAAC,WAAW,KAAK,SAAS,EAAE,CAAC;QACpC,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,aAAa,EAAE,kCAAkC,EAAE,CAAC;IACnF,CAAC;IACD,OAAO;QACL,EAAE,EAAE,IAAI;QACR,KAAK,EAAE,EAAE;QACT,WAAW,EAAE,KAAK,CAAC,WAAW;QAC9B,yEAAyE;QACzE,sBAAsB;QACtB,UAAU,EAAE,KAAK,CAAC,UAAU,IAAI,EAAE;KACnC,CAAC;AACJ,CAAC;AAED,kFAAkF;AAClF,MAAM,UAAU,0BAA0B,CAAC,IAAkB,EAAE,GAAG,GAAG,IAAI,CAAC,GAAG,EAAE;IAC7E,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,CAAC,gBAAgB,EAAE,IAAI,CAAC,UAAU,CAAC,CAAC;IAC7D,MAAM,MAAM,GACV,IAAI,CAAC,MAAM,KAAK,WAAW,CAAC,CAAC,CAAC,MAAM;QACpC,CAAC,CAAC,IAAI,CAAC,MAAM,KAAK,QAAQ,CAAC,CAAC,CAAC,SAAS;YACtC,CAAC,CAAC,UAAU,IAAI,CAAC,KAAK,IAAI,SAAS,EAAE,CAAC;IACxC,MAAM,MAAM,GAAG,kBAAkB,CAAC,IAAI,CAAC,CAAC;IACxC,OAAO;QACL,qBAAqB;QACrB,YAAY,IAAI,CAAC,EAAE,YAAY;QAC/B,IAAI,CAAC,UAAU,CAAC,CAAC,CAAC,gBAAgB,SAAS,CAAC,IAAI,CAAC,UAAU,CAAC,gBAAgB,CAAC,CAAC,CAAC,IAAI;QACnF,IAAI,CAAC,UAAU,CAAC,CAAC,CAAC,WAAW,SAAS,CAAC,IAAI,CAAC,UAAU,CAAC,WAAW,CAAC,CAAC,CAAC,IAAI;QACzE,WAAW,SAAS,CAAC,MAAM,CAAC,WAAW;QACvC,sBAAsB,SAAS,CAAC,IAAI,CAAC,YAAY,IAAI,IAAI,CAAC,EAAE,CAAC,KAAK,IAAI,CAAC,MAAM,MAAM,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,KAAK,UAC5G,IAAI,CAAC,aAAa,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,aAAa,kBAAkB,SAAS,CAAC,IAAI,CAAC,WAAW,IAAI,gBAAgB,CAAC,EAAE,CAAC,CAAC,CAAC,EACxH,YAAY;QACZ,WAAW,SAAS,CAAC,MAAM,CAAC,MAAM,GAAG,IAAI,CAAC,CAAC,CAAC,GAAG,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,kBAAkB,CAAC,CAAC,CAAC,MAAM,CAAC,WAAW;QAC3G,wBAAwB,IAAI,CAAC,WAAW,6BAA6B,IAAI,CAAC,cAAc,4BAA4B,SAAS,CAAC,IAAI,EAAE,GAAG,CAAC,wBAAwB;QAChK,sBAAsB;KACvB,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAC/B,CAAC"}
@@ -0,0 +1,39 @@
1
+ /**
2
+ * tool-description.ts — the model-facing description of the `SubagentWorkflow` tool.
3
+ *
4
+ * This is a deliberate port of Claude Code's `Workflow` tool description, not a
5
+ * paraphrase of it. The rule the text is held to: **match Claude Code's wording
6
+ * everywhere; deviate only in the specific clause where its sentence would be
7
+ * false about pi, and keep that deviation minimal and in its voice.** Wording
8
+ * parity is the point — a user who knows one tool should not have to relearn
9
+ * the other, and the orchestration patterns below are load-bearing guidance
10
+ * that gets used badly when compressed.
11
+ *
12
+ * Parts omitted because pi has no such feature: the `ultracode` opt-in, MCP
13
+ * tools reached through `ToolSearch`, the `agent-<id>.jsonl` resume fallback,
14
+ * and the `/config` workflow-size guideline.
15
+ *
16
+ * Clauses that had to deviate, each because Claude Code's is untrue here:
17
+ * - `schema` is pressure, not force — `toolChoice` is not plumbed through
18
+ * pi's `AgentSession`, so a child can decline and the call returns null.
19
+ * - `budget.total` is always null; pi has no token-target directive.
20
+ * - `parallel` propagates a fatal run error instead of folding it to null.
21
+ * - `effort` inherits the agent definition's level, then the parent's.
22
+ * - `isolation` removes the worktree on settle, changes kept on a branch.
23
+ * Additions with no upstream counterpart: `gate`, `resume`, `effort: "minimal"`,
24
+ * the saved-workflow directories, and the reject-unknown-options guarantee.
25
+ *
26
+ * Kept out of index.ts purely for size. `{{placeholder}}` tokens are rendered by
27
+ * the same substitution pass the Agent tool's description uses, so a
28
+ * user-authored override can interpolate the live agent roster.
29
+ */
30
+ /**
31
+ * Rendered with `{{typeList}}` substituted. Keep the prose accurate to what the
32
+ * runtime actually implements — documenting a global we do not ship is worse
33
+ * than documenting nothing, because the script only fails once it is running.
34
+ * `workflow-tool-description.test.ts` pins the parts that can drift: the
35
+ * `agent()` option set, the `resume` exclusions, the effort levels, the caps,
36
+ * and that every example here uses options the runtime actually accepts.
37
+ */
38
+ export declare const fullWorkflowToolDescription = "Execute a workflow script that orchestrates multiple subagents deterministically. Workflows run in the background \u2014 this tool returns immediately with a task ID, and you are notified when the workflow completes. Use /agents \u2192 Workflows to watch live progress.\n\nA workflow structures work across many agents \u2014 to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before committing), or to take on scale one context can't hold (migrations, audits, broad sweeps). The script is where you encode that structure: what fans out, what verifies, what synthesizes.\n\nONLY call this tool when the user has explicitly opted into multi-agent orchestration. Workflows can spawn dozens of agents and consume a large amount of tokens; the user must request that scale, not have it inferred. Explicit opt-in means one of:\n- The user directly asked you to run a workflow or use multi-agent orchestration in their own words (\"use a workflow\", \"run a workflow\", \"fan out agents\", \"orchestrate this with subagents\"). The ask must be in the user's words \u2014 a task that would merely benefit from a workflow does not count.\n- The user invoked a skill or slash command whose instructions tell you to call SubagentWorkflow.\n- The user asked you to run a specific named or saved workflow.\n\nFor any other task \u2014 even one that would clearly benefit from parallelism \u2014 do NOT call this tool. Use the Agent tool for individual subagents, or briefly describe what a multi-agent workflow could do and how much it would roughly cost, and ask the user whether to run it. Mention they can ask for one with \"use a workflow\" in a future message to skip the ask.\n\nWhen you do call it, the right move is often **hybrid**: scout inline first (list the files, find the channels, scope the diff) to discover the work-list, then call SubagentWorkflow to pipeline over it. You don't need to know the shape before the *task* \u2014 only before the *orchestration step*.\n\nCommon single-phase workflows you can chain across turns:\n- **Understand** \u2014 parallel readers over relevant subsystems \u2192 structured map\n- **Design** \u2014 judge panel of N independent approaches \u2192 scored synthesis\n- **Review** \u2014 dimensions \u2192 find \u2192 adversarially verify (example below)\n- **Research** \u2014 multi-modal sweep \u2192 deep-read \u2192 synthesize\n- **Migrate** \u2014 discover sites \u2192 transform each (worktree isolation) \u2192 verify\n\nFor larger work, run several in sequence \u2014 read each result before deciding the next phase. You stay in the loop; each workflow is one well-scoped fan-out.\n\nPass the script inline via `script` \u2014 do not Write it to a file first. Every invocation automatically persists its script to a file under the session directory and returns the path in the tool result. To iterate on a workflow, edit that file with Write/Edit and re-invoke SubagentWorkflow with `{scriptPath: \"<path>\"}` instead of resending the full script. A script you will run more than once belongs in `.pi/workflows/<name>.js` (or `.agents/workflows/`, or `<agent dir>/workflows/` for one that follows the user everywhere); call it with `name: \"<name>\"` instead of re-sending the source.\n\nEvery script must begin with `export const meta = {...}`:\n export const meta = {\n name: 'find-flaky-tests',\n description: 'Find flaky tests and propose fixes', // one-line, shown in permission dialog\n phases: [ // one entry per phase() call\n { title: 'Scan', detail: 'grep test logs for retries' },\n { title: 'Fix', detail: 'one agent per flaky test' },\n ],\n }\n // script body starts here \u2014 use agent()/parallel()/pipeline()/phase()/log()\n phase('Scan')\n const flaky = await agent('grep CI logs for retry markers', {schema: FLAKY_SCHEMA})\n ...\n\nThe `meta` object must be a PURE LITERAL \u2014 no variables, function calls, spreads, or template interpolation. Required fields: `name`, `description`. Optional: `whenToUse` (shown in the workflow list), `phases`. Use the SAME phase titles in meta.phases as in phase() calls \u2014 titles are matched exactly; a phase() call with no matching meta entry just gets its own progress group. Add `model` to a phase entry when that phase uses a specific model override.\n\nScript body hooks:\n- agent(prompt: string, opts?: {label?: string, phase?: string, schema?: object, model?: string, effort?: string, isolation?: 'worktree', agentType?: string, gate?: string, resume?: string}): Promise<any> \u2014 spawn a subagent. Without schema, returns its final text as a string. With schema (a JSON Schema), the subagent is given a StructuredOutput tool built from it and agent() returns the validated object \u2014 no parsing needed. A payload that does not match is rejected back to the child, which corrects it; a child that never answers through the tool gets one more prompt and then fails, so the call returns null \u2014 filter after every schema stage. Returns null if the user skips the agent mid-run or the subagent dies on a terminal API error after retries (filter with .filter(Boolean)). opts.label overrides the display label. opts.phase explicitly assigns this agent to a progress group (use this inside pipeline()/parallel() stages to avoid races on the global phase() state \u2014 same phase string \u2192 same group box). opts.model overrides the model for this agent call. Default to omitting it \u2014 the agent inherits the main-loop model (the resolved session model), which is almost always correct. Only set it when you're highly confident a different tier fits the task; when unsure, omit. opts.effort overrides the reasoning effort for this agent call ('minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max') \u2014 omit to inherit the agent definition's own level, then the parent's; use 'low' for cheap mechanical stages and higher tiers only for the hardest verify/judge stages. opts.isolation: 'worktree' runs the agent in a fresh git worktree \u2014 EXPENSIVE (setup time + disk per agent), use ONLY when agents mutate files in parallel and would otherwise conflict; the worktree is removed when the agent settles, its changes preserved on a branch. opts.gate: '<command>' runs a shell command after the agent finishes and requires it to pass \u2014 a non-zero exit marks the agent failed and the command's output becomes the error; prefer gate: 'npm test' over asking another agent whether the code looks right. opts.resume: '<label>' continues the child that ran under that label instead of starting fresh, so an iterative loop keeps its context \u2014 it cannot be combined with agentType, model, effort, isolation, gate or schema. opts.agentType uses a custom subagent type instead of the default workflow subagent \u2014 resolved from the same registry as the Agent tool; composes with schema. Available types:\n{{typeList}}\n- pipeline(items, stage1, stage2, ...): Promise<any[]> \u2014 run each item through all stages independently, NO barrier between stages. Item A can be in stage 3 while item B is still in stage 1. This is the DEFAULT for multi-stage work. Wall-clock = slowest single-item chain, not sum-of-slowest-per-stage. Every stage callback receives (prevResult, originalItem, index) \u2014 use originalItem/index in later stages to label work without threading context through stage 1's return value. A stage that throws drops that item to `null` and skips its remaining stages.\n- parallel(thunks: Array<() => Promise<any>>): Promise<any[]> \u2014 run tasks concurrently. This is a BARRIER: awaits all thunks before returning. A thunk that throws (or whose agent errors) resolves to `null` in the result array, so `.filter(Boolean)` before using the results; only a fatal run error \u2014 a cap breach, or a nested workflow that could not load \u2014 propagates instead of being folded into a null. Use ONLY when you genuinely need all results together.\n- log(message: string): void \u2014 emit a progress message to the user (shown as a narrator line above the progress tree)\n- phase(title: string): void \u2014 start a new phase; subsequent agent() calls are grouped under this title in the progress display\n- args: any \u2014 the value passed as SubagentWorkflow's `args` input, verbatim (undefined if not provided). Pass arrays/objects as actual JSON values in the tool call, NOT as a JSON-encoded string \u2014 `args: [\"a.ts\", \"b.ts\"]`, not `args: \"[\\\"a.ts\\\", ...]\"` (a stringified list reaches the script as one string, so `args.filter`/`args.map` throw). Use this to parameterize named workflows \u2014 e.g. pass a research question, target path, or config object directly instead of via a side-channel file.\n- budget: {total: number|null, spent(): number, remaining(): number} \u2014 `budget.total` is always null here: it comes from a token-target directive pi does not have, so guards like `while (budget.total && budget.remaining() > 50_000) { ... }` correctly do not fire rather than throwing on a missing global. `budget.spent()` returns output tokens spent by this run's agents. `budget.remaining()` returns `Infinity` with no target.\n- workflow(nameOrRef: string | {scriptPath: string}, args?: any): Promise<any> \u2014 run another workflow inline as a sub-step and return whatever it returns. Pass a name to invoke a saved workflow (same registry as {name: \"...\"}), or {scriptPath} to run a script file you Wrote earlier. The child shares this run's concurrency cap, agent counter, abort signal, and token budget \u2014 its agents appear under a \"\u25B8 name\" group in /agents \u2192 Workflows and its tokens count toward budget.spent(). The args param becomes the child's `args` global. Nesting is one level only: workflow() inside a child throws. Throws on unknown name / unreadable scriptPath / child syntax error; catch to handle gracefully.\n\nAny agent() option not listed above is rejected by name at the call.\n\nSubagents are told their final text IS the return value (not a human-facing message), so they return raw data. For structured output, use the schema option \u2014 validation happens at the tool-call layer so the model retries on mismatch.\n\nScripts are plain JavaScript, NOT TypeScript \u2014 type annotations (`: string[]`), interfaces, and generics fail to parse. The script body runs in an async context \u2014 use await directly. Standard JS built-ins (JSON, Math, Array, etc.) are available \u2014 EXCEPT `Date.now()`/`Math.random()`/argless `new Date()`, which throw (they would break resume); pass timestamps in via `args`, stamp results after the workflow returns, and for randomness vary the agent prompt/label by index. `eval` and `Function(...)` throw. No filesystem or Node.js API access.\n\nDEFAULT TO pipeline(). Only reach for a barrier (parallel between stages) when you genuinely need ALL prior-stage results together.\n\nA barrier is correct ONLY when stage N needs cross-item context from all of stage N-1:\n- Dedup/merge across the full result set before expensive downstream work\n- Early-exit if the total count is zero (\"0 bugs found \u2192 skip verification entirely\")\n- Stage N's prompt references \"the other findings\" for comparison\n\nA barrier is NOT justified by:\n- \"I need to flatten/map/filter first\" \u2014 do it inside a pipeline stage: pipeline(items, stageA, r => transform([r]).flat(), stageB)\n- \"The stages are conceptually separate\" \u2014 that's what pipeline() models. Separate stages \u2260 synchronized stages.\n- \"It's cleaner code\" \u2014 barrier latency is real. If 5 finders run and the slowest takes 3\u00D7 the fastest, a barrier wastes 2/3 of the fast finders' idle time.\n\nSmell test: if you wrote\n const a = await parallel(...)\n const b = transform(a) // flatten, map, filter \u2014 no cross-item dependency\n const c = await parallel(b.map(...))\nthat middle transform doesn't need the barrier. Rewrite as a pipeline with the transform inside a stage. When in doubt: pipeline.\n\nConcurrent agent() calls are capped at min(16, available CPUs - 2) per workflow \u2014 excess calls queue and run as slots free up. You can still pass 100 items to parallel()/pipeline() and they all complete; only ~10 run at any moment. Total agent count across a workflow's lifetime is capped at 1000 \u2014 a runaway-loop backstop set far above any real workflow. A single parallel()/pipeline() call accepts at most 4096 items; passing more is an explicit error, not a silent truncation.\n\nThe canonical multi-stage pattern \u2014 pipeline by default, each dimension verifies as soon as its review completes:\n export const meta = {\n name: 'review-changes',\n description: 'Review changed files across dimensions, verify each finding',\n phases: [{ title: 'Review' }, { title: 'Verify' }],\n }\n const DIMENSIONS = [{key: 'bugs', prompt: '...'}, {key: 'perf', prompt: '...'}]\n const results = await pipeline(\n DIMENSIONS,\n d => agent(d.prompt, {label: `review:${d.key}`, phase: 'Review', schema: FINDINGS_SCHEMA}),\n review => parallel(review.findings.map(f => () =>\n agent(`Adversarially verify: ${f.title}`, {label: `verify:${f.file}`, phase: 'Verify', schema: VERDICT_SCHEMA})\n .then(v => ({...f, verdict: v}))\n ))\n )\n const confirmed = results.flat().filter(Boolean).filter(f => f.verdict?.isReal)\n return { confirmed }\n // Dimension 'bugs' findings verify while dimension 'perf' is still reviewing. No wasted wall-clock.\n\nWhen a barrier IS correct \u2014 dedup across all findings before expensive verification:\n const all = await parallel(DIMENSIONS.map(d => () => agent(d.prompt, {schema: FINDINGS_SCHEMA})))\n const deduped = dedupeByFileAndLine(all.filter(Boolean).flatMap(r => r.findings)) // <-- genuinely needs ALL at once\n const verified = await parallel(deduped.map(f => () => agent(verifyPrompt(f), {schema: VERDICT_SCHEMA})))\n\nLoop-until-count pattern \u2014 accumulate to a target:\n const bugs = []\n while (bugs.length < 10) {\n const result = await agent(\"Find bugs in this codebase.\", {schema: BUGS_SCHEMA})\n bugs.push(...result.bugs)\n log(`${bugs.length}/10 found`)\n }\n\nGate-and-retry pattern \u2014 verify by running, and keep the agent's context across attempts:\n let fixed = await agent('Find and fix the failing test.', {label: 'fix', gate: 'npm test'})\n if (fixed === null) { // a non-zero exit failed the agent\n // Resume keeps everything the child already learned. It cannot carry the\n // gate, so re-verification needs its own gated call, in the same tree.\n fixed = await agent('`npm test` is still failing. Fix the cause.', {label: 'fix', resume: 'fix'})\n const verified = await agent('Run `npm test` and report the result. Change nothing.',\n {label: 'verify', phase: 'Verify', gate: 'npm test', effort: 'low'})\n return { passed: verified !== null, summary: fixed }\n }\n return { passed: true, summary: fixed }\n // An LLM judging whether a fix works is a weaker signal than the test suite.\n\nComposing patterns \u2014 exhaustive review (find \u2192 dedup vs seen \u2192 diverse-lens panel \u2192 loop-until-dry):\n const seen = new Set(), confirmed = []\n let dry = 0\n while (dry < 2) { // loop-until-dry\n const found = (await parallel(FINDERS.map(f => () => // barrier: collect all finders this round\n agent(f.prompt, {phase: 'Find', schema: BUGS})))).filter(Boolean).flatMap(r => r.bugs)\n const fresh = found.filter(b => !seen.has(key(b))) // dedup vs ALL seen \u2014 plain code, not an agent\n if (!fresh.length) { dry++; continue }\n dry = 0; fresh.forEach(b => seen.add(key(b)))\n const judged = await parallel(fresh.map(b => () => // every fresh bug judged concurrently...\n parallel(['correctness','security','repro'].map(lens => () => // ...each by 3 distinct lenses\n agent(`Judge \"${b.desc}\" via the ${lens} lens \u2014 real?`, {phase: 'Verify', schema: VERDICT})))\n .then(vs => ({ b, real: vs.filter(Boolean).filter(v => v.real).length >= 2 }))))\n confirmed.push(...judged.filter(v => v.real).map(v => v.b))\n }\n return confirmed\n // dedup vs `seen`, NOT `confirmed` \u2014 else judge-rejected findings reappear every round and it never converges.\n\nQuality patterns \u2014 common shapes; pick by task and compose freely:\n- Adversarial verify: spawn N independent skeptics per finding, each prompted to REFUTE. Kill if \u2265majority refute. Prevents plausible-but-wrong findings from surviving.\n const votes = await parallel(Array.from({length: 3}, () => () =>\n agent(`Try to refute: ${claim}. Default to refuted=true if uncertain.`, {schema: VERDICT})))\n const survives = votes.filter(Boolean).filter(v => !v.refuted).length >= 2\n- Verify by running, not by asking: when a claim is testable, `gate` it rather than asking another model whether it holds.\n- Perspective-diverse verify: when a finding can fail in more than one way, give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters \u2014 diversity catches failure modes redundancy can't.\n- Judge panel: generate N independent attempts from different angles (e.g. MVP-first, risk-first, user-first), score with parallel judges, synthesize from the winner while grafting the best ideas from runners-up. Beats one-attempt-iterated when the solution space is wide.\n- Loop-until-dry: for unknown-size discovery (bugs, issues, edge cases), keep spawning finders until K consecutive rounds return nothing new. Simple counters (while count < N) miss the tail.\n- Multi-modal sweep: parallel agents each searching a different way (by-container, by-content, by-entity, by-time). Each is blind to what the others surface; useful when one search angle won't find everything.\n- Completeness critic: a final agent that asks \"what's missing \u2014 modality not run, claim unverified, source unread?\" What it finds becomes the next round of work.\n- No silent caps: if a workflow bounds coverage (top-N, no-retry, sampling), `log()` what was dropped \u2014 silent truncation reads as \"covered everything\" when it didn't.\n\nScale to what the user asked for. \"find any bugs\" \u2192 a few finders, single-vote verify. \"thoroughly audit this\" or \"be comprehensive\" \u2192 larger finder pool, 3\u20135 vote adversarial pass, synthesis stage. When unsure, lean toward thoroughness for research/review/audit requests and toward brevity for quick checks.\n\nThese patterns aren't exhaustive \u2014 compose novel harnesses when the task calls for it (tournament brackets, self-repair loops, staged escalation, whatever fits).\n\nUse this tool for multi-step orchestration where control flow should be deterministic (loops, conditionals, fan-out) rather than model-driven.\n\n## Resume\n\nThe tool result includes a runId. To resume after a pause, kill, or script edit, relaunch with SubagentWorkflow({scriptPath, resumeFromRunId}) \u2014 the longest unchanged prefix of agent() calls returns cached results instantly; the first edited/new call and everything after it runs live. Same script + same args \u2192 100% cache hit. It is a prefix and not a lookup: a later call that still matches is not reused once an earlier one has changed. A journaled failure ends the prefix, so resuming a run that died at agent 5 retries exactly agent 5. Same session only, and the run must have finished \u2014 stop it from /agents \u2192 Workflows first. Before diagnosing why a completed workflow returned an empty or unexpected result, Read the run's `<run id>.workflow.jsonl` beside its script \u2014 it records each agent's actual return value; do not assume cached results are non-empty. Date.now()/Math.random()/new Date() are unavailable in scripts (they would break this) \u2014 stamp results after the workflow returns, or pass timestamps via args.";
39
+ //# sourceMappingURL=tool-description.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tool-description.d.ts","sourceRoot":"","sources":["../../src/workflow/tool-description.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAEH;;;;;;;GAOG;AACH,eAAO,MAAM,2BAA2B,4nnBAiK49B,CAAC"}
@@ -0,0 +1,200 @@
1
+ /**
2
+ * tool-description.ts — the model-facing description of the `SubagentWorkflow` tool.
3
+ *
4
+ * This is a deliberate port of Claude Code's `Workflow` tool description, not a
5
+ * paraphrase of it. The rule the text is held to: **match Claude Code's wording
6
+ * everywhere; deviate only in the specific clause where its sentence would be
7
+ * false about pi, and keep that deviation minimal and in its voice.** Wording
8
+ * parity is the point — a user who knows one tool should not have to relearn
9
+ * the other, and the orchestration patterns below are load-bearing guidance
10
+ * that gets used badly when compressed.
11
+ *
12
+ * Parts omitted because pi has no such feature: the `ultracode` opt-in, MCP
13
+ * tools reached through `ToolSearch`, the `agent-<id>.jsonl` resume fallback,
14
+ * and the `/config` workflow-size guideline.
15
+ *
16
+ * Clauses that had to deviate, each because Claude Code's is untrue here:
17
+ * - `schema` is pressure, not force — `toolChoice` is not plumbed through
18
+ * pi's `AgentSession`, so a child can decline and the call returns null.
19
+ * - `budget.total` is always null; pi has no token-target directive.
20
+ * - `parallel` propagates a fatal run error instead of folding it to null.
21
+ * - `effort` inherits the agent definition's level, then the parent's.
22
+ * - `isolation` removes the worktree on settle, changes kept on a branch.
23
+ * Additions with no upstream counterpart: `gate`, `resume`, `effort: "minimal"`,
24
+ * the saved-workflow directories, and the reject-unknown-options guarantee.
25
+ *
26
+ * Kept out of index.ts purely for size. `{{placeholder}}` tokens are rendered by
27
+ * the same substitution pass the Agent tool's description uses, so a
28
+ * user-authored override can interpolate the live agent roster.
29
+ */
30
+ /**
31
+ * Rendered with `{{typeList}}` substituted. Keep the prose accurate to what the
32
+ * runtime actually implements — documenting a global we do not ship is worse
33
+ * than documenting nothing, because the script only fails once it is running.
34
+ * `workflow-tool-description.test.ts` pins the parts that can drift: the
35
+ * `agent()` option set, the `resume` exclusions, the effort levels, the caps,
36
+ * and that every example here uses options the runtime actually accepts.
37
+ */
38
+ export const fullWorkflowToolDescription = `Execute a workflow script that orchestrates multiple subagents deterministically. Workflows run in the background — this tool returns immediately with a task ID, and you are notified when the workflow completes. Use /agents → Workflows to watch live progress.
39
+
40
+ A workflow structures work across many agents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before committing), or to take on scale one context can't hold (migrations, audits, broad sweeps). The script is where you encode that structure: what fans out, what verifies, what synthesizes.
41
+
42
+ ONLY call this tool when the user has explicitly opted into multi-agent orchestration. Workflows can spawn dozens of agents and consume a large amount of tokens; the user must request that scale, not have it inferred. Explicit opt-in means one of:
43
+ - The user directly asked you to run a workflow or use multi-agent orchestration in their own words ("use a workflow", "run a workflow", "fan out agents", "orchestrate this with subagents"). The ask must be in the user's words — a task that would merely benefit from a workflow does not count.
44
+ - The user invoked a skill or slash command whose instructions tell you to call SubagentWorkflow.
45
+ - The user asked you to run a specific named or saved workflow.
46
+
47
+ For any other task — even one that would clearly benefit from parallelism — do NOT call this tool. Use the Agent tool for individual subagents, or briefly describe what a multi-agent workflow could do and how much it would roughly cost, and ask the user whether to run it. Mention they can ask for one with "use a workflow" in a future message to skip the ask.
48
+
49
+ When you do call it, the right move is often **hybrid**: scout inline first (list the files, find the channels, scope the diff) to discover the work-list, then call SubagentWorkflow to pipeline over it. You don't need to know the shape before the *task* — only before the *orchestration step*.
50
+
51
+ Common single-phase workflows you can chain across turns:
52
+ - **Understand** — parallel readers over relevant subsystems → structured map
53
+ - **Design** — judge panel of N independent approaches → scored synthesis
54
+ - **Review** — dimensions → find → adversarially verify (example below)
55
+ - **Research** — multi-modal sweep → deep-read → synthesize
56
+ - **Migrate** — discover sites → transform each (worktree isolation) → verify
57
+
58
+ For larger work, run several in sequence — read each result before deciding the next phase. You stay in the loop; each workflow is one well-scoped fan-out.
59
+
60
+ Pass the script inline via \`script\` — do not Write it to a file first. Every invocation automatically persists its script to a file under the session directory and returns the path in the tool result. To iterate on a workflow, edit that file with Write/Edit and re-invoke SubagentWorkflow with \`{scriptPath: "<path>"}\` instead of resending the full script. A script you will run more than once belongs in \`.pi/workflows/<name>.js\` (or \`.agents/workflows/\`, or \`<agent dir>/workflows/\` for one that follows the user everywhere); call it with \`name: "<name>"\` instead of re-sending the source.
61
+
62
+ Every script must begin with \`export const meta = {...}\`:
63
+ export const meta = {
64
+ name: 'find-flaky-tests',
65
+ description: 'Find flaky tests and propose fixes', // one-line, shown in permission dialog
66
+ phases: [ // one entry per phase() call
67
+ { title: 'Scan', detail: 'grep test logs for retries' },
68
+ { title: 'Fix', detail: 'one agent per flaky test' },
69
+ ],
70
+ }
71
+ // script body starts here — use agent()/parallel()/pipeline()/phase()/log()
72
+ phase('Scan')
73
+ const flaky = await agent('grep CI logs for retry markers', {schema: FLAKY_SCHEMA})
74
+ ...
75
+
76
+ The \`meta\` object must be a PURE LITERAL — no variables, function calls, spreads, or template interpolation. Required fields: \`name\`, \`description\`. Optional: \`whenToUse\` (shown in the workflow list), \`phases\`. Use the SAME phase titles in meta.phases as in phase() calls — titles are matched exactly; a phase() call with no matching meta entry just gets its own progress group. Add \`model\` to a phase entry when that phase uses a specific model override.
77
+
78
+ Script body hooks:
79
+ - agent(prompt: string, opts?: {label?: string, phase?: string, schema?: object, model?: string, effort?: string, isolation?: 'worktree', agentType?: string, gate?: string, resume?: string}): Promise<any> — spawn a subagent. Without schema, returns its final text as a string. With schema (a JSON Schema), the subagent is given a StructuredOutput tool built from it and agent() returns the validated object — no parsing needed. A payload that does not match is rejected back to the child, which corrects it; a child that never answers through the tool gets one more prompt and then fails, so the call returns null — filter after every schema stage. Returns null if the user skips the agent mid-run or the subagent dies on a terminal API error after retries (filter with .filter(Boolean)). opts.label overrides the display label. opts.phase explicitly assigns this agent to a progress group (use this inside pipeline()/parallel() stages to avoid races on the global phase() state — same phase string → same group box). opts.model overrides the model for this agent call. Default to omitting it — the agent inherits the main-loop model (the resolved session model), which is almost always correct. Only set it when you're highly confident a different tier fits the task; when unsure, omit. opts.effort overrides the reasoning effort for this agent call ('minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max') — omit to inherit the agent definition's own level, then the parent's; use 'low' for cheap mechanical stages and higher tiers only for the hardest verify/judge stages. opts.isolation: 'worktree' runs the agent in a fresh git worktree — EXPENSIVE (setup time + disk per agent), use ONLY when agents mutate files in parallel and would otherwise conflict; the worktree is removed when the agent settles, its changes preserved on a branch. opts.gate: '<command>' runs a shell command after the agent finishes and requires it to pass — a non-zero exit marks the agent failed and the command's output becomes the error; prefer gate: 'npm test' over asking another agent whether the code looks right. opts.resume: '<label>' continues the child that ran under that label instead of starting fresh, so an iterative loop keeps its context — it cannot be combined with agentType, model, effort, isolation, gate or schema. opts.agentType uses a custom subagent type instead of the default workflow subagent — resolved from the same registry as the Agent tool; composes with schema. Available types:
80
+ {{typeList}}
81
+ - pipeline(items, stage1, stage2, ...): Promise<any[]> — run each item through all stages independently, NO barrier between stages. Item A can be in stage 3 while item B is still in stage 1. This is the DEFAULT for multi-stage work. Wall-clock = slowest single-item chain, not sum-of-slowest-per-stage. Every stage callback receives (prevResult, originalItem, index) — use originalItem/index in later stages to label work without threading context through stage 1's return value. A stage that throws drops that item to \`null\` and skips its remaining stages.
82
+ - parallel(thunks: Array<() => Promise<any>>): Promise<any[]> — run tasks concurrently. This is a BARRIER: awaits all thunks before returning. A thunk that throws (or whose agent errors) resolves to \`null\` in the result array, so \`.filter(Boolean)\` before using the results; only a fatal run error — a cap breach, or a nested workflow that could not load — propagates instead of being folded into a null. Use ONLY when you genuinely need all results together.
83
+ - log(message: string): void — emit a progress message to the user (shown as a narrator line above the progress tree)
84
+ - phase(title: string): void — start a new phase; subsequent agent() calls are grouped under this title in the progress display
85
+ - args: any — the value passed as SubagentWorkflow's \`args\` input, verbatim (undefined if not provided). Pass arrays/objects as actual JSON values in the tool call, NOT as a JSON-encoded string — \`args: ["a.ts", "b.ts"]\`, not \`args: "[\\"a.ts\\", ...]"\` (a stringified list reaches the script as one string, so \`args.filter\`/\`args.map\` throw). Use this to parameterize named workflows — e.g. pass a research question, target path, or config object directly instead of via a side-channel file.
86
+ - budget: {total: number|null, spent(): number, remaining(): number} — \`budget.total\` is always null here: it comes from a token-target directive pi does not have, so guards like \`while (budget.total && budget.remaining() > 50_000) { ... }\` correctly do not fire rather than throwing on a missing global. \`budget.spent()\` returns output tokens spent by this run's agents. \`budget.remaining()\` returns \`Infinity\` with no target.
87
+ - workflow(nameOrRef: string | {scriptPath: string}, args?: any): Promise<any> — run another workflow inline as a sub-step and return whatever it returns. Pass a name to invoke a saved workflow (same registry as {name: "..."}), or {scriptPath} to run a script file you Wrote earlier. The child shares this run's concurrency cap, agent counter, abort signal, and token budget — its agents appear under a "▸ name" group in /agents → Workflows and its tokens count toward budget.spent(). The args param becomes the child's \`args\` global. Nesting is one level only: workflow() inside a child throws. Throws on unknown name / unreadable scriptPath / child syntax error; catch to handle gracefully.
88
+
89
+ Any agent() option not listed above is rejected by name at the call.
90
+
91
+ Subagents are told their final text IS the return value (not a human-facing message), so they return raw data. For structured output, use the schema option — validation happens at the tool-call layer so the model retries on mismatch.
92
+
93
+ Scripts are plain JavaScript, NOT TypeScript — type annotations (\`: string[]\`), interfaces, and generics fail to parse. The script body runs in an async context — use await directly. Standard JS built-ins (JSON, Math, Array, etc.) are available — EXCEPT \`Date.now()\`/\`Math.random()\`/argless \`new Date()\`, which throw (they would break resume); pass timestamps in via \`args\`, stamp results after the workflow returns, and for randomness vary the agent prompt/label by index. \`eval\` and \`Function(...)\` throw. No filesystem or Node.js API access.
94
+
95
+ DEFAULT TO pipeline(). Only reach for a barrier (parallel between stages) when you genuinely need ALL prior-stage results together.
96
+
97
+ A barrier is correct ONLY when stage N needs cross-item context from all of stage N-1:
98
+ - Dedup/merge across the full result set before expensive downstream work
99
+ - Early-exit if the total count is zero ("0 bugs found → skip verification entirely")
100
+ - Stage N's prompt references "the other findings" for comparison
101
+
102
+ A barrier is NOT justified by:
103
+ - "I need to flatten/map/filter first" — do it inside a pipeline stage: pipeline(items, stageA, r => transform([r]).flat(), stageB)
104
+ - "The stages are conceptually separate" — that's what pipeline() models. Separate stages ≠ synchronized stages.
105
+ - "It's cleaner code" — barrier latency is real. If 5 finders run and the slowest takes 3× the fastest, a barrier wastes 2/3 of the fast finders' idle time.
106
+
107
+ Smell test: if you wrote
108
+ const a = await parallel(...)
109
+ const b = transform(a) // flatten, map, filter — no cross-item dependency
110
+ const c = await parallel(b.map(...))
111
+ that middle transform doesn't need the barrier. Rewrite as a pipeline with the transform inside a stage. When in doubt: pipeline.
112
+
113
+ Concurrent agent() calls are capped at min(16, available CPUs - 2) per workflow — excess calls queue and run as slots free up. You can still pass 100 items to parallel()/pipeline() and they all complete; only ~10 run at any moment. Total agent count across a workflow's lifetime is capped at 1000 — a runaway-loop backstop set far above any real workflow. A single parallel()/pipeline() call accepts at most 4096 items; passing more is an explicit error, not a silent truncation.
114
+
115
+ The canonical multi-stage pattern — pipeline by default, each dimension verifies as soon as its review completes:
116
+ export const meta = {
117
+ name: 'review-changes',
118
+ description: 'Review changed files across dimensions, verify each finding',
119
+ phases: [{ title: 'Review' }, { title: 'Verify' }],
120
+ }
121
+ const DIMENSIONS = [{key: 'bugs', prompt: '...'}, {key: 'perf', prompt: '...'}]
122
+ const results = await pipeline(
123
+ DIMENSIONS,
124
+ d => agent(d.prompt, {label: \`review:\${d.key}\`, phase: 'Review', schema: FINDINGS_SCHEMA}),
125
+ review => parallel(review.findings.map(f => () =>
126
+ agent(\`Adversarially verify: \${f.title}\`, {label: \`verify:\${f.file}\`, phase: 'Verify', schema: VERDICT_SCHEMA})
127
+ .then(v => ({...f, verdict: v}))
128
+ ))
129
+ )
130
+ const confirmed = results.flat().filter(Boolean).filter(f => f.verdict?.isReal)
131
+ return { confirmed }
132
+ // Dimension 'bugs' findings verify while dimension 'perf' is still reviewing. No wasted wall-clock.
133
+
134
+ When a barrier IS correct — dedup across all findings before expensive verification:
135
+ const all = await parallel(DIMENSIONS.map(d => () => agent(d.prompt, {schema: FINDINGS_SCHEMA})))
136
+ const deduped = dedupeByFileAndLine(all.filter(Boolean).flatMap(r => r.findings)) // <-- genuinely needs ALL at once
137
+ const verified = await parallel(deduped.map(f => () => agent(verifyPrompt(f), {schema: VERDICT_SCHEMA})))
138
+
139
+ Loop-until-count pattern — accumulate to a target:
140
+ const bugs = []
141
+ while (bugs.length < 10) {
142
+ const result = await agent("Find bugs in this codebase.", {schema: BUGS_SCHEMA})
143
+ bugs.push(...result.bugs)
144
+ log(\`\${bugs.length}/10 found\`)
145
+ }
146
+
147
+ Gate-and-retry pattern — verify by running, and keep the agent's context across attempts:
148
+ let fixed = await agent('Find and fix the failing test.', {label: 'fix', gate: 'npm test'})
149
+ if (fixed === null) { // a non-zero exit failed the agent
150
+ // Resume keeps everything the child already learned. It cannot carry the
151
+ // gate, so re-verification needs its own gated call, in the same tree.
152
+ fixed = await agent('\`npm test\` is still failing. Fix the cause.', {label: 'fix', resume: 'fix'})
153
+ const verified = await agent('Run \`npm test\` and report the result. Change nothing.',
154
+ {label: 'verify', phase: 'Verify', gate: 'npm test', effort: 'low'})
155
+ return { passed: verified !== null, summary: fixed }
156
+ }
157
+ return { passed: true, summary: fixed }
158
+ // An LLM judging whether a fix works is a weaker signal than the test suite.
159
+
160
+ Composing patterns — exhaustive review (find → dedup vs seen → diverse-lens panel → loop-until-dry):
161
+ const seen = new Set(), confirmed = []
162
+ let dry = 0
163
+ while (dry < 2) { // loop-until-dry
164
+ const found = (await parallel(FINDERS.map(f => () => // barrier: collect all finders this round
165
+ agent(f.prompt, {phase: 'Find', schema: BUGS})))).filter(Boolean).flatMap(r => r.bugs)
166
+ const fresh = found.filter(b => !seen.has(key(b))) // dedup vs ALL seen — plain code, not an agent
167
+ if (!fresh.length) { dry++; continue }
168
+ dry = 0; fresh.forEach(b => seen.add(key(b)))
169
+ const judged = await parallel(fresh.map(b => () => // every fresh bug judged concurrently...
170
+ parallel(['correctness','security','repro'].map(lens => () => // ...each by 3 distinct lenses
171
+ agent(\`Judge "\${b.desc}" via the \${lens} lens — real?\`, {phase: 'Verify', schema: VERDICT})))
172
+ .then(vs => ({ b, real: vs.filter(Boolean).filter(v => v.real).length >= 2 }))))
173
+ confirmed.push(...judged.filter(v => v.real).map(v => v.b))
174
+ }
175
+ return confirmed
176
+ // dedup vs \`seen\`, NOT \`confirmed\` — else judge-rejected findings reappear every round and it never converges.
177
+
178
+ Quality patterns — common shapes; pick by task and compose freely:
179
+ - Adversarial verify: spawn N independent skeptics per finding, each prompted to REFUTE. Kill if ≥majority refute. Prevents plausible-but-wrong findings from surviving.
180
+ const votes = await parallel(Array.from({length: 3}, () => () =>
181
+ agent(\`Try to refute: \${claim}. Default to refuted=true if uncertain.\`, {schema: VERDICT})))
182
+ const survives = votes.filter(Boolean).filter(v => !v.refuted).length >= 2
183
+ - Verify by running, not by asking: when a claim is testable, \`gate\` it rather than asking another model whether it holds.
184
+ - Perspective-diverse verify: when a finding can fail in more than one way, give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters — diversity catches failure modes redundancy can't.
185
+ - Judge panel: generate N independent attempts from different angles (e.g. MVP-first, risk-first, user-first), score with parallel judges, synthesize from the winner while grafting the best ideas from runners-up. Beats one-attempt-iterated when the solution space is wide.
186
+ - Loop-until-dry: for unknown-size discovery (bugs, issues, edge cases), keep spawning finders until K consecutive rounds return nothing new. Simple counters (while count < N) miss the tail.
187
+ - Multi-modal sweep: parallel agents each searching a different way (by-container, by-content, by-entity, by-time). Each is blind to what the others surface; useful when one search angle won't find everything.
188
+ - Completeness critic: a final agent that asks "what's missing — modality not run, claim unverified, source unread?" What it finds becomes the next round of work.
189
+ - No silent caps: if a workflow bounds coverage (top-N, no-retry, sampling), \`log()\` what was dropped — silent truncation reads as "covered everything" when it didn't.
190
+
191
+ Scale to what the user asked for. "find any bugs" → a few finders, single-vote verify. "thoroughly audit this" or "be comprehensive" → larger finder pool, 3–5 vote adversarial pass, synthesis stage. When unsure, lean toward thoroughness for research/review/audit requests and toward brevity for quick checks.
192
+
193
+ These patterns aren't exhaustive — compose novel harnesses when the task calls for it (tournament brackets, self-repair loops, staged escalation, whatever fits).
194
+
195
+ Use this tool for multi-step orchestration where control flow should be deterministic (loops, conditionals, fan-out) rather than model-driven.
196
+
197
+ ## Resume
198
+
199
+ The tool result includes a runId. To resume after a pause, kill, or script edit, relaunch with SubagentWorkflow({scriptPath, resumeFromRunId}) — the longest unchanged prefix of agent() calls returns cached results instantly; the first edited/new call and everything after it runs live. Same script + same args → 100% cache hit. It is a prefix and not a lookup: a later call that still matches is not reused once an earlier one has changed. A journaled failure ends the prefix, so resuming a run that died at agent 5 retries exactly agent 5. Same session only, and the run must have finished — stop it from /agents → Workflows first. Before diagnosing why a completed workflow returned an empty or unexpected result, Read the run's \`<run id>.workflow.jsonl\` beside its script — it records each agent's actual return value; do not assume cached results are non-empty. Date.now()/Math.random()/new Date() are unavailable in scripts (they would break this) — stamp results after the workflow returns, or pass timestamps via args.`;
200
+ //# sourceMappingURL=tool-description.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tool-description.js","sourceRoot":"","sources":["../../src/workflow/tool-description.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAEH;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,2BAA2B,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ogCAiKy9B,CAAC"}
@@ -0,0 +1,48 @@
1
+ /**
2
+ * worker-source.ts — the JavaScript that runs inside the workflow worker thread.
3
+ *
4
+ * The host spawns this with `new Worker(WORKER_SOURCE, { eval: true })`, so the
5
+ * source has to be an inlined string: `AGENTS.md` forbids dynamic `import()`,
6
+ * and a file path would have to survive bundling. Keeping it as a template
7
+ * literal costs editor tooling but nothing else — the worker is plain CommonJS
8
+ * JavaScript and never sees the TypeScript pipeline.
9
+ *
10
+ * Two boundaries stack here, and they are not the same boundary:
11
+ *
12
+ * host thread ←postMessage→ worker thread ←vm context→ workflow script
13
+ *
14
+ * The worker/host split exists for *killability*: `worker.terminate()` stops a
15
+ * runaway script mid-loop, which an in-process `vm` timeout cannot do once the
16
+ * script is inside an `await`. The vm context exists for *determinism and
17
+ * accident-avoidance*, not security — see the note on `codeGeneration` below.
18
+ *
19
+ * ## Why the context gets no host built-ins
20
+ *
21
+ * `vm.createContext(sandbox)` gives the script a fresh realm that already owns
22
+ * `Object`, `Array`, `JSON`, `Math`, `Date`, `Promise`, `Map`, `Set`. We inject
23
+ * *only* our own globals on top. Injecting host built-ins instead would hand the
24
+ * script `Object.constructor` → the **host** `Function`, i.e. a compiler for
25
+ * arbitrary host-realm code.
26
+ *
27
+ * That said: our injected globals are themselves host closures, so
28
+ * `agent.constructor` is still the host `Function`. The hygiene shrinks the
29
+ * surface; it does not close the hole. **`codeGeneration: { strings: false }` is
30
+ * the load-bearing defense** — it makes `Function("…")` and `eval("…")` throw
31
+ * `EvalError`, so a captured host `Function` cannot compile anything. Treat this
32
+ * as a determinism boundary, not a security boundary against a hostile script.
33
+ *
34
+ * ## Why determinism is a prelude and not a stub
35
+ *
36
+ * Because `Date` and `Math` come *from the realm*, they cannot be neutered by
37
+ * injection — there is nothing to inject over. So the compiled source is
38
+ * prefixed with a prelude that runs inside the realm and reassigns `Date.now`
39
+ * and `Math.random` in place, then lexically shadows `Date` with a subclass
40
+ * whose zero-argument constructor throws. Lexical shadowing rather than a global
41
+ * assignment because a `const` in the IIFE scope cannot be reached around.
42
+ *
43
+ * Determinism is enforced because a workflow's journal is replayed by prefix on
44
+ * resume: a script that reads the clock produces a different prefix on the
45
+ * second run and the replay silently diverges.
46
+ */
47
+ export declare const WORKER_SOURCE: string;
48
+ //# sourceMappingURL=worker-source.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"worker-source.d.ts","sourceRoot":"","sources":["../../src/workflow/worker-source.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6CG;AAuBH,eAAO,MAAM,aAAa,QAwsBzB,CAAC"}