bullswarm 0.11.1 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/docs/claude-dynamic-workflow-mechanics.md +112 -0
- package/docs/experiments/2026-08-29-ultracode-vs-bullswarm.md +106 -5
- package/package.json +1 -1
- package/skill/SKILL.md +23 -0
- package/src/help.js +2 -1
- package/src/lib/verify.js +23 -1
- package/src/workflow/cli.js +2 -0
- package/src/workflow/dashboard.js +2 -1
- package/src/workflow/decision.js +58 -2
- package/src/workflow/goal.js +35 -5
- package/src/workflow/runner.js +209 -12
- package/src/workflow/runtime.js +99 -31
- package/src/workflow/template.js +41 -9
- package/src/workflow/validate.js +10 -3
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,85 @@
|
|
|
1
1
|
# bullswarm changelog
|
|
2
2
|
|
|
3
|
+
## 0.12.0 — one decision is a whole program
|
|
4
|
+
|
|
5
|
+
Completes the convergence on Claude Code's dynamic-workflow mechanics
|
|
6
|
+
(`docs/claude-dynamic-workflow-mechanics.md` §1.11 and §4.2): the orchestrator
|
|
7
|
+
is positioned as the compiler of the goal into a program the runtime runs to
|
|
8
|
+
the end, and it is consulted again only at the program boundary (every action
|
|
9
|
+
finished, or the graph blocked).
|
|
10
|
+
|
|
11
|
+
- Data-driven fan-out in proposals: a planner `fanout` may carry
|
|
12
|
+
`itemsFrom: "outputs.<actionId>.outFile"` instead of inline `items`. The
|
|
13
|
+
producer becomes an implicit `dependsOn` and the runtime resolves the item
|
|
14
|
+
list when the producer finishes, so the planner no longer spends a round trip
|
|
15
|
+
waiting to see how many items discovery found. If the producer's output has
|
|
16
|
+
no parseable JSON array, the runtime runs ONE bounded, read-only extraction
|
|
17
|
+
action (`<fanoutId>-items`, `source: "runtime-extraction"`) over that output
|
|
18
|
+
before failing the fan-out truthfully. Resolved lists above
|
|
19
|
+
`maxItemsPerExpansion` fail the fan-out with the count. Events:
|
|
20
|
+
`action.items_resolved`, `action.items_extraction_requested`,
|
|
21
|
+
`action.items_extracted`.
|
|
22
|
+
- Repair policy on verify: `repair: { prompt, maxRounds (1–3), effort? }` on a
|
|
23
|
+
planner `verify`. When the verifier returns `ok:false`, the executor runs a
|
|
24
|
+
fix action (`<verifyId>-repair-<n>`, `source: "repair-policy"`) carrying the
|
|
25
|
+
verifier's concerns verbatim and re-runs the same verify, without a planner
|
|
26
|
+
turn. Dispatch or JSON-parse failures are not repaired. Events:
|
|
27
|
+
`action.repair_started`, `action.reverify_started`, `action.repaired`,
|
|
28
|
+
`action.reverify_rejected`, `action.repair_failed`.
|
|
29
|
+
- Fan-out artifact: a fan-out now writes `out-<id>-summary-*.md` (every item's
|
|
30
|
+
verdict plus an output excerpt) and records it as `outputs.<id>.outFile`, so a
|
|
31
|
+
verify can depend on a fan-out directly and `review` is inferred as usual.
|
|
32
|
+
- Fix: fan-out outputs stored the success COUNT in `ok`, so a dynamic action
|
|
33
|
+
depending on a fan-out could never become ready ("blocked by failed or
|
|
34
|
+
unresolved dependencies") and a fan-out never counted as a successful worker
|
|
35
|
+
for completion evidence. `ok` is now a boolean and the count moved to
|
|
36
|
+
`succeeded`; `fanoutSucceededCount()` reads pre-0.12 state files.
|
|
37
|
+
- Fix: the content gate rejected a worker whose whole answer is a JSON array
|
|
38
|
+
or object as an "announcement without substance", which is exactly what a
|
|
39
|
+
discovery step is told to return. Structured answers now pass
|
|
40
|
+
(`hasStructuredAnswer`).
|
|
41
|
+
- `parseJsonArray` prefers the trailing array, so prose containing brackets
|
|
42
|
+
before the list no longer poisons `itemsFrom`.
|
|
43
|
+
- Planner prompt: PLANNING DOCTRINE rewritten around "you are compiling the
|
|
44
|
+
goal into a program"; new data-driven fanout and repair skeletons; a
|
|
45
|
+
four-action program skeleton (discover → fanout(itemsFrom) → verify(repair)
|
|
46
|
+
→ verify-suite); `executionConstraints.programFeatures` and
|
|
47
|
+
`plannerConsultedOnlyAtProgramBoundary`. The goal orchestrator prompt is
|
|
48
|
+
reframed the same way.
|
|
49
|
+
- Literal double braces in prompts no longer kill actions. Only a known root
|
|
50
|
+
followed by dotted identifiers (`{{item}}`, `{{outputs.<id>.outFile}}`,
|
|
51
|
+
`{{inputs.x}}`, `{{runId}}`, `{{wfDir}}`) is a template ref; any other
|
|
52
|
+
`{{…}}` text (a JSDoc type such as `{{maxLength?: number}}`, Mustache, a JS
|
|
53
|
+
object in a template literal) is left exactly as written by the renderer
|
|
54
|
+
and ignored by the validator. Observed in the 0.11.1 comparison run: a
|
|
55
|
+
planner-authored verify prompt containing `{{maxLength?: number}}` failed
|
|
56
|
+
at render time with zero attempts. Relatedly, `verify` no longer
|
|
57
|
+
template-renders the artifact it reviews: only the reviewer instructions
|
|
58
|
+
are a template; the worker's report is appended verbatim.
|
|
59
|
+
- Doctrine: an `ok:true` verify is accepted; its concerns are informational.
|
|
60
|
+
The 0.11.1 comparison run spent a whole extra program round (7 actions,
|
|
61
|
+
~10 min) polishing "non-blocking" nits reported by verifiers that had
|
|
62
|
+
passed, which Claude's fix stage never does.
|
|
63
|
+
- Scout before compiling: `workflow goal` now starts with a read-only `scout`
|
|
64
|
+
run action (tree, manifest, test status, units of work with the files each
|
|
65
|
+
owns, shared files, risks; ends with a JSON array of unit names) so the
|
|
66
|
+
orchestrator's first program names real files and commands — the counterpart
|
|
67
|
+
of the inline scouting a Claude Code session does before authoring a
|
|
68
|
+
Workflow script. `--no-scout` skips it. A failed scout is non-fatal: the
|
|
69
|
+
planner still runs and sees `outputs.scout.ok=false` with the reason, and
|
|
70
|
+
the scout never counts as a delivery worker.
|
|
71
|
+
- The planner finally sees what workers said: every `outputs.<id>` in the
|
|
72
|
+
durable planner context carries `outputExcerpt` (up to 3 000 chars each,
|
|
73
|
+
36 000 total, newest first) instead of only `ok`/`why`/`outFile`.
|
|
74
|
+
- `workflow goal` default `maxItemsPerExpansion` raised 8 → 24 so a
|
|
75
|
+
data-driven fan-out over a medium repository does not fail on the bound.
|
|
76
|
+
- Known limitation: `itemsFrom` removes the planner turn, not the stage
|
|
77
|
+
barrier. A verify that depends on a data-driven fan-out waits for all items;
|
|
78
|
+
per-item verify overlap on discovered items would need chained
|
|
79
|
+
`stepTemplate`s (not in this release). For known items keep proposing N fix
|
|
80
|
+
+ N verify chains inline, which already overlap under the ready-set
|
|
81
|
+
scheduler.
|
|
82
|
+
|
|
3
83
|
## 0.11.1 — reliable `--watch` handoff
|
|
4
84
|
|
|
5
85
|
- `workflow goal --watch` no longer races the detached child: the watcher now
|
|
@@ -149,6 +149,53 @@ multi-phase work (understand → design → implement → review), that often me
|
|
|
149
149
|
several workflows in sequence — one per phase — so you stay in the loop between
|
|
150
150
|
them."
|
|
151
151
|
|
|
152
|
+
## 1.11 Upfront plan or mid-flight steering? — the answer
|
|
153
|
+
|
|
154
|
+
The question that decides how bullswarm should converge: does Claude prepare
|
|
155
|
+
all phases and parallelism **before** the workflow starts, or does it keep
|
|
156
|
+
planning **during** execution?
|
|
157
|
+
|
|
158
|
+
**Before. Entirely.** The orchestrator model writes one complete program, the
|
|
159
|
+
harness validates it, and then executes it without consulting the model again.
|
|
160
|
+
Evidence:
|
|
161
|
+
|
|
162
|
+
- **[SPEC]** the script is passed whole and parsed before any agent runs;
|
|
163
|
+
"Use this tool for multi-step orchestration where control flow should be
|
|
164
|
+
deterministic (loops, conditionals, fan-out) rather than model-driven";
|
|
165
|
+
"Workflows run in the background — this tool returns immediately … a
|
|
166
|
+
task-notification arrives when the workflow completes."
|
|
167
|
+
- **[OBSERVED, goal 2]** ~4 minutes of inline scouting and advisor review →
|
|
168
|
+
one 23 k-character script (probe → author → adversarial verify → fix-loop,
|
|
169
|
+
per module) → rejected on a parse error before any agent spawned → corrected
|
|
170
|
+
in 80 s → ~20 minutes of execution with six agents concurrent and **zero
|
|
171
|
+
orchestrator decisions**. The session's own words: "the harness will
|
|
172
|
+
re-invoke me when it completes, so polling would just burn tokens. Waiting."
|
|
173
|
+
|
|
174
|
+
What *looks* like mid-flight steering is pre-authored into the program:
|
|
175
|
+
|
|
176
|
+
1. **Data-driven shape.** A stage's structured result (`schema`) becomes the
|
|
177
|
+
next stage's items: `pipeline(discovery.failures, fix, verify)`,
|
|
178
|
+
loop-until-dry. The author fixes the *policy*; the runtime fixes the *size*.
|
|
179
|
+
2. **Repair as code.** `if (!verify.ok) { fix; verify }` bounded by a counter.
|
|
180
|
+
Every fix → re-verify handoff observed in the goal-2 journal was this `if`.
|
|
181
|
+
3. **Budget as data.** `budget.remaining()` scales loops; the ceiling is hard.
|
|
182
|
+
|
|
183
|
+
Model-level re-planning exists only **at workflow boundaries**: "run several
|
|
184
|
+
in sequence — read each result before deciding the next phase"; the hybrid
|
|
185
|
+
"scout inline first … then call Workflow"; and edit-and-resume ("the longest
|
|
186
|
+
unchanged prefix of agent() calls returns cached results instantly; the first
|
|
187
|
+
edited/new call and everything after it runs live"). Steering means stop, edit
|
|
188
|
+
the program, resume — never a per-step decision.
|
|
189
|
+
|
|
190
|
+
**Consequence for bullswarm.** Converge on *Direction A — a program, not a
|
|
191
|
+
step list*: one planning turn produces the complete graph **plus** the
|
|
192
|
+
adaptation policy, the runtime executes it to completion, and the planner is
|
|
193
|
+
consulted again only at the boundary (complete, or next program). Do not build
|
|
194
|
+
a continuous mid-flight steering loop (Direction B); Claude has none inside a
|
|
195
|
+
workflow, and bullswarm already has `steer`, cancel and resume for the
|
|
196
|
+
boundary-level levers. What the runtime still lacks to express a program is
|
|
197
|
+
listed in §4.2 (0.12.0 scope).
|
|
198
|
+
|
|
152
199
|
## 2. What that looks like from the outside **[OBSERVED]**
|
|
153
200
|
|
|
154
201
|
Filled from the experiment report as runs complete. Numbers here are copied
|
|
@@ -182,6 +229,8 @@ from `docs/experiments/2026-08-29-ultracode-vs-bullswarm.md`, never projected.
|
|
|
182
229
|
| Per-agent prompt | Self-contained, plus JSON schema enforced at tool layer | Planner-authored prompt; free-text answer, content-verified by heuristics; `verify` returns JSON verdict | Partial. Schema-enforced worker output is a candidate, not adopted yet |
|
|
183
230
|
| Failure handling | Loops in code; `null` on agent death | Planner replans (costly); 0.10.9 added corrective turns for invalid decisions and 0.11.0 recovers mis-shaped `verify.review` before dispatch | Improved; retry-in-code per action still absent |
|
|
184
231
|
| Determinism / resume | Journal of return values; prefix cache | Durable `state.json` + `events.jsonl` + action ledger; resume skips durable outputs | Equivalent |
|
|
232
|
+
| Data-driven fan-out | `pipeline(discovered.items, …)` — count unknown when the script is written | Decision schema forced inline `items`; the planner spent a turn waiting for discovery | **Fixed in 0.12.0** — `itemsFrom` on proposed fan-outs + one bounded extraction retry |
|
|
233
|
+
| Repair loops | `while`/retry in code | Planner replanned after every failed verify | **Fixed in 0.12.0** — `verify.repair` policy runs fix → re-verify inside the executor |
|
|
185
234
|
| Budget | Hard ceiling | Advisory targets (user decision) | Intentional difference |
|
|
186
235
|
| Observability | Progress tree, narrator, `/workflows` | `watch` heartbeat (semantic quiet + agent-output quiet since 0.10.9), `tui`, events | Comparable |
|
|
187
236
|
| Isolation | `isolation: 'worktree'` per agent | Shared `addDir`; planner-declared file ownership | Candidate |
|
|
@@ -212,6 +261,69 @@ from `docs/experiments/2026-08-29-ultracode-vs-bullswarm.md`, never projected.
|
|
|
212
261
|
validation (so the 0.10.9 corrective turn fixes it) instead of failing a
|
|
213
262
|
dispatch after a planning round trip.
|
|
214
263
|
|
|
264
|
+
### 4.2 Adopted in 0.12.0 — a *program* expressible in one decision
|
|
265
|
+
|
|
266
|
+
The user's framing for this release: position the orchestrator as the
|
|
267
|
+
**compiler** of the goal into a workflow program; the program drives every
|
|
268
|
+
phase and turn; the model is consulted again only at a boundary that needs
|
|
269
|
+
judgement — the same division of labour Claude Code uses between the script
|
|
270
|
+
author and the `Workflow` runtime.
|
|
271
|
+
|
|
272
|
+
1. **Data-driven fan-out in proposals** — shipped. A proposed `fanout` takes
|
|
273
|
+
`itemsFrom: "outputs.<actionId>.outFile"` (producer may be co-proposed; it
|
|
274
|
+
becomes an implicit `dependsOn`). The runtime resolves the list when the
|
|
275
|
+
producer finishes. This is Claude's `pipeline(discovery.failures, …)`: the
|
|
276
|
+
planner no longer spends a turn waiting to see how many items there are.
|
|
277
|
+
2. **Structured worker output** — shipped in its cheap form. Discovery workers
|
|
278
|
+
are told to end with a JSON array; `parseJsonArray` prefers the trailing
|
|
279
|
+
array; the content gate accepts a bare JSON array/object as substance; and
|
|
280
|
+
if the output still has no array the runtime runs ONE bounded, read-only
|
|
281
|
+
extraction action over it (never re-running the producer, which may have
|
|
282
|
+
mutated files). That is the "schema retry" of Claude's `StructuredOutput`,
|
|
283
|
+
done as a second cheap agent instead of a tool-layer retry. A general
|
|
284
|
+
`outputSchema` on run actions is still open (§5).
|
|
285
|
+
3. **Pre-authored repair** — shipped. `repair: { prompt, maxRounds }` on a
|
|
286
|
+
verify: verify-fail → `<verifyId>-repair-<n>` (concerns verbatim) →
|
|
287
|
+
re-verify, inside the executor. Claude's fix-loop as code.
|
|
288
|
+
4. **Boundary-only consultation** — the loop is now: decision 1 = the program;
|
|
289
|
+
the ready-set executor runs it to completion (fan-outs resolve, repairs run);
|
|
290
|
+
decision 2 = `complete` or the next program. The planner prompt says so
|
|
291
|
+
explicitly (`plannerConsultedOnlyAtProgramBoundary`), and the goal
|
|
292
|
+
orchestrator prompt is reframed as "compile the goal into a complete
|
|
293
|
+
workflow program".
|
|
294
|
+
5. **Scout, then compile** — shipped. Claude's author reads the repo inline
|
|
295
|
+
for ~4 min before writing the script; bullswarm's orchestrator was compiling
|
|
296
|
+
blind (goal text + cwd, forbidden to run commands, and the planner context
|
|
297
|
+
exposed no worker output text at all — only `ok`/`why`/`outFile`). `workflow
|
|
298
|
+
goal` now runs a read-only `scout` action first, and every output in the
|
|
299
|
+
planner context carries an `outputExcerpt`, so the first program is written
|
|
300
|
+
against a real survey and boundary decisions read what workers reported.
|
|
301
|
+
6. Two bugs found on the way that had silently blocked this shape in ≤ 0.11.1:
|
|
302
|
+
fan-out outputs recorded the success *count* in `ok`, so nothing could ever
|
|
303
|
+
depend on a fan-out (the ready-set test is `ok === true`); and the content
|
|
304
|
+
gate rejected a worker whose whole answer was a JSON array as an
|
|
305
|
+
"announcement without substance".
|
|
306
|
+
|
|
307
|
+
7. Two robustness gaps the 0.11.1 comparison run itself exposed, fixed
|
|
308
|
+
before 0.12.0 shipped: (a) a planner-authored verify prompt that quoted a
|
|
309
|
+
JSDoc type literally — `{{maxLength?: number}}` — was parsed as a template
|
|
310
|
+
ref and killed the action at render time with zero attempts, forcing an
|
|
311
|
+
extra planner turn to re-issue it. Only a known root plus dotted
|
|
312
|
+
identifiers is a ref now; other double-brace text is prompt content.
|
|
313
|
+
(Claude never has this class of bug: prompts are JS strings, the runtime
|
|
314
|
+
does no substitution.) (b) The new scout is a failable step ahead of the
|
|
315
|
+
planner; it is non-fatal by construction (`onError: continue`), the planner
|
|
316
|
+
sees `outputs.scout.ok=false` with the reason, and a run where only the
|
|
317
|
+
scout succeeded is `blocked`, never "delivered".
|
|
318
|
+
|
|
319
|
+
**Honest limitation.** `itemsFrom` removes the planner *turn*, not the stage
|
|
320
|
+
*barrier*: a verify depending on a data-driven fan-out waits for all items,
|
|
321
|
+
whereas Claude's `pipeline()` overlaps verify-B with fix-C for discovered items
|
|
322
|
+
too. Per-item overlap on unknown items would need a fan-out whose
|
|
323
|
+
`stepTemplate` is itself a chain — not in 0.12.0. For known items the planner
|
|
324
|
+
proposes N fix + N verify inline and the ready-set scheduler already overlaps
|
|
325
|
+
them.
|
|
326
|
+
|
|
215
327
|
## 5. Not adopted (yet), and why
|
|
216
328
|
|
|
217
329
|
- **Schema-enforced worker output.** bullswarm's content verification and the
|
|
@@ -221,15 +221,116 @@ session #1; goal submitted 17:24:4x Z. Observed driving sequence:
|
|
|
221
221
|
bullswarm 0.10.9 now has for planner decisions, except here the correction is
|
|
222
222
|
an inline retry inside one turn rather than a fresh planning process.
|
|
223
223
|
|
|
224
|
-
|
|
224
|
+
Execution numbers (from the workflow journal `wf_b235c760…/journal.jsonl` and
|
|
225
|
+
the session transcript; read-only, no interference):
|
|
225
226
|
|
|
226
|
-
|
|
227
|
+
| Measure | Value |
|
|
228
|
+
| --- | --- |
|
|
229
|
+
| Goal submitted → session idle | 17:24:46 → 18:22:45 Z = **58 min 0 s** |
|
|
230
|
+
| Inline scouting + advisor before the Workflow call | 4 min (17:24:46 → 17:28:42) |
|
|
231
|
+
| Workflow call rejected (parse error) → corrected resend | 94 s (17:28:42 → 17:30:16) |
|
|
232
|
+
| Workflow execution | 17:30:16 → 18:19:07 = **48 min 51 s**, 24 agents, 0 errors |
|
|
233
|
+
| Agents by stage | 6 probe, 6 author, 9 adversarial verify, 3 fix (slugify ×2, intervals ×1) |
|
|
234
|
+
| Max concurrent agents / mean parallelism | **6** (= the six items; cap was 8) / 3.1 |
|
|
235
|
+
| Orchestrator (session) model turns during execution | **0** — the session sat in "Waiting for 1 dynamic workflow to finish" |
|
|
236
|
+
| Session tool calls overall | 16 Bash, 2 Workflow, 1 ToolSearch, 0 Agent |
|
|
237
|
+
| Post-workflow inline verification | 18:19:15 → 18:22:44 (3.5 min): own `npm test`, comment-stripped code identity diff, executing every `@example`, link check, advisor |
|
|
238
|
+
| Session output tokens (orchestrator only) | 91 k; workers 534 k output, 52.9 M cache-read |
|
|
239
|
+
|
|
240
|
+
How the program actually ran (agent start → end):
|
|
227
241
|
|
|
228
|
-
|
|
242
|
+
```text
|
|
243
|
+
probe ×6 17:30:16 → 17:34:36 … 17:36:53 all six in flight at once
|
|
244
|
+
author ×6 17:34:36 → 17:41:20 … 17:46:59 each starts the second ITS probe ends (pipeline, no barrier)
|
|
245
|
+
verify ×6 17:41:20 → 17:45:48 … 17:52:08 each starts the second ITS author ends
|
|
246
|
+
fix slugify 17:46:45→17:55:10 · intervals 17:48:41→17:53:06 · slugify 18:01:34→18:11:52
|
|
247
|
+
re-verify intervals 17:53:07→17:58:36 · slugify 17:55:10→18:01:34 · slugify 18:11:52→18:19:07
|
|
248
|
+
```
|
|
229
249
|
|
|
230
|
-
|
|
250
|
+
Four of the six modules were completely done by 17:49; the remaining 30 min of
|
|
251
|
+
wall time was one module's (slugify) two-round fix loop, pre-authored in the
|
|
252
|
+
script as `while (!verdict.ok && rounds < N)`. No planner turn was spent on
|
|
253
|
+
"how many items", "did the verify pass", or "repair or not" — all three were
|
|
254
|
+
data-driven inside the program. This is the concrete shape 0.12.0 reproduces
|
|
255
|
+
(`itemsFrom`, `repair`).
|
|
231
256
|
|
|
232
|
-
(
|
|
257
|
+
Correctness audit of `g2-claude` (mine, read-only): `npm test` 168/168 (52 → 116
|
|
258
|
+
new); every existing `tests/<module>.test.js` SHA unchanged; every `src/*.js`
|
|
259
|
+
byte-identical to the base after stripping comment/blank lines (JSDoc only);
|
|
260
|
+
6 edge test files (16–24 tests each), 6 docs pages + `docs/README.md` index,
|
|
261
|
+
`@example` on every export. No deliverable missing.
|
|
262
|
+
|
|
263
|
+
### bullswarm 0.11.1 (installed binary) — same goal, `g2-bs-v2`
|
|
264
|
+
|
|
265
|
+
Launched 18:23:39 Z via `run-bs-g2-v2.sh` (`--concurrency 8 --max-agents 40
|
|
266
|
+
--max-expansion-rounds 8`, orchestrator `claude-code`, single pool, Fable
|
|
267
|
+
excluded). Started only after the Claude session went idle so the two never
|
|
268
|
+
competed for the machine.
|
|
269
|
+
|
|
270
|
+
Run `wf-mtda5qq5-c6166f`, isolated `BULLSWARM_HOME`, single pool
|
|
271
|
+
`claude-code` / `claude-opus-5`, `--concurrency 8 --max-agents 40
|
|
272
|
+
--max-expansion-rounds 8 --foreground`, launched 18:23:42 Z on a pristine copy
|
|
273
|
+
of the fixture. Observed read-only from `state.json` / `events.jsonl`.
|
|
274
|
+
|
|
275
|
+
**Timeline**
|
|
276
|
+
|
|
277
|
+
| When (Z) | What happened |
|
|
278
|
+
| --- | --- |
|
|
279
|
+
| 18:23:42 → 18:27:55 | Planner turn 1 (253 s). ONE decision carrying the whole graph: **14 actions** — `module-{csv,duration,intervals,lru,semver,slugify}` (run, no deps), `verify-<module>` ×6 (each depending only on its own module), `docs-index` (depends on all six modules), `verify-suite` (depends on `docs-index` + all six verifies, `review: outputs.docs-index.outFile`). |
|
|
280
|
+
| 18:27:55 | All six module writers start together — **6 concurrent workers** (cap 8, six items). |
|
|
281
|
+
| 18:33:17 → 18:36:39 | Writers finish one by one; each `verify-<module>` starts the moment its own writer finishes (ready-set scheduler); `docs-index` starts at 18:36:40 when the last writer lands. |
|
|
282
|
+
| 18:35:26 | **`verify-slugify` dies with zero attempts**: `template ref "{{maxLength?: number}}" unresolved at render time`. The planner had quoted a JSDoc record type literally in the prompt; the renderer treated the double braces as a template ref. `verify-suite` is then blocked by a failed dependency. |
|
|
283
|
+
| 18:41 → 18:46:01 | Planner turn 2 (~5 min). It diagnosed the render-time death correctly ("two consecutive opening curly braces … treated as an unresolved reference") and proposed **7 actions**: `slugify-recheck` + `verify-slugify-2`, `polish-semver` + `verify-semver-2`, `polish-lru` + `verify-lru-2`, `final-suite`. The two `polish-*` actions react to *non-blocking* nits reported by verifiers that had **passed** (`verify-semver` and `verify-lru` were `ok:true`). |
|
|
284
|
+
| 18:46:01 → 19:03:06 | Remediation round runs (3 fixes → 3 re-verifies → `final-suite`), max 3 concurrent. |
|
|
285
|
+
| 19:03:06 → 19:04:56 | Planner turn 3 (110 s): `complete`, verified, no concerns. |
|
|
286
|
+
|
|
287
|
+
What this run settles, before the numbers: 0.11.x's planning doctrine already
|
|
288
|
+
produces the Claude shape — one decision = the whole graph, six writers in
|
|
289
|
+
parallel, per-item verify overlapping other items' writes, a final
|
|
290
|
+
whole-suite verify at the end. The remaining differences are (a) a runtime
|
|
291
|
+
robustness bug (literal braces), (b) repair happening as a *planner turn*
|
|
292
|
+
rather than inside the program, (c) the planner treating informational
|
|
293
|
+
concerns as work, and (d) no scout before the first program (the planner
|
|
294
|
+
compiled from the goal text alone — correctly here, because the goal names the
|
|
295
|
+
six modules).
|
|
296
|
+
|
|
297
|
+
**Numbers** (from `state.json`/`events.jsonl` via `metrics-bullswarm.mjs`; audit
|
|
298
|
+
via `audit-fixture.sh`, read-only, after the run ended)
|
|
299
|
+
|
|
300
|
+
| Measure | bullswarm 0.11.1 | Claude Code #2 (for reference) |
|
|
301
|
+
| --- | --- | --- |
|
|
302
|
+
| Goal submitted → terminal | 18:23:42 → 19:04:56 Z = **41 min 14 s** (2 474 s) | 58 min 0 s (48 min 51 s of it inside `Workflow`) |
|
|
303
|
+
| Planner / orchestrator time | **667 s = 27 % of wall**, 3 turns (253 s, 304 s, 110 s) | 4 min scouting + 94 s script fix before execution; **0 turns during** the 48 min 51 s execution |
|
|
304
|
+
| Agents dispatched | **22** (19 workers + 3 planner) | 24 workers (6 probe, 6 author, 9 verify, 3 fix) |
|
|
305
|
+
| Max concurrent / mean parallelism | **6** / 2.74 | 6 / 3.1 |
|
|
306
|
+
| Actions per decision | 14, 7, 0 | one script (24 agents) |
|
|
307
|
+
| Actions that died without running | 2 (`verify-slugify` render-time template ref; `verify-suite` blocked by it) | 0 |
|
|
308
|
+
| Remediation rounds | 1 (as a planner turn) | fix loops inside the script (3 fix agents) |
|
|
309
|
+
| Outcome | `completed`, verified, no concerns | done, self-verified inline |
|
|
310
|
+
| Tests after | **130/130** (52 existing + 78 new: 15/14/12/11/13/13 per module) | 168/168 (52 + 116 new: 24/18/19/21/… per module) |
|
|
311
|
+
| Existing `tests/*.test.js` | byte-identical to base (all 7) | byte-identical (all 7) |
|
|
312
|
+
| `src/` changes | comment-only in all 6 modules (0 non-comment line diffs) | comment-only in all 7 files |
|
|
313
|
+
| Deliverables | 6 edge suites, 6 docs pages, `docs/README.md` index (6 links) — all present | same set, all present |
|
|
314
|
+
| Tokens | 148 k *estimated* (utf8 bytes/4 — provider gave no usage; not comparable) | orchestrator 91 k output; workers 534 k output, 52.9 M cache-read (real usage) |
|
|
315
|
+
|
|
316
|
+
Two things worth noting beyond the table. (1) The run *modified the fixture
|
|
317
|
+
to route around the tool's bug*: the planner's remediation rewrote
|
|
318
|
+
`src/slugify.js`'s JSDoc from `@param {{maxLength?: number}} [options]` to the
|
|
319
|
+
dotted `@param {number} [options.maxLength=0]` form so that no later verify
|
|
320
|
+
would trip on the double braces. Comment-only, goal-compliant, but a
|
|
321
|
+
runtime defect leaked into the deliverable. Root cause is two-fold and both
|
|
322
|
+
are fixed in 0.12.0: the renderer treated any `{{…}}` as a ref, and `verify`
|
|
323
|
+
template-rendered the *review artifact* (a worker's report, arbitrary text)
|
|
324
|
+
together with its instructions. (2) The remediation round spent two of its
|
|
325
|
+
three fixes on "non-blocking" nits from verifiers that had returned
|
|
326
|
+
`ok:true`; 0.12.0's doctrine tells the planner those are informational.
|
|
327
|
+
|
|
328
|
+
### bullswarm 0.12.0 (installed binary) — same goal, fresh copy `g2-bs-v3`
|
|
329
|
+
|
|
330
|
+
(pending — after the 0.12.0 release)
|
|
331
|
+
|
|
332
|
+
The originally planned 0.10.9 goal-2 run was dropped at the user's request
|
|
333
|
+
(2026-08-29): the installed latest is the only baseline that matters.
|
|
233
334
|
|
|
234
335
|
## Behaviour differences observed
|
|
235
336
|
|
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -340,6 +340,29 @@ per-item fix→verify chains plus one whole-system verify — because each plann
|
|
|
340
340
|
turn is a full orchestrator round trip. See
|
|
341
341
|
`docs/claude-dynamic-workflow-mechanics.md` for the model this follows.
|
|
342
342
|
|
|
343
|
+
Since 0.12.0 a decision is meant to be a whole *program*: the runtime executes
|
|
344
|
+
it to the end and consults the planner again only at the program boundary
|
|
345
|
+
(every action finished, or the graph blocked). Two planner-level features make
|
|
346
|
+
that expressible without extra turns:
|
|
347
|
+
|
|
348
|
+
- `fanout.itemsFrom: "outputs.<actionId>.outFile"` — fan out over the JSON
|
|
349
|
+
array an earlier (or co-proposed) action ends its output with. The producer
|
|
350
|
+
becomes an implicit dependency and the list is resolved at execution time. A
|
|
351
|
+
producer that answered in prose gets one bounded read-only extraction action
|
|
352
|
+
(`<fanoutId>-items`, `source: "runtime-extraction"`) before the fan-out fails
|
|
353
|
+
truthfully.
|
|
354
|
+
- `verify.repair: { prompt, maxRounds }` — when the verifier returns
|
|
355
|
+
`ok:false`, the executor runs `<verifyId>-repair-<n>` (`source:
|
|
356
|
+
"repair-policy"`) with the verifier's concerns verbatim and re-runs the same
|
|
357
|
+
verify, up to `maxRounds` (1–3), without a planner turn.
|
|
358
|
+
|
|
359
|
+
Every fan-out records a summary artifact as `outputs.<id>.outFile` and a
|
|
360
|
+
boolean `ok` (item count in `succeeded`), so a verify may depend on a fan-out
|
|
361
|
+
directly. The planner context carries `outputs.<id>.outputExcerpt` (what each
|
|
362
|
+
finished action reported), and `workflow goal` runs a read-only `scout` action
|
|
363
|
+
first (`--no-scout` to skip) so the first program is compiled from a real
|
|
364
|
+
survey of the repository rather than from the goal text alone.
|
|
365
|
+
|
|
343
366
|
Allowed planner decisions are `proceed`, `complete`, `needs_more_work`,
|
|
344
367
|
`retry`, `escalate`, `wait_for_approval`, and `stop`. Expansion decisions must
|
|
345
368
|
contain bounded actions; malformed or over-budget output executes nothing.
|
package/src/help.js
CHANGED
|
@@ -585,7 +585,8 @@ const workflowGoalText = rich({
|
|
|
585
585
|
{ flag: '--max-agents <n>', desc: 'planning target for total dispatched agents (soft, not a hard stop)', default: '30 (max 500)' },
|
|
586
586
|
{ flag: '--max-expansion-rounds <n>', desc: 'planning target for planner replanning rounds', default: '8 (max 50)' },
|
|
587
587
|
{ flag: '--max-actions <n>', desc: 'planning target for total dispatched actions', default: '40 (max 1000)' },
|
|
588
|
-
{ flag: '--max-items-per-expansion <n>', desc: 'cap on fanout items
|
|
588
|
+
{ flag: '--max-items-per-expansion <n>', desc: 'cap on fanout items per planner round, inline or resolved from itemsFrom at execution time', default: '24 (max 100)' },
|
|
589
|
+
{ flag: '--no-scout', desc: 'skip the read-only scout action that surveys the repository (tree, manifest, test status, units of work) before the orchestrator compiles its first program', default: 'scout runs first' },
|
|
589
590
|
{ flag: '--max-workflow-seconds <n>', desc: 'planning target for total wall-clock seconds', default: '3600 (max 86400)' },
|
|
590
591
|
{ flag: '--concurrency <n>', desc: 'max parallel dispatches; dependency-ready actions from one decision run concurrently up to this cap', default: '8 (max 16)' },
|
|
591
592
|
{ flag: '--retry-attempts <0..3>', desc: 'same-pool retries per failed action', default: '1' },
|
package/src/lib/verify.js
CHANGED
|
@@ -120,6 +120,28 @@ function hasVerifyJson(text) {
|
|
|
120
120
|
}
|
|
121
121
|
}
|
|
122
122
|
|
|
123
|
+
/**
|
|
124
|
+
* A structured answer: the output is, or ends with, a JSON array (a discovery
|
|
125
|
+
* step's item list, possibly empty) or is a single JSON object. Such output is
|
|
126
|
+
* substance by construction; the prose heuristics must not reject it as an
|
|
127
|
+
* announcement.
|
|
128
|
+
*/
|
|
129
|
+
export function hasStructuredAnswer(text) {
|
|
130
|
+
const trimmed = text.trim();
|
|
131
|
+
if (trimmed.startsWith('{') && trimmed.endsWith('}')) {
|
|
132
|
+
try { return typeof JSON.parse(trimmed) === 'object'; } catch { /* not a single object */ }
|
|
133
|
+
}
|
|
134
|
+
const end = trimmed.lastIndexOf(']');
|
|
135
|
+
if (end === -1 || trimmed.slice(end + 1).trim().length > 0) return false;
|
|
136
|
+
for (let start = trimmed.lastIndexOf('[', end); start !== -1; start = trimmed.lastIndexOf('[', start - 1)) {
|
|
137
|
+
try {
|
|
138
|
+
if (Array.isArray(JSON.parse(trimmed.slice(start, end + 1)))) return true;
|
|
139
|
+
} catch { /* keep widening */ }
|
|
140
|
+
if (start === 0) break;
|
|
141
|
+
}
|
|
142
|
+
return false;
|
|
143
|
+
}
|
|
144
|
+
|
|
123
145
|
/**
|
|
124
146
|
* Judge delegate output content.
|
|
125
147
|
* @param {string} text full delegate output
|
|
@@ -134,7 +156,7 @@ export function judgeContent(text, { exitCode, expectWork = true, acceptVerifyJs
|
|
|
134
156
|
if (scanForFailure(text)) {
|
|
135
157
|
return { verdict: 'fail', why: 'failure pattern at output head' };
|
|
136
158
|
}
|
|
137
|
-
if (expectWork && !looksLikeWork(text) && !(acceptVerifyJson && hasVerifyJson(text))) {
|
|
159
|
+
if (expectWork && !looksLikeWork(text) && !(acceptVerifyJson && hasVerifyJson(text)) && !hasStructuredAnswer(text)) {
|
|
138
160
|
return { verdict: 'intent_only', why: 'announcement without substance' };
|
|
139
161
|
}
|
|
140
162
|
return { verdict: 'pass', why: 'content passed all gates' };
|
package/src/workflow/cli.js
CHANGED
|
@@ -372,6 +372,7 @@ async function wfGoal(opts) {
|
|
|
372
372
|
cwd: opts.cwd ?? process.cwd(),
|
|
373
373
|
orchestrator,
|
|
374
374
|
settings: goalSettings(opts),
|
|
375
|
+
scout: !opts.noScout,
|
|
375
376
|
worktreeIsolation: loadState(BULLSWARM_DIR()).config?.worktreeIsolation ?? 'agent-decides',
|
|
376
377
|
});
|
|
377
378
|
} catch (err) {
|
|
@@ -622,6 +623,7 @@ function parseFlags(argv) {
|
|
|
622
623
|
const a = argv[i];
|
|
623
624
|
if (a === '--json') out.json = true;
|
|
624
625
|
else if (a === '--quiet') out.quiet = true;
|
|
626
|
+
else if (a === '--no-scout') out.noScout = true;
|
|
625
627
|
else if (a === '--resume') out.resume = argv[++i];
|
|
626
628
|
else if (a === '--after') out.after = argv[++i];
|
|
627
629
|
else if (a === '--input') {
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// It deliberately uses only ANSI sequences and Node's standard streams.
|
|
3
3
|
|
|
4
4
|
import { readFileSync, writeFileSync, existsSync } from 'node:fs';
|
|
5
|
+
import { fanoutSucceededCount } from './runner.js';
|
|
5
6
|
import { join } from 'node:path';
|
|
6
7
|
import { listRuns, resolveRunId } from './short-id.js';
|
|
7
8
|
import { appendEvent, readEvents } from './events.js';
|
|
@@ -64,7 +65,7 @@ export function dashboardRows(bullswarmDir) {
|
|
|
64
65
|
const state = r.state ?? {};
|
|
65
66
|
const steps = state.steps ?? [];
|
|
66
67
|
const fanout = Object.values(state.outputs ?? {}).filter((v) => v?.items).reduce((acc, v) => ({
|
|
67
|
-
total: acc.total + (v.total ?? 0), ok: acc.ok + (v
|
|
68
|
+
total: acc.total + (v.total ?? 0), ok: acc.ok + fanoutSucceededCount(v), failed: acc.failed + (v.failed ?? 0),
|
|
68
69
|
}), { total: 0, ok: 0, failed: 0 });
|
|
69
70
|
return {
|
|
70
71
|
...r,
|
package/src/workflow/decision.js
CHANGED
|
@@ -38,6 +38,14 @@ export function parseDecisionText(text) {
|
|
|
38
38
|
// dispatch: instructions move to prompt, and a single-dependency verify reviews
|
|
39
39
|
// its dependency's artifact.
|
|
40
40
|
export const REVIEW_PATH_RE = /^outputs\.([A-Za-z0-9_-]+(?:\[\d+\])?)\.outFile$/;
|
|
41
|
+
// Data-driven fan-out source: the artifact of an earlier (or co-proposed)
|
|
42
|
+
// action whose output ends with a JSON array of items.
|
|
43
|
+
export const ITEMS_FROM_RE = /^outputs\.([A-Za-z0-9_-]+)(?:\.outFile)?$/;
|
|
44
|
+
export const REPAIR_MAX_ROUNDS = 3;
|
|
45
|
+
|
|
46
|
+
export function looksLikeItemsFromPath(value) {
|
|
47
|
+
return typeof value === 'string' && ITEMS_FROM_RE.test(value.trim());
|
|
48
|
+
}
|
|
41
49
|
|
|
42
50
|
export function looksLikeReviewPath(value) {
|
|
43
51
|
return typeof value === 'string' && REVIEW_PATH_RE.test(value.trim());
|
|
@@ -48,6 +56,17 @@ export function normalizeDecisionProposal(proposal) {
|
|
|
48
56
|
return {
|
|
49
57
|
...proposal,
|
|
50
58
|
actions: proposal.actions.map((action) => {
|
|
59
|
+
if (action?.type === 'fanout' && !Array.isArray(action.items) && looksLikeItemsFromPath(action.itemsFrom)) {
|
|
60
|
+
// A fanout fed by an artifact implicitly depends on the producer.
|
|
61
|
+
const itemsFrom = action.itemsFrom.trim();
|
|
62
|
+
const producer = ITEMS_FROM_RE.exec(itemsFrom)[1];
|
|
63
|
+
const dependsOn = Array.isArray(action.dependsOn) ? action.dependsOn : [];
|
|
64
|
+
return {
|
|
65
|
+
...action,
|
|
66
|
+
itemsFrom,
|
|
67
|
+
...(dependsOn.includes(producer) || producer === action.id ? {} : { dependsOn: [...dependsOn, producer] }),
|
|
68
|
+
};
|
|
69
|
+
}
|
|
51
70
|
if (action?.type !== 'verify') return action;
|
|
52
71
|
const singleDependency = Array.isArray(action.dependsOn) && action.dependsOn.length === 1
|
|
53
72
|
? action.dependsOn[0] : null;
|
|
@@ -137,8 +156,23 @@ export function validateDecisionProposal(proposal, {
|
|
|
137
156
|
issues.push(`${at} needs a prompt`);
|
|
138
157
|
}
|
|
139
158
|
if (action.type === 'fanout') {
|
|
140
|
-
|
|
141
|
-
|
|
159
|
+
const hasItems = Array.isArray(action.items);
|
|
160
|
+
const hasItemsFrom = action.itemsFrom != null;
|
|
161
|
+
if (!hasItems && !hasItemsFrom) {
|
|
162
|
+
issues.push(`${at}.items must be an inline array, or ${at}.itemsFrom must be "outputs.<actionId>.outFile" naming the action whose output ends with the JSON array of items`);
|
|
163
|
+
} else if (hasItems) {
|
|
164
|
+
proposedItems += action.items.length;
|
|
165
|
+
if (hasItemsFrom) issues.push(`${at} must use either items or itemsFrom, not both`);
|
|
166
|
+
}
|
|
167
|
+
if (hasItemsFrom && !hasItems) {
|
|
168
|
+
if (!looksLikeItemsFromPath(action.itemsFrom)) {
|
|
169
|
+
issues.push(`${at}.itemsFrom must be a dotted artifact path like "outputs.<actionId>.outFile" (the action whose output ends with a JSON array), not inline items, instructions, or a filesystem path`);
|
|
170
|
+
} else {
|
|
171
|
+
const producer = ITEMS_FROM_RE.exec(action.itemsFrom.trim())[1];
|
|
172
|
+
if (producer === action.id) issues.push(`${at}.itemsFrom cannot reference the fanout itself`);
|
|
173
|
+
else if (!known.has(producer) && !proposedIds.has(producer)) issues.push(`${at}.itemsFrom references unknown action "${producer}"`);
|
|
174
|
+
}
|
|
175
|
+
}
|
|
142
176
|
if (!action.stepTemplate || typeof action.stepTemplate !== 'object') issues.push(`${at}.stepTemplate is required`);
|
|
143
177
|
for (const runtimeOwned of ['pool', 'addDir', 'taskFile']) {
|
|
144
178
|
if (action.stepTemplate?.[runtimeOwned] != null) {
|
|
@@ -146,6 +180,28 @@ export function validateDecisionProposal(proposal, {
|
|
|
146
180
|
}
|
|
147
181
|
}
|
|
148
182
|
}
|
|
183
|
+
if (action.repair != null && action.type !== 'verify') {
|
|
184
|
+
issues.push(`${at}.repair is only valid on verify actions`);
|
|
185
|
+
}
|
|
186
|
+
if (action.type === 'verify' && action.repair != null) {
|
|
187
|
+
const repair = action.repair;
|
|
188
|
+
if (!repair || typeof repair !== 'object' || Array.isArray(repair)) {
|
|
189
|
+
issues.push(`${at}.repair must be an object like {"prompt":"<how to fix what the verifier rejects>","maxRounds":1}`);
|
|
190
|
+
} else {
|
|
191
|
+
if (typeof repair.prompt !== 'string' || !repair.prompt.trim()) {
|
|
192
|
+
issues.push(`${at}.repair.prompt must be a non-empty string telling a worker how to fix what the verifier rejected`);
|
|
193
|
+
}
|
|
194
|
+
if (repair.maxRounds != null && !(Number.isInteger(repair.maxRounds) && repair.maxRounds >= 1 && repair.maxRounds <= REPAIR_MAX_ROUNDS)) {
|
|
195
|
+
issues.push(`${at}.repair.maxRounds must be an integer from 1 to ${REPAIR_MAX_ROUNDS}`);
|
|
196
|
+
}
|
|
197
|
+
if (repair.effort != null && !['high', 'medium', 'low'].includes(repair.effort)) {
|
|
198
|
+
issues.push(`${at}.repair.effort must be high|medium|low`);
|
|
199
|
+
}
|
|
200
|
+
for (const runtimeOwned of ['pool', 'addDir', 'taskFile']) {
|
|
201
|
+
if (repair[runtimeOwned] != null) issues.push(`${at}.repair.${runtimeOwned} is runtime-owned and cannot be proposed by a planner`);
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
}
|
|
149
205
|
if (action.type === 'verify') {
|
|
150
206
|
if (typeof action.review !== 'string') {
|
|
151
207
|
issues.push(`${at}.review is required: a verify with one dependsOn reviews that artifact automatically; with several, set review to "outputs.<actionId>.outFile" and put reviewer instructions in prompt`);
|
package/src/workflow/goal.js
CHANGED
|
@@ -9,17 +9,18 @@ const NAME_RE = /^[a-z0-9][a-z0-9-]*$/;
|
|
|
9
9
|
|
|
10
10
|
export const AUTONOMOUS_ORCHESTRATOR_PROMPT = [
|
|
11
11
|
'You are the autonomous orchestrator for the user goal in the durable workflow context.',
|
|
12
|
+
'Your job is to compile the goal into a complete workflow program. The runtime executes every action you propose, in dependency order and in parallel, without consulting you, and calls you again only at the program boundary: when every action has finished or the graph is blocked.',
|
|
12
13
|
'Own the workflow from initial decomposition through implementation and independent verification.',
|
|
13
14
|
'The user did not author a graph and must not be asked to steer routine execution.',
|
|
14
15
|
'This is a control-plane decision thread, not a worker assignment. Do not invoke Bullswarm, run shell commands, call tools, or modify repository files. Use only the durable context supplied below and return the requested decision JSON; delegate all inspection, implementation, and verification as bounded actions.',
|
|
15
16
|
'',
|
|
16
17
|
'At every checkpoint:',
|
|
17
18
|
'1. Observe the intent, completed actions, artifacts, failures, verification results, available capabilities, and remaining budget.',
|
|
18
|
-
'2. If evidence is insufficient, return needs_more_work with the COMPLETE
|
|
19
|
+
'2. If evidence is insufficient, return needs_more_work with the COMPLETE program: every bounded run, fanout, and verify action you can see now, not the smallest step. Independent actions run concurrently; dependent actions start as soon as their dependencies succeed. One planning round trip costs minutes, so a decision with one action when several are obvious is the expensive choice, and anything decidable by data (how many items, whether a check passed, whether to repair once) belongs in the program, not in a later decision.',
|
|
19
20
|
'3. Give workers self-contained prompts with the exact goal, absolute working directory, the files they may edit (and that they must not touch others), the expected artifact, and the exact acceptance command. A worker sees only its own prompt.',
|
|
20
21
|
'4. Assign every action a short kebab-case phase name such as discover, fix, verify-items, or verify-suite. Phases are forward-only: never append new work to a phase that already finished.',
|
|
21
|
-
'5. Use dependsOn only for real data or same-file ordering dependencies. For N
|
|
22
|
-
' A verify action with exactly one dependency automatically reviews that dependency artifact; you do not need to supply a review path.',
|
|
22
|
+
'5. Use dependsOn only for real data or same-file ordering dependencies. For N known items propose N fix actions and N verify actions (each verify depending only on its own fix) plus one final verify depending on all of them; use fanout with inline items when every item needs the identical prompt. When the item count is unknown, propose a discovery run whose prompt ends with "RETURN ONLY a JSON array of <items>" and a fanout with itemsFrom "outputs.<discovery-id>.outFile", so the runtime fans out the moment discovery finishes.',
|
|
23
|
+
' A verify action with exactly one dependency automatically reviews that dependency artifact (a fan-out artifact summarises every item); you do not need to supply a review path. Give every verify a repair policy {"prompt": "<how to fix what the verifier rejects>", "maxRounds": 1-3} so a rejected verdict is fixed and re-checked inside the program instead of costing another checkpoint.',
|
|
23
24
|
'6. Recover from a failed action with a new bounded action in a new phase when useful; do not repeat an identical failed plan.',
|
|
24
25
|
'7. Require concrete verification of changed behavior. For code changes, obtain relevant test or inspection evidence before completion.',
|
|
25
26
|
'8. Return complete only when durable outputs prove the original goal and its acceptance checks are satisfied.',
|
|
@@ -33,6 +34,28 @@ export const AUTONOMOUS_ORCHESTRATOR_PROMPT = [
|
|
|
33
34
|
'Return stop when unresolved concerns make verified completion disproportionate or when a concrete safety, authority, capability, dependency, or external blocker remains. Stop returns a qualified final outcome; it is not a blanket workflow failure when useful work exists.',
|
|
34
35
|
].join('\n');
|
|
35
36
|
|
|
37
|
+
// Read-only survey that runs before the orchestrator's first decision, so the
|
|
38
|
+
// program it compiles names real files, modules, and commands instead of
|
|
39
|
+
// guessing — the equivalent of the inline scouting a Claude Code session does
|
|
40
|
+
// before authoring a Workflow script.
|
|
41
|
+
export function scoutPrompt(goal, cwd) {
|
|
42
|
+
return [
|
|
43
|
+
'You are the read-only SCOUT for an autonomous workflow. Another agent will turn the goal below into a program of parallel worker actions using ONLY your report, so be concrete and complete.',
|
|
44
|
+
`Working directory (absolute): ${cwd}`,
|
|
45
|
+
`Goal: ${goal}`,
|
|
46
|
+
'',
|
|
47
|
+
'Survey what the goal touches. Do NOT modify, create, or delete any file; do not install dependencies; do not commit.',
|
|
48
|
+
'Report under exactly these headings, at most ~80 lines total:',
|
|
49
|
+
'TREE: the directory tree to depth 3 (skip node_modules, .git, build output), one entry per line.',
|
|
50
|
+
'MANIFEST: package/build manifest facts that matter (name, language/runtime, test command, lint/format command, module system).',
|
|
51
|
+
'TEST STATUS: run the test command once and report the exact pass/fail counts and any failing test names.',
|
|
52
|
+
'UNITS OF WORK: one bullet per independent item the goal implies (module, file, finding, page). For each: the exact files it owns, the exact focused command that proves it is done, and anything already present.',
|
|
53
|
+
'SHARED FILES: files that more than one unit would touch (indexes, barrels, README tables, config) and therefore must be edited by one action after the others.',
|
|
54
|
+
'RISKS: anything that constrains the plan (files that must not change, flaky tests, missing tools, ambiguous requirements).',
|
|
55
|
+
'Finally, END your output with a JSON array of the unit-of-work names in UNITS OF WORK, e.g. ["csv","duration"]. Nothing after the array.',
|
|
56
|
+
].join('\n');
|
|
57
|
+
}
|
|
58
|
+
|
|
36
59
|
function positiveInt(value, fallback, { min = 1, max = Number.MAX_SAFE_INTEGER } = {}) {
|
|
37
60
|
if (value == null) return fallback;
|
|
38
61
|
const parsed = Number(value);
|
|
@@ -48,6 +71,7 @@ export function buildGoalWorkflow({
|
|
|
48
71
|
orchestrator = null,
|
|
49
72
|
name = null,
|
|
50
73
|
settings = {},
|
|
74
|
+
scout = true,
|
|
51
75
|
worktreeIsolation = 'agent-decides',
|
|
52
76
|
} = {}) {
|
|
53
77
|
if (typeof goal !== 'string' || !goal.trim()) {
|
|
@@ -64,7 +88,7 @@ export function buildGoalWorkflow({
|
|
|
64
88
|
const maxAgents = positiveInt(settings.maxAgents, 30, { max: 500 });
|
|
65
89
|
const maxExpansionRounds = positiveInt(settings.maxExpansionRounds, 8, { max: 50 });
|
|
66
90
|
const maxActions = positiveInt(settings.maxActions, 40, { max: 1000 });
|
|
67
|
-
const maxItemsPerExpansion = positiveInt(settings.maxItemsPerExpansion,
|
|
91
|
+
const maxItemsPerExpansion = positiveInt(settings.maxItemsPerExpansion, 24, { max: 100 });
|
|
68
92
|
const maxWorkflowSeconds = positiveInt(settings.maxWorkflowSeconds, 3600, { max: 86_400 });
|
|
69
93
|
const concurrency = positiveInt(settings.concurrency, 8, { max: 16 });
|
|
70
94
|
const retryAttempts = positiveInt(settings.retryAttempts, 1, { min: 0, max: 3 });
|
|
@@ -112,7 +136,13 @@ export function buildGoalWorkflow({
|
|
|
112
136
|
},
|
|
113
137
|
phases: [{
|
|
114
138
|
name: 'autonomous-delivery',
|
|
115
|
-
steps: [{
|
|
139
|
+
steps: [...(scout === true ? [{
|
|
140
|
+
id: 'scout',
|
|
141
|
+
type: 'run',
|
|
142
|
+
lane: 'analyze',
|
|
143
|
+
addDir: targetDir,
|
|
144
|
+
prompt: scoutPrompt(goal.trim(), targetDir),
|
|
145
|
+
}] : []), {
|
|
116
146
|
id: 'orchestrator',
|
|
117
147
|
type: 'decide',
|
|
118
148
|
...(orchestrator ? { pool: orchestrator } : {}),
|
package/src/workflow/runner.js
CHANGED
|
@@ -11,8 +11,10 @@ import { randomBytes } from 'node:crypto';
|
|
|
11
11
|
import { validateWorkflow } from './validate.js';
|
|
12
12
|
import { WorkflowRuntime } from './runtime.js';
|
|
13
13
|
import { generateShortId, listRuns } from './short-id.js';
|
|
14
|
+
import { extractItems } from './template.js';
|
|
14
15
|
import {
|
|
15
16
|
validateDecisionProposal, normalizeDecisionProposal, DecisionValidationError,
|
|
17
|
+
ITEMS_FROM_RE,
|
|
16
18
|
} from './decision.js';
|
|
17
19
|
|
|
18
20
|
export function loadWorkflow(pathOrName, searchDirs) {
|
|
@@ -459,6 +461,10 @@ export async function runWorkflow(opts) {
|
|
|
459
461
|
}
|
|
460
462
|
}
|
|
461
463
|
|
|
464
|
+
function isScoutAction(action) {
|
|
465
|
+
return action?.id === 'scout' && action?.parentId == null && action?.kind === 'run';
|
|
466
|
+
}
|
|
467
|
+
|
|
462
468
|
function actionOutputOk(action, outputs) {
|
|
463
469
|
return Object.hasOwn(outputs ?? {}, action.id)
|
|
464
470
|
? outputs[action.id]?.ok === true
|
|
@@ -503,8 +509,11 @@ function responseExcerpt(text, limit = 1200) {
|
|
|
503
509
|
function terminalPlannerOutcome(state, gate, reason) {
|
|
504
510
|
const dynamicActions = (state.actionLedger ?? []).filter((action) => action.parentId === gate.id);
|
|
505
511
|
const observedActions = (state.actionLedger ?? []).filter((action) => action.kind !== 'decide');
|
|
512
|
+
// A read-only scout (goal survey) is evidence for the planner, never a
|
|
513
|
+
// delivery: a run where only the scout succeeded is blocked, not delivered.
|
|
506
514
|
const usefulWorkers = observedActions.filter((action) =>
|
|
507
|
-
action.kind !== 'verify' &&
|
|
515
|
+
action.kind !== 'verify' && !isScoutAction(action)
|
|
516
|
+
&& action.status === 'succeeded' && actionOutputOk(action, state.outputs));
|
|
508
517
|
const latestWorker = usefulWorkers.at(-1) ?? null;
|
|
509
518
|
const concerns = completionEvidenceGaps(
|
|
510
519
|
dynamicActions,
|
|
@@ -551,6 +560,163 @@ async function runDecisionLoop({ runtime, gate, phase, state, retryAttempts }) {
|
|
|
551
560
|
// dependencies finish, not when the whole wave finishes. Real concurrency is
|
|
552
561
|
// capped by the runtime's global dispatch limiter (settings.concurrency), so
|
|
553
562
|
// this mirrors a pipeline: item B's verify overlaps item C's fix.
|
|
563
|
+
const latestDecisionSequence = () => state.decisions?.at?.(-1)?.sequence ?? null;
|
|
564
|
+
const runProgramAction = async (definition, source) => {
|
|
565
|
+
state.plan.actions.push({
|
|
566
|
+
id: definition.id,
|
|
567
|
+
kind: definition.type,
|
|
568
|
+
phase: executionPhase(definition),
|
|
569
|
+
dependsOn: definition.dependsOn ?? [],
|
|
570
|
+
source,
|
|
571
|
+
decisionSequence: latestDecisionSequence(),
|
|
572
|
+
definition,
|
|
573
|
+
});
|
|
574
|
+
state.currentStep = { id: definition.id, type: definition.type, phase: executionPhase(definition) };
|
|
575
|
+
runtime.persist();
|
|
576
|
+
let result;
|
|
577
|
+
try {
|
|
578
|
+
result = await runtime.runStep(
|
|
579
|
+
{ ...definition, parentId: gate.id, _dynamic: true },
|
|
580
|
+
{ phase: executionPhase(definition), retryAttempts },
|
|
581
|
+
);
|
|
582
|
+
} catch (err) {
|
|
583
|
+
result = { ok: false, why: err.message };
|
|
584
|
+
state.outputs[definition.id] = result;
|
|
585
|
+
runtime.setActionStatus(definition, { phase: executionPhase(definition) }, 'failed_terminal', {
|
|
586
|
+
finishedAt: new Date().toISOString(), why: err.message,
|
|
587
|
+
});
|
|
588
|
+
runtime.emit('action.failed', { actionId: definition.id, status: 'failed_terminal', why: err.message });
|
|
589
|
+
}
|
|
590
|
+
state.steps.push({
|
|
591
|
+
phase: executionPhase(definition), stepId: definition.id, type: definition.type,
|
|
592
|
+
ok: result.ok, why: result.why ?? null, dynamic: true, source,
|
|
593
|
+
});
|
|
594
|
+
runtime.persist();
|
|
595
|
+
return result;
|
|
596
|
+
};
|
|
597
|
+
|
|
598
|
+
// Resolve a proposed fanout's items from the producer artifact named by
|
|
599
|
+
// itemsFrom. If the producer did not end with a parseable JSON array, run
|
|
600
|
+
// ONE bounded, read-only extraction action over its output before giving up;
|
|
601
|
+
// never re-run the producer (it may have mutated files).
|
|
602
|
+
const resolveProposedItems = async (action) => {
|
|
603
|
+
const itemsFrom = action.itemsFrom.trim();
|
|
604
|
+
const producerId = ITEMS_FROM_RE.exec(itemsFrom)?.[1] ?? null;
|
|
605
|
+
const limit = Number(settings.maxItemsPerExpansion ?? 50) || 50;
|
|
606
|
+
const attempt = (path) => {
|
|
607
|
+
try { return { ok: true, items: extractItems(state, path) }; } catch (err) { return { ok: false, why: err.message }; }
|
|
608
|
+
};
|
|
609
|
+
const finish = (outcome) => {
|
|
610
|
+
if (outcome.items.length > limit) {
|
|
611
|
+
return { ok: false, why: `fanout "${action.id}" resolved ${outcome.items.length} items from ${itemsFrom}, exceeding maxItemsPerExpansion=${limit}; propose a narrower discovery or split the fan-out` };
|
|
612
|
+
}
|
|
613
|
+
runtime.emit('action.items_resolved', { actionId: action.id, itemsFrom, count: outcome.items.length });
|
|
614
|
+
return outcome;
|
|
615
|
+
};
|
|
616
|
+
let resolved = attempt(itemsFrom);
|
|
617
|
+
if (!resolved.ok) {
|
|
618
|
+
const producer = producerId ? state.outputs[producerId] : null;
|
|
619
|
+
let sourceText = typeof producer?.outputText === 'string' ? producer.outputText : '';
|
|
620
|
+
try {
|
|
621
|
+
if (producer?.outFile && existsSync(producer.outFile)) sourceText = readFileSync(producer.outFile, 'utf8');
|
|
622
|
+
} catch { /* fall back to the recorded excerpt */ }
|
|
623
|
+
const excerpt = sourceText.length > 20_000 ? `${sourceText.slice(0, 20_000)}\n[truncated]` : sourceText;
|
|
624
|
+
const extractionId = `${action.id}-items`;
|
|
625
|
+
if (state.outputs[extractionId]?.ok === true) {
|
|
626
|
+
// Resume: the extraction already ran durably; reuse its artifact.
|
|
627
|
+
resolved = attempt(`outputs.${extractionId}.outFile`);
|
|
628
|
+
if (resolved.ok) return finish(resolved);
|
|
629
|
+
}
|
|
630
|
+
runtime.emit('action.items_extraction_requested', {
|
|
631
|
+
actionId: action.id, producerId, extractionActionId: extractionId, why: resolved.why,
|
|
632
|
+
});
|
|
633
|
+
const extraction = {
|
|
634
|
+
id: extractionId,
|
|
635
|
+
type: 'run',
|
|
636
|
+
phase: action.phase,
|
|
637
|
+
...(action.lane != null ? { lane: action.lane } : {}),
|
|
638
|
+
...(action.addDir != null ? { addDir: action.addDir } : {}),
|
|
639
|
+
...(action.timeoutSec != null ? { timeoutSec: action.timeoutSec } : {}),
|
|
640
|
+
dependsOn: producerId ? [producerId] : [],
|
|
641
|
+
prompt: [
|
|
642
|
+
`A previous step ("${producerId ?? itemsFrom}") was supposed to end its output with a JSON array of items, but no JSON array could be parsed (${resolved.why}).`,
|
|
643
|
+
'Below is that step\'s complete output. Extract the list of items it describes and RETURN ONLY a JSON array: no prose, no markdown fences, nothing before "[" or after "]".',
|
|
644
|
+
'Each element must be a short string or a small flat object, exactly as the text describes it. If the text describes no list at all, return [].',
|
|
645
|
+
'Do not run commands and do not modify any file; this is a read-only extraction.',
|
|
646
|
+
'',
|
|
647
|
+
'---- BEGIN STEP OUTPUT ----',
|
|
648
|
+
excerpt,
|
|
649
|
+
'---- END STEP OUTPUT ----',
|
|
650
|
+
].join('\n'),
|
|
651
|
+
};
|
|
652
|
+
const extracted = await runProgramAction(extraction, 'runtime-extraction');
|
|
653
|
+
if (!extracted.ok) {
|
|
654
|
+
return { ok: false, why: `fanout "${action.id}" items could not be resolved from ${itemsFrom} (${resolved.why}); extraction action "${extractionId}" failed: ${extracted.why}` };
|
|
655
|
+
}
|
|
656
|
+
resolved = attempt(`outputs.${extractionId}.outFile`);
|
|
657
|
+
if (!resolved.ok) {
|
|
658
|
+
return { ok: false, why: `fanout "${action.id}" items could not be resolved from ${itemsFrom}; extraction action "${extractionId}" returned no JSON array either (${resolved.why})` };
|
|
659
|
+
}
|
|
660
|
+
runtime.emit('action.items_extracted', { actionId: action.id, extractionActionId: extractionId, count: resolved.items.length });
|
|
661
|
+
}
|
|
662
|
+
return finish(resolved);
|
|
663
|
+
};
|
|
664
|
+
|
|
665
|
+
// Bounded repair loop for a verify that returned a real ok:false verdict:
|
|
666
|
+
// run a fix action carrying the verifier's concerns verbatim, then re-run the
|
|
667
|
+
// same verify. Dispatch or parse failures are not repairable and return as-is.
|
|
668
|
+
const repairAndReverify = async (action, firstResult) => {
|
|
669
|
+
const maxRounds = Math.min(3, Math.max(1, Number(action.repair.maxRounds) || 1));
|
|
670
|
+
// Rounds already spent (a resumed run re-verifies first) count against the bound.
|
|
671
|
+
const priorRounds = (state.plan?.actions ?? [])
|
|
672
|
+
.filter((entry) => entry.source === 'repair-policy' && entry.dependsOn?.[0] === action.id).length;
|
|
673
|
+
let result = firstResult;
|
|
674
|
+
for (let round = priorRounds + 1; round <= maxRounds && result.ok === false; round++) {
|
|
675
|
+
const verdict = state.outputs[action.id]?.verify;
|
|
676
|
+
if (!verdict || typeof verdict !== 'object') break;
|
|
677
|
+
const concerns = Array.isArray(verdict.concerns)
|
|
678
|
+
? verdict.concerns.filter((entry) => typeof entry === 'string' && entry.trim()) : [];
|
|
679
|
+
const repairId = `${action.id}-repair-${round}`;
|
|
680
|
+
const repair = {
|
|
681
|
+
id: repairId,
|
|
682
|
+
type: 'run',
|
|
683
|
+
phase: action.phase,
|
|
684
|
+
...(action.lane != null ? { lane: action.lane } : {}),
|
|
685
|
+
...(action.addDir != null ? { addDir: action.addDir } : {}),
|
|
686
|
+
...(action.timeoutSec != null ? { timeoutSec: action.timeoutSec } : {}),
|
|
687
|
+
...(action.repair.effort != null ? { effort: action.repair.effort } : {}),
|
|
688
|
+
...(action.requiresCapabilities != null ? { requiresCapabilities: action.requiresCapabilities } : {}),
|
|
689
|
+
dependsOn: [action.id],
|
|
690
|
+
prompt: [
|
|
691
|
+
action.repair.prompt.trim(),
|
|
692
|
+
'',
|
|
693
|
+
`An independent verifier ("${action.id}") rejected the previous result${
|
|
694
|
+
typeof verdict.summary === 'string' && verdict.summary.trim() ? `: ${verdict.summary.trim()}` : '.'}`,
|
|
695
|
+
...(concerns.length ? ['Concerns to resolve (verbatim from the verifier):', ...concerns.map((entry) => `- ${entry}`)] : []),
|
|
696
|
+
'',
|
|
697
|
+
`Repair round ${round} of ${maxRounds}. Resolve every concern above, re-run the acceptance command yourself, and report exactly what changed with evidence.`,
|
|
698
|
+
].join('\n'),
|
|
699
|
+
};
|
|
700
|
+
runtime.emit('action.repair_started', { verifyId: action.id, repairId, round, maxRounds, concerns });
|
|
701
|
+
const repaired = await runProgramAction(repair, 'repair-policy');
|
|
702
|
+
if (!repaired.ok) {
|
|
703
|
+
runtime.emit('action.repair_failed', { verifyId: action.id, repairId, round, why: repaired.why });
|
|
704
|
+
break;
|
|
705
|
+
}
|
|
706
|
+
runtime.emit('action.reverify_started', { verifyId: action.id, repairId, round });
|
|
707
|
+
state.currentStep = { id: action.id, type: action.type, phase: executionPhase(action) };
|
|
708
|
+
runtime.persist();
|
|
709
|
+
result = await runtime.runStep(
|
|
710
|
+
{ ...action, parentId: gate.id, _dynamic: true },
|
|
711
|
+
{ phase: executionPhase(action), retryAttempts },
|
|
712
|
+
);
|
|
713
|
+
runtime.emit(result.ok ? 'action.repaired' : 'action.reverify_rejected', {
|
|
714
|
+
verifyId: action.id, repairId, round, ok: result.ok, why: result.why ?? null,
|
|
715
|
+
});
|
|
716
|
+
}
|
|
717
|
+
return result;
|
|
718
|
+
};
|
|
719
|
+
|
|
554
720
|
const executeActions = async (actions) => {
|
|
555
721
|
const pending = new Map(actions.map((action) => [action.id, action]));
|
|
556
722
|
const running = new Map();
|
|
@@ -561,16 +727,40 @@ async function runDecisionLoop({ runtime, gate, phase, state, retryAttempts }) {
|
|
|
561
727
|
}
|
|
562
728
|
let result;
|
|
563
729
|
try {
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
730
|
+
let executable = action;
|
|
731
|
+
// Data-driven fan-out: the item list comes from an earlier action's
|
|
732
|
+
// artifact, resolved now that the producer has finished. No planner
|
|
733
|
+
// turn is needed to turn "discovered N things" into N workers.
|
|
734
|
+
if (action.type === 'fanout' && !Array.isArray(action.items) && typeof action.itemsFrom === 'string') {
|
|
735
|
+
const resolved = await resolveProposedItems(action);
|
|
736
|
+
if (!resolved.ok) {
|
|
737
|
+
result = { ok: false, why: resolved.why, itemsUnresolved: true };
|
|
738
|
+
state.outputs[action.id] = result;
|
|
739
|
+
runtime.setActionStatus(action, { phase: executionPhase(action) }, 'failed_terminal', {
|
|
740
|
+
finishedAt: new Date().toISOString(), why: resolved.why,
|
|
741
|
+
});
|
|
742
|
+
runtime.emit('action.failed', { actionId: action.id, status: 'failed_terminal', why: resolved.why });
|
|
743
|
+
} else {
|
|
744
|
+
executable = { ...action, items: resolved.items, itemsResolvedFrom: action.itemsFrom.trim() };
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
if (!result) {
|
|
748
|
+
state.currentStep = {
|
|
749
|
+
id: action.id,
|
|
750
|
+
type: action.type,
|
|
751
|
+
phase: executionPhase(action),
|
|
752
|
+
};
|
|
753
|
+
runtime.persist();
|
|
754
|
+
result = await runtime.runStep(
|
|
755
|
+
{ ...executable, parentId: gate.id, _dynamic: true },
|
|
756
|
+
{ phase: executionPhase(action), retryAttempts },
|
|
757
|
+
);
|
|
758
|
+
// Repair policy: a verify that carries `repair` fixes and re-checks
|
|
759
|
+
// inside the executor instead of returning to the planner.
|
|
760
|
+
if (action.type === 'verify' && result.ok === false && action.repair && typeof action.repair === 'object') {
|
|
761
|
+
result = await repairAndReverify(action, result);
|
|
762
|
+
}
|
|
763
|
+
}
|
|
574
764
|
} catch (err) {
|
|
575
765
|
// A post-artifact observer failure must not rewrite a durably
|
|
576
766
|
// successful action as failed. Propagate it to the gate boundary;
|
|
@@ -852,6 +1042,13 @@ async function runDecisionLoop({ runtime, gate, phase, state, retryAttempts }) {
|
|
|
852
1042
|
}
|
|
853
1043
|
}
|
|
854
1044
|
|
|
1045
|
+
/** Item success count of a fan-out output; tolerates pre-0.12 state where `ok` held the count. */
|
|
1046
|
+
export function fanoutSucceededCount(output) {
|
|
1047
|
+
if (typeof output?.succeeded === 'number') return output.succeeded;
|
|
1048
|
+
if (typeof output?.ok === 'number') return output.ok;
|
|
1049
|
+
return output?.ok === true ? (output.total ?? 0) : 0;
|
|
1050
|
+
}
|
|
1051
|
+
|
|
855
1052
|
export function buildReport(state, doc, runDir) {
|
|
856
1053
|
const stepResults = state.steps ?? [];
|
|
857
1054
|
const fanoutSteps = Object.entries(state.outputs ?? {}).filter(
|
|
@@ -860,7 +1057,7 @@ export function buildReport(state, doc, runDir) {
|
|
|
860
1057
|
let fanoutOk = 0;
|
|
861
1058
|
let fanoutFailed = 0;
|
|
862
1059
|
for (const [, v] of fanoutSteps) {
|
|
863
|
-
fanoutOk += v
|
|
1060
|
+
fanoutOk += fanoutSucceededCount(v);
|
|
864
1061
|
fanoutFailed += v.failed ?? 0;
|
|
865
1062
|
}
|
|
866
1063
|
const simpleOk = stepResults.filter((s) => s.ok).length;
|
package/src/workflow/runtime.js
CHANGED
|
@@ -40,6 +40,11 @@ import { resolveDispatchModel } from '../lib/strategy.js';
|
|
|
40
40
|
// Persisting full transcripts bloat state.json on long workflows. The
|
|
41
41
|
// full text is always on disk in the per-step outFile.
|
|
42
42
|
export const OUTPUT_TEXT_CAP_BYTES = 64 * 1024;
|
|
43
|
+
// Per-item excerpt kept in a fan-out's durable summary artifact.
|
|
44
|
+
const FANOUT_ITEM_EXCERPT_BYTES = 6_000;
|
|
45
|
+
// What the planner sees of each action's output (per output / all outputs).
|
|
46
|
+
const PLANNER_EXCERPT_CHARS = 3_000;
|
|
47
|
+
const PLANNER_EXCERPT_TOTAL_CHARS = 36_000;
|
|
43
48
|
|
|
44
49
|
export function plannerBudgetContext(budget = {}) {
|
|
45
50
|
const dispatchesUsedBeforePlanner = Number(budget.dispatchesUsed ?? 0);
|
|
@@ -846,22 +851,26 @@ export class WorkflowRuntime {
|
|
|
846
851
|
'{"ok": <true|false>, "concerns": [<string>...], "summary": <string>}.',
|
|
847
852
|
'No prose and no markdown fences. Set ok:true only when the requested checks actually pass.',
|
|
848
853
|
].join('\n');
|
|
849
|
-
|
|
854
|
+
// Only the reviewer INSTRUCTIONS are a template. The review target is a
|
|
855
|
+
// worker's artifact — arbitrary text that routinely contains code, JSDoc
|
|
856
|
+
// types and other double-brace sequences — and is appended verbatim,
|
|
857
|
+
// never rendered. (Observed 2026-08-28: a worker report quoting
|
|
858
|
+
// `{{maxLength?: number}}` killed its verify at render time.)
|
|
859
|
+
const rendered = renderDeep({
|
|
850
860
|
lane: step.lane ?? 'analyze',
|
|
851
861
|
addDir: step.addDir,
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
const taskText = rendered.prompt;
|
|
862
|
+
prompt: reviewInstructions,
|
|
863
|
+
}, scope);
|
|
864
|
+
// A custom prompt changes the review instructions, never the review
|
|
865
|
+
// input. Always append the resolved artifact so the skeptic receives
|
|
866
|
+
// the thing it is meant to judge.
|
|
867
|
+
const taskText = [
|
|
868
|
+
rendered.prompt,
|
|
869
|
+
'',
|
|
870
|
+
'---- BEGIN REVIEW TARGET ----',
|
|
871
|
+
reviewedText,
|
|
872
|
+
'---- END REVIEW TARGET ----',
|
|
873
|
+
].join('\n');
|
|
865
874
|
const targetDir = rendered.addDir ? String(rendered.addDir).replace(/^~/, process.env.HOME ?? '') : process.cwd();
|
|
866
875
|
|
|
867
876
|
const stamp = `${step.id}-${Date.now().toString(36)}`;
|
|
@@ -933,14 +942,32 @@ export class WorkflowRuntime {
|
|
|
933
942
|
decisionSequence: steering.decisionSequence,
|
|
934
943
|
});
|
|
935
944
|
}
|
|
936
|
-
|
|
945
|
+
// The planner sees what each action actually said, not just ok/why: an
|
|
946
|
+
// excerpt of every output, newest first, under a total character budget so
|
|
947
|
+
// long runs stay within the planner's context.
|
|
948
|
+
let excerptBudget = PLANNER_EXCERPT_TOTAL_CHARS;
|
|
949
|
+
const excerptFor = (output) => {
|
|
950
|
+
const text = typeof output?.outputText === 'string' ? output.outputText.trim() : '';
|
|
951
|
+
if (!text) return { outputExcerpt: null };
|
|
952
|
+
if (excerptBudget <= 0) return { outputExcerpt: null, outputExcerptOmitted: true, outputChars: text.length };
|
|
953
|
+
const limit = Math.min(PLANNER_EXCERPT_CHARS, excerptBudget);
|
|
954
|
+
const excerpt = text.length > limit ? `${text.slice(0, limit)}\n…[${text.length - limit} more chars in outFile]` : text;
|
|
955
|
+
excerptBudget -= excerpt.length;
|
|
956
|
+
return { outputExcerpt: excerpt, outputChars: text.length };
|
|
957
|
+
};
|
|
958
|
+
const outputEntries = Object.entries(this.state.outputs ?? {});
|
|
959
|
+
const excerpts = new Map(outputEntries.slice().reverse().map(([id, output]) => [id, excerptFor(output)]));
|
|
960
|
+
const outputs = Object.fromEntries(outputEntries.map(([id, output]) => [id, {
|
|
937
961
|
ok: output?.ok,
|
|
938
962
|
why: output?.why ?? null,
|
|
939
963
|
pool: output?.pool ?? null,
|
|
940
964
|
outFile: output?.outFile ?? null,
|
|
941
965
|
verify: output?.verify ?? null,
|
|
942
966
|
total: output?.total,
|
|
967
|
+
succeeded: output?.succeeded,
|
|
943
968
|
failed: output?.failed,
|
|
969
|
+
itemsFrom: output?.itemsFrom,
|
|
970
|
+
...excerpts.get(id),
|
|
944
971
|
}]));
|
|
945
972
|
const actionForPlanner = (action) => ({
|
|
946
973
|
id: action.id,
|
|
@@ -989,6 +1016,8 @@ export class WorkflowRuntime {
|
|
|
989
1016
|
executionConstraints: {
|
|
990
1017
|
concurrency: Number(this.state.settings?.concurrency ?? 1) || 1,
|
|
991
1018
|
readySiblingsRunConcurrently: true,
|
|
1019
|
+
programFeatures: ['itemsFrom', 'repair'],
|
|
1020
|
+
plannerConsultedOnlyAtProgramBoundary: true,
|
|
992
1021
|
actionTimeoutSec: Number(step.actionDefaults?.timeoutSec ?? step.timeoutSec) || null,
|
|
993
1022
|
actionTimeoutIsExplicitOptIn: step.actionDefaults?.timeoutSec != null || step.timeoutSec != null,
|
|
994
1023
|
retryAttempts: Number(this.state.settings?.retryAttempts ?? 0),
|
|
@@ -1038,27 +1067,37 @@ export class WorkflowRuntime {
|
|
|
1038
1067
|
'Every proposed action MUST use the field "type" (never "kind").',
|
|
1039
1068
|
'Every action MUST include a forward-only kebab-case "phase". Never reuse a name listed in closedPhases.',
|
|
1040
1069
|
'',
|
|
1041
|
-
'PLANNING DOCTRINE —
|
|
1042
|
-
'-
|
|
1043
|
-
'-
|
|
1044
|
-
'-
|
|
1070
|
+
'PLANNING DOCTRINE — you are compiling the goal into a PROGRAM, not choosing the next step:',
|
|
1071
|
+
'- The runtime executes your whole decision to completion without consulting you: a ready-set scheduler starts every action whose dependsOn have all succeeded, concurrently up to executionConstraints.concurrency, and starts each dependent the moment its own dependencies finish. You are consulted again only at the program boundary, when every action has finished or the graph is blocked. Each consultation is a separate process round trip (typically 1-2 minutes), so anything decidable by data must be encoded in the program, never deferred to a later decision.',
|
|
1072
|
+
'- Propose the COMPLETE dependency graph you can see now: discovery, per-item work, per-item verification, and the final whole-system verification, all in ONE decision. A decision carrying a single action when several are obvious wastes a round trip.',
|
|
1073
|
+
'- Unknown item count: never spend a decision to learn how many items there are. Propose a discovery run action whose prompt ends with "RETURN ONLY a JSON array of <items>", plus a fanout with "itemsFrom":"outputs.<discovery-id>.outFile" whose stepTemplate.prompt uses {{item}}. The runtime resolves the list when discovery finishes (with one bounded read-only extraction retry if the output is not a clean array) and fans out immediately.',
|
|
1074
|
+
'- Verification failures: give each verify a "repair" policy {"prompt":"<how to fix what the verifier rejects>","maxRounds":1-3}. When the verifier returns ok:false, the runtime runs a fix action carrying the verifier concerns verbatim and re-runs the same verify, inside the program. Only verifies still failing after their rounds come back to you.',
|
|
1075
|
+
'- A verify that returned ok:true is accepted. Its concerns are informational (overlaps, wording nits, "non-blocking" notes): do not spend a program round polishing them unless the goal text itself demands it. Only ok:false verifies are work.',
|
|
1076
|
+
'- Per-item chains: for N known items propose N focused run actions plus N verify actions, each verify depending only on its own run, so verifying one item overlaps with fixing another; add one final verify depending on all of them. For items discovered at run time use the discovery → fanout → verify shape above.',
|
|
1045
1077
|
'- File ownership: every action prompt must name exactly which files it may edit and state that it must not touch any other file. Two actions that must edit the same file MUST be ordered with dependsOn; never let concurrent actions write the same file.',
|
|
1046
1078
|
'- Self-contained prompts: a worker sees only its own prompt, never this context. Each prompt must state the absolute working directory, what to read, what to change, the exact command that proves success, and what to report back. Prefer many small parallel actions over one large serial one.',
|
|
1079
|
+
'- Read before you compile: outputs.<id>.outputExcerpt is what each finished action actually reported (outputs.scout, when present, is a read-only survey of the repository: tree, manifest, test status, units of work, shared files, risks). Name real files, modules, and commands from it in your program instead of guessing.',
|
|
1047
1080
|
'',
|
|
1048
1081
|
'Action skeletons (copy the shape exactly; every field shown is required unless marked optional):',
|
|
1049
1082
|
' run: {"id":"bounded-action","type":"run","phase":"implement","prompt":"Do bounded work.","dependsOn":["prior-action"]}',
|
|
1050
1083
|
' fanout: {"id":"per-item-check","type":"fanout","phase":"inspect","items":["alpha","beta"],"stepTemplate":{"prompt":"Inspect {{item}} and report concrete evidence."},"dependsOn":["prior-action"]}',
|
|
1051
|
-
'
|
|
1084
|
+
' fanout (data-driven): {"id":"per-module-fix","type":"fanout","phase":"fix","itemsFrom":"outputs.discover-modules.outFile","stepTemplate":{"prompt":"In /abs/repo fix only the module {{item}}; run its focused test; report the diff summary."}}',
|
|
1085
|
+
' verify: {"id":"independent-check","type":"verify","phase":"verify","prompt":"Independently re-run the tests and report pass/fail with evidence.","dependsOn":["bounded-action"],"repair":{"prompt":"In /abs/repo fix the failing behaviour the verifier reports, editing only the files named in the concerns, then re-run the tests.","maxRounds":1}}',
|
|
1052
1086
|
'verify semantics: the reviewer receives the artifact of the action named in review, which the runtime infers as outputs.<the single dependsOn>.outFile; put the reviewer INSTRUCTIONS in prompt. A verify with several dependsOn must set review explicitly to "outputs.<actionId>.outFile". review is never instructions or a filesystem path.',
|
|
1087
|
+
'Program skeleton (discovery → data-driven fan-out → verify with repair → final whole-suite check, all in ONE decision; the runtime runs it to the end without you):',
|
|
1088
|
+
' [{"id":"discover-modules","type":"run","phase":"discover","prompt":"In /abs/repo list every module under src/ whose test in tests/ fails. Do not edit anything. RETURN ONLY a JSON array of module names, e.g. [\\"alpha\\",\\"beta\\"]."},',
|
|
1089
|
+
' {"id":"fix-module","type":"fanout","phase":"fix","itemsFrom":"outputs.discover-modules.outFile","stepTemplate":{"prompt":"In /abs/repo edit only src/{{item}}.js so tests/{{item}}.test.js passes; run node --test tests/{{item}}.test.js; report the diff summary."}},',
|
|
1090
|
+
' {"id":"verify-modules","type":"verify","phase":"verify-items","prompt":"For every module in the reviewed fan-out summary re-run node --test tests/<module>.test.js in /abs/repo and confirm tests/ is unchanged.","dependsOn":["fix-module"],"repair":{"prompt":"In /abs/repo fix the modules the verifier lists, editing only their src files, and re-run their tests.","maxRounds":2}},',
|
|
1091
|
+
' {"id":"verify-suite","type":"verify","phase":"verify-suite","prompt":"Run the full npm test in /abs/repo and report pass/fail counts.","dependsOn":["verify-modules"]}]',
|
|
1053
1092
|
'Graph skeleton (two parallel fix→verify chains plus a final whole-suite check, all in ONE decision):',
|
|
1054
1093
|
' [{"id":"fix-alpha","type":"run","phase":"fix","prompt":"In /abs/repo edit only src/alpha.js so tests/alpha.test.js passes; run node --test tests/alpha.test.js; report the diff summary."},',
|
|
1055
1094
|
' {"id":"fix-beta","type":"run","phase":"fix","prompt":"In /abs/repo edit only src/beta.js so tests/beta.test.js passes; run node --test tests/beta.test.js; report the diff summary."},',
|
|
1056
1095
|
' {"id":"verify-alpha","type":"verify","phase":"verify-items","prompt":"Re-run node --test tests/alpha.test.js in /abs/repo and confirm tests/ is unchanged.","dependsOn":["fix-alpha"]},',
|
|
1057
1096
|
' {"id":"verify-beta","type":"verify","phase":"verify-items","prompt":"Re-run node --test tests/beta.test.js in /abs/repo and confirm tests/ is unchanged.","dependsOn":["fix-beta"]},',
|
|
1058
1097
|
' {"id":"verify-suite","type":"verify","phase":"verify-suite","prompt":"Run the full npm test in /abs/repo and report pass/fail counts.","review":"outputs.fix-beta.outFile","dependsOn":["verify-alpha","verify-beta"]}]',
|
|
1059
|
-
'fanout
|
|
1098
|
+
'fanout needs stepTemplate (an object whose prompt uses {{item}}) plus EITHER inline items OR itemsFrom ("outputs.<actionId>.outFile", an action whose output ends with a JSON array; the producer becomes an implicit dependency). A fan-out artifact (outputs.<fanoutId>.outFile) is a summary of every item result, so a verify may depend on a fanout directly. verify.review MUST be a string. dependsOn is optional and may only name existing or newly proposed action IDs.',
|
|
1060
1099
|
'Do not propose pool, addDir, or taskFile; those are runtime-owned and any such proposal is rejected.',
|
|
1061
|
-
'New actions may only be type run, fanout
|
|
1100
|
+
'New actions may only be type run, fanout, or verify. The runtime validates every proposal and returns rejected proposals to you with the exact issues for a bounded correction turn.',
|
|
1062
1101
|
'If executionConstraints.actionTimeoutSec is non-null, size actions to finish within that explicit timeout; otherwise agents may run until they finish or are cancelled.',
|
|
1063
1102
|
'Agent-count, workflow-duration, and expansion-round budgets are advisory planning targets, never hard stop conditions. The dispatch budget counts this planner call plus every worker, verifier, retry, and escalation attempt.',
|
|
1064
1103
|
'As expansion headroom approaches zero, strongly prefer convergence: consolidate existing artifacts, avoid optional investigation, and return complete when verification supports it. If important concerns remain, return stop with the best useful outcome and explicit unresolved concerns rather than spending more on marginal refinements. Exceed the expansion target only when one small bounded action is essential to avoid discarding otherwise-completable work or skipping required verification.',
|
|
@@ -1239,11 +1278,23 @@ export class WorkflowRuntime {
|
|
|
1239
1278
|
|
|
1240
1279
|
const oks = results.filter((r) => r?.verdict?.ok === true).length;
|
|
1241
1280
|
if (this.state.cancelRequested) failures += items.length - results.filter(Boolean).length;
|
|
1281
|
+
// A fan-out is itself an artifact: write a durable summary of every item
|
|
1282
|
+
// result so a verify (or a later fan-out) can depend on the fan-out
|
|
1283
|
+
// directly via outputs.<fanoutId>.outFile, the same way it depends on a run.
|
|
1284
|
+
const summary = this.writeFanoutSummary(step, items, results, oks);
|
|
1285
|
+
// `ok` is a boolean like every other output (so dependents and the
|
|
1286
|
+
// ready-set scheduler can test outputs.<id>.ok === true); the item count
|
|
1287
|
+
// lives in `succeeded`.
|
|
1242
1288
|
this.state.outputs[step.id] = {
|
|
1243
1289
|
total: items.length,
|
|
1244
|
-
ok:
|
|
1290
|
+
ok: failures === 0,
|
|
1291
|
+
succeeded: oks,
|
|
1245
1292
|
failed: items.length - oks,
|
|
1246
1293
|
items: results,
|
|
1294
|
+
...(step.itemsResolvedFrom ? { itemsFrom: step.itemsResolvedFrom } : {}),
|
|
1295
|
+
outFile: summary.outFile,
|
|
1296
|
+
outputText: summary.outputText,
|
|
1297
|
+
outputTruncated: summary.truncated || undefined,
|
|
1247
1298
|
};
|
|
1248
1299
|
parentAction.status = failures === 0 ? 'succeeded' : 'failed_terminal';
|
|
1249
1300
|
parentAction.itemsCompleted = oks;
|
|
@@ -1256,6 +1307,31 @@ export class WorkflowRuntime {
|
|
|
1256
1307
|
return { ok: failures === 0, results };
|
|
1257
1308
|
}
|
|
1258
1309
|
|
|
1310
|
+
writeFanoutSummary(step, items, results, oks) {
|
|
1311
|
+
const lines = [`# fanout ${step.id}: ${oks}/${items.length} items ok`, ''];
|
|
1312
|
+
for (const [index, entry] of results.entries()) {
|
|
1313
|
+
const item = entry?.item ?? items[index];
|
|
1314
|
+
const label = typeof item === 'string' ? item : JSON.stringify(item);
|
|
1315
|
+
const verdict = entry?.verdict;
|
|
1316
|
+
lines.push(`## [${index}] ${label} — ${verdict?.ok === true ? 'ok' : `failed: ${verdict?.why ?? 'no result'}`}`);
|
|
1317
|
+
if (entry?.outFile) lines.push(`artifact: ${entry.outFile}`);
|
|
1318
|
+
let text = '';
|
|
1319
|
+
try {
|
|
1320
|
+
if (entry?.outFile && existsSync(entry.outFile)) text = readFileSync(entry.outFile, 'utf8');
|
|
1321
|
+
} catch { /* the per-item artifact is optional in the summary */ }
|
|
1322
|
+
if (text.trim()) {
|
|
1323
|
+
lines.push('', text.length > FANOUT_ITEM_EXCERPT_BYTES
|
|
1324
|
+
? `${text.slice(0, FANOUT_ITEM_EXCERPT_BYTES)}\n[truncated]` : text.trimEnd());
|
|
1325
|
+
}
|
|
1326
|
+
lines.push('');
|
|
1327
|
+
}
|
|
1328
|
+
const outFile = join(this.runDir, `out-${step.id}-summary-${Date.now().toString(36)}.md`);
|
|
1329
|
+
const full = lines.join('\n');
|
|
1330
|
+
try { writeFileSync(outFile, full); } catch { /* summary is best-effort; items remain in state */ }
|
|
1331
|
+
const truncated = full.length > OUTPUT_TEXT_CAP_BYTES;
|
|
1332
|
+
return { outFile, outputText: truncated ? full.slice(0, OUTPUT_TEXT_CAP_BYTES) : full, truncated };
|
|
1333
|
+
}
|
|
1334
|
+
|
|
1259
1335
|
recordOutput(stepId, verdict, paths) {
|
|
1260
1336
|
let outputText = null;
|
|
1261
1337
|
let truncated = false;
|
|
@@ -1282,11 +1358,3 @@ export class WorkflowRuntime {
|
|
|
1282
1358
|
this.persist();
|
|
1283
1359
|
}
|
|
1284
1360
|
}
|
|
1285
|
-
|
|
1286
|
-
function renderTemplate0(str, scope) {
|
|
1287
|
-
return str.replace(/\{\{\s*([^}]+?)\s*\}\}/g, (_, ref) => {
|
|
1288
|
-
const v = ref.trim().split('.').reduce((acc, k) => (acc == null ? undefined : acc[k]), scope);
|
|
1289
|
-
if (v === undefined) throw new Error(`unresolved ref {{${ref.trim()}}}`);
|
|
1290
|
-
return typeof v === 'string' ? v : JSON.stringify(v);
|
|
1291
|
-
});
|
|
1292
|
-
}
|
package/src/workflow/template.js
CHANGED
|
@@ -10,14 +10,32 @@ export function getPath(obj, path) {
|
|
|
10
10
|
return path.split('.').reduce((acc, key) => (acc == null ? undefined : acc[key]), obj);
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
+
/**
|
|
14
|
+
* The only things bullswarm treats as template refs: a known root followed by
|
|
15
|
+
* dotted identifier segments. Anything else between double braces — a JSDoc
|
|
16
|
+
* type like `{{maxLength?: number}}`, a Mustache/Jinja snippet, a JS object in
|
|
17
|
+
* a template literal — is ordinary prompt text and is left exactly as written.
|
|
18
|
+
* (Observed 2026-08-28: a planner-authored verify prompt containing the literal
|
|
19
|
+
* `{{maxLength?: number}}` killed the action at render time with zero attempts.)
|
|
20
|
+
*/
|
|
21
|
+
export const TEMPLATE_REF_ROOTS = ['item', 'inputs', 'outputs', 'runId', 'wfDir', 'steps'];
|
|
22
|
+
const TEMPLATE_REF_GRAMMAR = new RegExp(`^(?:${TEMPLATE_REF_ROOTS.join('|')})(?:\\.[A-Za-z0-9_-]+)*$`);
|
|
23
|
+
export const TEMPLATE_TOKEN_RE = /\{\{\s*([^}]+?)\s*\}\}/g;
|
|
24
|
+
|
|
25
|
+
export function isTemplateRef(ref) {
|
|
26
|
+
return typeof ref === 'string' && TEMPLATE_REF_GRAMMAR.test(ref.trim());
|
|
27
|
+
}
|
|
28
|
+
|
|
13
29
|
/**
|
|
14
30
|
* Expand {{ref}} tokens in a string against a scope object.
|
|
15
31
|
* Supports {{item}}, {{item.path.to.field}}, {{inputs.x}}, {{outputs.stepId}},
|
|
16
|
-
* {{runId}}, {{wfDir}}. Non-string values are JSON-stringified.
|
|
32
|
+
* {{runId}}, {{wfDir}}. Non-string values are JSON-stringified. Double-brace
|
|
33
|
+
* text that is not a ref (see isTemplateRef) is returned untouched.
|
|
17
34
|
*/
|
|
18
35
|
export function renderTemplate(str, scope) {
|
|
19
36
|
if (typeof str !== 'string') return str;
|
|
20
|
-
return str.replace(
|
|
37
|
+
return str.replace(TEMPLATE_TOKEN_RE, (match, ref) => {
|
|
38
|
+
if (!isTemplateRef(ref)) return match;
|
|
21
39
|
const v = getPath(scope, ref.trim());
|
|
22
40
|
if (v === undefined) {
|
|
23
41
|
throw new Error(`template ref "{{${ref.trim()}}}" unresolved at render time`);
|
|
@@ -89,13 +107,27 @@ export function extractItems(state, itemsFrom) {
|
|
|
89
107
|
|
|
90
108
|
/** Parse the first JSON array found in a text blob, tolerating prose around it. */
|
|
91
109
|
export function parseJsonArray(text) {
|
|
92
|
-
|
|
110
|
+
if (typeof text !== 'string') return null;
|
|
111
|
+
const tryParse = (slice) => {
|
|
112
|
+
try {
|
|
113
|
+
const arr = JSON.parse(slice);
|
|
114
|
+
return Array.isArray(arr) ? arr : null;
|
|
115
|
+
} catch {
|
|
116
|
+
return null;
|
|
117
|
+
}
|
|
118
|
+
};
|
|
119
|
+
// Workers are told to END their output with the array, so prefer the
|
|
120
|
+
// trailing array: from the last "]" walk "[" candidates right-to-left until
|
|
121
|
+
// one parses. Prose that itself contains brackets ("[see below]") then no
|
|
122
|
+
// longer poisons the parse.
|
|
93
123
|
const end = text.lastIndexOf(']');
|
|
94
|
-
if (
|
|
95
|
-
|
|
96
|
-
const
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
return null;
|
|
124
|
+
if (end === -1) return null;
|
|
125
|
+
for (let start = text.lastIndexOf('[', end); start !== -1; start = text.lastIndexOf('[', start - 1)) {
|
|
126
|
+
const parsed = tryParse(text.slice(start, end + 1));
|
|
127
|
+
if (parsed) return parsed;
|
|
128
|
+
if (start === 0) break;
|
|
100
129
|
}
|
|
130
|
+
// Fallback: the widest span (first "[" to last "]").
|
|
131
|
+
const first = text.indexOf('[');
|
|
132
|
+
return first !== -1 && first < end ? tryParse(text.slice(first, end + 1)) : null;
|
|
101
133
|
}
|
package/src/workflow/validate.js
CHANGED
|
@@ -7,6 +7,8 @@
|
|
|
7
7
|
// W3. Lanes and pinned pools are checked against the live registry so a
|
|
8
8
|
// typo never burns quota discovering itself mid-run.
|
|
9
9
|
|
|
10
|
+
import { TEMPLATE_TOKEN_RE, isTemplateRef } from './template.js';
|
|
11
|
+
|
|
10
12
|
const LANES = ['analyze', 'build', 'chore'];
|
|
11
13
|
const ON_ERROR = ['continue', 'fail', 'skip-phase'];
|
|
12
14
|
const STEP_TYPES = ['run', 'fanout', 'verify', 'decide'];
|
|
@@ -24,12 +26,17 @@ function collect(issues, ok, msg) {
|
|
|
24
26
|
return ok;
|
|
25
27
|
}
|
|
26
28
|
|
|
27
|
-
/**
|
|
29
|
+
/**
|
|
30
|
+
* Extract {{ref}} tokens from a string. Only grammar-conforming refs count
|
|
31
|
+
* (known root + dotted identifiers); other double-brace text is prompt content.
|
|
32
|
+
*/
|
|
28
33
|
export function templateRefs(str) {
|
|
29
34
|
const out = [];
|
|
30
|
-
const re =
|
|
35
|
+
const re = new RegExp(TEMPLATE_TOKEN_RE.source, 'g');
|
|
31
36
|
let m;
|
|
32
|
-
while ((m = re.exec(str)) !== null)
|
|
37
|
+
while ((m = re.exec(str)) !== null) {
|
|
38
|
+
if (isTemplateRef(m[1])) out.push(m[1].trim());
|
|
39
|
+
}
|
|
33
40
|
return out;
|
|
34
41
|
}
|
|
35
42
|
|