bullswarm 0.13.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/docs/claude-dynamic-workflow-mechanics.md +42 -10
- package/docs/experiments/2026-08-29-ultracode-vs-bullswarm.md +82 -17
- package/docs/planner-prompt-audit-2026-08-29.md +157 -0
- package/package.json +1 -1
- package/skill/SKILL.md +5 -0
- package/src/workflow/decision.js +18 -4
- package/src/workflow/goal.js +42 -23
- package/src/workflow/runner.js +13 -3
- package/src/workflow/runtime.js +222 -89
- package/src/workflow/schema.js +80 -0
- package/src/workflow/template.js +20 -7
- package/src/workflow/validate.js +13 -2
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,53 @@
|
|
|
1
1
|
# bullswarm changelog
|
|
2
2
|
|
|
3
|
+
## 0.14.0 — structured worker output, compact planner contract
|
|
4
|
+
|
|
5
|
+
- A verify whose reply cannot be parsed as the verdict JSON gets ONE bounded
|
|
6
|
+
re-ask (event `verify.verdict_retry`) before its failure can reach a planner
|
|
7
|
+
boundary — observed on run `ejk9w2`: one unparseable verdict cost a full
|
|
8
|
+
planner turn plus ~8 minutes of re-proving a passing state.
|
|
9
|
+
- Planner contract amendments from the same run's observations: a verify is
|
|
10
|
+
scoped to what can be true at its point in the graph (later-scheduled work is
|
|
11
|
+
not a defect; cosmetic mismatches are concerns, never ok:false); when the
|
|
12
|
+
goal's acceptance checks pass the planner returns complete instead of adding
|
|
13
|
+
polish actions; restored the shared-working-tree, redundant-verification,
|
|
14
|
+
and operatorSteering guidance dropped by the contract merge.
|
|
15
|
+
- "Full" planner-context excerpts (scout, new-since-last-decision, failing
|
|
16
|
+
verifies) obey the per-excerpt and total budgets again; the compaction must
|
|
17
|
+
never rebuild the 163 k-char contexts it replaced.
|
|
18
|
+
- Planner context and contract compacted: complete emitted planner task text up to the durable-context marker, worktree-isolation suffix included **OBSERVED** `16,316 -> 5,208` characters, and a sample turn-2 durable context **COMPUTED** `163,000 -> 23,547` characters by replacing full attempt records with compact ledger rows and retaining full output excerpts only for new/scout or `ok:false` verify actions.
|
|
19
|
+
- Planner `run` actions and fan-out `stepTemplate`s may declare an optional
|
|
20
|
+
`outputSchema`, an object-typed JSON-Schema subset. The runtime tells the
|
|
21
|
+
worker to end its output with one matching JSON object, parses and validates
|
|
22
|
+
it, and persists a `run` result as `outputs.<id>.data` with `schemaOk: true`;
|
|
23
|
+
fan-out results store those fields inside each `outputs.<fanoutId>.items[]`
|
|
24
|
+
entry. Successful validation emits `action.output_validated`.
|
|
25
|
+
- Schema failures emit `action.output_schema_retry` and receive exactly one
|
|
26
|
+
bounded retry with the validation errors and the previous output tail. If
|
|
27
|
+
that retry also fails, the action remains `ok:false`, records
|
|
28
|
+
`schemaOk:false` and `schemaErrors`, and keeps the output text with the
|
|
29
|
+
reason `output did not match outputSchema: <errors>`. Resumed runs do not
|
|
30
|
+
re-dispatch actions already marked `schemaOk:true`.
|
|
31
|
+
- Dependent prompts can render `{{outputs.<id>.data.<field>}}`, and
|
|
32
|
+
`fanout.itemsFrom` accepts `outputs.<id>.data.items` without an extraction
|
|
33
|
+
agent when the array is already present. Planner decision validation rejects
|
|
34
|
+
`outputSchema` on a proposed `verify` because verify has a fixed verdict
|
|
35
|
+
shape.
|
|
36
|
+
|
|
37
|
+
## 0.13.2 — user text is never a template
|
|
38
|
+
|
|
39
|
+
- `workflow goal` failed before anything ran when the goal text quoted
|
|
40
|
+
something shaped like a template ref (`{{outputs.x.data.field}}` in a goal
|
|
41
|
+
*about* templates): the goal was spliced into the scout prompt and the
|
|
42
|
+
workflow validator rejected the ref as unresolvable — "autonomous workflow
|
|
43
|
+
invalid (nothing ran)". The goal is now a declared input (`inputs.goal`)
|
|
44
|
+
inserted at render time, so nothing in user text is ever parsed.
|
|
45
|
+
- A grammar-valid ref with nothing behind it no longer kills the action at
|
|
46
|
+
render time. It is left literally in the prompt and reported as
|
|
47
|
+
`template.unresolved_ref { actionId, ref }`; planner-authored prompts may
|
|
48
|
+
quote refs as text, and a worker can usually still act on the literal.
|
|
49
|
+
`renderTemplate(str, scope, { strict: true })` keeps the old hard failure.
|
|
50
|
+
|
|
3
51
|
## 0.13.1 — a repaired verify counts as verification of its repair
|
|
4
52
|
|
|
5
53
|
- `completionEvidenceGaps` accepted a verify as evidence for the latest worker
|
|
@@ -216,7 +216,15 @@ from `docs/experiments/2026-08-29-ultracode-vs-bullswarm.md`, never projected.
|
|
|
216
216
|
policy and race on the barrel file." The policy it cites was a caution line
|
|
217
217
|
in the planner prompt; a concurrency cap of 8 was available and unused.
|
|
218
218
|
Discovery alone then ran 458 s. (Final numbers: experiment report.)
|
|
219
|
-
-
|
|
219
|
+
- Final comparison on the 6-module fixture (experiment report for the full
|
|
220
|
+
tables). Goal 2 (known items): Claude 58 min, 24 agents, 0 orchestrator
|
|
221
|
+
turns during execution, parallelism 3.1 · bullswarm 0.11.1 41 min, 3 planner
|
|
222
|
+
turns (27 %), max 6 concurrent, parallelism 2.74 · bullswarm 0.12.1 48 min of
|
|
223
|
+
execution after a 3 h quota wait it survived, 4 planner turns (two caused by
|
|
224
|
+
the bug fixed in 0.13.1), max 7 concurrent, one live repair round. Goal 3
|
|
225
|
+
(discovered items) on 0.13.1: 28 min 42 s, **one planner turn**, the runtime
|
|
226
|
+
recorded `complete` itself, exactly the three unguarded modules fixed,
|
|
227
|
+
75/75 tests.
|
|
220
228
|
|
|
221
229
|
## 3. bullswarm today, mechanic by mechanic
|
|
222
230
|
|
|
@@ -226,7 +234,7 @@ from `docs/experiments/2026-08-29-ultracode-vs-bullswarm.md`, never projected.
|
|
|
226
234
|
| Phases | Labels for grouping; never synchronise | Forward-only kebab-case names per action; also just labels | None |
|
|
227
235
|
| Parallelism | `pipeline` default, `parallel` barrier; cap min(16, CPUs−2) | `executeActions` ran dependency-ready siblings **serially** (`runner.js:558`); only `fanout` items ran concurrently; goal default concurrency 3 | **Fixed in 0.11.0** — ready-set scheduler + default 8 |
|
|
228
236
|
| Planner bias | Script author is told to fan out and default to pipeline | Goal prompt said "return needs_more_work with the **smallest useful set** of bounded … actions" (`goal.js:18`) and planner prompt said "keep actions cohesive" | **Fixed in 0.11.0** — "propose the COMPLETE dependency graph", per-item fix→verify chains, file ownership, self-contained prompts |
|
|
229
|
-
| Per-agent prompt | Self-contained, plus JSON schema enforced at tool layer | Planner-authored prompt;
|
|
237
|
+
| Per-agent prompt | Self-contained, plus JSON schema enforced at tool layer | Planner-authored prompt; `outputSchema` validates structured worker data, while `verify` retains its fixed JSON verdict | Adopted for declared schemas; tool-layer enforcement remains a difference |
|
|
230
238
|
| Failure handling | Loops in code; `null` on agent death | Planner replans (costly); 0.10.9 added corrective turns for invalid decisions and 0.11.0 recovers mis-shaped `verify.review` before dispatch | Improved; retry-in-code per action still absent |
|
|
231
239
|
| Determinism / resume | Journal of return values; prefix cache | Durable `state.json` + `events.jsonl` + action ledger; resume skips durable outputs | Equivalent |
|
|
232
240
|
| Data-driven fan-out | `pipeline(discovered.items, …)` — count unknown when the script is written | Decision schema forced inline `items`; the planner spent a turn waiting for discovery | **Fixed in 0.12.0** — `itemsFrom` on proposed fan-outs + one bounded extraction retry |
|
|
@@ -280,8 +288,7 @@ author and the `Workflow` runtime.
|
|
|
280
288
|
if the output still has no array the runtime runs ONE bounded, read-only
|
|
281
289
|
extraction action over it (never re-running the producer, which may have
|
|
282
290
|
mutated files). That is the "schema retry" of Claude's `StructuredOutput`,
|
|
283
|
-
done as a second cheap agent instead of a tool-layer retry.
|
|
284
|
-
`outputSchema` on run actions is still open (§5).
|
|
291
|
+
done as a second cheap agent instead of a tool-layer retry.
|
|
285
292
|
3. **Pre-authored repair** — shipped. `repair: { prompt, maxRounds }` on a
|
|
286
293
|
verify: verify-fail → `<verifyId>-repair-<n>` (concerns verbatim) →
|
|
287
294
|
re-verify, inside the executor. Claude's fix-loop as code.
|
|
@@ -316,7 +323,7 @@ author and the `Workflow` runtime.
|
|
|
316
323
|
sees `outputs.scout.ok=false` with the reason, and a run where only the
|
|
317
324
|
scout succeeded is `blocked`, never "delivered".
|
|
318
325
|
|
|
319
|
-
8. **Program-level completion** (0.13.0
|
|
326
|
+
8. **Program-level completion** (0.13.0) —
|
|
320
327
|
`completion: { when: "all-actions-ok", reason }` on a program. Claude's
|
|
321
328
|
script simply returns when its code is done; bullswarm still spent a final
|
|
322
329
|
planner turn (110–250 s measured) to say `complete` after a clean run. Now
|
|
@@ -324,10 +331,39 @@ author and the `Workflow` runtime.
|
|
|
324
331
|
never below the completion policy) and consults the planner only when
|
|
325
332
|
something failed. With 0.12.0's repair-in-program this makes a clean run
|
|
326
333
|
**one planner turn**: compile, execute, done — Claude's "0 orchestrator turns
|
|
327
|
-
during execution" for the passing case.
|
|
334
|
+
during execution" for the passing case. **[OBSERVED]** goal-3 run
|
|
335
|
+
`wf-mtdkvx0k` (0.13.1): the planner attached the predicate on its own, the
|
|
336
|
+
runtime emitted `decision.auto_completed` (`source: program-completion`),
|
|
337
|
+
one planner process for the whole 28 min run.
|
|
328
338
|
9. **Rate limits are waited for** (0.12.1): a burst-gated provider parks the
|
|
329
339
|
dispatch in `waiting_for_quota` until the window resets instead of failing
|
|
330
340
|
the run in 4 s, which is what the first 0.12.0 comparison launch did.
|
|
341
|
+
**[OBSERVED]** goal-2 run `wf-mtdcghw0`: parked at 95 % for 3 h 2 min,
|
|
342
|
+
dispatched 17 s after the provider reset.
|
|
343
|
+
10. **A repair is verified by its verify's re-run** (0.13.1). The executor's
|
|
344
|
+
repair loop creates `<verify>-repair-N` *depending on* the verify, then runs
|
|
345
|
+
the verify again; the completion-evidence check only followed
|
|
346
|
+
`verify.dependsOn` and so never saw a repair as verified. **[OBSERVED]** on
|
|
347
|
+
`wf-mtdcghw0`: a clean `complete` rejected, three more planner turns
|
|
348
|
+
(~11 min) to re-prove a passed re-verify. In Claude's model this bug cannot
|
|
349
|
+
exist — the script's `while (!ok)` loop *is* the evidence — which is the
|
|
350
|
+
general lesson: every piece of control flow bullswarm moves from planner
|
|
351
|
+
into runtime needs its evidence rule moved with it.
|
|
352
|
+
11. **[SPEC] Schema-enforced worker output** — a planner `run` action or fan-out
|
|
353
|
+
`stepTemplate` may declare an object-typed `outputSchema` subset. The
|
|
354
|
+
runtime appends instructions for one trailing matching JSON object, with no
|
|
355
|
+
prose or markdown fences after it, then parses and validates the object.
|
|
356
|
+
A successful `run` persists `outputs.<id>.data` and `schemaOk: true`; a
|
|
357
|
+
fan-out stores those schema results inside each
|
|
358
|
+
`outputs.<fanoutId>.items[]` entry. Both emit `action.output_validated`. A mismatch emits
|
|
359
|
+
`action.output_schema_retry` and gets exactly one bounded retry carrying
|
|
360
|
+
the validation errors and the previous output tail; a second mismatch
|
|
361
|
+
fails the action while retaining its output text and recording
|
|
362
|
+
`schemaOk:false` and `schemaErrors`. Dependent prompts can render data
|
|
363
|
+
fields, and `fanout.itemsFrom` can consume `outputs.<id>.data.items` without
|
|
364
|
+
extraction when it is already an array. Planner decision validation rejects
|
|
365
|
+
`outputSchema` on a proposed `verify` because verify has a fixed verdict
|
|
366
|
+
shape.
|
|
331
367
|
|
|
332
368
|
**Honest limitation.** `itemsFrom` removes the planner *turn*, not the stage
|
|
333
369
|
*barrier*: a verify depending on a data-driven fan-out waits for all items,
|
|
@@ -339,10 +375,6 @@ them.
|
|
|
339
375
|
|
|
340
376
|
## 5. Not adopted (yet), and why
|
|
341
377
|
|
|
342
|
-
- **Schema-enforced worker output.** bullswarm's content verification and the
|
|
343
|
-
JSON `verify` verdict cover the failure mode today; adding per-action
|
|
344
|
-
`outputSchema` is the next step if planners keep re-asking workers for
|
|
345
|
-
structure.
|
|
346
378
|
- **Per-action worktree isolation.** File ownership declared by the planner is
|
|
347
379
|
cheaper and matches Claude's own guidance ("EXPENSIVE … use ONLY when agents
|
|
348
380
|
mutate files in parallel and would otherwise conflict").
|
|
@@ -325,7 +325,7 @@ together with its instructions. (2) The remediation round spent two of its
|
|
|
325
325
|
three fixes on "non-blocking" nits from verifiers that had returned
|
|
326
326
|
`ok:true`; 0.12.0's doctrine tells the planner those are informational.
|
|
327
327
|
|
|
328
|
-
### bullswarm 0.12.0 (installed binary) — same goal, fresh copy `g2-bs-v3`
|
|
328
|
+
### bullswarm 0.12.0 → 0.12.1 (installed binary) — same goal, fresh copy `g2-bs-v3`
|
|
329
329
|
|
|
330
330
|
**First launch, 19:09:48 Z, installed 0.12.0 — failed in 4 s.** Both the
|
|
331
331
|
scout and the orchestrator were `failed_terminal` with `no eligible pool`
|
|
@@ -393,27 +393,73 @@ Take the bug and the dependency slip out and this run is ~29 min of execution wi
|
|
|
393
393
|
The originally planned 0.10.9 goal-2 run was dropped at the user's request
|
|
394
394
|
(2026-08-29): the installed latest is the only baseline that matters.
|
|
395
395
|
|
|
396
|
+
### bullswarm 0.13.1 (installed binary) — goal 3, discovery-shaped, fresh copy `g3-bs-v3`
|
|
397
|
+
|
|
398
|
+
Goal 3 was written to exercise what goal 2 cannot: an **unknown item list**.
|
|
399
|
+
"Some — not all — of the exported functions accept a wrong-typed argument and
|
|
400
|
+
misbehave. Find out which modules actually have this problem (probe every
|
|
401
|
+
export; keep only the misbehaving modules), then for EACH affected module only:
|
|
402
|
+
add top-of-function argument validation (TypeError naming function, parameter,
|
|
403
|
+
expected type; behaviour for valid input unchanged) and `tests/<module>.guards.test.js`
|
|
404
|
+
(node:test, one test per guard). Do not modify existing tests or touch modules
|
|
405
|
+
that already validate. Finish with `npm test` passing and report exactly which
|
|
406
|
+
modules you changed and which you left alone, with evidence." Same fixture
|
|
407
|
+
family, same single pool (`claude-opus-5`), 0.13.1 installed after the goal-2
|
|
408
|
+
run ended so the binary each run used is unambiguous. Run `wf-mtdkvx0k-c40480`,
|
|
409
|
+
23:23:56 → 23:52:41 Z.
|
|
410
|
+
|
|
411
|
+
| when (Z) | what |
|
|
412
|
+
|---|---|
|
|
413
|
+
| 23:23:59 → 23:28:05 | scout (246 s): probed all six modules; found exactly three misbehaving (csv, slugify, semver) with per-function evidence |
|
|
414
|
+
| 23:28:05 → 23:32:59 | planner turn 1 (294 s): **one 9-action program with `completion: {when: "all-actions-ok"}`** — `fix-{csv,slugify,semver}` + `audit-remaining` (independently re-probe duration/intervals/lru/index) in parallel, each with its own `verify-*` carrying `repair {maxRounds: 2}`, then `verify-suite` |
|
|
415
|
+
| 23:32:59 | 4 workers started in the same second |
|
|
416
|
+
| 23:37:04 → 23:39:31 | each `verify-<m>` started as its own fix finished (pipeline, no barrier) |
|
|
417
|
+
| 23:42:44 → 23:52:41 | `verify-suite` (597 s) ok:true |
|
|
418
|
+
| 23:52:41 | **runtime recorded `complete` itself** — `decision.auto_completed`, `source: program-completion`; no second planner process |
|
|
419
|
+
|
|
420
|
+
| metric | value |
|
|
421
|
+
|---|---|
|
|
422
|
+
| wall | **28 min 42 s** (1 722 s), no quota wait |
|
|
423
|
+
| planner turns / seconds | **1 / 294 s (17 %)** |
|
|
424
|
+
| dispatches / max concurrent / parallelism | 11 / 4 / 1.82 (four items → four chains; width was item-bound, cap 8 unused) |
|
|
425
|
+
| repairs | 0 needed (every verify ok:true first time) |
|
|
426
|
+
| actions by source | planner 9; completion recorded by the runtime |
|
|
427
|
+
| result (audit-fixture.sh + `npm test`) | `src/csv.js`, `src/semver.js`, `src/slugify.js` modified (24/9/9 non-comment lines); duration/intervals/lru/index untouched; 3 new `*.guards.test.js`; existing tests byte-identical; **75/75** (52 + 23) |
|
|
428
|
+
| tokens (estimate) | 51 422 |
|
|
429
|
+
|
|
430
|
+
Two notes. First, the planner did **not** use `fanout.itemsFrom` — it inlined
|
|
431
|
+
the three modules the scout had already named and gave the "not yet confirmed"
|
|
432
|
+
half of the repo to one `audit-remaining` worker. That is the right call (the
|
|
433
|
+
scout had done the discovery), and it is exactly what Claude's author does when
|
|
434
|
+
it discovers the list inline before writing the script; `itemsFrom` stays the
|
|
435
|
+
tool for lists that only exist after a worker runs. Second, `verify-others` and
|
|
436
|
+
`verify-suite` both flagged `lru` throwing `RangeError` rather than `TypeError`
|
|
437
|
+
for a wrong-typed capacity and both correctly treated it as informational (the
|
|
438
|
+
existing test pins `RangeError`): passing-with-nits produced no extra work,
|
|
439
|
+
as the doctrine intends.
|
|
440
|
+
|
|
396
441
|
## Behaviour differences observed
|
|
397
442
|
|
|
398
443
|
Same goal, same fixture, same model (Opus for every worker and for bullswarm's
|
|
399
444
|
planner; the Claude session's author was Opus too). Read left to right: what
|
|
400
445
|
Claude did, what bullswarm 0.11.1 did on the identical run, and what 0.12.x
|
|
401
|
-
now does about it
|
|
402
|
-
|
|
446
|
+
now does about it. Every 0.12.x/0.13.x cell is unit-tested; cells marked
|
|
447
|
+
**observed** were also seen live in the `g2-bs-v3` (0.12.1) and `g3-bs-v3`
|
|
448
|
+
(0.13.1) runs above.
|
|
403
449
|
|
|
404
|
-
| Dimension | Claude Code `Workflow` (ultracode) — observed | bullswarm 0.11.1 — observed | bullswarm 0.12.
|
|
450
|
+
| Dimension | Claude Code `Workflow` (ultracode) — observed | bullswarm 0.11.1 — observed | bullswarm 0.12.1 / 0.13.1 |
|
|
405
451
|
| --- | --- | --- | --- |
|
|
406
|
-
| Who plans, and when | The session author read every file and ran the tests inline (4 min), then wrote **one script** (23 k chars, 5 `agent()` sites). **0 orchestrator turns during the 48 min 51 s of execution.** | The planner compiled the **whole 14-action graph in one decision** (253 s) — but blind: goal text + cwd only, no repo survey, no worker output text in its context. Consulted **3 times** (253 s, 304 s, 110 s) = **27 % of wall**. | Read-only `scout` action before the planner; `outputExcerpt` of every finished action in the planner context; prompt reframed as "compile the goal into a PROGRAM"; planner told it is consulted only at the program boundary. |
|
|
452
|
+
| Who plans, and when | The session author read every file and ran the tests inline (4 min), then wrote **one script** (23 k chars, 5 `agent()` sites). **0 orchestrator turns during the 48 min 51 s of execution.** | The planner compiled the **whole 14-action graph in one decision** (253 s) — but blind: goal text + cwd only, no repo survey, no worker output text in its context. Consulted **3 times** (253 s, 304 s, 110 s) = **27 % of wall**. | Read-only `scout` action before the planner; `outputExcerpt` of every finished action in the planner context; prompt reframed as "compile the goal into a PROGRAM"; planner told it is consulted only at the program boundary. **Observed:** both runs compiled the whole program on turn 1 from the scout's survey; goal 3 ran on **one planner turn** (0.13.0 self-completion). |
|
|
407
453
|
| Item discovery | `pipeline(MODULES, probe, author, verify, fix-loop)` over a known list; when a list is unknown Claude discovers it inline *before* writing the script. | Goal named the six modules → inlined them. Nothing to discover here. | `fanout.itemsFrom: "outputs.<discovery>.outFile"` resolved at run time (+ one bounded read-only extraction retry), so an unknown item count never costs a planner turn. |
|
|
408
454
|
| Parallel width and overlap | **6 concurrent** (= six items, cap 8), mean parallelism 3.1. Per-item pipeline: author-B starts the second probe-B ends; no barriers. | **6 concurrent**, mean parallelism 2.74. Ready-set scheduler: each `verify-<m>` started the second its own `module-<m>` finished; `docs-index` waited for all six by design. | Unchanged for known items. Limitation stays: a verify on a *discovered* fan-out waits for all items (no per-item chain inside a fan-out yet). |
|
|
409
|
-
| Verify → fix | Fix loops **pre-authored in code** (`while (!verdict.ok && rounds < N)`): slugify ×2, intervals ×1, all inside the script; 9 verifies, 3 fixes, 0 planner involvement. | A failed/blocked verify came back to the **planner** (turn 2, 304 s), which authored `slugify-recheck` + `verify-slugify-2`. Round trip ≈ 5 min before the fix even started. | `verify.repair { prompt, maxRounds 1–3 }` — the executor runs `<verify>-repair-<n>` with the concerns verbatim and re-runs the same verify; only still-failing verifies return to the planner. |
|
|
410
|
-
| Passing verifies with nits | Schema-forced `{ok, issues}`; the script fixes only when `!ok`. Nits on passing modules were ignored. | Planner spent **2 of 3 remediation fixes** (`polish-semver`, `polish-lru`) on "non-blocking" notes from verifiers that had returned `ok:true` — an extra ~10 min program round. | Doctrine line: an `ok:true` verify is accepted; its concerns are informational. |
|
|
411
|
-
| Robustness to content | Prompts are JS strings; the runtime substitutes nothing. A parse error in the *script* was caught by the harness and corrected inline in 94 s. | The template renderer parsed **any** `{{…}}` — in a planner prompt *and* in the review artifact it appended. `verify-slugify` died at render time with **0 attempts**, blocked `verify-suite`, cost a planner round, and the fix **rewrote fixture source** (JSDoc) to dodge the bug. | Only a known root + dotted identifiers is a template ref; other double braces are text. `verify` appends the reviewed artifact verbatim, never rendered. |
|
|
412
|
-
| Provider rate limits | Agents retry on API errors; a terminal error resolves the agent to `null`, the script keeps going. The session waits. | Pool burst-gated (5h window 91 %) → the whole run **failed in 4 s** with `no eligible pool`, no reset time named (first 0.12.0 launch, 19:09 Z). | 0.12.1: `waiting_for_quota` stage, meter re-read every 60 s, continue when the window resets; fail only after reset + 10 min grace, naming pool / usage / reset time. |
|
|
455
|
+
| Verify → fix | Fix loops **pre-authored in code** (`while (!verdict.ok && rounds < N)`): slugify ×2, intervals ×1, all inside the script; 9 verifies, 3 fixes, 0 planner involvement. | A failed/blocked verify came back to the **planner** (turn 2, 304 s), which authored `slugify-recheck` + `verify-slugify-2`. Round trip ≈ 5 min before the fix even started. | `verify.repair { prompt, maxRounds 1–3 }` — the executor runs `<verify>-repair-<n>` with the concerns verbatim and re-runs the same verify; only still-failing verifies return to the planner. **Observed** (goal 2): `verify-full-delivery` failed on a missing `docs/README.md`, the repair wrote it and the re-verify passed, ~9 min, no planner turn. Exposed the 0.13.1 bug (repair never counted as verified). |
|
|
456
|
+
| Passing verifies with nits | Schema-forced `{ok, issues}`; the script fixes only when `!ok`. Nits on passing modules were ignored. | Planner spent **2 of 3 remediation fixes** (`polish-semver`, `polish-lru`) on "non-blocking" notes from verifiers that had returned `ok:true` — an extra ~10 min program round. | Doctrine line: an `ok:true` verify is accepted; its concerns are informational. **Observed:** 6 + 4 passing verifies with concerns in the two runs, zero polish actions. |
|
|
457
|
+
| Robustness to content | Prompts are JS strings; the runtime substitutes nothing. A parse error in the *script* was caught by the harness and corrected inline in 94 s. | The template renderer parsed **any** `{{…}}` — in a planner prompt *and* in the review artifact it appended. `verify-slugify` died at render time with **0 attempts**, blocked `verify-suite`, cost a planner round, and the fix **rewrote fixture source** (JSDoc) to dodge the bug. | Only a known root + dotted identifiers is a template ref; other double braces are text. `verify` appends the reviewed artifact verbatim, never rendered. **Observed:** same pristine `slugify.js` with `{{maxLength?: number}}`, `verify-slugify` ran and passed. |
|
|
458
|
+
| Provider rate limits | Agents retry on API errors; a terminal error resolves the agent to `null`, the script keeps going. The session waits. | Pool burst-gated (5h window 91 %) → the whole run **failed in 4 s** with `no eligible pool`, no reset time named (first 0.12.0 launch, 19:09 Z). | 0.12.1: `waiting_for_quota` stage, meter re-read every 60 s, continue when the window resets; fail only after reset + 10 min grace, naming pool / usage / reset time. **Observed:** waited 3 h 2 min at 95 %, resumed 17 s after the reset, no operator action. |
|
|
413
459
|
| Structured worker output | `schema:` forces a `StructuredOutput` tool call; mismatches retry at the tool layer, so the script never parses prose. | Prose "content gate" (`looksLikeWork`); a bare JSON array answer was rejected as an "announcement"; the verify verdict is the only structured channel. | Content gate accepts JSON; `parseJsonArray` prefers the trailing array; one extraction action when discovery output has no array. A general `outputSchema` on run actions is still open. |
|
|
414
460
|
| Failure semantics | In code: `parallel()` never rejects, a throwing stage drops its item to `null`, `.filter(Boolean)`. | Runtime `onError: continue` per step; failed dependencies block dependents; blocked graph → planner. | Same, plus: a failed scout is non-fatal; fan-out `ok` is a boolean so dependents can wait on a whole fan-out. |
|
|
415
|
-
| Outcome quality (audits, read-only) | 168/168 tests (52 + 116 new); existing tests byte-identical; `src` comment-only; every deliverable present. | 130/130 tests (52 + 78 new); existing tests byte-identical; `src` comment-only; every deliverable present. |
|
|
416
|
-
| Time and agents | 58 min end to end; 24 agents. | 41 min end to end; 22 dispatches. Faster because it wrote fewer tests per module (11–15 vs 16–24) and skipped Claude's probe stage — not because it orchestrated better. | — |
|
|
461
|
+
| Outcome quality (audits, read-only) | 168/168 tests (52 + 116 new); existing tests byte-identical; `src` comment-only; every deliverable present. | 130/130 tests (52 + 78 new); existing tests byte-identical; `src` comment-only; every deliverable present. | Goal 2 on 0.12.1: 120/120 (52 + 68); existing tests byte-identical; `src` comment-only; every deliverable present. Goal 3 on 0.13.1: exactly the three unguarded modules changed, 75/75. |
|
|
462
|
+
| Time and agents | 58 min end to end; 24 agents. | 41 min end to end; 22 dispatches. Faster because it wrote fewer tests per module (11–15 vs 16–24) and skipped Claude's probe stage — not because it orchestrated better. | Goal 2 on 0.12.1: 48 min of execution (+3 h quota wait), 22 dispatches, max 7 concurrent — ~11 min of it spent on the 0.13.1 bug and ~9 min on one repair round. Goal 3 on 0.13.1: 28 min 42 s, 11 dispatches, 1 planner turn. |
|
|
417
463
|
|
|
418
464
|
The short version: after 0.11.x the *shape* already matched (one decision =
|
|
419
465
|
whole graph, six in parallel, per-item verify overlap). What still separated
|
|
@@ -425,8 +471,9 @@ the runtime.
|
|
|
425
471
|
|
|
426
472
|
## What to change in bullswarm
|
|
427
473
|
|
|
428
|
-
**Shipped in this cycle** (0.12.0 `c1b71a8`, 0.12.1 `beeed94
|
|
429
|
-
unit
|
|
474
|
+
**Shipped in this cycle** (0.12.0 `c1b71a8`, 0.12.1 `beeed94`, 0.13.0
|
|
475
|
+
`6e5f620`, 0.13.1 `1bf0840` — all released to npm and installed; unit suite
|
|
476
|
+
299/299):
|
|
430
477
|
|
|
431
478
|
1. Orchestrator as **compiler**: prompt reframed; planner consulted only at the
|
|
432
479
|
program boundary; `programFeatures: ['itemsFrom', 'repair']` advertised.
|
|
@@ -442,11 +489,18 @@ unit tests, 295/295):
|
|
|
442
489
|
8. Template refs are grammar-checked; review artifacts are never rendered.
|
|
443
490
|
9. Doctrine: `ok:true` verifies are accepted; concerns are informational.
|
|
444
491
|
10. Burst-gated providers are waited for (`waiting_for_quota`), not failed.
|
|
445
|
-
11. **Program-level completion predicate** (
|
|
446
|
-
the 0.12.1 observation run is in flight): `completion: { when:
|
|
492
|
+
11. **Program-level completion predicate** (0.13.0): `completion: { when:
|
|
447
493
|
"all-actions-ok", reason }` lets a clean program record its own `complete`
|
|
448
|
-
decision — no final planner turn
|
|
449
|
-
|
|
494
|
+
decision — no final planner turn just to say so. Anything failing still
|
|
495
|
+
returns to the planner. **Observed** on goal 3: the planner attached it
|
|
496
|
+
unprompted, the runtime recorded `complete` (`source: program-completion`)
|
|
497
|
+
at 23:52:41 Z, one planner turn for the whole run.
|
|
498
|
+
12. **A repaired verify counts as verification of its repair** (0.13.1): the
|
|
499
|
+
completion-evidence check only followed `verify.dependsOn`; a repair action
|
|
500
|
+
depends on its verify (reverse edge), so after a clean repair round every
|
|
501
|
+
`complete` was rejected as "missing a successful verification of latest
|
|
502
|
+
worker <verify>-repair-1" — observed on goal 2 (three extra planner turns,
|
|
503
|
+
~11 min) and it would have blocked item 11 in the same situation.
|
|
450
504
|
|
|
451
505
|
**Still open, in priority order** (each is a measured gap, not a guess):
|
|
452
506
|
|
|
@@ -467,3 +521,14 @@ unit tests, 295/295):
|
|
|
467
521
|
whatever the planner wrote. A default skeptic framing in the verify wrapper
|
|
468
522
|
is cheap and would have caught nothing extra here — listed for parity, not
|
|
469
523
|
urgency.
|
|
524
|
+
5. **Dependency slips by the planner.** Goal 2's `write-docs-index` was
|
|
525
|
+
compiled with `dependsOn: []` although it reads the six `docs/<m>.md` files
|
|
526
|
+
the builders create; it ran first and found nothing. The verify + repair
|
|
527
|
+
loop recovered it (~9 min). Claude has the same failure class (a mis-ordered
|
|
528
|
+
`pipeline` stage) and the same recovery. A doctrine line — "an action that
|
|
529
|
+
reads another proposed action's deliverable must depend on it" — is free;
|
|
530
|
+
a deterministic check is not possible without declared outputs, which would
|
|
531
|
+
be a small schema addition (`produces: [paths]`).
|
|
532
|
+
6. **Record `completion` on the decision.** The planner artifact carries the
|
|
533
|
+
predicate but `state.decisions[]` does not, so `watch`/metrics cannot show
|
|
534
|
+
that a program declared itself self-completing until it does. One field.
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# Planner prompt and context audit — 2026-08-29
|
|
2
|
+
|
|
3
|
+
Question from the user: after the 0.11 → 0.14 iterations, does the instruction
|
|
4
|
+
set given to the orchestrator/planner still make sense, or should it be
|
|
5
|
+
simplified or refactored?
|
|
6
|
+
|
|
7
|
+
Method: measure what the planner actually receives, not what the source files
|
|
8
|
+
look like. The two planner task files of dogfood run `wf-mtdq1l9v-ed22fe`
|
|
9
|
+
(`75t4n2`, bullswarm building `outputSchema` in its own repo, runtime 0.13.2)
|
|
10
|
+
are the specimens; sizes are characters of the task text
|
|
11
|
+
(`task-orchestrator-*.md`), tokens ≈ chars / 4.
|
|
12
|
+
|
|
13
|
+
## 1. What one planner turn receives
|
|
14
|
+
|
|
15
|
+
| section | source | turn 1 | turn 2 |
|
|
16
|
+
| --- | --- | ---: | ---: |
|
|
17
|
+
| orchestrator prompt + worktree line | `goal.js` `AUTONOMOUS_ORCHESTRATOR_PROMPT` | 5 000 | 5 000 |
|
|
18
|
+
| PLANNING DOCTRINE (11 bullets) | `runtime.js` runDecision | 3 605 | 3 605 |
|
|
19
|
+
| Action skeletons (6 shapes + verify semantics) | `runtime.js` | 1 342 | 1 342 |
|
|
20
|
+
| Program skeleton (discovery → fan-out → verify → suite) | `runtime.js` | 1 217 | 1 217 |
|
|
21
|
+
| Graph skeleton (two chains + suite) + fanout paragraph + runtime-owned line | `runtime.js` | 4 054 | 4 054 |
|
|
22
|
+
| durable context (JSON) | `runtime.js` plannerContext | 17 474 | 162 946 |
|
|
23
|
+
| **total** | | **32 730** (~8 k tokens) | **178 452** (~45 k tokens) |
|
|
24
|
+
|
|
25
|
+
The fixed prefix is 15.2 k chars on every turn. The durable context grew
|
|
26
|
+
**9×** between turn 1 and turn 2 of the same run.
|
|
27
|
+
|
|
28
|
+
### Where the 163 k of turn 2 went
|
|
29
|
+
|
|
30
|
+
| key | chars | what it is |
|
|
31
|
+
| --- | ---: | --- |
|
|
32
|
+
| `completedActions` (19 entries) | 66 700 | every finished action **with its full attempt records**: routing candidates and pace numbers, usage, pricing table, child pid, timings — ~3.5 k per action |
|
|
33
|
+
| `outputs` (19 entries) | 41 560 | `outputExcerpt` of ~3 k chars for *every* finished action, including ones that finished ok and were already verified in the previous program |
|
|
34
|
+
| `intent` | 6 657 | the goal text (6.5 k) + cwd + policy — needed, once |
|
|
35
|
+
| `failures` | 1 073 | the two blocked actions — duplicates of `completedActions` entries |
|
|
36
|
+
| `availablePools`, `budget`, `executionConstraints`, `closedPhases`, … | ~1 800 | fine |
|
|
37
|
+
|
|
38
|
+
## 2. What is said more than once
|
|
39
|
+
|
|
40
|
+
Reading the prefix as the planner does, the same rules appear two or three
|
|
41
|
+
times in different words:
|
|
42
|
+
|
|
43
|
+
| rule | orchestrator prompt | doctrine | skeletons |
|
|
44
|
+
| --- | --- | --- | --- |
|
|
45
|
+
| propose the whole program, one round trip costs minutes | item 2 | bullets 1, 2 | "all in ONE decision" ×2 |
|
|
46
|
+
| N items → N run + N verify + one suite verify | item 5 | "Per-item chains" | Graph skeleton |
|
|
47
|
+
| unknown items → discovery + `itemsFrom` fan-out | item 5 | "Unknown item count" | Program skeleton + fanout paragraph |
|
|
48
|
+
| every verify gets a `repair` policy | item 5 | "Verification failures" | verify skeleton |
|
|
49
|
+
| `completion: all-actions-ok` on a clean program | item 5 | "Self-completing programs" | — |
|
|
50
|
+
| self-contained worker prompts, file ownership | item 3 | "File ownership", "Self-contained prompts" | — |
|
|
51
|
+
| don't propose pool/addDir/taskFile | closing line | — | final line |
|
|
52
|
+
|
|
53
|
+
Item 5 of the orchestrator prompt alone is 1 050 chars and restates four
|
|
54
|
+
doctrine bullets. The two program skeletons both end in the same
|
|
55
|
+
`verify-items → verify-suite` tail.
|
|
56
|
+
|
|
57
|
+
## 3. Does it matter? Measured
|
|
58
|
+
|
|
59
|
+
- Turn 1 (32.7 k chars) took **637 s**; turn 2 (178 k chars) took **390 s**.
|
|
60
|
+
Latency is therefore dominated by the model's reasoning on the goal, not by
|
|
61
|
+
context size — the 6.5 k-char goal and a 12-action program cost more thinking
|
|
62
|
+
than reading 45 k tokens. Trimming context is a **cost** and **attention**
|
|
63
|
+
lever, not primarily a latency lever.
|
|
64
|
+
- Cost: turn 2 read ~45 k tokens to emit a ~1 k-token decision. At Opus
|
|
65
|
+
prices that is ~$0.25 per boundary; a run with four boundaries (goal 2 on
|
|
66
|
+
0.12.1) spends more on re-reading attempt metadata than on the decisions.
|
|
67
|
+
- Attention: the planner's turn-2 reason correctly diagnosed the blocked
|
|
68
|
+
graph, so quality did not visibly suffer here — but 64 k chars of pricing
|
|
69
|
+
tables and routing candidates are noise it must skip to find the two
|
|
70
|
+
`ok:false` concerns that matter.
|
|
71
|
+
- Behaviour observed in three runs (goal 2, goal 3, dogfood): every rule the
|
|
72
|
+
prefix repeats was followed on the first turn (whole program, per-item
|
|
73
|
+
chains, repair policies, `completion`). No observed decision needed a rule
|
|
74
|
+
to be stated twice.
|
|
75
|
+
|
|
76
|
+
## 4. Recommendation
|
|
77
|
+
|
|
78
|
+
Yes — refactor, in two independent pieces, both measurable:
|
|
79
|
+
|
|
80
|
+
**A. Compact the durable context (the 9× growth).** Planner-facing ledger rows
|
|
81
|
+
instead of raw ledger entries: `{ id, type, phase, status, pool, durationSec,
|
|
82
|
+
attempts, why }` (~150 chars; 19 actions → ~3 k instead of 66.7 k). Keep a
|
|
83
|
+
full `outputExcerpt` only for actions finished **since the last decision** and
|
|
84
|
+
for every `ok:false` verify; older ok actions get a one-line summary (id, ok,
|
|
85
|
+
first 200 chars). Replace `failures` with the ids of failing actions (their
|
|
86
|
+
full entry already sits in the ledger). Expected turn-2 context: ~25 k chars
|
|
87
|
+
instead of 163 k. Pure runtime change; no planner behaviour change intended.
|
|
88
|
+
|
|
89
|
+
**B. One contract instead of three overlapping texts.** Merge
|
|
90
|
+
`AUTONOMOUS_ORCHESTRATOR_PROMPT` and the doctrine bullets into a single ordered
|
|
91
|
+
list of ~10 rules (target ≤ 4 k chars, from 8.1 k), each stated once with its
|
|
92
|
+
reason; keep exactly two JSON examples — the action shapes list and one
|
|
93
|
+
complete program (discovery → data-driven fan-out → per-item verify with
|
|
94
|
+
repair → suite verify, with `completion`) — and delete the second program
|
|
95
|
+
skeleton (target ≤ 3 k, from 6.6 k). Total prefix ≤ 7 k chars, from 15.2 k.
|
|
96
|
+
|
|
97
|
+
Acceptance for both: unit tests on the context builder (row shape, excerpt
|
|
98
|
+
policy by decision sequence) and on the prompt (each rule appears once; the
|
|
99
|
+
skeleton assertions in `tests/workflow-adaptive.test.js` updated); then one
|
|
100
|
+
re-run of goal 3 on the same fixture (baseline 0.13.1: 28 min 42 s, 1 planner
|
|
101
|
+
turn, 294 s) to confirm the decision shape is unchanged and record the new
|
|
102
|
+
per-turn size.
|
|
103
|
+
|
|
104
|
+
Not recommended: cutting the goal text or the scout excerpt from the context —
|
|
105
|
+
both were used verbatim by every first-turn program observed.
|
|
106
|
+
|
|
107
|
+
## 5. Outcome
|
|
108
|
+
|
|
109
|
+
Measurements below were taken after the refactor from the current source and
|
|
110
|
+
from the committed source saved into `/tmp/goal-before.mjs` and
|
|
111
|
+
`/tmp/runtime-before.js`, using the same temporary measurement script. The
|
|
112
|
+
canonical prefix is the complete emitted planner task text counted from its
|
|
113
|
+
first character up to (not including) the durable-context marker, with the
|
|
114
|
+
worktree-isolation suffix included. The committed baseline predates the named
|
|
115
|
+
section exports, so its emitted prefix was reconstructed from the committed
|
|
116
|
+
`runtime.js` task-text assembly and the committed `AUTONOMOUS_ORCHESTRATOR_PROMPT`.
|
|
117
|
+
|
|
118
|
+
- Complete emitted planner task text up to the durable-context marker,
|
|
119
|
+
worktree-isolation suffix included: **OBSERVED**, `16,316` characters before
|
|
120
|
+
and `5,208` characters after. Command: `node /tmp/measure-planner.mjs`.
|
|
121
|
+
- Static planner task-prefix array through the durable-context boundary:
|
|
122
|
+
**OBSERVED**, `16,209` characters before and `5,101` characters after. The
|
|
123
|
+
after value is 107 characters shorter because it excludes the unchanged
|
|
124
|
+
worktree-isolation suffix; this is a secondary source-level measurement, not
|
|
125
|
+
the canonical emitted-prefix headline. Command: `node /tmp/measure-planner.mjs`.
|
|
126
|
+
- `PLANNER_RULES_SECTION`: **OBSERVED**, `2,202` characters after. Command:
|
|
127
|
+
`node /tmp/measure-planner.mjs`.
|
|
128
|
+
- `PLANNER_EXAMPLES_SECTION`: **OBSERVED**, `1,867` characters after. Command:
|
|
129
|
+
`node /tmp/measure-planner.mjs`.
|
|
130
|
+
- `AUTONOMOUS_ORCHESTRATOR_PROMPT`: **OBSERVED**, `4,670` characters after;
|
|
131
|
+
the committed before source had no separately exported rules or examples
|
|
132
|
+
sections, so separate before-section sizes are **NOT AVAILABLE**, not
|
|
133
|
+
inferred. Command: `node /tmp/measure-planner.mjs`.
|
|
134
|
+
- Sample durable context: **OBSERVED** baseline `163,000` characters in the
|
|
135
|
+
audit's rounded turn-2 durable-context total (the detailed table records
|
|
136
|
+
`162,946`; the full turn-2 task was `178,452`), and **COMPUTED** `23,547`
|
|
137
|
+
characters after. The computed sample applies 19
|
|
138
|
+
compact ledger rows at 150 characters each, keeps a 3,000-character scout
|
|
139
|
+
excerpt and two 3,000-character failing-verify excerpts, truncates the
|
|
140
|
+
other 16 action excerpts to 200 characters, represents two failures as
|
|
141
|
+
20-character IDs, and retains the audit's 6,657-character intent and
|
|
142
|
+
1,800-character other-context components. Command: `node /tmp/compute-context.mjs`.
|
|
143
|
+
|
|
144
|
+
Deliverables:
|
|
145
|
+
|
|
146
|
+
- Durable planner context shrank because completed actions are compact ledger
|
|
147
|
+
rows, failures are IDs, and stale successful output is truncated.
|
|
148
|
+
- Planner contract shrank because overlapping prompt/doctrine/skeleton text is
|
|
149
|
+
now one ordered rules section plus exactly two JSON examples, single-sourced
|
|
150
|
+
in `src/workflow/goal.js`.
|
|
151
|
+
- Runtime prompt construction shrank because `src/workflow/runtime.js` imports
|
|
152
|
+
the shared contract instead of carrying a duplicate doctrine and graph
|
|
153
|
+
skeleton.
|
|
154
|
+
- `skill/SKILL.md` was left unchanged: it documents the general durable
|
|
155
|
+
context and the separate run-state/TUI attempt view, but does not document a
|
|
156
|
+
renamed/dropped planner-context field shape such as the old attempt records
|
|
157
|
+
or an old `failures` representation.
|
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -361,6 +361,11 @@ that expressible without extra turns:
|
|
|
361
361
|
(`source: "program-completion"`, event `decision.auto_completed`) and the run
|
|
362
362
|
ends without another planner turn; a failing action emits
|
|
363
363
|
`decision.completion_predicate_unmet` and the boundary returns to the planner.
|
|
364
|
+
- `outputSchema` on a `run` or fan-out `stepTemplate` — declare it when a
|
|
365
|
+
downstream action needs reliable structured data, such as an object to render
|
|
366
|
+
into a dependent prompt or an `items` array for `fanout.itemsFrom`; leave it
|
|
367
|
+
off for ordinary prose; planner proposals must not put it on `verify`, whose
|
|
368
|
+
verdict shape is fixed.
|
|
364
369
|
|
|
365
370
|
Every fan-out records a summary artifact as `outputs.<id>.outFile` and a
|
|
366
371
|
boolean `ok` (item count in `succeeded`), so a verify may depend on a fan-out
|
package/src/workflow/decision.js
CHANGED
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
// or dispatch any new action.
|
|
4
4
|
|
|
5
5
|
export const DECISION_SCHEMA_VERSION = 'bullswarm.workflow.decision.v1';
|
|
6
|
+
import { isValidOutputSchema } from './schema.js';
|
|
6
7
|
export const DECISIONS = new Set([
|
|
7
8
|
'proceed', 'complete', 'needs_more_work', 'retry', 'escalate',
|
|
8
9
|
'wait_for_approval', 'stop',
|
|
@@ -40,7 +41,7 @@ export function parseDecisionText(text) {
|
|
|
40
41
|
export const REVIEW_PATH_RE = /^outputs\.([A-Za-z0-9_-]+(?:\[\d+\])?)\.outFile$/;
|
|
41
42
|
// Data-driven fan-out source: the artifact of an earlier (or co-proposed)
|
|
42
43
|
// action whose output ends with a JSON array of items.
|
|
43
|
-
export const ITEMS_FROM_RE = /^outputs\.([A-Za-z0-9_-]+)(?:\.outFile)?$/;
|
|
44
|
+
export const ITEMS_FROM_RE = /^outputs\.([A-Za-z0-9_-]+)(?:\.data\.([A-Za-z0-9_-]+)|\.outFile)?$/;
|
|
44
45
|
export const REPAIR_MAX_ROUNDS = 3;
|
|
45
46
|
// Program-level completion predicates a planner may attach to a program so the
|
|
46
47
|
// runtime can record completion itself when every action finishes ok.
|
|
@@ -179,7 +180,15 @@ export function validateDecisionProposal(proposal, {
|
|
|
179
180
|
if (action.type === 'run' && typeof action.prompt !== 'string') {
|
|
180
181
|
issues.push(`${at} needs a prompt`);
|
|
181
182
|
}
|
|
182
|
-
|
|
183
|
+
if (action.outputSchema !== undefined) {
|
|
184
|
+
if (action.type !== 'run') issues.push(`${at}.outputSchema is only allowed on run actions`);
|
|
185
|
+
else {
|
|
186
|
+
const schema = isValidOutputSchema(action.outputSchema);
|
|
187
|
+
if (!schema.ok) issues.push(...schema.issues.map((issue) => `${at}.outputSchema: ${issue}`));
|
|
188
|
+
else if (action.outputSchema.type !== 'object') issues.push(`${at}.outputSchema.type must be "object"`);
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
if (action.type === 'fanout') {
|
|
183
192
|
const hasItems = Array.isArray(action.items);
|
|
184
193
|
const hasItemsFrom = action.itemsFrom != null;
|
|
185
194
|
if (!hasItems && !hasItemsFrom) {
|
|
@@ -198,10 +207,15 @@ export function validateDecisionProposal(proposal, {
|
|
|
198
207
|
}
|
|
199
208
|
}
|
|
200
209
|
if (!action.stepTemplate || typeof action.stepTemplate !== 'object') issues.push(`${at}.stepTemplate is required`);
|
|
201
|
-
|
|
210
|
+
for (const runtimeOwned of ['pool', 'addDir', 'taskFile']) {
|
|
202
211
|
if (action.stepTemplate?.[runtimeOwned] != null) {
|
|
203
212
|
issues.push(`${at}.stepTemplate.${runtimeOwned} is runtime-owned and cannot be proposed by a planner`);
|
|
204
|
-
|
|
213
|
+
}
|
|
214
|
+
if (action.stepTemplate?.outputSchema !== undefined) {
|
|
215
|
+
const schema = isValidOutputSchema(action.stepTemplate.outputSchema);
|
|
216
|
+
if (!schema.ok) issues.push(...schema.issues.map((issue) => `${at}.stepTemplate.outputSchema: ${issue}`));
|
|
217
|
+
else if (action.stepTemplate.outputSchema.type !== 'object') issues.push(`${at}.stepTemplate.outputSchema.type must be "object"`);
|
|
218
|
+
}
|
|
205
219
|
}
|
|
206
220
|
}
|
|
207
221
|
if (action.repair != null && action.type !== 'verify') {
|