bullswarm 0.36.0 → 0.37.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/AGENTS.md +20 -6
  2. package/CHANGELOG.md +234 -0
  3. package/README.md +3 -3
  4. package/data/openrouter-benchmarks.json +4803 -4740
  5. package/docs/design/redesign-mechanics-principles-options.md +1 -1
  6. package/docs/design/tidy-0.35.1/frames/real-stats-model-120.txt +1 -1
  7. package/docs/design/tidy-0.35.1/frames/real-stats-model-200.txt +1 -1
  8. package/docs/design/tidy-0.35.1/frames/real-stats-model-55.txt +1 -1
  9. package/docs/design/tidy-0.35.1/frames/real-stats-spending-120.txt +1 -1
  10. package/docs/design/tidy-0.35.1/frames/real-stats-spending-200.txt +1 -1
  11. package/docs/design/tidy-0.35.1/frames/real-stats-spending-55.txt +1 -1
  12. package/docs/guide/concepts.md +4 -2
  13. package/docs/guide/getting-started.md +19 -10
  14. package/docs/guide/observing.md +100 -21
  15. package/docs/guide/routing.md +4 -4
  16. package/docs/guide/run.md +13 -7
  17. package/docs/guide/workflows.md +158 -9
  18. package/docs/integrations/claude-code.md +1 -1
  19. package/docs/integrations/issue-watcher.md +2 -2
  20. package/docs/reference/cli.md +15 -6
  21. package/docs/reference/configuration.md +3 -6
  22. package/docs/reference/program.md +128 -8
  23. package/docs/reference/providers.md +1 -1
  24. package/docs/reference/result.md +23 -21
  25. package/mcp/server.mjs +1 -1
  26. package/mods/bullswarm/README.md +3 -3
  27. package/mods/bullswarm/hooks/hooks.json +1 -1
  28. package/mods/bullswarm/hooks/pane.tsx +1 -1
  29. package/mods/bullswarm/hooks/pool-rows.tsx +2 -4
  30. package/mods/bullswarm/hooks/pools.ts +3 -9
  31. package/mods/bullswarm/hooks/register.ts +22 -11
  32. package/mods/bullswarm/hooks/route.ts +15 -7
  33. package/mods/bullswarm/hooks/runs.ts +19 -1
  34. package/mods/bullswarm/hooks/step.ts +4 -0
  35. package/mods/bullswarm/hooks/strip.tsx +5 -3
  36. package/mods/bullswarm/hooks/verdict.ts +128 -8
  37. package/mods/bullswarm/types/index.d.ts +13 -4
  38. package/package.json +1 -1
  39. package/skill/SKILL.md +342 -450
  40. package/skill/references/operations.md +130 -31
  41. package/skill/references/patterns.md +313 -0
  42. package/skill/references/program.md +207 -436
  43. package/src/cli.js +43 -502
  44. package/src/help.js +192 -61
  45. package/src/lib/attempt-stream.js +33 -1
  46. package/src/lib/attempt-usage.js +235 -0
  47. package/src/lib/bounded-capture.js +59 -0
  48. package/src/lib/cli-flags.js +7 -3
  49. package/src/lib/clone.js +2 -0
  50. package/src/lib/config.js +0 -3
  51. package/src/lib/provider-errors.js +182 -0
  52. package/src/lib/retention.js +1 -1
  53. package/src/lib/route.js +28 -123
  54. package/src/lib/run-delegate.js +340 -0
  55. package/src/lib/run-step.js +163 -0
  56. package/src/lib/stale.js +2 -9
  57. package/src/lib/state.js +3 -3
  58. package/src/lib/strategy.js +1 -1
  59. package/src/lib/tasks.js +9 -5
  60. package/src/lib/watch.js +43 -1053
  61. package/src/lib/worker-argv.js +98 -0
  62. package/src/lib/worker-report.js +129 -0
  63. package/src/meters/framework.js +0 -4
  64. package/src/meters/registry.js +0 -16
  65. package/src/provider-cli.js +2 -1
  66. package/src/providers/echo/echo-worker.mjs +20 -2
  67. package/src/strategy-cli.js +0 -2
  68. package/src/workflow/answers.js +279 -0
  69. package/src/workflow/attempt-bytes.js +57 -0
  70. package/src/workflow/attempt-record.js +137 -0
  71. package/src/workflow/budget-model.js +46 -157
  72. package/src/workflow/caller-planner.js +167 -0
  73. package/src/workflow/cli-capabilities.js +137 -0
  74. package/src/workflow/cli-goal-document.js +95 -0
  75. package/src/workflow/cli-goal.js +251 -0
  76. package/src/workflow/cli-inspect.js +254 -0
  77. package/src/workflow/cli-launch.js +271 -0
  78. package/src/workflow/cli-plan.js +556 -0
  79. package/src/workflow/cli-pool-checks.js +74 -0
  80. package/src/workflow/cli-program-checks.js +125 -0
  81. package/src/workflow/cli-run-lookup.js +46 -0
  82. package/src/workflow/cli-run-verbs.js +258 -0
  83. package/src/workflow/cli-step-verbs.js +499 -0
  84. package/src/workflow/cli-steps.js +573 -0
  85. package/src/workflow/cli.js +20 -2440
  86. package/src/workflow/contract-v3.js +129 -0
  87. package/src/workflow/dash-kit.js +4 -38
  88. package/src/workflow/dashboard.js +5 -177
  89. package/src/workflow/day-key.js +51 -0
  90. package/src/workflow/earlier-work.js +56 -0
  91. package/src/workflow/evidence-runner.js +1 -1
  92. package/src/workflow/fleet-view.js +1 -5
  93. package/src/workflow/folder-walk.js +89 -0
  94. package/src/workflow/gates-loops.js +752 -0
  95. package/src/workflow/goal-column.js +60 -0
  96. package/src/workflow/history-view.js +23 -21
  97. package/src/workflow/history.js +35 -72
  98. package/src/workflow/home-model.js +80 -198
  99. package/src/workflow/home-view.js +32 -9
  100. package/src/workflow/kernel-resume.js +129 -0
  101. package/src/workflow/ledger.js +2 -4
  102. package/src/workflow/metrics-legacy.js +127 -0
  103. package/src/workflow/metrics.js +798 -0
  104. package/src/workflow/needs-you.js +18 -5
  105. package/src/workflow/no-pool-why.js +60 -0
  106. package/src/workflow/ownership.js +1 -1
  107. package/src/workflow/pick-preview.js +119 -0
  108. package/src/workflow/program-v3.js +522 -0
  109. package/src/workflow/reprice.js +1 -4
  110. package/src/workflow/retry-handoff.js +108 -0
  111. package/src/workflow/revision-v3.js +168 -0
  112. package/src/workflow/rollup.js +25 -329
  113. package/src/workflow/run-control.js +329 -0
  114. package/src/workflow/run-counts.js +27 -0
  115. package/src/workflow/run-features.js +22 -3
  116. package/src/workflow/run-model.js +67 -321
  117. package/src/workflow/run-one-step.js +65 -0
  118. package/src/workflow/run-verdict.js +184 -0
  119. package/src/workflow/run-view.js +65 -28
  120. package/src/workflow/runs-cli.js +26 -6
  121. package/src/workflow/runs-view.js +4 -102
  122. package/src/workflow/stats-model.js +204 -567
  123. package/src/workflow/stats-view.js +2 -1
  124. package/src/workflow/status.js +3 -0
  125. package/src/workflow/step-change-hint.js +22 -0
  126. package/src/workflow/step-model.js +27 -84
  127. package/src/workflow/step-prompts.js +255 -0
  128. package/src/workflow/step-route.js +38 -10
  129. package/src/workflow/step-view.js +7 -0
  130. package/src/workflow/step-vocabulary.js +6 -1
  131. package/src/workflow/time-box.js +39 -3
  132. package/src/workflow/usage-ledger.js +254 -0
  133. package/src/workflow/usage-preference.js +2 -2
  134. package/src/workflow/v2-dispatch.js +140 -43
  135. package/src/workflow/v2-outcome.js +141 -38
  136. package/src/workflow/v2-planner.js +28 -5
  137. package/src/workflow/v2-presentation.js +14 -2
  138. package/src/workflow/v2-revision.js +48 -6
  139. package/src/workflow/v2-runtime.js +73 -1499
  140. package/src/workflow/v2-scheduler.js +11 -5
  141. package/src/workflow/v2-state.js +155 -78
  142. package/src/workflow/v3-display.js +196 -0
  143. package/src/workflow/v3-phases.js +122 -0
  144. package/src/workflow/v3-timeline.js +84 -0
  145. package/src/workflow/verify-rounds.js +29 -7
  146. package/src/workflow/watch-cli.js +73 -12
  147. package/src/workflow/workflow-flags.js +92 -0
package/AGENTS.md CHANGED
@@ -30,6 +30,15 @@ step, the dispatched planner or the preflight scout alike; the
30
30
  needs-you block with `step rerun --avoid` and `step accept`; the per-step
31
31
  `route`; `verifyRounds` counts fixes (default 1); reviews are placed only by
32
32
  route. Runs started earlier keep their rules (`features.json`).
33
+ 0.37.0 (program v3, the generic model) has landed: a step is a run, and a
34
+ workflow composes steps, phases, gates and loops. `bullswarm run` is a
35
+ one-step workflow. New programs are `bullswarm.workflow.program.v3`: a step
36
+ passes by facts (clean exit, deliverable produced, evidence passed, answer
37
+ matching its schema), a gate waits for `workflow continue`, a loop reruns its
38
+ steps until one condition holds (at most 5 rounds), and new work is added
39
+ with `workflow add`, never by editing the run's steps. v3 reports facts per
40
+ step and has no requirement IDs, `evidenceFor` or `verifyRounds`. v2 programs
41
+ and saved runs keep running and replaying as before.
33
42
 
34
43
  ## Non-negotiable doctrine
35
44
 
@@ -82,12 +91,17 @@ route. Runs started earlier keep their rules (`features.json`).
82
91
  7. New goal workflows are caller-planned programs in a shared workspace.
83
92
  `bullswarm workflow goal --program` executes the graph; `--orchestrator`
84
93
  explicitly delegates planning. File territories are advisory scheduling
85
- hints. A failed check gets one fix step and one re-review
86
- (`defaults.verifyRounds`, default 1, 0-3), then the caller, who takes over
87
- through the watcher's needs-you block: rerun elsewhere, change the step,
88
- take over, or accept anyway. `verified` separately records requirement
89
- evidence. `--isolation` opts into strict per-worker worktrees. Saved V2
90
- runs preserve their original semantics (`features.json`).
94
+ hints. In a v3 program (the format for new work) a check is an ordinary
95
+ step, and fixing until it passes is a loop the caller declares (`loops`,
96
+ `until` one condition, `maxRounds` 1-5); a loop out of rounds or a gate
97
+ waits for the caller (`workflow continue`), and a failed step goes to the
98
+ caller through the watcher's needs-you block: rerun elsewhere, add steps
99
+ (`workflow add`), take over, or accept anyway. v3 validate refuses
100
+ `defaults.verifyRounds`. In a v2 program a failed check gets one fix step
101
+ and one re-review (`defaults.verifyRounds`, default 1, 0-3), then the
102
+ caller, and `verified` separately records requirement evidence.
103
+ `--isolation` opts into strict per-worker worktrees. Saved V2 runs
104
+ preserve their original semantics (`features.json`).
91
105
  8. Historical authored-graph runs remain visible as read-only `legacy` rows.
92
106
  Their executor was removed in 0.27.0; driving commands fail closed before
93
107
  dispatch and historical run directories remain untouched.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,240 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ ## 0.37.0 — One model: steps, phases, gates and loops
6
+
7
+ - 0.37.0 in one line: a step is a run, and a workflow composes steps, phases,
8
+ gates and loops. `bullswarm run` is a one-step workflow; `bullswarm workflow
9
+ goal --program` runs a program you write. New programs are
10
+ `bullswarm.workflow.program.v3`; v2 programs and saved runs still run and
11
+ replay as before.
12
+ - workflow: program v3 has four blocks. `steps` (the work: `id`, `prompt`,
13
+ `dependsOn`, and optional `phase`, `label`, `lane`, `effort`, `reasoning`,
14
+ `route`, `answer`, `evidence`, `deliverable`, `files`, `retry` 0 or 1,
15
+ `timeBox`), `phase` (a label that groups steps and changes nothing else),
16
+ `gates` (`{id, dependsOn, when?, note?}`: the steps behind it wait for
17
+ `bullswarm workflow continue <run> <gate>`, or only when its condition
18
+ holds) and `loops` (`{id, steps, until, maxRounds}` 1-5: the steps run again
19
+ until the condition holds). There is one condition form:
20
+ `{step, field, equals?}` on a required boolean of the step's answer, or
21
+ `{step, evidence: "passed"}`. A check is an ordinary step with an `answer`
22
+ schema and/or `evidence`; roles, kinds, requirement IDs, `evidenceFor` and
23
+ `verifyRounds` belong to v2 programs and are refused in v3.
24
+ - workflow: `answer` asks a step to write JSON to a file Bullswarm names; the
25
+ file (at most 256 KiB), not the reply, is checked against the schema (a
26
+ mismatch is failure kind `schema`), handed to the steps that depend on it,
27
+ read by conditions and printed by `watch`, `wait` and `runs result`. The
28
+ text of `runs result` on a v3 run prints `# answer <step> <json>` for each
29
+ step that declares one (and no `# gaps` line), and `runs result --json
30
+ --summary` carries it as the step's `answer` (`answerBytes` in its place when
31
+ the answer's JSON is over 1 KiB; the full result holds it). A finished step
32
+ whose answer passed its schema reads `answer checked`, not proven: a
33
+ well-formed answer is the worker's claim, and only a command or schema check
34
+ Bullswarm runs itself proves a step (the proof line counts such steps apart,
35
+ `N answer checked: <steps>`).
36
+ - workflow: new verbs for v3 runs. `bullswarm workflow add <run> --steps
37
+ part.json` (or `--from-answer <step>`) appends steps, gates and loops without
38
+ changing anything the run has, and reopens a finished run. `bullswarm
39
+ workflow wait <run> <id...>` reads until the named steps, gates or loops
40
+ settle and prints their facts and answers (exit 0, 1 when one failed or the
41
+ run stopped short, 2 on a timeout or unknown id). `bullswarm workflow
42
+ continue <run> <gate|loop> [--rounds <1-5>]` passes a waiting gate or gives
43
+ a loop more rounds. A run with only waiting gates or loops left is parked
44
+ with status `waiting`; `watch --until trouble` wakes on it and prints the
45
+ `continue` command.
46
+ - workflow: a v3 run's steps are never edited. `plan revise` on a v3 run
47
+ accepts only reruns and otherwise refuses with "a v3 run's steps cannot be
48
+ added, changed or removed with plan revise in this build; add steps, gates
49
+ or loops with `bullswarm workflow add` …". The needs-you block and the
50
+ refusals on a v3 run offer `add steps` / `then wait` (`bullswarm workflow add
51
+ <run> --steps part.json`, then `bullswarm workflow wait <run> <added ids>`)
52
+ where a v2 run offers `change the step`.
53
+ - workflow: a completed v3 run hands nothing back and prints no `your call:`
54
+ (a v3 run is never "verified": every step succeeding is what it reports, and
55
+ its `runs result --json --summary` carries no `verified` field). A
56
+ partial v3 run's `restart` line names the run's folder: `start a new run:
57
+ bullswarm workflow goal "<goal>" --cwd <run folder> --program <file.json>`.
58
+ - workflow: `bullswarm workflow plan contract` prints the v3 contract
59
+ (`bullswarm.workflow.contract.v3`: fields, rules and an example that
60
+ validates); `--v2` prints the contract of old programs. `plan validate`'s
61
+ launch line names the absolute path of the program file you gave it.
62
+ - workflow: a v3 build or chore step in a folder that is not a git repository
63
+ is now checked for a change: Bullswarm lists the folder before and after the
64
+ worker (up to 5,000 files and 64 MiB, `.git` and `node_modules` skipped), and
65
+ a step that changed nothing fails `not-produced`. In a larger folder the
66
+ change stays unchecked, as before; name the step's `files` there.
67
+ - workflow: `bullswarm workflow goal --program` launches print `watch
68
+ --until trouble` for a v3 run, since a gate or loop can stop it for you.
69
+ - run: `bullswarm run` is a one-step workflow. Its flags become a one-step
70
+ program that the workflow kernel runs in the foreground, with no scout and no
71
+ planner, and the run is recorded under `workflows/<id>/` with its rollup, so
72
+ `bullswarm workflow runs --all` and `workflow runs show <shortId>` list it
73
+ with the other runs. It no longer writes a separate task entry to the
74
+ decision log: the step's attempt writes the one entry, as every workflow
75
+ step does. The verdict keeps `ok`, `why`, `failureKind`, `retryAfter`,
76
+ `pick` and `outFile`, and adds `runId`, `shortId`, `answer`, `answerCheck`
77
+ and `attempts`; `why` is the run's own fact line (`all 1 step succeeded`, or
78
+ the step's failure). The `--json` verdict is compact (about 50 lines): the
79
+ last attempt's full usage (`meta.usage`, often 200 lines of cost detail) is
80
+ replaced by `usage`, a short summary of the whole run (attempts, minutes,
81
+ tokens, money with its basis), and the last field, `details`, names
82
+ `bullswarm workflow runs result <shortId> --json`, where per-attempt cost
83
+ lives. A caller that reads only the tail still finds its run. The text
84
+ verdict ends with `run: <shortId> · details: bullswarm workflow runs result
85
+ <shortId>`.
86
+ - run: a run gets the step's one automatic retry (a crash on another pool
87
+ that can run it, a failed answer check on the same pool with the errors
88
+ attached). `--no-retry` gives one attempt, as before. A usage limit is
89
+ still never retried.
90
+ - run: a run passes by facts, as every workflow step does: the exit, its
91
+ deliverable and, with the new `--answer-schema <file>`, a JSON answer the
92
+ worker writes to a file Bullswarm names and Bullswarm checks against the
93
+ schema (a mismatch is failure kind `schema`). A `build` or `chore` run must
94
+ change a file, else it fails `not-produced`; ask a question on
95
+ `--lane analyze`. A file git ignores counts when its bytes change (a
96
+ deliverable in an `out/` that `.gitignore` lists), except tool caches and
97
+ build output (`dist`, `build`, `coverage`, `target`, `*.log`,
98
+ `*.tsbuildinfo`), and no worker prompt names a commit as a way to produce,
99
+ so a retry is never steered into committing.
100
+ - workflow: `workflow runs` lists a goal that starts with a folder ("Work in
101
+ /path/proj. Read every ticket…") by the words after the folder, or by the
102
+ next line when the folder is alone on its line, so runs with the same
103
+ lead-in can be told apart. A bare leading file path stays, shown by its
104
+ name ("parser.js: fix …").
105
+ - workflow: a `## Not done` line that says `none`, `nothing` or `n/a` and
106
+ then gives a reason or a scope (`- none for this build step. The review
107
+ rounds belong to other actions…`, `- None — all done.`, `- nothing left`)
108
+ is not an unfinished item, so the step no longer reads `returned early · 1
109
+ not done`. A line that goes on to name work (`- none of the tests pass`,
110
+ `- None, except the README update.`) still counts.
111
+ - run: new routing flags `--avoid-pool`, `--use-provider` and
112
+ `--avoid-provider` (comma-separated), the same filters a workflow step's
113
+ `route` takes. `--dry-run` prints the kernel's own first pick for the step.
114
+ - run: keep-on-caller and incumbency are gone. The calling agent is never a
115
+ pool of its own run, so `keepOnClaude` is no longer in the verdict;
116
+ `--no-caller` is accepted with a one-line notice for this release and
117
+ will be removed. A run at
118
+ the recursion depth limit is a plain refusal: `ok: false`, failure kind
119
+ `depth`. Every step is routed by one rule, so nothing reads or writes
120
+ `incumbents` in `state.json` or `config.callerName` (a new home no longer
121
+ writes it), and the fleet view and the Claude Code mod no longer show
122
+ "incumbent for".
123
+ - run: `--timeout <seconds>` kills the worker as before and the verdict reads
124
+ `timeout after <N>s` with failure kind `interrupted` (a process failure), so
125
+ it gets the one automatic retry on another pool; `--no-retry` gives one
126
+ attempt. A new failure kind was not added: `interrupted` already has the
127
+ right retry, and one more kind would change every failure table for one flag.
128
+ - mod: routed Claude Code subagents run on lane `analyze` with one attempt: a
129
+ build run must change a file, so a subagent that only answers would fail
130
+ `not-produced`.
131
+ - dashboard: a v3 run reads by its blocks. The Run page groups its steps by
132
+ their `phase` (plan boxes and timeline rules carry the phase name), draws each
133
+ gate and loop as a row of its own (`loop polish · passed in round 2 of 3 ·
134
+ check's evidence passed`, `gate approve · waiting for you · <note>` with its
135
+ `continue` command), tags each loop step's attempt with its round, and prints
136
+ each attempt's checked answer under it. A run parked at a gate or loop reads
137
+ `waiting at gate approve` with `next: bullswarm workflow continue <run>
138
+ <gate>` on Run, Runs and Home. A one-step run shows no plan boxes and no
139
+ phase frame. A v3 run or step never reads `verified` or `not verified`; the
140
+ Step page leads its result with the answer. `bullswarm workflow runs --json`
141
+ lists a parked run's `waitingFor`.
142
+ - run: the verdict carries `notDone`, `{count, items}` (the first five
143
+ items, each at most 160 characters), when the worker's report listed
144
+ unfinished items under `## Not done`, and `null` otherwise; the text verdict
145
+ prints `worker left 2 items not done: <item>; <item>`. `ok` and `why` are
146
+ unchanged: a step that returned early still succeeded by its facts, and the
147
+ caller now sees what it left. The mod's verdict note and routed-subagent
148
+ notice say the same.
149
+ - mod: the verdict note of `bullswarm run --answer-schema` carries the checked
150
+ answer (or the failed check); a launched workflow is pointed at `workflow
151
+ watch <run> --until trouble` instead of `--next`; the strip and prompt
152
+ context list a parked run as `waiting at gate <id>` with its continue
153
+ command; a `workflow watch` or `workflow wait` result gets a note naming a
154
+ waiting gate or loop, the loop's round and the checked answers.
155
+ - stats: the totals line and Home's figures name single runs apart from
156
+ workflows (the totals line reads `3 runs · 5 workflows`, Home reads
157
+ `Runs: 3 · workflows 5 · verified 2 (40%)`). Home's verified count and share
158
+ are of the v2 workflows alone, and Home leaves them out when the period has
159
+ no v2 workflow (`Runs: 3 · workflows 2`). The overview carries `oneStepRuns`,
160
+ `verifiableRuns` and `verifiedRuns`, and a v3 rollup carries
161
+ `programFormat: 3` and, for a one-step run, `oneStep: true`.
162
+ - stats: single runs count. Stats, Budget and History read the single runs the
163
+ decision log still holds (older ones) and the one-step workflows new runs
164
+ record, so totals from this version on include `bullswarm run` work that
165
+ earlier versions left out.
166
+ - stats: one definition of each number on every page. A run belongs to the
167
+ day it finished (History counted it on its start day). A money total exists
168
+ only when every attempt in the scope was priced, and the priced subtotal
169
+ always travels beside it; Budget's `apiEquivalentUsd` is the pool's whole
170
+ amount for the period instead of a sum of per-pool amounts, and History's
171
+ day spend is the priced subtotal, equal to the Stats trend for that day.
172
+ Worker minutes and usage come from one per-attempt record, so Stats numbers
173
+ move only by rounding (up to 0.01 worker-minutes).
174
+ - fixed (candidate QA): `workflow add` with a step whose `phase` names a
175
+ phase the run already has failed `presentation is missing program action
176
+ <step>`; the step now joins that phase.
177
+ - fixed (candidate QA): a step whose route left no pool said `every pool that
178
+ could run it shares a provider` when a pool of another provider could run
179
+ it one tier up. The reason now names the step's lane and tier and each other
180
+ provider's pool with what ruled it out (`claude-code (claude-code): no model
181
+ on the low tier for analyze work (has high)`), and the step records
182
+ `routeWhy` and `routeCandidates` (`{pool, provider, excluded}`) in its
183
+ failure and in `result.json`, which were null before.
184
+ - fixed (candidate QA): the result summary named at most four steps in
185
+ `answerCheckedSteps` beside a larger `answerChecked` count; it now names
186
+ every one (and every `acceptedSteps` entry). A v3 step whose route left no
187
+ pool reads `retryable: false` with `noPool: true` in `handback.unfinished`,
188
+ since resume fails it the same way until a pool passes its route at its
189
+ tier.
190
+ - fixed (candidate QA): `bullswarm workflow plan contract` with no goal
191
+ printed a usage error; the v3 contract now prints with `goal: null` and
192
+ `'<goal>'` in its commands. `--v2` still needs the goal.
193
+ - fixed (candidate QA): a loop you continued after it ran out of rounds read
194
+ as passed (`✓ loop polish passed`). It now reads `→ loop polish continued by
195
+ the caller after 3 of 3 rounds (condition not met)` in `workflow continue`,
196
+ watch, wait and the Run page; `result.json` records `loops: [{id, outcome,
197
+ rounds, maxRounds}]` (`outcome` is `passed`, `continued-unmet`,
198
+ `out-of-rounds`, `blocked` or `pending`), names a continued loop at the end
199
+ of `reason`, and the summary and proof line carry it.
200
+ - fixed (candidate QA): `watch --until trouble|outcome` printed nothing until
201
+ a wake; it now prints one start line, `watching <run> until trouble · <n>
202
+ steps`.
203
+ - fixed (candidate QA): in a run under the failure rule, an attempt that
204
+ failed and was retried is recorded as `failed` (its `attempt.finished`
205
+ event says `willRetry: true`), not `interrupted`. Earlier runs keep their
206
+ label.
207
+ - fixed (candidate QA): a loop writer that changed no file in round 2 or later
208
+ passed, carrying round 1's work; a loop round is new work now, so it fails
209
+ `not-produced` like round 1. `workflow wait` said `deliverable files
210
+ produced` for a step whose attempt changed nothing and carried earlier work;
211
+ `wait` and `watch` now say `carried from an earlier attempt`.
212
+ - workflow: `watch --timeout <seconds>` exits 0 after that long with no wake,
213
+ printing `⧖ watch timed out after <s>s · the run is still <status>` and a
214
+ `next:` line with `--after`, `--since` and `--timeout`, so a foreground watch
215
+ ends before the caller's tool call is killed and loses no wake.
216
+ - skill: the run-or-workflow choice comes first. One worker that can hold the
217
+ whole input and make one deliverable (a 40-ticket triage, a research brief,
218
+ one feature) is one `bullswarm run`, with `--answer-schema` for a checkable
219
+ answer; do not split an input into chunks unless one worker cannot hold it.
220
+ A loop's critique asks only for what the sources can show, rounds are capped
221
+ at 2 unless a round is cheap, and a watch runs in the foreground when the
222
+ caller's harness cannot wake it when a background process ends. Pattern 5
223
+ (triage) is one run, or one classify step with a gate for uncertain
224
+ tickets.
225
+ - removed: keep-on-caller (`keepOnClaude` in the verdict), incumbency (the
226
+ `incumbents` state and the "incumbent for" lines), and the rule that a run
227
+ never waits for its caller: a v3 run waits at the gates and loops you
228
+ declare, and nowhere else.
229
+ - upgrade: nothing to migrate. v2 programs (`bullswarm.workflow.program.v2`)
230
+ still validate and run, `plan contract --v2` prints their contract, and runs
231
+ saved by earlier versions keep their rules when resumed or viewed. Write new
232
+ programs as v3 (`skill/references/patterns.md` has five that validate as
233
+ printed). Scripts that parsed `keepOnClaude` from a `run` verdict, or passed
234
+ `--no-caller`, should drop them; scripts that read a `run` verdict gain
235
+ `runId`, `shortId`, `answer`, `answerCheck` and `attempts`. A `build` or
236
+ `chore` run that changes nothing now fails `not-produced`, in or outside a
237
+ git repository: ask questions on `--lane analyze`.
238
+
5
239
  ## 0.36.0 — Usage limits go back to the caller; each account bills itself
6
240
 
7
241
  - known issues in this release: a limit marker whose reset was guessed still
package/README.md CHANGED
@@ -36,11 +36,11 @@ flowchart LR
36
36
 
37
37
  - **Spends quota by pace.** Each task goes to the plan with the most spare quota for how far its window has run, so a plan that is behind, or about to reset with quota left, gets used first.
38
38
  - **Picks models for you.** Tasks come in three lanes (`analyze`, `build`, `chore`). Setup asks each CLI which models it offers and suggests the newest one for each effort level, so you don't have to update settings every time a vendor ships a model.
39
- - **Runs one task or a whole workflow.** `bullswarm run` sends one task to one agent and returns a verdict. For bigger goals your main agent writes a plan; Bullswarm runs the independent steps in parallel across agents, then integration, and a review by a different agent when the plan asks for one.
40
- - **Checks the work, not the exit code.** A delegate saying "done" isn't enough. Bullswarm reads what it actually produced, and a workflow finishing is kept separate from its requirements being verified.
39
+ - **Runs one task or a whole workflow.** `bullswarm run` is a one-step workflow: one task to one agent, one automatic retry, and an optional answer checked against a JSON schema. For bigger goals your main agent writes a program from four blocks: steps (each with an optional checked answer), phases that group them, gates where the run waits for you, and loops that repeat steps until a condition holds, such as "until the tests pass". Independent steps run in parallel across agents, and a step can be routed away from the agents that did the work it checks.
40
+ - **Checks the work, not the exit code.** A delegate saying "done" isn't enough. Bullswarm reads what it actually produced, runs the checks a step declares (a command, a schema), checks each answer itself, and labels every finished step by what backs it: `proven by command` (a check Bullswarm ran passed), `answer checked` (the answer is well formed, which is a claim, not proof), or `unproven`. A run reports these facts, not a verdict; `verified` requirements belong to old v2 programs only.
41
41
  - **Shows everything live.** A terminal dashboard covers quota, history, running workflows and each agent's individual turns.
42
42
  - **Works with the agent you already use.** The `/bullswarm` skill teaches Claude Code, Codex and Grok when to delegate. There's also an early-access Claude Code Mod that shows runs and usage inside Claude Code.
43
- - **Lets you steer mid-run.** Add, change, remove or rerun steps while a workflow is running, or pause and resume it.
43
+ - **Lets you steer mid-run.** Add steps, gates or loops while a workflow runs (`workflow add`, also straight from a step's answer), continue a waiting gate or give a loop more rounds, rerun a step, or pause and resume it.
44
44
  - **Extends with providers.** Claude Code, Codex and Grok are built in. Other CLIs can be added as providers without touching the core.
45
45
 
46
46
  ## See it