bullswarm 0.35.6 → 0.37.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/AGENTS.md +84 -15
  2. package/CHANGELOG.md +544 -1
  3. package/README.md +4 -4
  4. package/bin/check-schema.js +78 -0
  5. package/data/openrouter-benchmarks.json +8356 -8079
  6. package/docs/design/patterns/README.md +66 -0
  7. package/docs/design/patterns/feature-slices.md +58 -0
  8. package/docs/design/patterns/many-angles.md +57 -0
  9. package/docs/design/patterns/triage-at-scale.md +57 -0
  10. package/docs/design/redesign-mechanics-principles-options.md +418 -248
  11. package/docs/design/tidy-0.35.1/frames/colour/real-run-finished-200.txt +6 -6
  12. package/docs/design/tidy-0.35.1/frames/colour/real-run-finished-55.txt +6 -6
  13. package/docs/design/tidy-0.35.1/frames/colour/real-run-running-200.txt +5 -5
  14. package/docs/design/tidy-0.35.1/frames/colour/real-run-running-55.txt +5 -5
  15. package/docs/design/tidy-0.35.1/frames/real-run-finished-120.txt +5 -5
  16. package/docs/design/tidy-0.35.1/frames/real-run-finished-200.txt +6 -6
  17. package/docs/design/tidy-0.35.1/frames/real-run-finished-55.txt +6 -6
  18. package/docs/design/tidy-0.35.1/frames/real-run-running-120.txt +4 -4
  19. package/docs/design/tidy-0.35.1/frames/real-run-running-200.txt +5 -5
  20. package/docs/design/tidy-0.35.1/frames/real-run-running-55.txt +5 -5
  21. package/docs/design/tidy-0.35.1/frames/real-stats-model-120.txt +1 -1
  22. package/docs/design/tidy-0.35.1/frames/real-stats-model-200.txt +1 -1
  23. package/docs/design/tidy-0.35.1/frames/real-stats-model-55.txt +1 -1
  24. package/docs/design/tidy-0.35.1/frames/real-stats-spending-120.txt +1 -1
  25. package/docs/design/tidy-0.35.1/frames/real-stats-spending-200.txt +1 -1
  26. package/docs/design/tidy-0.35.1/frames/real-stats-spending-55.txt +1 -1
  27. package/docs/guide/concepts.md +28 -6
  28. package/docs/guide/getting-started.md +19 -10
  29. package/docs/guide/index.md +2 -2
  30. package/docs/guide/observing.md +124 -32
  31. package/docs/guide/playbook.md +1 -1
  32. package/docs/guide/routing.md +120 -60
  33. package/docs/guide/run.md +15 -9
  34. package/docs/guide/workflows.md +396 -45
  35. package/docs/integrations/claude-code.md +2 -2
  36. package/docs/integrations/issue-watcher.md +2 -2
  37. package/docs/reference/cli.md +101 -111
  38. package/docs/reference/configuration.md +5 -8
  39. package/docs/reference/program.md +341 -50
  40. package/docs/reference/providers.md +5 -5
  41. package/docs/reference/result.md +70 -50
  42. package/mcp/server.mjs +3 -3
  43. package/mods/bullswarm/README.md +3 -3
  44. package/mods/bullswarm/hooks/hooks.json +1 -1
  45. package/mods/bullswarm/hooks/pane.tsx +1 -1
  46. package/mods/bullswarm/hooks/pool-rows.tsx +2 -4
  47. package/mods/bullswarm/hooks/pools.ts +3 -9
  48. package/mods/bullswarm/hooks/register.ts +22 -11
  49. package/mods/bullswarm/hooks/route.ts +15 -7
  50. package/mods/bullswarm/hooks/runs.ts +19 -1
  51. package/mods/bullswarm/hooks/step.ts +4 -0
  52. package/mods/bullswarm/hooks/strip.tsx +5 -3
  53. package/mods/bullswarm/hooks/verdict.ts +128 -8
  54. package/mods/bullswarm/types/index.d.ts +13 -4
  55. package/package.json +1 -1
  56. package/providers/contrib/command-code/connector-history.json +11 -2
  57. package/providers/contrib/command-code/connector.json +7 -3
  58. package/providers/contrib/opencode/connector-history.json +5 -1
  59. package/providers/contrib/opencode/connector.json +2 -2
  60. package/skill/SKILL.md +361 -271
  61. package/skill/references/operations.md +414 -99
  62. package/skill/references/patterns.md +313 -0
  63. package/skill/references/program.md +221 -143
  64. package/src/cli.js +62 -694
  65. package/src/help.js +337 -141
  66. package/src/lib/attempt-stream.js +33 -1
  67. package/src/lib/attempt-usage.js +235 -0
  68. package/src/lib/auth-signatures.js +3 -3
  69. package/src/lib/bounded-capture.js +59 -0
  70. package/src/lib/cli-flags.js +9 -5
  71. package/src/lib/clone.js +2 -0
  72. package/src/lib/config.js +19 -36
  73. package/src/lib/pool-labels.js +0 -15
  74. package/src/lib/probe.js +1 -1
  75. package/src/lib/provider-errors.js +182 -0
  76. package/src/lib/providers.js +0 -20
  77. package/src/lib/quota.js +116 -272
  78. package/src/lib/retention.js +1 -1
  79. package/src/lib/route.js +74 -227
  80. package/src/lib/run-delegate.js +340 -0
  81. package/src/lib/run-step.js +163 -0
  82. package/src/lib/stale.js +83 -24
  83. package/src/lib/state.js +12 -255
  84. package/src/lib/strategy.js +4 -4
  85. package/src/lib/tasks.js +9 -5
  86. package/src/lib/transcripts/claude-code.js +0 -40
  87. package/src/lib/transcripts/codex.js +0 -48
  88. package/src/lib/transcripts/grok.js +0 -1
  89. package/src/lib/usage-basis.js +0 -4
  90. package/src/lib/watch.js +125 -1070
  91. package/src/lib/worker-argv.js +98 -0
  92. package/src/lib/worker-report.js +129 -0
  93. package/src/meters/framework.js +92 -11
  94. package/src/meters/registry.js +39 -49
  95. package/src/provider-cli.js +2 -1
  96. package/src/providers/_schema.json +3 -3
  97. package/src/providers/claude-code/connector-history.json +4 -1
  98. package/src/providers/claude-code/connector.json +4 -4
  99. package/src/providers/claude-code/provider.mjs +3 -42
  100. package/src/providers/codex/connector-history.json +6 -1
  101. package/src/providers/codex/connector.json +7 -3
  102. package/src/providers/echo/echo-worker.mjs +20 -2
  103. package/src/providers/grok/connector-history.json +5 -1
  104. package/src/providers/grok/connector.json +4 -2
  105. package/src/setup.js +2 -15
  106. package/src/strategy-cli.js +0 -2
  107. package/src/workflow/action-validator.js +275 -53
  108. package/src/workflow/answers.js +279 -0
  109. package/src/workflow/attempt-bytes.js +57 -0
  110. package/src/workflow/attempt-record.js +137 -0
  111. package/src/workflow/budget-model.js +46 -157
  112. package/src/workflow/budget-view.js +5 -1
  113. package/src/workflow/caller-planner.js +167 -0
  114. package/src/workflow/cli-capabilities.js +137 -0
  115. package/src/workflow/cli-goal-document.js +95 -0
  116. package/src/workflow/cli-goal.js +251 -0
  117. package/src/workflow/cli-inspect.js +254 -0
  118. package/src/workflow/cli-launch.js +271 -0
  119. package/src/workflow/cli-plan.js +556 -0
  120. package/src/workflow/cli-pool-checks.js +74 -0
  121. package/src/workflow/cli-program-checks.js +125 -0
  122. package/src/workflow/cli-run-lookup.js +46 -0
  123. package/src/workflow/cli-run-verbs.js +258 -0
  124. package/src/workflow/cli-step-verbs.js +499 -0
  125. package/src/workflow/cli-steps.js +573 -0
  126. package/src/workflow/cli.js +20 -2005
  127. package/src/workflow/contract-v3.js +129 -0
  128. package/src/workflow/dash-kit.js +14 -183
  129. package/src/workflow/dashboard.js +19 -365
  130. package/src/workflow/day-key.js +51 -0
  131. package/src/workflow/earlier-work.js +56 -0
  132. package/src/workflow/evidence-output.js +0 -2
  133. package/src/workflow/evidence-runner.js +798 -0
  134. package/src/workflow/fleet-view.js +3 -21
  135. package/src/workflow/folder-walk.js +89 -0
  136. package/src/workflow/gates-loops.js +752 -0
  137. package/src/workflow/goal-column.js +60 -0
  138. package/src/workflow/history-view.js +25 -156
  139. package/src/workflow/history.js +35 -72
  140. package/src/workflow/home-model.js +81 -200
  141. package/src/workflow/home-view.js +38 -254
  142. package/src/workflow/kernel-resume.js +129 -0
  143. package/src/workflow/ledger.js +47 -7
  144. package/src/workflow/metrics-legacy.js +127 -0
  145. package/src/workflow/metrics.js +798 -0
  146. package/src/workflow/needs-you.js +468 -0
  147. package/src/workflow/no-pool-why.js +60 -0
  148. package/src/workflow/ownership.js +3 -6
  149. package/src/workflow/pick-preview.js +119 -0
  150. package/src/workflow/pool-refresh.js +2 -2
  151. package/src/workflow/program-v3.js +522 -0
  152. package/src/workflow/reprice.js +2 -8
  153. package/src/workflow/retry-handoff.js +108 -0
  154. package/src/workflow/revision-v3.js +168 -0
  155. package/src/workflow/rollup.js +25 -329
  156. package/src/workflow/run-control.js +329 -0
  157. package/src/workflow/run-counts.js +27 -0
  158. package/src/workflow/run-features.js +74 -0
  159. package/src/workflow/run-model.js +70 -323
  160. package/src/workflow/run-one-step.js +65 -0
  161. package/src/workflow/run-verdict.js +184 -0
  162. package/src/workflow/run-view.js +183 -665
  163. package/src/workflow/runs-cli.js +40 -10
  164. package/src/workflow/runs-view.js +4 -102
  165. package/src/workflow/schema-check.js +429 -0
  166. package/src/workflow/stat-kit.js +0 -108
  167. package/src/workflow/stats-model.js +204 -567
  168. package/src/workflow/stats-view.js +2 -5
  169. package/src/workflow/status.js +3 -0
  170. package/src/workflow/step-change-hint.js +22 -0
  171. package/src/workflow/step-model.js +50 -244
  172. package/src/workflow/step-prompts.js +255 -0
  173. package/src/workflow/step-route.js +463 -0
  174. package/src/workflow/step-view.js +16 -3
  175. package/src/workflow/step-vocabulary.js +286 -0
  176. package/src/workflow/task-step.js +0 -4
  177. package/src/workflow/time-box.js +56 -7
  178. package/src/workflow/usage-ledger.js +254 -0
  179. package/src/workflow/usage-preference.js +2 -2
  180. package/src/workflow/usage-view.js +11 -131
  181. package/src/workflow/v2-dispatch.js +1283 -223
  182. package/src/workflow/v2-outcome.js +676 -73
  183. package/src/workflow/v2-planner.js +222 -44
  184. package/src/workflow/v2-presentation.js +17 -3
  185. package/src/workflow/v2-revision.js +235 -12
  186. package/src/workflow/v2-runtime.js +545 -1261
  187. package/src/workflow/v2-scheduler.js +37 -18
  188. package/src/workflow/v2-state.js +298 -87
  189. package/src/workflow/v2-workspace.js +18 -4
  190. package/src/workflow/v3-display.js +196 -0
  191. package/src/workflow/v3-phases.js +122 -0
  192. package/src/workflow/v3-timeline.js +84 -0
  193. package/src/workflow/verify-rounds.js +461 -61
  194. package/src/workflow/watch-cli.js +342 -90
  195. package/src/workflow/workflow-flags.js +92 -0
package/AGENTS.md CHANGED
@@ -8,31 +8,100 @@ A CLI that routes bounded tasks to whichever coding-agent CLI subscription
8
8
  has the most quota headroom, paced by live provider meters, verified by
9
9
  content. Published as `bullswarm` on npm.
10
10
 
11
+ ## Redesign in progress (2026-09)
12
+
13
+ The core is being redesigned around facts-only mechanics, four mandatory
14
+ principles, caller-chosen options and a pattern library, for any kind of work
15
+ rather than code only. The design, the decisions taken and the staged build
16
+ plan are in `docs/design/redesign-mechanics-principles-options.md`, and draft
17
+ pattern cards are in `docs/design/patterns/`. Build in the plan's stage order.
18
+ The doctrine below stays in force until the stage that changes an item lands.
19
+ The redesign rewords items 1, 5, 6 and 7, and each stage updates this file.
20
+ Stage 1 (step vocabulary) has landed: program-mode steps may state a role and a
21
+ deliverable, each kind belongs to one role and keeps its exact routing, and
22
+ the no-op gate is now 'declared deliverable not produced' (failure kind
23
+ `not-produced`), measured over the whole step. Stage 2 (evidence v1) has landed:
24
+ program steps may declare command and schema `evidence` that the kernel runs
25
+ after the worker, a failure is `failed-evidence` with one same-pool retry, and
26
+ finished steps in new runs are labelled `proven by …` or `finished · unproven`.
27
+ Stage 3 (failure rule and routing constraints) has landed: one automatic retry
28
+ per step, then the caller; a usage limit goes straight to the caller, from a
29
+ step, the dispatched planner or the preflight scout alike; the
30
+ needs-you block with `step rerun --avoid` and `step accept`; the per-step
31
+ `route`; `verifyRounds` counts fixes (default 1); reviews are placed only by
32
+ route. Runs started earlier keep their rules (`features.json`).
33
+ 0.37.0 (program v3, the generic model) has landed: a step is a run, and a
34
+ workflow composes steps, phases, gates and loops. `bullswarm run` is a
35
+ one-step workflow. New programs are `bullswarm.workflow.program.v3`: a step
36
+ passes by facts (clean exit, deliverable produced, evidence passed, answer
37
+ matching its schema), a gate waits for `workflow continue`, a loop reruns its
38
+ steps until one condition holds (at most 5 rounds), and new work is added
39
+ with `workflow add`, never by editing the run's steps. v3 reports facts per
40
+ step and has no requirement IDs, `evidenceFor` or `verifyRounds`. v2 programs
41
+ and saved runs keep running and replaying as before.
42
+
11
43
  ## Non-negotiable doctrine
12
44
 
13
- 1. Judge delegate output by CONTENT, not exit code (see `src/lib/verify.js`).
45
+ 1. Judge delegate output by what can be checked, never by the delegate's exit code or its own report: the content (`src/lib/verify.js`) and, when a step declares them, the command and schema evidence Bullswarm runs itself (`src/workflow/evidence-runner.js`).
14
46
  2. Pace by meter surplus = elapsed% (from provider resets_at) − used%.
15
47
  Weekly/monthly windows pace; 5h windows are burst gates only (M1–M5 in
16
- `src/meters/framework.js`).
48
+ `src/meters/framework.js`). Any metered window at 100% (5h, weekly or
49
+ monthly) keeps its pool out of every pick until that window resets.
17
50
  3. Provider quirks live in the provider's directory (`src/providers/<name>/`,
18
51
  `providers/contrib/<name>/`, or `~/.bullswarm/providers/<name>/`), never in
19
52
  core logic (see `docs/reference/providers.md`).
20
- 4. Quarantine always auto-releases; recursion depth is core-owned via env
21
- (`BULLSWARM_DEPTH`).
22
- 5. Workflow dispatches must honor the same guarantees as single runs:
23
- `BULLSWARM_DEPTH` is propagated, burst-gated pools are excluded, and
24
- auth verdicts quarantine the pool + append to the shared decision log
25
- (R6/R7/R8 in `src/workflow/v2-dispatch.js`).
26
- 6. Adversarial verification is a first-class primitive: an action naming
27
- requirements in `evidenceFor` is dispatched under an evidence contract and
28
- judges them from the durable artifact, so a requirement is only verified by
29
- work someone else inspected (R-skeptic).
53
+ 4. A spent or dead pool is never remembered across steps: nothing pauses or
54
+ benches a pool, and every pick reads the live meters. The one fact that
55
+ outlives a step is the 100% refusal marker a usage limit writes when the
56
+ meter cannot be read, and it counts only when its reset was named or
57
+ measured, never guessed (`refusalResetKnown` in `src/meters/framework.js`).
58
+ Recursion depth is core-owned via env (`BULLSWARM_DEPTH`).
59
+ 5. Workflow dispatches honor the same guarantees as single runs, in
60
+ `src/workflow/v2-dispatch.js`: `BULLSWARM_DEPTH` is checked and propagated
61
+ (`assertDepthAllowed` and `childDepthEnv` in `dispatchV2Action`), pools at
62
+ a spent window are excluded (`preparePools`), and a sign-in failure is
63
+ failure kind `auth`: the step's retry skips every pool in the dead
64
+ credential's group (`upstreamGroupOf`), a choice held for that dispatch
65
+ only and never stored. A step's `route`, the run's pin and the step's
66
+ capability tier are hard filters applied before pace ranks what is left. The failure rule is one automatic retry per step (a
67
+ process failure on another eligible pool, a gate failure on the same pool
68
+ with the failure attached), then the caller. An `act` step is never retried
69
+ once its worker started. A usage limit (a spent 5-hour or weekly window, or
70
+ no credit left) ends the step and sends it to the caller: no wait, no
71
+ automatic move, no retry. The pool's meter is re-read after it
72
+ (when that read fails, the refusal marker counts the pool as full only until
73
+ a reset the provider named or a reading measured, never a guessed one), so
74
+ a window at 100% keeps that pool out of later steps. Finding no capable pool
75
+ free at the pick sends the step to the caller too. Nothing waits for a
76
+ pool inside a run; only a transient rate limit backs off on the same pool,
77
+ at most twice (20 s, then 60 s, or a named wait of at most 2 minutes), then
78
+ goes to the caller.
79
+ Only a failed step's dependents wait. The dispatched planner and the
80
+ preflight scout follow the same usage-limit rule (`usageLimitsToCaller` in
81
+ `dispatchV2Action`): they stop and the run tells the caller, with no
82
+ automatic move to another pool.
83
+ 6. Review is a caller option, recorded as a fact. A step naming requirements
84
+ in `evidenceFor` is dispatched under the evidence contract and judges them
85
+ from the durable artifact. Where it runs is the caller's choice through
86
+ `route` (`independentOf`, `providers`, `pools`); Bullswarm never moves a
87
+ review on its own, and records who reviewed (pool, model, provider) and
88
+ whether that provider also wrote the work. A caller's `step accept` is
89
+ recorded as evidence `choice` and never makes a requirement verified.
90
+ Runs started before this rule keep automatic writer avoidance (R12/R13).
30
91
  7. New goal workflows are caller-planned programs in a shared workspace.
31
92
  `bullswarm workflow goal --program` executes the graph; `--orchestrator`
32
93
  explicitly delegates planning. File territories are advisory scheduling
33
- hints, and the graph finishes without automatic gap rounds. `verified`
34
- separately records requirement evidence. `--isolation` opts into strict
35
- per-worker worktrees. Saved V2 runs preserve their original semantics.
94
+ hints. In a v3 program (the format for new work) a check is an ordinary
95
+ step, and fixing until it passes is a loop the caller declares (`loops`,
96
+ `until` one condition, `maxRounds` 1-5); a loop out of rounds or a gate
97
+ waits for the caller (`workflow continue`), and a failed step goes to the
98
+ caller through the watcher's needs-you block: rerun elsewhere, add steps
99
+ (`workflow add`), take over, or accept anyway. v3 validate refuses
100
+ `defaults.verifyRounds`. In a v2 program a failed check gets one fix step
101
+ and one re-review (`defaults.verifyRounds`, default 1, 0-3), then the
102
+ caller, and `verified` separately records requirement evidence.
103
+ `--isolation` opts into strict per-worker worktrees. Saved V2 runs
104
+ preserve their original semantics (`features.json`).
36
105
  8. Historical authored-graph runs remain visible as read-only `legacy` rows.
37
106
  Their executor was removed in 0.27.0; driving commands fail closed before
38
107
  dispatch and historical run directories remain untouched.