bullswarm 0.35.6 → 0.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +70 -15
- package/CHANGELOG.md +310 -1
- package/README.md +2 -2
- package/bin/check-schema.js +78 -0
- package/data/openrouter-benchmarks.json +8205 -7991
- package/docs/design/patterns/README.md +66 -0
- package/docs/design/patterns/feature-slices.md +58 -0
- package/docs/design/patterns/many-angles.md +57 -0
- package/docs/design/patterns/triage-at-scale.md +57 -0
- package/docs/design/redesign-mechanics-principles-options.md +418 -248
- package/docs/design/tidy-0.35.1/frames/colour/real-run-finished-200.txt +6 -6
- package/docs/design/tidy-0.35.1/frames/colour/real-run-finished-55.txt +6 -6
- package/docs/design/tidy-0.35.1/frames/colour/real-run-running-200.txt +5 -5
- package/docs/design/tidy-0.35.1/frames/colour/real-run-running-55.txt +5 -5
- package/docs/design/tidy-0.35.1/frames/real-run-finished-120.txt +5 -5
- package/docs/design/tidy-0.35.1/frames/real-run-finished-200.txt +6 -6
- package/docs/design/tidy-0.35.1/frames/real-run-finished-55.txt +6 -6
- package/docs/design/tidy-0.35.1/frames/real-run-running-120.txt +4 -4
- package/docs/design/tidy-0.35.1/frames/real-run-running-200.txt +5 -5
- package/docs/design/tidy-0.35.1/frames/real-run-running-55.txt +5 -5
- package/docs/guide/concepts.md +25 -5
- package/docs/guide/index.md +2 -2
- package/docs/guide/observing.md +24 -11
- package/docs/guide/playbook.md +1 -1
- package/docs/guide/routing.md +117 -57
- package/docs/guide/run.md +2 -2
- package/docs/guide/workflows.md +238 -36
- package/docs/integrations/claude-code.md +1 -1
- package/docs/reference/cli.md +89 -108
- package/docs/reference/configuration.md +3 -3
- package/docs/reference/program.md +218 -47
- package/docs/reference/providers.md +4 -4
- package/docs/reference/result.md +52 -34
- package/mcp/server.mjs +2 -2
- package/package.json +1 -1
- package/providers/contrib/command-code/connector-history.json +11 -2
- package/providers/contrib/command-code/connector.json +7 -3
- package/providers/contrib/opencode/connector-history.json +5 -1
- package/providers/contrib/opencode/connector.json +2 -2
- package/skill/SKILL.md +259 -61
- package/skill/references/operations.md +298 -82
- package/skill/references/program.md +356 -49
- package/src/cli.js +42 -215
- package/src/help.js +154 -89
- package/src/lib/auth-signatures.js +3 -3
- package/src/lib/cli-flags.js +2 -2
- package/src/lib/config.js +19 -33
- package/src/lib/pool-labels.js +0 -15
- package/src/lib/probe.js +1 -1
- package/src/lib/providers.js +0 -20
- package/src/lib/quota.js +116 -272
- package/src/lib/route.js +51 -109
- package/src/lib/stale.js +81 -15
- package/src/lib/state.js +9 -252
- package/src/lib/strategy.js +3 -3
- package/src/lib/transcripts/claude-code.js +0 -40
- package/src/lib/transcripts/codex.js +0 -48
- package/src/lib/transcripts/grok.js +0 -1
- package/src/lib/usage-basis.js +0 -4
- package/src/lib/watch.js +144 -79
- package/src/meters/framework.js +92 -7
- package/src/meters/registry.js +39 -33
- package/src/providers/_schema.json +3 -3
- package/src/providers/claude-code/connector-history.json +4 -1
- package/src/providers/claude-code/connector.json +4 -4
- package/src/providers/claude-code/provider.mjs +3 -42
- package/src/providers/codex/connector-history.json +6 -1
- package/src/providers/codex/connector.json +7 -3
- package/src/providers/grok/connector-history.json +5 -1
- package/src/providers/grok/connector.json +4 -2
- package/src/setup.js +2 -15
- package/src/workflow/action-validator.js +275 -53
- package/src/workflow/budget-view.js +5 -1
- package/src/workflow/cli.js +543 -108
- package/src/workflow/dash-kit.js +13 -148
- package/src/workflow/dashboard.js +15 -189
- package/src/workflow/evidence-output.js +0 -2
- package/src/workflow/evidence-runner.js +798 -0
- package/src/workflow/fleet-view.js +2 -16
- package/src/workflow/history-view.js +2 -135
- package/src/workflow/home-model.js +1 -2
- package/src/workflow/home-view.js +6 -245
- package/src/workflow/ledger.js +45 -3
- package/src/workflow/needs-you.js +455 -0
- package/src/workflow/ownership.js +2 -5
- package/src/workflow/pool-refresh.js +2 -2
- package/src/workflow/reprice.js +1 -4
- package/src/workflow/run-features.js +55 -0
- package/src/workflow/run-model.js +3 -2
- package/src/workflow/run-view.js +123 -642
- package/src/workflow/runs-cli.js +14 -4
- package/src/workflow/schema-check.js +429 -0
- package/src/workflow/stat-kit.js +0 -108
- package/src/workflow/stats-view.js +0 -4
- package/src/workflow/step-model.js +23 -160
- package/src/workflow/step-route.js +435 -0
- package/src/workflow/step-view.js +9 -3
- package/src/workflow/step-vocabulary.js +281 -0
- package/src/workflow/task-step.js +0 -4
- package/src/workflow/time-box.js +17 -4
- package/src/workflow/usage-view.js +11 -131
- package/src/workflow/v2-dispatch.js +1168 -205
- package/src/workflow/v2-outcome.js +551 -51
- package/src/workflow/v2-planner.js +195 -40
- package/src/workflow/v2-presentation.js +3 -1
- package/src/workflow/v2-revision.js +189 -8
- package/src/workflow/v2-runtime.js +813 -103
- package/src/workflow/v2-scheduler.js +26 -13
- package/src/workflow/v2-state.js +148 -14
- package/src/workflow/v2-workspace.js +18 -4
- package/src/workflow/verify-rounds.js +434 -56
- package/src/workflow/watch-cli.js +274 -83
package/AGENTS.md
CHANGED
|
@@ -8,31 +8,86 @@ A CLI that routes bounded tasks to whichever coding-agent CLI subscription
|
|
|
8
8
|
has the most quota headroom, paced by live provider meters, verified by
|
|
9
9
|
content. Published as `bullswarm` on npm.
|
|
10
10
|
|
|
11
|
+
## Redesign in progress (2026-09)
|
|
12
|
+
|
|
13
|
+
The core is being redesigned around facts-only mechanics, four mandatory
|
|
14
|
+
principles, caller-chosen options and a pattern library, for any kind of work
|
|
15
|
+
rather than code only. The design, the decisions taken and the staged build
|
|
16
|
+
plan are in `docs/design/redesign-mechanics-principles-options.md`, and draft
|
|
17
|
+
pattern cards are in `docs/design/patterns/`. Build in the plan's stage order.
|
|
18
|
+
The doctrine below stays in force until the stage that changes an item lands.
|
|
19
|
+
The redesign rewords items 1, 5, 6 and 7, and each stage updates this file.
|
|
20
|
+
Stage 1 (step vocabulary) has landed: program-mode steps may state a role and a
|
|
21
|
+
deliverable, each kind belongs to one role and keeps its exact routing, and
|
|
22
|
+
the no-op gate is now 'declared deliverable not produced' (failure kind
|
|
23
|
+
`not-produced`), measured over the whole step. Stage 2 (evidence v1) has landed:
|
|
24
|
+
program steps may declare command and schema `evidence` that the kernel runs
|
|
25
|
+
after the worker, a failure is `failed-evidence` with one same-pool retry, and
|
|
26
|
+
finished steps in new runs are labelled `proven by …` or `finished · unproven`.
|
|
27
|
+
Stage 3 (failure rule and routing constraints) has landed: one automatic retry
|
|
28
|
+
per step, then the caller; a usage limit goes straight to the caller, from a
|
|
29
|
+
step, the dispatched planner or the preflight scout alike; the
|
|
30
|
+
needs-you block with `step rerun --avoid` and `step accept`; the per-step
|
|
31
|
+
`route`; `verifyRounds` counts fixes (default 1); reviews are placed only by
|
|
32
|
+
route. Runs started earlier keep their rules (`features.json`).
|
|
33
|
+
|
|
11
34
|
## Non-negotiable doctrine
|
|
12
35
|
|
|
13
|
-
1. Judge delegate output by
|
|
36
|
+
1. Judge delegate output by what can be checked, never by the delegate's exit code or its own report: the content (`src/lib/verify.js`) and, when a step declares them, the command and schema evidence Bullswarm runs itself (`src/workflow/evidence-runner.js`).
|
|
14
37
|
2. Pace by meter surplus = elapsed% (from provider resets_at) − used%.
|
|
15
38
|
Weekly/monthly windows pace; 5h windows are burst gates only (M1–M5 in
|
|
16
|
-
`src/meters/framework.js`).
|
|
39
|
+
`src/meters/framework.js`). Any metered window at 100% (5h, weekly or
|
|
40
|
+
monthly) keeps its pool out of every pick until that window resets.
|
|
17
41
|
3. Provider quirks live in the provider's directory (`src/providers/<name>/`,
|
|
18
42
|
`providers/contrib/<name>/`, or `~/.bullswarm/providers/<name>/`), never in
|
|
19
43
|
core logic (see `docs/reference/providers.md`).
|
|
20
|
-
4.
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
44
|
+
4. A spent or dead pool is never remembered across steps: nothing pauses or
|
|
45
|
+
benches a pool, and every pick reads the live meters. The one fact that
|
|
46
|
+
outlives a step is the 100% refusal marker a usage limit writes when the
|
|
47
|
+
meter cannot be read, and it counts only when its reset was named or
|
|
48
|
+
measured, never guessed (`refusalResetKnown` in `src/meters/framework.js`).
|
|
49
|
+
Recursion depth is core-owned via env (`BULLSWARM_DEPTH`).
|
|
50
|
+
5. Workflow dispatches honor the same guarantees as single runs, in
|
|
51
|
+
`src/workflow/v2-dispatch.js`: `BULLSWARM_DEPTH` is checked and propagated
|
|
52
|
+
(`assertDepthAllowed` and `childDepthEnv` in `dispatchV2Action`), pools at
|
|
53
|
+
a spent window are excluded (`preparePools`), and a sign-in failure is
|
|
54
|
+
failure kind `auth`: the step's retry skips every pool in the dead
|
|
55
|
+
credential's group (`upstreamGroupOf`), a choice held for that dispatch
|
|
56
|
+
only and never stored. A step's `route`, the run's pin and the step's
|
|
57
|
+
capability tier are hard filters applied before pace ranks what is left. The failure rule is one automatic retry per step (a
|
|
58
|
+
process failure on another eligible pool, a gate failure on the same pool
|
|
59
|
+
with the failure attached), then the caller. An `act` step is never retried
|
|
60
|
+
once its worker started. A usage limit (a spent 5-hour or weekly window, or
|
|
61
|
+
no credit left) ends the step and sends it to the caller: no wait, no
|
|
62
|
+
automatic move, no retry. The pool's meter is re-read after it
|
|
63
|
+
(when that read fails, the refusal marker counts the pool as full only until
|
|
64
|
+
a reset the provider named or a reading measured, never a guessed one), so
|
|
65
|
+
a window at 100% keeps that pool out of later steps. Finding no capable pool
|
|
66
|
+
free at the pick sends the step to the caller too. Nothing waits for a
|
|
67
|
+
pool inside a run; only a transient rate limit backs off on the same pool,
|
|
68
|
+
at most twice (20 s, then 60 s, or a named wait of at most 2 minutes), then
|
|
69
|
+
goes to the caller.
|
|
70
|
+
Only a failed step's dependents wait. The dispatched planner and the
|
|
71
|
+
preflight scout follow the same usage-limit rule (`usageLimitsToCaller` in
|
|
72
|
+
`dispatchV2Action`): they stop and the run tells the caller, with no
|
|
73
|
+
automatic move to another pool.
|
|
74
|
+
6. Review is a caller option, recorded as a fact. A step naming requirements
|
|
75
|
+
in `evidenceFor` is dispatched under the evidence contract and judges them
|
|
76
|
+
from the durable artifact. Where it runs is the caller's choice through
|
|
77
|
+
`route` (`independentOf`, `providers`, `pools`); Bullswarm never moves a
|
|
78
|
+
review on its own, and records who reviewed (pool, model, provider) and
|
|
79
|
+
whether that provider also wrote the work. A caller's `step accept` is
|
|
80
|
+
recorded as evidence `choice` and never makes a requirement verified.
|
|
81
|
+
Runs started before this rule keep automatic writer avoidance (R12/R13).
|
|
30
82
|
7. New goal workflows are caller-planned programs in a shared workspace.
|
|
31
83
|
`bullswarm workflow goal --program` executes the graph; `--orchestrator`
|
|
32
84
|
explicitly delegates planning. File territories are advisory scheduling
|
|
33
|
-
hints
|
|
34
|
-
|
|
35
|
-
|
|
85
|
+
hints. A failed check gets one fix step and one re-review
|
|
86
|
+
(`defaults.verifyRounds`, default 1, 0-3), then the caller, who takes over
|
|
87
|
+
through the watcher's needs-you block: rerun elsewhere, change the step,
|
|
88
|
+
take over, or accept anyway. `verified` separately records requirement
|
|
89
|
+
evidence. `--isolation` opts into strict per-worker worktrees. Saved V2
|
|
90
|
+
runs preserve their original semantics (`features.json`).
|
|
36
91
|
8. Historical authored-graph runs remain visible as read-only `legacy` rows.
|
|
37
92
|
Their executor was removed in 0.27.0; driving commands fail closed before
|
|
38
93
|
dispatch and historical run directories remain untouched.
|
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,316 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
## 0.36.0 — Usage limits go back to the caller; each account bills itself
|
|
6
|
+
|
|
7
|
+
- known issues in this release: a limit marker whose reset was guessed still
|
|
8
|
+
counts as a 100% reading in the spend forecast, so it can rank a pool lower
|
|
9
|
+
for up to 5 hours (it never keeps the pool out); the Claude Code mod's pool
|
|
10
|
+
panes still read the removed pausing fields, so a pool held by a usage limit
|
|
11
|
+
can show as ready there; the preflight scout and the dispatched planner can
|
|
12
|
+
make up to three attempts (a correction plus a retry); `strategy
|
|
13
|
+
set-pausing` exits 2 with the general usage text instead of naming the
|
|
14
|
+
removed command.
|
|
15
|
+
- workflow: every step gets one automatic retry, then comes back to you. A
|
|
16
|
+
crashed, silent or signed-out worker is retried once on another pool that can
|
|
17
|
+
run the step (the same pool when it is the only one, except after a sign-in
|
|
18
|
+
failure). A failed gate (declared evidence, a deliverable not produced, a
|
|
19
|
+
report in the wrong format, or output judged failed) is retried once on the
|
|
20
|
+
same pool with the failure attached. A usage limit is never retried, moved or
|
|
21
|
+
waited out: the step comes back to you (see below). An `act` step (one that
|
|
22
|
+
changes the outside world) is never retried automatically once its worker has
|
|
23
|
+
started. Steps that do not depend on a failed step keep running.
|
|
24
|
+
`--retry-attempts` (0-3, default 1) sets the retries per step. Runs started
|
|
25
|
+
before this version keep their old retry rules when resumed.
|
|
26
|
+
- workflow: when a step needs you, `watch` prints one block: what failed, each
|
|
27
|
+
try, what is still running and what waits on it, and four commands (rerun
|
|
28
|
+
elsewhere, change the step, take over, accept anyway). Two commands are new.
|
|
29
|
+
`bullswarm workflow step rerun <run> <step> --avoid <pool>` runs a step again
|
|
30
|
+
off the pools you name, with its last attempt's handoff; the pools stay in the
|
|
31
|
+
step's route. `bullswarm workflow step accept <run> <step> --reason "…"`
|
|
32
|
+
accepts a failed step, or a check's failing requirements, as your choice: its
|
|
33
|
+
dependents run, the step reads `accepted by choice`, the record says "choice"
|
|
34
|
+
and never counts as proof (the end-of-run proof line counts it apart from
|
|
35
|
+
proven steps), and rerunning the step undoes it. A run that ends with a
|
|
36
|
+
failed step, or with a requirement its review loop left failing, lists
|
|
37
|
+
`rerun` and `accept` under `your call:`; for a requirement they name the
|
|
38
|
+
check that judged it (`--avoid <pool>`, `--requirement <id>`).
|
|
39
|
+
- workflow: a usage limit, or no pool free, sends the step back to you; a step
|
|
40
|
+
never waits for a pool. A spent usage window (5-hour, weekly or monthly), or
|
|
41
|
+
no credit left, ends the step at once: no wait, no move to another pool, no
|
|
42
|
+
retry, even for a limit notice that names no reset. The pool's meter is
|
|
43
|
+
then read again at once (when it cannot be read, the pool counts as full
|
|
44
|
+
until its reset only when the provider named that reset or an earlier meter
|
|
45
|
+
reading gave it), so a window it shows at 100% keeps the pool out of later
|
|
46
|
+
steps until that window resets; a single `bullswarm run` does the same.
|
|
47
|
+
When every pool that can run the step is nearly spent, or at a 5-hour,
|
|
48
|
+
weekly or monthly limit, the step stops as well (`no pool with quota to
|
|
49
|
+
spare: <pool> at its 5-hour limit until …`, `<pool> nearly spent (forecast
|
|
50
|
+
…%) until …`, or `no pool free: …`; a pool at its 5-hour limit no longer
|
|
51
|
+
reads as "no enabled pool has a model on the tier"); when another pool is
|
|
52
|
+
free, it takes the step as usual. A
|
|
53
|
+
retry the step was promised that finds no free pool keeps its own failure
|
|
54
|
+
and says why (`· no retry: <pool> <reason>; …`). The rest of the run keeps
|
|
55
|
+
going. When a return time is known, the needs-you block shows `back at
|
|
56
|
+
<time>` (the failed pool's reset, the end of a rate limit's named wait, or
|
|
57
|
+
the earliest known return of the pools that can run the step) and a `wait
|
|
58
|
+
for it` command (`after <time>: bullswarm workflow step rerun <run>
|
|
59
|
+
<step>`); with `--jsonl` they are `backAt` and `options.waitForIt`. The
|
|
60
|
+
watch's usage-limit line ends `back to you`. A short "too many requests"
|
|
61
|
+
rate limit still backs off on the same pool, at most twice (20 s, then 60 s,
|
|
62
|
+
or the wait it names when that is at most 2 minutes), before the step comes
|
|
63
|
+
back to you; a try after one reads `· after a rate-limit backoff`, and the
|
|
64
|
+
block's header says `backed off twice` (never `not retried`: a backoff is
|
|
65
|
+
not the step's retry). One that names a longer wait comes back to you at
|
|
66
|
+
once, and so does one whose pool runs out during the backoff. Runs started
|
|
67
|
+
before this version keep their old rules.
|
|
68
|
+
- pools: Bullswarm no longer pauses or benches a pool, and never remembers a
|
|
69
|
+
spent or dead pool from one step to the next. `bullswarm strategy
|
|
70
|
+
set-pausing`, `bullswarm pools resume`, and the `PAUSED`, `BENCHED`,
|
|
71
|
+
`strikes=` and `automatic pausing: off` lines in `bullswarm pools` are gone;
|
|
72
|
+
a pause or bench record an earlier version left in `state.json` is ignored.
|
|
73
|
+
What replaces them: every pick reads the live meters, so a window at 100%
|
|
74
|
+
(5-hour, weekly or monthly) keeps a pool out until that window resets, and
|
|
75
|
+
nothing else does. A usage limit goes back to the caller, and the pool's
|
|
76
|
+
meter is read again at once; when it cannot be read, the pool counts as full
|
|
77
|
+
only until a reset the provider named or an earlier reading gave, and a
|
|
78
|
+
reset nobody knows keeps no pool out (the meter column then reads
|
|
79
|
+
`[refused <age> · reset unknown]`; such a reset used to be guessed as a
|
|
80
|
+
whole window from the refusal, which could keep a pool with no meter reader
|
|
81
|
+
out for up to a week, or a month). After a sign-in failure the step's one
|
|
82
|
+
automatic retry goes to a pool that does not share the dead credential; the
|
|
83
|
+
next step routes as usual. A stall, a provider error or a failed free-model
|
|
84
|
+
probe no longer counts a strike: the step's retry, or the pick, goes to
|
|
85
|
+
another pool. Before, a proven usage limit paused the pool until its reset,
|
|
86
|
+
a sign-in failure paused the pool and every pool sharing its credential for
|
|
87
|
+
10 minutes, and a second stall or failure in a row benched a pool for 10
|
|
88
|
+
minutes, until the pause ran out, `pools resume` lifted it or `set-pausing
|
|
89
|
+
off` turned pausing off. For scripts: `pools --json` drops `pausing`,
|
|
90
|
+
`quarantine` and `pauseWhy`; a `run --json` verdict drops `quarantineHint`,
|
|
91
|
+
`quarantineUntil`, `quarantineSource`, `quarantinedUntil` and
|
|
92
|
+
`quarantinedSiblings`, names its limit decision `usageLimit` (it was
|
|
93
|
+
`quotaPause`), and carries `retryAfter` when a usage limit's reset is
|
|
94
|
+
known; a run no longer writes `pool.benched` events, and `workflow watch`
|
|
95
|
+
skips one a saved run holds, prints no `paused until` or `not paused` on
|
|
96
|
+
its usage-limit line, and with `--jsonl` drops `paused` and `proof` from
|
|
97
|
+
`attempt.quota`; `bullswarm health` no longer reports paused pools
|
|
98
|
+
(`quarantined`, `quarantineCluster`) or exits 1 for them; and neither
|
|
99
|
+
`pools` nor `health` changes `state.json` any more.
|
|
100
|
+
- workflow: the dispatched Workflow Planner (`--orchestrator`) and the
|
|
101
|
+
preflight scout (`--scout`, or the scout before a dispatched planner) follow
|
|
102
|
+
the same rule. A usage limit, a rate limit still there after its short
|
|
103
|
+
backoff, or no free pool stops it and tells you; it no longer moves to
|
|
104
|
+
another pool by itself. The run finishes `partial` with `the workflow
|
|
105
|
+
planner stopped on a usage limit: <why> · back at <time> · your call: resume
|
|
106
|
+
after <time> with bullswarm workflow resume <run>, plan it yourself with
|
|
107
|
+
bullswarm workflow plan revise <run> --program <file.json>, or start a new
|
|
108
|
+
run` (`stopped: no pool free` when a pool was out for another reason or no
|
|
109
|
+
pool can run it at all), and `planner.finished` and
|
|
110
|
+
`preflight.scout_finished` carry `retryAfter` (the scout's also
|
|
111
|
+
`runContinues`: whether the run goes on without its report). The watch
|
|
112
|
+
prints the planner's stop as `✗ planner stopped · out of quota on <pool> ·
|
|
113
|
+
back at <time>` (it used to read `× planning attempt rejected · …`). A scout
|
|
114
|
+
with no program after it ends the run the same way (`the preflight scout
|
|
115
|
+
stopped on a usage limit: …`). A scout before your own program lets the run
|
|
116
|
+
go on without its report, and the watch prints `⚠ preflight scout stopped ·
|
|
117
|
+
out of quota on <pool> · back at <time> · the run continues without its
|
|
118
|
+
report`. `--until trouble` wakes on either stop. The Run page's timeline has
|
|
119
|
+
a row for either stop: `[Workflow Planner] planner stopped · <label> on
|
|
120
|
+
<pool> · back at <time>` or `Scout stopped · <label> on <pool> · back at
|
|
121
|
+
<time>` (ending `· the run continues without its report` when the run goes
|
|
122
|
+
on without it), where `<label>` is `out of quota`, `rate limited` or `no
|
|
123
|
+
eligible pool`, as in the watch. A sign-in failure, a provider error or a worker that
|
|
124
|
+
died at start still moves the planner or the scout to another pool.
|
|
125
|
+
`bullswarm run` still stops after its one attempt on a usage limit. Runs
|
|
126
|
+
started before this version keep their old rules.
|
|
127
|
+
- workflow: when a usage limit or no free pool stopped the planner or scout
|
|
128
|
+
and ended the run, `workflow resume <run>` runs it again, and the run goes
|
|
129
|
+
on from there. Run it after the `back at` time in the stop reason; it
|
|
130
|
+
prints `✓ reopened the partial run <run>; running again: the workflow
|
|
131
|
+
planner` (or `the preflight scout`). Before that time it can stop the same
|
|
132
|
+
way, and it adds `note: the workflow planner stopped with its pool back at
|
|
133
|
+
<time>; run before then, it can fail the same way again`. When no return
|
|
134
|
+
time is known, the reason says `bullswarm workflow resume <run> once a pool
|
|
135
|
+
is free`. The result's `retry` option names it too (`bullswarm workflow
|
|
136
|
+
resume <run> after <time> (reruns the workflow planner)`). `plan revise`
|
|
137
|
+
with your own program and a new run stay the other choices. A scout the run
|
|
138
|
+
went on without (one before your own program) is not run again. Any run that
|
|
139
|
+
`workflow resume` reopens now reads `run reopened from <status> by workflow
|
|
140
|
+
resume` in the watch; it used to say `by a plan revision` whichever command
|
|
141
|
+
reopened the run.
|
|
142
|
+
- workflow: a pool that is nearly spent (the router's "expiring but
|
|
143
|
+
draining": its weekly or monthly window closes soon and the step would push
|
|
144
|
+
it past its limit) is no longer given a step in runs started by this
|
|
145
|
+
version, even when it is the only pool left; the router used to pick it as a
|
|
146
|
+
last resort. The step takes another pool that can run it, or comes back to
|
|
147
|
+
you when none is free (`no pool with quota to spare: <pool> nearly spent
|
|
148
|
+
(forecast …%) until …`). The dispatched planner and the preflight scout
|
|
149
|
+
follow the same rule, and their schema correction, and the one retry on the
|
|
150
|
+
same pool they get when no other pool can run them, never go back to a pool
|
|
151
|
+
that has become nearly spent: the correction moves to another free pool, and
|
|
152
|
+
with none free they stop and tell you, with that pool's `nearly spent`
|
|
153
|
+
reason. A pool the caller pinned is exempt: the run's `--worker-pool` (for a
|
|
154
|
+
step or the scout), a route that allows only that pool (for a step), and
|
|
155
|
+
`--orchestrator <pool> --orchestrator-strict` (for the planner).
|
|
156
|
+
- routing: a pool whose weekly or monthly window reads 100% is no longer
|
|
157
|
+
picked until that window resets, by `bullswarm run` or any workflow. Only a
|
|
158
|
+
5-hour window at 100% used to keep a pool out, so a spent week whose reset
|
|
159
|
+
was more than a day away was still given work that then failed on the
|
|
160
|
+
limit. In a workflow started by this version, a step with no other pool free
|
|
161
|
+
comes back to you with `<pool> at its weekly limit until <time>` (or
|
|
162
|
+
`monthly`) in its `why`, and `bullswarm run`
|
|
163
|
+
names such a pool in its reason as `(burst-gated: <pool> at its weekly
|
|
164
|
+
limit)`.
|
|
165
|
+
- routing: the last-mile reason no longer promises a retry. A pool near its
|
|
166
|
+
5-hour limit that still gets the work reads `last mile: <pool> 88.1% of 5h,
|
|
167
|
+
a limit mid-attempt goes back to the caller`; it used to end `handoff covers
|
|
168
|
+
the wall`.
|
|
169
|
+
- workflow: `step rerun` and `resume` after a failure the pool caused (a
|
|
170
|
+
sign-in failure, a provider error, or a worker that died before it answered
|
|
171
|
+
or changed a file) start on another pool when one can take the step now.
|
|
172
|
+
Before, the rerun's first pick knew nothing of the failure and could go
|
|
173
|
+
straight back to the pool whose sign-in had just failed. A usage limit is
|
|
174
|
+
not one of these: a rerun after one is routed as usual (`--avoid <pool>`
|
|
175
|
+
keeps it off that pool).
|
|
176
|
+
- workflow: a step can say where it runs with `route`: `pools` and `providers`
|
|
177
|
+
to use or avoid, and `independentOf` (earlier steps, or "writers" on a check)
|
|
178
|
+
whose providers it must not use. It is applied before quota pacing, and the
|
|
179
|
+
kernel's fix and re-review steps inherit it. Reviews are no longer moved away
|
|
180
|
+
from writers' pools on their own; route a check when you want an independent
|
|
181
|
+
reviewer. Every review records who reviewed (pool, model, provider) and
|
|
182
|
+
whether that provider also wrote the work. Runs started before this version
|
|
183
|
+
keep the automatic placement.
|
|
184
|
+
- workflow: a failed check gets one fix step and one re-review, then comes
|
|
185
|
+
back to you. `defaults.verifyRounds` counts those fix cycles (0-3, default
|
|
186
|
+
1; 0 means no automatic fix). A plan file written for an earlier version
|
|
187
|
+
changes meaning: `verifyRounds: 1` used to mean one review round and no fix,
|
|
188
|
+
and now means one fix and one re-review; set 0 for review only. `plan
|
|
189
|
+
validate`, launch and `plan revise` print a note when a plan sets it. The
|
|
190
|
+
fix step also runs the command evidence of the steps it repairs. When every
|
|
191
|
+
step affecting a failed requirement declares a `report`, the repair step is
|
|
192
|
+
a read-only `analyze` step whose deliverable is `report` and which owns no
|
|
193
|
+
files. A requirement an `act` step affects gets no repair: the loop stops
|
|
194
|
+
with `stoppedBy: act-step`. Runs started before this version keep their
|
|
195
|
+
budget of up to 3 review rounds.
|
|
196
|
+
- workflow: a program-mode step can say what it does and what it leaves
|
|
197
|
+
behind. `role` is one of `investigate`, `produce`, `transform`, `combine`,
|
|
198
|
+
`check` or `act`, and `deliverable` is `files`, `report`, `data`, `media`
|
|
199
|
+
or `outward` (`data` and `media` name exact `paths`). A role sets lane and
|
|
200
|
+
effort the way `kind` does. Every kind still validates and belongs to one
|
|
201
|
+
role, keeping its own lane, effort and gate: `mechanical` is a
|
|
202
|
+
transform, `io-read` and `architecture` investigate, `implement` produces,
|
|
203
|
+
`integration` and `digest` combine, `check` and `adversarial-acceptance`
|
|
204
|
+
check. A step may give both only when they agree, and then only the kind is
|
|
205
|
+
stored, so annotating an exported plan changes nothing. An `act` step works
|
|
206
|
+
outside the workspace (send, post, publish, deploy): it may not edit
|
|
207
|
+
workspace files, and the kernel never repairs a requirement it affects.
|
|
208
|
+
- workflow: a step whose declared deliverable was not produced fails as
|
|
209
|
+
`not-produced`. A `files` deliverable needs a changed file or a new commit.
|
|
210
|
+
A deliverable with `paths` needs every path present and at least one
|
|
211
|
+
written during the step, checked without git, so git-ignored output counts
|
|
212
|
+
in a shared workspace (an isolated run refuses a git-ignored deliverable
|
|
213
|
+
path, and any run refuses a path that names a directory). A `report` must
|
|
214
|
+
not be empty, and that includes a `combine` step's report. `act` steps, and
|
|
215
|
+
`combine` steps with a `files` deliverable and no paths, are not judged.
|
|
216
|
+
The check covers the whole step, so a retry or rerun that finds the work
|
|
217
|
+
already done is not failed; a rerun of a step that failed `not-produced` is
|
|
218
|
+
judged again. A program written only with kinds or lanes
|
|
219
|
+
validates exactly as before. One behavior is new in runs started by this
|
|
220
|
+
version: a build-lane step other than `integration` that changes no file
|
|
221
|
+
and makes no commit fails as `not-produced`; 0.35.6 recorded it as
|
|
222
|
+
succeeded. Chore and analyze steps with no declared deliverable are not
|
|
223
|
+
judged. Runs started before this version keep their original rules when
|
|
224
|
+
resumed. The failure gets one retry on the same pool with the failure
|
|
225
|
+
attached, then comes back to you; `workflow resume` never reruns it. The
|
|
226
|
+
skill and planner rules make the gate a `check` and keep commit and PR steps
|
|
227
|
+
`mechanical`.
|
|
228
|
+
- workflow: a program step can declare `evidence`, checks Bullswarm runs
|
|
229
|
+
itself after the worker finishes. `{"type":"command","cmd":"npm test -- tests/x.test.js"}` passes on exit code 0; `{"type":"schema","file":"out/records.json","schema":"schemas/record.json"}` passes when the JSON
|
|
230
|
+
(or JSONL) file matches the schema. Up to 5 per step; each is stopped after
|
|
231
|
+
`timeoutSec` (default 120, at most 600). The worker sees them in its brief.
|
|
232
|
+
They run in the step's workspace once the "not produced" check has passed,
|
|
233
|
+
in their own process group, and their full output is saved next to the
|
|
234
|
+
step's task. A check sees `BULLSWARM_EVIDENCE=1`, `BULLSWARM_STEP_ID`,
|
|
235
|
+
`BULLSWARM_STEP_OUTPUT` and `BULLSWARM_RUN_DIR`; `"file": "$output"`
|
|
236
|
+
checks the step's own final response. A check that changes the step's files
|
|
237
|
+
fails; untracked by-products are reported, not failed, and in an isolated
|
|
238
|
+
copy the ones a check created are removed and the ones it changed are put
|
|
239
|
+
back, so they never reach the merge. A failing check
|
|
240
|
+
fails the attempt as `failed-evidence`: Bullswarm retries once on the same
|
|
241
|
+
pool with the check's output attached, and after that the step fails and
|
|
242
|
+
comes back to you (`workflow resume` does not rerun it). A failed check on an
|
|
243
|
+
act step, and a check that cannot run (a missing or unsupported schema),
|
|
244
|
+
come straight back to you, and so does an act step whose checks were
|
|
245
|
+
stopped, because its worker may already have acted. Only the caller declares
|
|
246
|
+
evidence; a dispatched planner cannot, and review steps and digests take
|
|
247
|
+
none. Each check's result is in `runs result <id> --json` under
|
|
248
|
+
`actions[].evidenceResults`; when the worker fails first, no check runs and
|
|
249
|
+
that value is `null`. Schema checks cover the JSON Schema subset listed in
|
|
250
|
+
the program reference; any other keyword is refused, never skipped, and a
|
|
251
|
+
JSONL file with no records fails.
|
|
252
|
+
`node <package>/bin/check-schema.js <file> <schema>` runs the same check by
|
|
253
|
+
hand.
|
|
254
|
+
- workflow: each finished step in a new run says what backs it: `proven by
|
|
255
|
+
command`, `schema` or `review` (a check step passed every requirement the
|
|
256
|
+
step affects), `review pending` while a check step still covers them, or
|
|
257
|
+
`finished · unproven`. `watch`, `runs result` and the
|
|
258
|
+
result summary show it, with a one-line count at the end of the run. Runs
|
|
259
|
+
started before this version show no labels.
|
|
260
|
+
- workflow: an attempt's diff file is named from its task file
|
|
261
|
+
(`diff-<action>-attempt-<n>.txt` beside `task-<action>-attempt-<n>.md`), so
|
|
262
|
+
a rerun writes a new diff and leaves the earlier attempt's diff in place.
|
|
263
|
+
- providers: running out of credit is a usage limit for every provider, not a
|
|
264
|
+
failure. Claude Code's `Credit balance is too low` used to bench the pool as
|
|
265
|
+
a broken sign-in; Codex (`You're out of credits`, `You hit your spend cap`),
|
|
266
|
+
Command Code (`You have insufficient credits`, `Premium credits exhausted`,
|
|
267
|
+
`You've reached today's limit on …`) and OpenCode (`Quota exceeded. Check
|
|
268
|
+
your plan and billing details.`) messages were not recognised at all, and a
|
|
269
|
+
spent Grok Build balance (`API error (status 402 Payment Required): Grok
|
|
270
|
+
Build usage balance exhausted`, in an error event with exit code 0) was
|
|
271
|
+
accepted as a finished reply, so the step failed on its missing report and
|
|
272
|
+
was retried on grok itself. Each is now read as a usage limit; in a
|
|
273
|
+
workflow started by this version the step comes back to you. HTTP 402
|
|
274
|
+
counts as an error-shaped line for every provider. The phrases come from
|
|
275
|
+
each CLI's own strings (claude 2.1.282, codex 0.155.1, command-code 1.65.0,
|
|
276
|
+
opencode 1.18.31).
|
|
277
|
+
- run, workflow: limit and sign-in notices are read only where the provider
|
|
278
|
+
reports them, with their own kind. A long reply that quotes sign-in or limit
|
|
279
|
+
wording (for example a review of the auth checks that mentions
|
|
280
|
+
`unauthorized`) is no longer failed as a sign-in failure or a usage limit:
|
|
281
|
+
Claude Code repeats the whole reply in its final record, and a reply long
|
|
282
|
+
enough to be shortened in the live view was read as the provider's own
|
|
283
|
+
error. A usage-limit or rate-limit notice inside the provider's stream error
|
|
284
|
+
event keeps its limit kind; it used to be read as a provider error.
|
|
285
|
+
- workers bill the account of the pool they run on. When the caller itself ran
|
|
286
|
+
under a Claude home (`CLAUDE_CONFIG_DIR` set), every `claude-code` pool
|
|
287
|
+
spawned with the caller's home instead of its own, so every Claude pool
|
|
288
|
+
billed, and hit the limits of, the caller's one account. A pool's own
|
|
289
|
+
settings now win over the caller's environment; Bullswarm's own
|
|
290
|
+
`BULLSWARM_*` variables still come from the caller.
|
|
291
|
+
- command-code: headless runs no longer pass `--tools-all`. The default tool
|
|
292
|
+
set still reads, writes, searches and runs shell commands. The flag was what
|
|
293
|
+
added `enter_plan_mode`, so a headless run can no longer switch itself into
|
|
294
|
+
plan mode. `--yolo` is unchanged.
|
|
295
|
+
- dashboard: a Run timeline row and the live block label the routing tier as
|
|
296
|
+
`tier <effort>` and, when the attempt recorded one, the applied level as
|
|
297
|
+
`reasoning <level>`. A narrow timeline row keeps the tier and drops the
|
|
298
|
+
model and reasoning; the narrow live block keeps the model and its
|
|
299
|
+
reasoning and drops the tier. The selected timeline row, the live block and
|
|
300
|
+
the Step header list a step's stored not-done items, clipped to the width; a
|
|
301
|
+
collapsed timeline row still shows only `returned early · N not done`.
|
|
302
|
+
- planner: the skill, the program reference, and the program-mode planner
|
|
303
|
+
rules say `dependsOn` is how a writer waits for a file or contract it needs,
|
|
304
|
+
that a behavior and its focused test stay in one action, and that the full
|
|
305
|
+
browser gate, the commit, and the PR are separate steps after integration,
|
|
306
|
+
with an explicit `timeBox` on the browser step.
|
|
307
|
+
- watch: the stale check for a writing step that owns no files reads the
|
|
308
|
+
changed files `git status` lists, including a file edited again through a
|
|
309
|
+
shell command. A step that declares owned files is still scored from those
|
|
310
|
+
files only. When git cannot be read, that check contributes nothing.
|
|
311
|
+
- watch: event mode no longer keeps every new event in memory until the next
|
|
312
|
+
heartbeat. An opted-in heartbeat still prints the event count and the action
|
|
313
|
+
count. Plain `workflow watch <runId>` follows until the outcome.
|
|
314
|
+
|
|
5
315
|
## 0.35.6 — set reasoning per tier and per model from the setup screen
|
|
6
316
|
|
|
7
317
|
- setup: the setup screen (`bullswarm setup`, `strategy tui`) can now set
|
|
@@ -687,7 +997,6 @@ under `docs/design/prototype-frames-0.33.1/` (`budget-55.txt`,
|
|
|
687
997
|
`budget-120.txt`, `home-today-55.txt`, `home-today-120.txt` and their READMEs),
|
|
688
998
|
drawn from the real pool figures of 2026-09-18.
|
|
689
999
|
|
|
690
|
-
|
|
691
1000
|
## 0.33.0 — the dashboard release
|
|
692
1001
|
|
|
693
1002
|
- dashboard: the phone `Home` no longer hides data behind its width. The
|
package/README.md
CHANGED
|
@@ -36,7 +36,7 @@ flowchart LR
|
|
|
36
36
|
|
|
37
37
|
- **Spends quota by pace.** Each task goes to the plan with the most spare quota for how far its window has run, so a plan that is behind, or about to reset with quota left, gets used first.
|
|
38
38
|
- **Picks models for you.** Tasks come in three lanes (`analyze`, `build`, `chore`). Setup asks each CLI which models it offers and suggests the newest one for each effort level, so you don't have to update settings every time a vendor ships a model.
|
|
39
|
-
- **Runs one task or a whole workflow.** `bullswarm run` sends one task to one agent and returns a verdict. For bigger goals your main agent writes a plan; Bullswarm runs the independent steps in parallel across agents, then integration and a
|
|
39
|
+
- **Runs one task or a whole workflow.** `bullswarm run` sends one task to one agent and returns a verdict. For bigger goals your main agent writes a plan; Bullswarm runs the independent steps in parallel across agents, then integration, and a review by a different agent when the plan asks for one.
|
|
40
40
|
- **Checks the work, not the exit code.** A delegate saying "done" isn't enough. Bullswarm reads what it actually produced, and a workflow finishing is kept separate from its requirements being verified.
|
|
41
41
|
- **Shows everything live.** A terminal dashboard covers quota, history, running workflows and each agent's individual turns.
|
|
42
42
|
- **Works with the agent you already use.** The `/bullswarm` skill teaches Claude Code, Codex and Grok when to delegate. There's also an early-access Claude Code Mod that shows runs and usage inside Claude Code.
|
|
@@ -62,7 +62,7 @@ More screens, including phone-sized ones, are in the [gallery](https://bulls-wor
|
|
|
62
62
|
| **One agent's built-in subagents or workflows** | Everything draws on that one plan's quota, the same vendor grades its own work, and your process is tied to that vendor's feature. |
|
|
63
63
|
| **An API gateway that re-exposes your subscriptions** | Turning a consumer subscription into a generic API endpoint can conflict with provider terms, and you lose each agent's own tools and harness. |
|
|
64
64
|
| **Switching tools by hand** | You become the scheduler, and the quota you did not get to still expires. |
|
|
65
|
-
| **Bullswarm** | Drives each vendor's own headless CLI—`claude -p`, `codex exec`, `grok -p`—the way those CLIs are meant to be scripted, with the accounts you already signed in. Work lands where quota is spare, and a different agent
|
|
65
|
+
| **Bullswarm** | Drives each vendor's own headless CLI—`claude -p`, `codex exec`, `grok -p`—the way those CLIs are meant to be scripted, with the accounts you already signed in. Work lands where quota is spare, and a different agent can check it. |
|
|
66
66
|
|
|
67
67
|
Bullswarm never proxies a subscription as an API and never collects or shares your vendor credentials.
|
|
68
68
|
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// check-schema <file> <schema> [--json] [--format json|jsonl] [--unfence]
|
|
3
|
+
// Everything after `--` is positional, so a path may start with '-'.
|
|
4
|
+
// Exit 0 valid, 1 invalid, 2 cannot check (or a usage error). The evidence
|
|
5
|
+
// runner calls this by absolute path; it has no package.json bin entry.
|
|
6
|
+
|
|
7
|
+
import { checkSchemaFiles } from '../src/workflow/schema-check.js';
|
|
8
|
+
|
|
9
|
+
const USAGE = 'usage: check-schema <file> <schema> [--json] [--format json|jsonl] [--unfence]';
|
|
10
|
+
const PRINTED_ERRORS = 20;
|
|
11
|
+
|
|
12
|
+
function parseArgs(argv) {
|
|
13
|
+
const positional = [];
|
|
14
|
+
const options = { json: false, format: undefined, unfence: false };
|
|
15
|
+
for (let index = 0; index < argv.length; index += 1) {
|
|
16
|
+
const arg = argv[index];
|
|
17
|
+
if (arg === '--') { positional.push(...argv.slice(index + 1)); break; }
|
|
18
|
+
if (arg === '--json') options.json = true;
|
|
19
|
+
else if (arg === '--unfence') options.unfence = true;
|
|
20
|
+
else if (arg === '--format') {
|
|
21
|
+
const value = argv[index += 1];
|
|
22
|
+
if (value !== 'json' && value !== 'jsonl') return null;
|
|
23
|
+
options.format = value;
|
|
24
|
+
} else if (arg.startsWith('--format=')) {
|
|
25
|
+
const value = arg.slice('--format='.length);
|
|
26
|
+
if (value !== 'json' && value !== 'jsonl') return null;
|
|
27
|
+
options.format = value;
|
|
28
|
+
} else if (arg.startsWith('-') && arg !== '-') return null;
|
|
29
|
+
else positional.push(arg);
|
|
30
|
+
}
|
|
31
|
+
if (positional.length !== 2) return null;
|
|
32
|
+
return { file: positional[0], schema: positional[1], ...options };
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function main(argv) {
|
|
36
|
+
const args = parseArgs(argv);
|
|
37
|
+
if (!args) {
|
|
38
|
+
if (argv.includes('--json')) {
|
|
39
|
+
process.stdout.write(`${JSON.stringify({ ok: false, exit: 2, errorCount: 0, errors: [], notes: [], why: USAGE, fault: 'check' })}\n`);
|
|
40
|
+
} else {
|
|
41
|
+
process.stderr.write(`${USAGE}\n`);
|
|
42
|
+
}
|
|
43
|
+
return 2;
|
|
44
|
+
}
|
|
45
|
+
let result;
|
|
46
|
+
try {
|
|
47
|
+
result = checkSchemaFiles({ cwd: process.cwd(), file: args.file, schema: args.schema, format: args.format, unfence: args.unfence });
|
|
48
|
+
} catch (error) {
|
|
49
|
+
result = { exit: 2, errorCount: 0, errors: [], notes: [], why: `checker error: ${error.message}`, fault: 'check' };
|
|
50
|
+
}
|
|
51
|
+
if (args.json) {
|
|
52
|
+
process.stdout.write(`${JSON.stringify({
|
|
53
|
+
ok: result.exit === 0,
|
|
54
|
+
exit: result.exit,
|
|
55
|
+
errorCount: result.errorCount,
|
|
56
|
+
errors: result.errors.slice(0, PRINTED_ERRORS),
|
|
57
|
+
notes: result.notes,
|
|
58
|
+
why: result.why,
|
|
59
|
+
fault: result.exit === 2 ? result.fault : null,
|
|
60
|
+
})}\n`);
|
|
61
|
+
return result.exit;
|
|
62
|
+
}
|
|
63
|
+
const lines = [];
|
|
64
|
+
if (result.exit === 0) lines.push(`${args.file} matches ${args.schema}`);
|
|
65
|
+
else if (result.exit === 1) {
|
|
66
|
+
const count = result.why.replace(/^not valid: /, '');
|
|
67
|
+
lines.push(`${args.file} does not match ${args.schema}: ${count}`);
|
|
68
|
+
for (const error of result.errors.slice(0, PRINTED_ERRORS)) lines.push(` ${error}`);
|
|
69
|
+
if (result.errors.length > PRINTED_ERRORS) lines.push(` … and ${result.errors.length - PRINTED_ERRORS} more`);
|
|
70
|
+
} else {
|
|
71
|
+
lines.push(`cannot check ${args.file} against ${args.schema}: ${result.why}`);
|
|
72
|
+
}
|
|
73
|
+
for (const note of result.notes) lines.push(`note: ${note}`);
|
|
74
|
+
(result.exit === 2 ? process.stderr : process.stdout).write(`${lines.join('\n')}\n`);
|
|
75
|
+
return result.exit;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
process.exitCode = main(process.argv.slice(2));
|