bullswarm 0.10.7 → 0.10.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/CHANGELOG.md +37 -0
- package/README.md +21 -5
- package/connectors/claude-code.json +1 -0
- package/package.json +1 -1
- package/skill/SKILL.md +26 -3
- package/src/cli.js +31 -14
- package/src/help.js +1222 -119
- package/src/integrate.js +2 -8
- package/src/lib/release.js +10 -9
- package/src/lib/watch.js +1 -1
- package/src/setup.js +8 -0
- package/src/strategy-cli.js +9 -20
- package/src/workflow/cli.js +33 -26
- package/src/workflow/draft-cli.js +20 -38
- package/src/workflow/goal.js +1 -0
- package/src/workflow/runner.js +120 -21
- package/src/workflow/runs-cli.js +8 -19
- package/src/workflow/runtime.js +21 -2
- package/src/workflow/validate.js +1 -0
- package/src/workflow/watch-cli.js +105 -4
package/AGENTS.md
CHANGED
|
@@ -37,7 +37,7 @@ content. Published as `bullswarm` on npm.
|
|
|
37
37
|
## Development
|
|
38
38
|
|
|
39
39
|
```bash
|
|
40
|
-
npm test #
|
|
40
|
+
npm test # full suite, no network needed (meters read from cache)
|
|
41
41
|
node bin/bullswarm.js doctor --json # readiness report
|
|
42
42
|
node bin/bullswarm.js workflow list # discover workflows
|
|
43
43
|
node bin/bullswarm.js workflow runs # ongoing workflow instances
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,42 @@
|
|
|
1
1
|
# bullswarm changelog
|
|
2
2
|
|
|
3
|
+
## 0.10.9 — planner self-correction and honest silence
|
|
4
|
+
|
|
5
|
+
- An invalid or non-JSON orchestrator decision no longer fails the run. The
|
|
6
|
+
runtime feeds the exact validator issues, the rejected proposal, and a
|
|
7
|
+
response excerpt back to the same orchestrator thread for bounded corrective
|
|
8
|
+
turns (`settings.maxPlannerCorrections`, default 2, emits
|
|
9
|
+
`decision.correction_requested`), then benches that pool as orchestrator for
|
|
10
|
+
the rest of the run and escalates to one other eligible pool
|
|
11
|
+
(`decision.orchestrator_escalated`), and only then settles on a qualified
|
|
12
|
+
`completed_with_concerns`/`blocked` outcome. The decide action's ledger status
|
|
13
|
+
now agrees with that outcome (`failed_retryable` while correcting,
|
|
14
|
+
`failed_terminal` on exhaustion) instead of reporting `succeeded`.
|
|
15
|
+
- The planner prompt now shows complete `run`, `fanout` (`items` +
|
|
16
|
+
`stepTemplate.prompt` with `{{item}}`), and `verify` (`review`) skeletons, so
|
|
17
|
+
the first proposal can match what `decision.js` validates.
|
|
18
|
+
- `workflow watch` separates two silences: `quiet` counts durable workflow
|
|
19
|
+
events, and a new `agent output … ago` figure (JSONL `transportQuietForSec`)
|
|
20
|
+
counts raw output from live agents, so a thinking agent and a dead one look
|
|
21
|
+
different on the same heartbeat line.
|
|
22
|
+
- Removed the dead thin-leaf help renderer left over from the help unification
|
|
23
|
+
and stopped hard-coding the test count in AGENTS.md.
|
|
24
|
+
|
|
25
|
+
## 0.10.8 — quieter monitoring and resilient orchestration
|
|
26
|
+
|
|
27
|
+
- Made `workflow watch` aggregate low-level activity into compact interval
|
|
28
|
+
heartbeats by default, with event/action deltas, quiet duration, prompt
|
|
29
|
+
semantic transitions, terminal result handoff, and `--verbose` drill-down.
|
|
30
|
+
- Implemented the documented top-level `run --prompt` form and standardized
|
|
31
|
+
usage errors as exit 2 across run, workflow drafts/runs, and strategy paths.
|
|
32
|
+
- Added explicit goal-resume orchestrator pin/unpin behavior and strengthened
|
|
33
|
+
the orchestrator as a control-plane-only decision thread.
|
|
34
|
+
- Recognize Claude's exact `Failed to authenticate` response as an auth failure
|
|
35
|
+
and migrate the connector signature additively without replacing local
|
|
36
|
+
connector customization.
|
|
37
|
+
- Expanded non-network CLI, health, release, watch, resume, auth, help, and
|
|
38
|
+
documentation coverage. The full suite now contains 271 tests.
|
|
39
|
+
|
|
3
40
|
## 0.10.7 — clearer orchestration overview
|
|
4
41
|
|
|
5
42
|
- Replaced the orchestrator trace dump with a summary-first view showing what
|
package/README.md
CHANGED
|
@@ -67,6 +67,7 @@ bullswarm setup # re-run or repair
|
|
|
67
67
|
bullswarm pools # meter state, pace position, quarantine status
|
|
68
68
|
bullswarm strategy refresh --apply --yes # approve capability-aware tier autopilot
|
|
69
69
|
bullswarm run --lane analyze --add-dir ~/some-repo --task-file /tmp/t.md --json
|
|
70
|
+
bullswarm run --lane analyze --add-dir ~/some-repo --prompt "Inspect the parser" --json
|
|
70
71
|
bullswarm workflow goal "Fix the failing tests and verify the change" --cwd ~/some-repo
|
|
71
72
|
bullswarm health # re-judge saved outputs; catch gate failures
|
|
72
73
|
```
|
|
@@ -84,6 +85,17 @@ bullswarm health # re-judge saved outputs; catch gate failures
|
|
|
84
85
|
| `doctor` | Machine-readable readiness report; self-heals on first call |
|
|
85
86
|
| `workflow` | Start an autonomous goal, or run / validate / draft / inspect explicit workflows and their live instances. |
|
|
86
87
|
|
|
88
|
+
Discover and validate workflow definitions without executing them:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
bullswarm workflow list
|
|
92
|
+
bullswarm workflow list --json
|
|
93
|
+
bullswarm workflow validate workflows/my-workflow.json
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
`workflow goal --request <path>` and `--run-id <id>` are internal detached-runner
|
|
97
|
+
resume plumbing. Normal callers should provide a goal or use `--resume <shortId|runId>`.
|
|
98
|
+
|
|
87
99
|
## Model strategy and invocation telemetry
|
|
88
100
|
|
|
89
101
|
Bullswarm can inventory the models exposed by installed agent CLIs and combine
|
|
@@ -260,16 +272,20 @@ auditing completed runs.
|
|
|
260
272
|
|
|
261
273
|
### Live workflow dashboard
|
|
262
274
|
|
|
263
|
-
For ordinary observation, use the non-interactive watcher.
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
275
|
+
For ordinary observation, use the non-interactive watcher. Human output is one
|
|
276
|
+
compact aggregate line per semantic change and a 60-second heartbeat while
|
|
277
|
+
otherwise quiet. Each line reports status/location, events and agent actions
|
|
278
|
+
captured since the preceding sample, and quiet duration. It does not repeat
|
|
279
|
+
command or response excerpts. Use `--verbose` for the detailed per-agent and
|
|
280
|
+
last-action view. Compact terminal output reports the overall attempt count and
|
|
281
|
+
elapsed time; `--verbose` includes every attempt's agent/model, outcome, and
|
|
282
|
+
tokens. This keeps agent monitoring cheap while retaining a drill-down path.
|
|
268
283
|
|
|
269
284
|
```bash
|
|
270
285
|
bullswarm workflow watch <shortId>
|
|
271
286
|
bullswarm workflow watch <shortId> --jsonl # automation-friendly stream
|
|
272
287
|
bullswarm workflow watch <shortId> --once # one current/terminal snapshot
|
|
288
|
+
bullswarm workflow watch <shortId> --verbose # detailed agent/action view
|
|
273
289
|
```
|
|
274
290
|
|
|
275
291
|
`workflow tui` is the interactive, Claude-style `/workflows` view. For an
|
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -59,6 +59,10 @@ bullswarm run --lane <analyze|build|chore> \
|
|
|
59
59
|
--add-dir <abs/path/to/repo> \
|
|
60
60
|
--task-file <abs/path/to/task.md> \
|
|
61
61
|
--json
|
|
62
|
+
|
|
63
|
+
# Equivalent explicit prompt form (also accepts trailing task text)
|
|
64
|
+
bullswarm run --lane analyze --add-dir <abs/path/to/repo> \
|
|
65
|
+
--prompt "Inspect and verify the parser" --json
|
|
62
66
|
```
|
|
63
67
|
|
|
64
68
|
The verdict shape:
|
|
@@ -202,8 +206,15 @@ bullswarm workflow runs --historical --since yesterday --until today
|
|
|
202
206
|
bullswarm workflow runs show <shortId> # state + report + summary
|
|
203
207
|
bullswarm workflow runs result <shortId> --json # stable delivery for the caller
|
|
204
208
|
bullswarm workflow runs delete <shortId> --yes
|
|
209
|
+
|
|
210
|
+
bullswarm workflow list
|
|
211
|
+
bullswarm workflow validate workflows/demo.json
|
|
205
212
|
```
|
|
206
213
|
|
|
214
|
+
`workflow goal --request <path>` and `--run-id <id>` are internal detached-runner
|
|
215
|
+
plumbing. Ordinary callers should provide a goal or resume with
|
|
216
|
+
`--resume <shortId|runId>`.
|
|
217
|
+
|
|
207
218
|
Historical time ranges filter the workflow's initiation time (`startedAt`),
|
|
208
219
|
not completion time. `--since` is inclusive and `--until` is exclusive, with
|
|
209
220
|
`--from`/`--to` and `--started-after`/`--started-before` aliases. Bounds accept
|
|
@@ -279,9 +290,13 @@ reason, and child-process termination evidence. `workflow tui` displays the
|
|
|
279
290
|
same information interactively. `workflow events` supports replay after a
|
|
280
291
|
monotonic sequence cursor.
|
|
281
292
|
|
|
282
|
-
Prefer `workflow watch` for ordinary monitoring;
|
|
283
|
-
|
|
284
|
-
|
|
293
|
+
Prefer `workflow watch` for ordinary monitoring; default human text is a compact
|
|
294
|
+
aggregate heartbeat with status/location, interval event and agent-action counts,
|
|
295
|
+
and quiet duration. It does not repeat command or response excerpts. Add
|
|
296
|
+
`--verbose` for the detailed agent and last-action view. `--jsonl` keeps the
|
|
297
|
+
machine-readable snapshot stream. Compact terminal text includes overall timing
|
|
298
|
+
and the exact `workflow runs result <shortId> --json` next command; `--verbose`
|
|
299
|
+
adds per-attempt timing and token details. `workflow steer` is an optional durable queue for autonomous workflows:
|
|
285
300
|
it never changes the active worker and is delivered only to the next
|
|
286
301
|
not-yet-started decision gate. Steering cannot expand the original authority,
|
|
287
302
|
weaken verification, or bypass proposal validation.
|
|
@@ -320,6 +335,14 @@ execute -> observe -> decide -> validate proposal -> append -> execute -> observ
|
|
|
320
335
|
Allowed planner decisions are `proceed`, `complete`, `needs_more_work`,
|
|
321
336
|
`retry`, `escalate`, `wait_for_approval`, and `stop`. Expansion decisions must
|
|
322
337
|
contain bounded actions; malformed or over-budget output executes nothing.
|
|
338
|
+
A malformed or non-JSON decision does not fail the run: the runtime returns
|
|
339
|
+
the exact validator issues to the same orchestrator thread for up to
|
|
340
|
+
`settings.maxPlannerCorrections` (default 2) corrective turns
|
|
341
|
+
(`decision.correction_requested` events), then benches that pool as
|
|
342
|
+
orchestrator for the rest of the run and tries one other eligible pool
|
|
343
|
+
(`decision.orchestrator_escalated`), and only then settles on a qualified
|
|
344
|
+
`completed_with_concerns`/`blocked` outcome. Safety bounds (`maxActions`,
|
|
345
|
+
`maxItemsPerExpansion`) still terminate immediately with a qualified outcome.
|
|
323
346
|
`complete` still requires successful delivery and verification. `stop` returns
|
|
324
347
|
`completed_with_concerns` with a ready best-effort artifact when useful work
|
|
325
348
|
exists, or `blocked` when it does not; neither hides failed verification.
|
package/src/cli.js
CHANGED
|
@@ -17,7 +17,7 @@ import { release } from './lib/release.js';
|
|
|
17
17
|
import { cmdWorkflow } from './workflow/cli.js';
|
|
18
18
|
import { cmdStrategy, maybeRefreshStrategy } from './strategy-cli.js';
|
|
19
19
|
import { cmdIntegrate, installIntegration } from './integrate.js';
|
|
20
|
-
import { helpForArgs } from './help.js';
|
|
20
|
+
import { helpForArgs, usageLine } from './help.js';
|
|
21
21
|
import { resolveDispatchModel } from './lib/strategy.js';
|
|
22
22
|
|
|
23
23
|
export function getBullswarmDir() {
|
|
@@ -96,6 +96,34 @@ async function cmdRun(opts) {
|
|
|
96
96
|
}
|
|
97
97
|
const targetDir = resolve(opts['add-dir'] ?? process.cwd());
|
|
98
98
|
|
|
99
|
+
// Validate task input before routing so malformed invocations never fall
|
|
100
|
+
// through to the caller-session path.
|
|
101
|
+
if (opts['task-file'] === true || opts.prompt === true ||
|
|
102
|
+
(opts['task-file'] != null && typeof opts['task-file'] !== 'string') ||
|
|
103
|
+
(opts.prompt != null && typeof opts.prompt !== 'string')) {
|
|
104
|
+
console.error('usage: --prompt and --task-file require a value');
|
|
105
|
+
return 2;
|
|
106
|
+
}
|
|
107
|
+
if (opts['task-file'] && opts.prompt != null) {
|
|
108
|
+
console.error('usage: choose one of --prompt, --task-file, or trailing task text');
|
|
109
|
+
return 2;
|
|
110
|
+
}
|
|
111
|
+
if (opts['task-file'] && opts.rest.length) {
|
|
112
|
+
console.error('usage: choose one of --task-file or trailing task text');
|
|
113
|
+
return 2;
|
|
114
|
+
}
|
|
115
|
+
if (opts.prompt != null && opts.rest.length) {
|
|
116
|
+
console.error('usage: choose one of --prompt or trailing task text');
|
|
117
|
+
return 2;
|
|
118
|
+
}
|
|
119
|
+
const taskText = opts['task-file']
|
|
120
|
+
? readFileSync(opts['task-file'], 'utf8')
|
|
121
|
+
: opts.prompt ?? opts.rest.join(' ');
|
|
122
|
+
if (!taskText.trim()) {
|
|
123
|
+
console.error('empty task: pass --task-file, --prompt, or the task as arguments');
|
|
124
|
+
return 2;
|
|
125
|
+
}
|
|
126
|
+
|
|
99
127
|
// Recursion guard FIRST — core-owned, env handshake.
|
|
100
128
|
let state = loadState(getBullswarmDir());
|
|
101
129
|
try {
|
|
@@ -166,15 +194,6 @@ async function cmdRun(opts) {
|
|
|
166
194
|
subscription: poolView.subscription ?? connector.subscription ?? null,
|
|
167
195
|
};
|
|
168
196
|
|
|
169
|
-
// Task text: --task-file content or stdin string.
|
|
170
|
-
const taskText = opts['task-file']
|
|
171
|
-
? readFileSync(opts['task-file'], 'utf8')
|
|
172
|
-
: opts.rest.join(' ');
|
|
173
|
-
if (!taskText.trim()) {
|
|
174
|
-
console.error('empty task: pass --task-file or the task as arguments');
|
|
175
|
-
return 2;
|
|
176
|
-
}
|
|
177
|
-
|
|
178
197
|
const stamp = `${Date.now()}-${Math.random().toString(36).slice(2, 7)}`;
|
|
179
198
|
const runDir = join(getBullswarmDir(), 'runs');
|
|
180
199
|
mkdirSync(runDir, { recursive: true });
|
|
@@ -455,9 +474,7 @@ export async function main(argv) {
|
|
|
455
474
|
case 'release':
|
|
456
475
|
return cmdRelease(opts);
|
|
457
476
|
default:
|
|
458
|
-
console.error(
|
|
459
|
-
`unknown verb "${verb}". try: setup | integrate | run | health | pools | strategy | doctor | workflow | runs | version | release`,
|
|
460
|
-
);
|
|
477
|
+
console.error(`unknown verb "${verb}". Run "bullswarm --help" for the list of commands.`);
|
|
461
478
|
return 2;
|
|
462
479
|
}
|
|
463
480
|
}
|
|
@@ -467,7 +484,7 @@ export async function main(argv) {
|
|
|
467
484
|
function cmdRelease(opts) {
|
|
468
485
|
const kind = opts.rest[0];
|
|
469
486
|
if (!['patch', 'minor', 'major'].includes(kind)) {
|
|
470
|
-
console.error(
|
|
487
|
+
console.error(`usage: ${usageLine(['release'])}`);
|
|
471
488
|
return 2;
|
|
472
489
|
}
|
|
473
490
|
try {
|