bullswarm 0.38.3 → 0.38.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/docs/guide/observing.md +2 -2
- package/docs/guide/routing.md +2 -2
- package/docs/guide/workflows.md +1 -0
- package/docs/reference/cli.md +1 -1
- package/docs/reference/program.md +1 -1
- package/package.json +1 -1
- package/skill/SKILL.md +12 -4
- package/skill/references/patterns.md +2 -0
- package/skill/references/program.md +5 -2
- package/skill/references/recovery.md +5 -1
- package/src/help.js +12 -5
- package/src/lib/cli-flags.js +1 -1
- package/src/workflow/cli-plan.js +12 -1
- package/src/workflow/cli-steps.js +1 -1
- package/src/workflow/cli.js +1 -1
- package/src/workflow/contract-v3.js +1 -1
- package/src/workflow/control-nodes.js +47 -0
- package/src/workflow/dashboard.js +3 -21
- package/src/workflow/format-duration.js +12 -0
- package/src/workflow/gates-loops.js +3 -42
- package/src/workflow/home-view.js +7 -19
- package/src/workflow/legacy-run-state.js +7 -0
- package/src/workflow/needs-you.js +1 -1
- package/src/workflow/plan-try-checks.js +160 -0
- package/src/workflow/reprice-cli.js +168 -0
- package/src/workflow/reprice.js +3 -164
- package/src/workflow/run-model.js +2 -1
- package/src/workflow/run-view.js +8 -32
- package/src/workflow/runs-view.js +3 -1
- package/src/workflow/short-id.js +2 -7
- package/src/workflow/step-model.js +3 -7
- package/src/workflow/step-view.js +1 -1
- package/src/workflow/v2-cancellation.js +1 -4
- package/src/workflow/v2-revision.js +1 -1
- package/src/workflow/v3-timeline.js +2 -1
- package/src/workflow/watch-cli.js +3 -11
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
## 0.38.6 — no import cycles
|
|
6
|
+
|
|
7
|
+
- internal: no module under `src/` imports itself back through its imports any more (13 import cycles removed by moving four small helpers into their own files; `reprice`'s command-line part is now `src/workflow/reprice-cli.js`). Behaviour is unchanged; a new test fails if a cycle comes back.
|
|
8
|
+
|
|
9
|
+
## 0.38.5 — plan validate --try-checks (0.38.4 republished)
|
|
10
|
+
|
|
11
|
+
- release: 0.38.4 was tagged but never published to npm (a test read only the `## Unreleased` changelog section, which the release empties); 0.38.5 ships 0.38.4's changes. The docs site builds again: three wrapped lines that began with a `<placeholder>` were read as HTML.
|
|
12
|
+
|
|
13
|
+
## 0.38.4 — plan validate --try-checks
|
|
14
|
+
|
|
15
|
+
- plan validate: `--try-checks` runs each step's command check once, now,
|
|
16
|
+
against the current tree, the way a step runs it (same shell, folder,
|
|
17
|
+
environment and timeout), and prints one `try` line per check with its exit
|
|
18
|
+
and last output line (or `timed out after <n>s`) and the names of any files
|
|
19
|
+
it changed, ignored files and folders outside git included. A check that
|
|
20
|
+
exits non-zero or times out also prints the last 20 lines of its output,
|
|
21
|
+
indented under the try line, so the error is visible. A check that reads the step's output is not tried, and the results never
|
|
22
|
+
change the exit code. Without the flag validate still runs nothing; when the
|
|
23
|
+
program has command checks it now says so (`checks not run · add
|
|
24
|
+
--try-checks …`, and `checks: {tried: false, commands}` in `--json`). Use it
|
|
25
|
+
for checks that are safe to run now: a missing module or command on a `try`
|
|
26
|
+
line means the check itself is wrong.
|
|
27
|
+
- docs: `workflow step accept` on a failed step inside a loop does not end the
|
|
28
|
+
loop; to stop a loop early let it run out of rounds, then
|
|
29
|
+
`workflow continue <run> <loop>` (recorded as condition not met).
|
|
30
|
+
|
|
5
31
|
## 0.38.3 — Blind reviews and a reorganized skill
|
|
6
32
|
|
|
7
33
|
- steps: a v3 step may declare `"blindTo": ["<step id>", ...]`, naming steps
|
package/docs/guide/observing.md
CHANGED
|
@@ -264,8 +264,8 @@ A real run on grok, parked at its gate, on the Run page at 120 columns:
|
|
|
264
264
|
continued without more rounds reads `→ continued by the caller after N of
|
|
265
265
|
max rounds (condition not met)`: it never passed. Each attempt of a
|
|
266
266
|
loop step says its round (`write · round 2`).
|
|
267
|
-
- **A gate row** heads the phase of the steps behind it and reads
|
|
268
|
-
<steps>`, `waiting for you · <note>` with its `continue` command, `passed ·
|
|
267
|
+
- **A gate row** heads the phase of the steps behind it and reads
|
|
268
|
+
`waits after <steps>`, `waiting for you · <note>` with its `continue` command, `passed ·
|
|
269
269
|
continued by the caller`, or `skipped · <condition> does not hold` when its
|
|
270
270
|
`when` condition did not hold.
|
|
271
271
|
- **Answers.** `answer {…}` is the checked answer (it matched the step's schema);
|
package/docs/guide/routing.md
CHANGED
|
@@ -198,8 +198,8 @@ well: as `quota` when every reason is a usage limit, else as `unavailable`. Its
|
|
|
198
198
|
spare: pool-a at its 5-hour limit until <time>; pool-b at its weekly limit
|
|
199
199
|
until <time>`, and `back at` is the earliest known return among them. A retry
|
|
200
200
|
the step was promised (after a crash, a sign-in failure or a failed gate) that
|
|
201
|
-
finds no free pool keeps its own failure and ends its `why` with
|
|
202
|
-
<pool> <reason>; …`. When another pool that can run the step is free, routing
|
|
201
|
+
finds no free pool keeps its own failure and ends its `why` with
|
|
202
|
+
`· no retry: <pool> <reason>; …`. When another pool that can run the step is free, routing
|
|
203
203
|
picks it as usual.
|
|
204
204
|
|
|
205
205
|
A run saved by 0.37.x may end with `the workflow planner stopped on a usage
|
package/docs/guide/workflows.md
CHANGED
|
@@ -108,6 +108,7 @@ bullswarm workflow plan validate "Make the acme tests pass" --cwd=/private/tmp/v
|
|
|
108
108
|
✓ program v3 valid: 2 steps, 0 gates, 1 loop (nothing launched)
|
|
109
109
|
fix build/medium deliverable=files
|
|
110
110
|
check analyze/medium evidence=command answer after fix
|
|
111
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
111
112
|
loop until-green steps fix, check · until check's evidence passed · at most 3 rounds
|
|
112
113
|
launch bullswarm workflow goal 'Make the acme tests pass' --cwd /private/tmp/v37fix/acme --program /private/tmp/v37fix/loop.json --json
|
|
113
114
|
```
|
package/docs/reference/cli.md
CHANGED
|
@@ -255,7 +255,7 @@ Trailing `<task text...>` is mutually exclusive with `--prompt` and `--task-file
|
|
|
255
255
|
| `--avoid-provider <provider,...>` | never route to pools of these providers | unset |
|
|
256
256
|
| `--dry-run` | print the kernel's routing pick, the forecast, and the exact command that would be spawned, without spawning, recording a run, registering an assignment, or writing the decision log | off (dispatches for real) |
|
|
257
257
|
| `--no-caller` | accepted and ignored for one release: the calling agent is never a pool of its own run | removed in 0.37.0 |
|
|
258
|
-
| `--json` | print the compact machine-readable verdict; top-level `pool` and `model` name who ran the last attempt (null when nothing was dispatched; `pick` keeps the same two beside the command); its last field, details, is the command for the full record and per-attempt cost (bullswarm workflow runs result <shortId> --json) | human-readable summary ending with the run id and that command |
|
|
258
|
+
| `--json` | print the compact machine-readable verdict; top-level `pool` and `model` name who ran the last attempt (null when nothing was dispatched; `pick` keeps the same two beside the command); its last field, details, is the command for the full record and per-attempt cost (`bullswarm workflow runs result <shortId> --json`) | human-readable summary ending with the run id and that command |
|
|
259
259
|
|
|
260
260
|
A run is a one-step workflow, recorded under `workflows/<id>/` (`bullswarm workflow runs --all`). It gets the step's one automatic retry unless `--no-retry`; a usage limit exits 1 with no retry, and the pool's meter is read again at once, so a window it shows at 100% keeps the pool out of later picks until that window resets. Nothing else about a failed pool is remembered. A build or chore run must change a file (else failure kind `not-produced`). The JSON shape is in [Result envelope](/reference/result).
|
|
261
261
|
|
|
@@ -367,7 +367,7 @@ Checks receive `BULLSWARM_EVIDENCE=1`, `BULLSWARM_STEP_ID`, `BULLSWARM_STEP_OUTP
|
|
|
367
367
|
|
|
368
368
|
Checks must be read-only. Before the first item and after each one, Bullswarm hashes the deliverable: the step's `ownedFiles`, declared deliverable paths and saved response, plus every tracked file and HEAD in an isolated copy or for a build or chore step with no `ownedFiles` (it runs alone). A change fails the item with `changed the deliverable: …`, and the later items do not run. Untracked by-products there are recorded as `touched` rather than failed; declare a new file as a deliverable path when it must be protected. In an isolated copy, after the last item, Bullswarm removes the files the checks created and puts back untracked files they rewrote or deleted (up to 16 MB in total), so none of them is merged back. One it cannot put back is named in an attempt note, `check by-product not restored: <paths>`, and is left out of the ownership check and the merge-back. Elsewhere HEAD is not compared: when it moves while an item runs (another step may have committed), the item records `headMoved: true` as a fact and does not fail.
|
|
369
369
|
|
|
370
|
-
Scope a command to its step. Put a whole-suite command on a step that runs alone or last, and pass `--run` or `CI=1` yourself when a test runner watches files. Run every check once
|
|
370
|
+
Scope a command to its step. Put a whole-suite command on a step that runs alone or last, and pass `--run` or `CI=1` yourself when a test runner watches files. Run every check once before launch (`workflow plan validate --try-checks` runs each command check once in the workspace, for checks that are safe to run now) and give it a generous timeout: changing a wrong check amends the step and reruns its worker. A suite that runs longer than 600 seconds cannot be one item: split it into several items or keep it in the step's prompt. To prove finished work without rerunning it, add a `check` step with its own `evidence`; it does not rerun the work. An old kernel refuses `evidence`; pause, revise, then resume.
|
|
371
371
|
|
|
372
372
|
Each check's result is in the full `bullswarm workflow runs result <id> --json` envelope (not `--summary`) under `actions[].evidenceResults`: `status`, `exit`, `tail`, `why` and the log path. `bullswarm workflow action show <id> <step>` shows the same for each attempt under `attempts[].evidenceResults`. Each item's full output is in `evidence-<step>-attempt-<n>-<k>.log` in the run directory.
|
|
373
373
|
|
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -90,6 +90,7 @@ validate:
|
|
|
90
90
|
✓ program v3 valid: 2 steps, 0 gates, 0 loops (nothing launched)
|
|
91
91
|
build build/medium deliverable=files evidence=command
|
|
92
92
|
review analyze/medium answer after build route: independent of build blind to build
|
|
93
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
93
94
|
```
|
|
94
95
|
|
|
95
96
|
Keep the goal in a file and pass it as `"$(cat goal.txt)"` to both commands,
|
|
@@ -121,7 +122,11 @@ it a `--timeout` under your tool's time limit (`--until trouble --timeout 100`
|
|
|
121
122
|
for 2 minutes). A restart without `--after` attaches at the newest event and
|
|
122
123
|
skips wakes in between. Never end your turn while a run you own is still
|
|
123
124
|
running. At a gate: `bullswarm workflow continue <shortId> <gate>`; a loop out
|
|
124
|
-
of rounds takes `--rounds <1-5>`. `
|
|
125
|
+
of rounds takes `--rounds <1-5>`. `workflow step accept` on a failed step
|
|
126
|
+
inside a loop does not end the loop (the next round still starts); to stop a
|
|
127
|
+
loop early let it run out of rounds (`maxRounds`), then
|
|
128
|
+
`workflow continue <shortId> <loop>`, recorded as condition not met.
|
|
129
|
+
`watch --until trouble` also wakes on
|
|
125
130
|
`steering received` (a person left guidance: decide what it means and add
|
|
126
131
|
steps).
|
|
127
132
|
|
|
@@ -145,9 +150,12 @@ Lessons from real runs; each holds for this version.
|
|
|
145
150
|
the other. Give each writer its own files and tell it to keep other
|
|
146
151
|
workers' edits; keep a breaking rename in one step, not parallel with its
|
|
147
152
|
consumers, which would build against the old name.
|
|
148
|
-
5. **Checks are facts.** Put anything a machine can say in `evidence
|
|
149
|
-
|
|
150
|
-
|
|
153
|
+
5. **Checks are facts.** Put anything a machine can say in `evidence`. When
|
|
154
|
+
the checks are safe to run now (not ones that write, deploy, call paid
|
|
155
|
+
services or take long), run `bullswarm workflow plan validate … --try-checks`
|
|
156
|
+
before launch and read each `try` line: an error such as a missing module
|
|
157
|
+
or command means the check itself is wrong, and fixing a wrong check later
|
|
158
|
+
reruns the worker. Workers share one tree, so a later step can undo what an earlier
|
|
151
159
|
step's check proved; put the final check where nothing runs after it
|
|
152
160
|
(after integration, or on the step a gate waits behind).
|
|
153
161
|
6. **Reviews that hold.** State the contract as numbered checks, make any
|
|
@@ -122,6 +122,7 @@ validate:
|
|
|
122
122
|
✓ program v3 valid: 2 steps, 0 gates, 1 loop (nothing launched)
|
|
123
123
|
fix build/medium deliverable=files
|
|
124
124
|
check analyze/medium evidence=command answer after fix
|
|
125
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
125
126
|
loop until-green steps fix, check · until check's evidence passed · at most 3 rounds
|
|
126
127
|
```
|
|
127
128
|
|
|
@@ -258,6 +259,7 @@ validate (plan and the added steps as one program):
|
|
|
258
259
|
slice-writer build/medium deliverable=files after pick-slices
|
|
259
260
|
slice-command build/medium deliverable=files after pick-slices
|
|
260
261
|
check analyze/medium deliverable=report evidence=command after slice-writer, slice-command
|
|
262
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
261
263
|
gate pick-slices after plan · waits for you · Read the slices, add one build step per slice and a check, then continue
|
|
262
264
|
```
|
|
263
265
|
|
|
@@ -318,8 +318,11 @@ Rules the fields above do not show on their own:
|
|
|
318
318
|
tell it to keep other workers' edits.
|
|
319
319
|
- **Evidence: checks Bullswarm runs.** Add a command or schema check for
|
|
320
320
|
anything a machine can check (`"evidence": [{"type": "command", "cmd": "npm
|
|
321
|
-
test", "timeoutSec": 300}]`, at most 5 items).
|
|
322
|
-
launch
|
|
321
|
+
test", "timeoutSec": 300}]`, at most 5 items). Try the checks before
|
|
322
|
+
launch with `bullswarm workflow plan validate … --try-checks` when they are
|
|
323
|
+
safe to run now (it runs each command check once in the workspace and
|
|
324
|
+
prints a `try` line; a check that reads `$output` is not tried), because
|
|
325
|
+
fixing a wrong check reruns the worker.
|
|
323
326
|
- **Steps that must not repeat.** Sending or publishing is `"deliverable":
|
|
324
327
|
"outward"` with `"retry": 0`; an outward step is never retried once its
|
|
325
328
|
worker started.
|
|
@@ -236,7 +236,11 @@ finishes: the handback").
|
|
|
236
236
|
**An accept is a choice, never proof.** `step accept` lets the step's
|
|
237
237
|
dependents run, and the step reads `accepted by choice` and is counted apart
|
|
238
238
|
in the proof line (`N accepted by choice: <steps>`). Report it as your
|
|
239
|
-
decision, never as verification.
|
|
239
|
+
decision, never as verification. `step accept` on a failed step inside a
|
|
240
|
+
loop does not end the loop (the next round still starts); to stop a loop
|
|
241
|
+
early let it run out of rounds (`maxRounds`), then
|
|
242
|
+
`bullswarm workflow continue <shortId> <loop>`, which is recorded as
|
|
243
|
+
condition not met.
|
|
240
244
|
|
|
241
245
|
## Other stopping rules
|
|
242
246
|
|
package/src/help.js
CHANGED
|
@@ -1090,18 +1090,19 @@ const workflowPlanText = rich({
|
|
|
1090
1090
|
});
|
|
1091
1091
|
|
|
1092
1092
|
const workflowPlanValidateText = rich({
|
|
1093
|
-
usage: 'bullswarm workflow plan validate "<goal>" --program <file.json> [--cwd <dir>] [--summary <text>] [--json]',
|
|
1093
|
+
usage: 'bullswarm workflow plan validate "<goal>" --program <file.json> [--cwd <dir>] [--summary <text>] [--try-checks] [--json]',
|
|
1094
1094
|
purpose: 'Check a program you authored against the exact contract a launch would enforce, without '
|
|
1095
1095
|
+ 'creating a run: the same validator and the same preview state as '
|
|
1096
1096
|
+ 'workflow goal --program. Exit 0 prints the accepted actions and the launch line; exit 2 prints '
|
|
1097
1097
|
+ 'every validator issue so you can fix the file and re-run. A bullswarm.workflow.program.v2 is refused '
|
|
1098
|
-
+ '(exit 2), as a launch refuses it.',
|
|
1098
|
+
+ '(exit 2), as a launch refuses it. Command checks are not run unless you add --try-checks.',
|
|
1099
1099
|
args: [{ name: '"<goal>"', desc: 'the goal text exactly as it will be passed to workflow goal' }],
|
|
1100
1100
|
options: [
|
|
1101
1101
|
{ flag: '--program <file.json>', desc: 'a bare bullswarm.workflow.program.v3 document, or one inside a planner response envelope', default: 'required' },
|
|
1102
1102
|
{ flag: '--cwd <dir>', desc: 'working directory the goal will execute in (must exist)', default: 'current directory' },
|
|
1103
1103
|
{ flag: '--summary <text>', desc: 'one-line summary recorded for a bare program document', default: 'derived from the action purposes' },
|
|
1104
|
-
{ flag: '--
|
|
1104
|
+
{ flag: '--try-checks', desc: 'run each step\'s command check once now, in --cwd, the way a step runs it, and print a try line per check (exit, last output line when it passes, and any files it changed; a check that fails or times out also shows the last 20 lines of its output); a check that reads the step\'s output is not tried. Only for checks that are safe to run now: it may take time, touch the network or cost money, and a check that writes changes your tree (nothing is restored). The results never change the exit code', default: 'off (checks are counted, not run)' },
|
|
1105
|
+
{ flag: '--json', desc: 'print the acceptance document ({action: "plan-valid", requirements, program, checks?, next}) or the refusal ({error: "program-invalid", issues, next}) as JSON', default: 'human summary' },
|
|
1105
1106
|
{ flag: '--isolation', desc: 'validate against strict per-worker worktree isolation', default: 'off (shared workspace)' },
|
|
1106
1107
|
{ flag: '--worker-pool <pool|auto>', desc: 'pin the worker pool the preview routes with', default: 'auto (routing decides per action)' },
|
|
1107
1108
|
{ flag: '--worker-model <model|auto>', desc: 'pin the worker model the preview routes with', default: 'auto' },
|
|
@@ -1112,8 +1113,14 @@ const workflowPlanValidateText = rich({
|
|
|
1112
1113
|
{ flag: '--concurrency <n>', desc: 'execution concurrency for the previewed run', default: '4' },
|
|
1113
1114
|
{ flag: '--retry-attempts <0..3>', desc: 'automatic retries per step before it comes back to you (process failures on another pool, gate failures on the same pool with the failure attached)', default: '1' },
|
|
1114
1115
|
],
|
|
1115
|
-
safety: [
|
|
1116
|
-
|
|
1116
|
+
safety: [
|
|
1117
|
+
'read-only without --try-checks — nothing is launched, dispatched, or written; the exit code is the verdict (0 valid, 2 invalid, 1 bad cwd)',
|
|
1118
|
+
'with --try-checks it runs your command checks in --cwd (never a worker, never in the run home); whatever they do, they do: use it only for checks that are safe to run now',
|
|
1119
|
+
],
|
|
1120
|
+
examples: [
|
|
1121
|
+
{ cmd: 'bullswarm workflow plan validate "1. Fix the parser. 2. Update the docs." --cwd . --program plan.json --json' },
|
|
1122
|
+
{ cmd: 'bullswarm workflow plan validate "Fix the parser" --cwd . --program plan.json --try-checks', note: 'a missing module or command on a try line means the check itself is wrong' },
|
|
1123
|
+
],
|
|
1117
1124
|
next: 'bullswarm workflow goal "<same goal>" --cwd <dir> --program plan.json once it validates.',
|
|
1118
1125
|
});
|
|
1119
1126
|
|
package/src/lib/cli-flags.js
CHANGED
|
@@ -146,7 +146,7 @@ const TABLE = {
|
|
|
146
146
|
'concurrency', 'retry-attempts',
|
|
147
147
|
],
|
|
148
148
|
'workflow plan validate': [
|
|
149
|
-
'program', 'cwd', 'summary', 'json', 'isolation', 'worker-pool',
|
|
149
|
+
'program', 'cwd', 'summary', 'json', 'try-checks', 'isolation', 'worker-pool',
|
|
150
150
|
'worker-model', 'worker-reasoning', 'max-agents', 'max-actions',
|
|
151
151
|
'max-expansion-rounds', 'concurrency', 'retry-attempts',
|
|
152
152
|
],
|
package/src/workflow/cli-plan.js
CHANGED
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
goalNextCommands, refuseProgramInvalid, loadCallerProgram, previewValidateInitialProgram, printAdvisories,
|
|
21
21
|
ProgramV2RefusedError, refuseProgramV2,
|
|
22
22
|
} from './cli-program-checks.js';
|
|
23
|
+
import { CHECKS_NOT_RUN_LINE, commandCheckCount, tryCommandChecks, tryLine } from './plan-try-checks.js';
|
|
23
24
|
|
|
24
25
|
// Flags that only make sense on a launch have no meaning for the read-only
|
|
25
26
|
// planning commands.
|
|
@@ -149,6 +150,11 @@ async function planValidate(opts) {
|
|
|
149
150
|
}
|
|
150
151
|
if (workspaceIssues.length) return refuseProgramInvalid(goal, opts, workspaceIssues, { message: 'program invalid against the contract (nothing launched)' });
|
|
151
152
|
const next = goalNextCommands(goal, doc.intent.cwd, opts);
|
|
153
|
+
// Command checks run only when the caller asks (--try-checks): a check may
|
|
154
|
+
// be slow, cost money or touch the network. Their results never change the
|
|
155
|
+
// exit code.
|
|
156
|
+
const commands = commandCheckCount(accepted.program.actions);
|
|
157
|
+
const tried = commands && opts['try-checks'] ? await tryCommandChecks(accepted.program.actions, { cwd: doc.intent.cwd }) : null;
|
|
152
158
|
const payload = {
|
|
153
159
|
action: 'plan-valid',
|
|
154
160
|
requirements: doc.intent.requirements,
|
|
@@ -173,6 +179,7 @@ async function planValidate(opts) {
|
|
|
173
179
|
// valid program so a caller can read it without probing for the key.
|
|
174
180
|
advisories: programAdvisories(accepted.program, { requirements: null })
|
|
175
181
|
.map((item) => ({ ...item, message: v3IssueWording(item.message) })),
|
|
182
|
+
...(commands ? { checks: tried ? { tried: true, results: tried } : { tried: false, commands } } : {}),
|
|
176
183
|
next: { launch: next.launch },
|
|
177
184
|
};
|
|
178
185
|
if (opts.json) console.log(JSON.stringify(payload, null, 2));
|
|
@@ -180,7 +187,11 @@ async function planValidate(opts) {
|
|
|
180
187
|
const control = programControl(accepted.program);
|
|
181
188
|
const count = (n, word) => `${n} ${word}${n === 1 ? '' : 's'}`;
|
|
182
189
|
console.log(`✓ program v3 valid: ${count(payload.program.actions.length, 'step')}, ${count(control.gates.length, 'gate')}, ${count(control.loops.length, 'loop')} (nothing launched)`);
|
|
183
|
-
for (const action of payload.program.actions)
|
|
190
|
+
for (const action of payload.program.actions) {
|
|
191
|
+
console.log(` ${action.id.padEnd(24)} ${action.lane}/${action.effort}${action.deliverable ? ` deliverable=${action.deliverable.type}${action.deliverable.paths?.length ? `:${action.deliverable.paths.join(',')}` : ''}` : ''}${action.evidence ? ` evidence=${action.evidence.map((item) => item.type).join(',')}` : ''}${action.reasoning ? ` reasoning=${action.reasoning}` : ''}${action.answer ? ' answer' : ''}${action.dependsOn.length ? ` after ${action.dependsOn.join(', ')}` : ''}${action.route ? ` route: ${routeSummary(action.route)}` : ''}${action.blindTo?.length ? ` blind to ${action.blindTo.join(', ')}` : ''}`);
|
|
192
|
+
for (const result of tried ?? []) if (result.step === action.id) console.log(tryLine(result));
|
|
193
|
+
}
|
|
194
|
+
if (commands && !tried) console.log(CHECKS_NOT_RUN_LINE);
|
|
184
195
|
for (const line of controlSummaryLines(control)) console.log(line);
|
|
185
196
|
printAdvisories(payload.advisories);
|
|
186
197
|
console.log(` launch ${next.launch}`);
|
|
@@ -36,7 +36,7 @@ import { workspacePathIssues } from './v2-planner.js';
|
|
|
36
36
|
import { acquireKernelLease } from './v2-process.js';
|
|
37
37
|
import { reviseV2Program } from './run-control.js';
|
|
38
38
|
import { deserializeV2DurableState, serializeV2DurableState } from './v2-state.js';
|
|
39
|
-
import { formatDuration } from './
|
|
39
|
+
import { formatDuration } from './format-duration.js';
|
|
40
40
|
|
|
41
41
|
function readState(runDir) {
|
|
42
42
|
return deserializeV2DurableState(readFileSync(join(runDir, 'state.json'), 'utf8'));
|
package/src/workflow/cli.js
CHANGED
|
@@ -7,7 +7,7 @@ import { runDashboard, dashboardJson, overviewSnapshot } from './dashboard.js';
|
|
|
7
7
|
import { wfAdd, wfContinue, wfWait } from './cli-steps.js';
|
|
8
8
|
import { helpText, usageLine } from '../help.js';
|
|
9
9
|
import { flagName, unknownFlagExit } from '../lib/cli-flags.js';
|
|
10
|
-
import { cmdReprice } from './reprice.js';
|
|
10
|
+
import { cmdReprice } from './reprice-cli.js';
|
|
11
11
|
import { resolvePoolId } from '../lib/pool-labels.js';
|
|
12
12
|
import { workflowHelpPath, parseFlags, flagErrors } from './workflow-flags.js';
|
|
13
13
|
import { BULLSWARM_DIR, legacyRunRefusal, legacyRunSummary, legacySummaryLines } from './cli-run-lookup.js';
|
|
@@ -123,7 +123,7 @@ export function buildV3Contract({ goal, cwd, next, workerReasoning = null }) {
|
|
|
123
123
|
timeoutSec: { default: EVIDENCE_DEFAULT_TIMEOUT_SEC, max: EVIDENCE_MAX_TIMEOUT_SEC },
|
|
124
124
|
schemaKeywords: [...SCHEMA_ASSERTED_KEYWORDS], schemaIgnored: [...SCHEMA_IGNORED_KEYWORDS],
|
|
125
125
|
schemaFormats: ['json', 'jsonl'], outputFile: '$output', env: [...EVIDENCE_ENV_KEYS], checker: CHECKER_PATH,
|
|
126
|
-
note: 'checks are read-only: a check that changes the deliverable fails; the answer schema and a schema check accept the same keywords',
|
|
126
|
+
note: 'checks are read-only: a check that changes the deliverable fails; the answer schema and a schema check accept the same keywords; workflow plan validate --try-checks runs each command check once against the current tree before launch (only for checks safe to run now; a check that reads $output is not tried)',
|
|
127
127
|
},
|
|
128
128
|
notV3: 'purpose, affects, evidenceFor, kind, role, inputs, produces and defaults.verifyRounds belong to v2 programs; a check is an ordinary step with an answer and/or evidence',
|
|
129
129
|
},
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
// A v3 run's control nodes: the gates and loops its program declares, and
|
|
2
|
+
// the record of each (stored, or the initial one before the kernel's first
|
|
3
|
+
// pass). gates-loops.js moves them; the revision walk reads them too.
|
|
4
|
+
|
|
5
|
+
import { PROGRAM_V3_SCHEMA_VERSION } from './program-v3.js';
|
|
6
|
+
|
|
7
|
+
/** The run's gates and loops ({gates, loops}), or null for a v2 run or a v3 run with none. */
|
|
8
|
+
export function controlOf(state) {
|
|
9
|
+
const program = state?.program;
|
|
10
|
+
if (program?.schemaVersion !== PROGRAM_V3_SCHEMA_VERSION) return null;
|
|
11
|
+
const gates = program.control?.gates ?? [];
|
|
12
|
+
const loops = program.control?.loops ?? [];
|
|
13
|
+
return gates.length || loops.length ? { gates, loops } : null;
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function defaultRecord(node, type) {
|
|
17
|
+
return type === 'loop'
|
|
18
|
+
? { id: node.id, type, status: 'pending', at: null, round: 1, maxRounds: node.maxRounds }
|
|
19
|
+
: { id: node.id, type, status: 'pending', at: null };
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Every control node's record, stored or (before the kernel's first pass) the initial one. */
|
|
23
|
+
export function controlRecords(state) {
|
|
24
|
+
const control = controlOf(state);
|
|
25
|
+
if (!control) return [];
|
|
26
|
+
const stored = new Map((state.controlNodes ?? []).map((record) => [record.id, record]));
|
|
27
|
+
return [
|
|
28
|
+
...control.loops.map((loop) => stored.get(loop.id) ?? defaultRecord(loop, 'loop')),
|
|
29
|
+
...control.gates.map((gate) => stored.get(gate.id) ?? defaultRecord(gate, 'gate')),
|
|
30
|
+
];
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* The revision walk's edges through the run's gates and loops ([from, to]): a
|
|
35
|
+
* gate's dependencies lead to the gate, a loop's steps to the loop, so a step
|
|
36
|
+
* rerun or accepted before one reaches the steps behind it. A node that has
|
|
37
|
+
* passed is left out: the steps behind it would start before the rerun step.
|
|
38
|
+
*/
|
|
39
|
+
export function controlReachEdges(state) {
|
|
40
|
+
const control = controlOf(state);
|
|
41
|
+
if (!control) return [];
|
|
42
|
+
const status = new Map(controlRecords(state).map((record) => [record.id, record.status]));
|
|
43
|
+
const edges = [];
|
|
44
|
+
for (const loop of control.loops) if (status.get(loop.id) !== 'passed') for (const id of loop.steps) edges.push([id, loop.id]);
|
|
45
|
+
for (const gate of control.gates) if (status.get(gate.id) !== 'passed') for (const id of gate.dependsOn) edges.push([id, gate.id]);
|
|
46
|
+
return edges;
|
|
47
|
+
}
|
|
@@ -24,12 +24,7 @@ import { readRollups, rollupIndexPath } from './rollup.js';
|
|
|
24
24
|
import { loadState } from '../lib/state.js';
|
|
25
25
|
import { listTasks, taskKey } from '../lib/tasks.js';
|
|
26
26
|
import { filterDashboardRows } from './runs-view.js';
|
|
27
|
-
import {
|
|
28
|
-
workflowPanelModel,
|
|
29
|
-
planProgress,
|
|
30
|
-
planStripParts,
|
|
31
|
-
runEconomics,
|
|
32
|
-
} from './run-model.js';
|
|
27
|
+
import { workflowPanelModel } from './run-model.js';
|
|
33
28
|
import {
|
|
34
29
|
renderWorkflowOverviewPanel,
|
|
35
30
|
workflowTimelineLines,
|
|
@@ -55,18 +50,9 @@ import { requestCancel } from './dashboard-cancel.js';
|
|
|
55
50
|
import { dashboardRows, activeDashboardRows, detailRow } from './dashboard-rows.js';
|
|
56
51
|
import { renderDetails } from './dashboard-details.js';
|
|
57
52
|
import { renderDashboardPage } from './dashboard-render.js';
|
|
58
|
-
export { dimText, tint, truncate, visibleLength, meterAnsi, blank, strong, inverseText } from './dashboard-ansi.js';
|
|
59
|
-
export { clamp } from './dashboard-clamp.js';
|
|
60
|
-
export { reasoningText, clockText, formatBytes, durationText, moneyText, minutesText, compactUsage, tokenText, ageText, clockAt } from './dashboard-value-text.js';
|
|
61
|
-
export { stateStatus, stateStartedAt, stateFinishedAt, TERMINAL_ACTIONS, workflowRunLabel } from './dashboard-run-state.js';
|
|
62
|
-
export { statusIcon, workflowStatusIcon } from './dashboard-status.js';
|
|
63
|
-
export { SIDEBAR_WIDTH, renderPanel, joinPanels, panelWindow, panelCell, dimLine, selectLine, timelineText } from './dashboard-panels.js';
|
|
64
|
-
export { pushView, pushColumns } from './dashboard-frame.js';
|
|
65
|
-
export { PERIOD_ITEMS } from './dashboard-pages.js';
|
|
66
53
|
export { requestCancel } from './dashboard-cancel.js';
|
|
67
54
|
export { dashboardRows } from './dashboard-rows.js';
|
|
68
55
|
export { readLicencePerDay, writeClipboard, activeDashboardRows, renderDetails };
|
|
69
|
-
export { agentDetailLines, taskPreview, outcomePreview, wrapLines } from './dashboard-agent-detail.js';
|
|
70
56
|
export { renderDashboardPage } from './dashboard-render.js';
|
|
71
57
|
|
|
72
58
|
/** The operating-system-command introducer and its terminator, for OSC 52. */
|
|
@@ -1991,12 +1977,8 @@ export function dashboardJson(bullswarmDir, { all = false, token = null, cancel
|
|
|
1991
1977
|
return { action: 'list', count: runs.length, runs };
|
|
1992
1978
|
}
|
|
1993
1979
|
|
|
1994
|
-
//
|
|
1995
|
-
//
|
|
1996
|
-
// dashboard import path while page-specific functions move out of this file.
|
|
1980
|
+
// Callers that read the run panel through the dashboard keep this import
|
|
1981
|
+
// path; the page modules import it from run-model.js, the module that owns it.
|
|
1997
1982
|
export {
|
|
1998
1983
|
workflowPanelModel,
|
|
1999
|
-
runEconomics,
|
|
2000
|
-
planProgress,
|
|
2001
|
-
planStripParts,
|
|
2002
1984
|
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
// A duration in seconds as the watch lines print it: `42s`, `3m05s`, `1h07m`.
|
|
2
|
+
|
|
3
|
+
export function formatDuration(seconds) {
|
|
4
|
+
if (!Number.isFinite(seconds)) return '?';
|
|
5
|
+
const total = Math.max(0, Math.round(seconds));
|
|
6
|
+
const hours = Math.floor(total / 3600);
|
|
7
|
+
const minutes = Math.floor((total % 3600) / 60);
|
|
8
|
+
const secs = total % 60;
|
|
9
|
+
if (hours) return `${hours}h${String(minutes).padStart(2, '0')}m`;
|
|
10
|
+
if (minutes) return `${minutes}m${String(secs).padStart(2, '0')}s`;
|
|
11
|
+
return `${secs}s`;
|
|
12
|
+
}
|
|
@@ -30,6 +30,7 @@ import { randomUUID } from 'node:crypto';
|
|
|
30
30
|
import { existsSync, readdirSync, readFileSync, renameSync, writeFileSync } from 'node:fs';
|
|
31
31
|
import { join } from 'node:path';
|
|
32
32
|
import { appendEvent } from './events.js';
|
|
33
|
+
import { controlOf, controlRecords } from './control-nodes.js';
|
|
33
34
|
import { PROGRAM_V3_SCHEMA_VERSION } from './program-v3.js';
|
|
34
35
|
import {
|
|
35
36
|
commitV2Revision, exportV2Plan, planV2Revision, rejectedRevisionRecord, removeStaleReceipts, revisionEventPayload,
|
|
@@ -46,6 +47,8 @@ const ANSWER_SHOWN_CHARS = 4000;
|
|
|
46
47
|
const TAIL_SHOWN_LINES = 20;
|
|
47
48
|
const LINE_SHOWN_CHARS = 200;
|
|
48
49
|
|
|
50
|
+
export { controlOf, controlRecords };
|
|
51
|
+
|
|
49
52
|
// --- the program's control nodes ---------------------------------------------
|
|
50
53
|
|
|
51
54
|
/**
|
|
@@ -77,32 +80,6 @@ export function continuedLoopText(id, rounds, maxRounds, { by = true } = {}) {
|
|
|
77
80
|
return `loop ${id} continued${by ? ' by the caller' : ''} after ${rounds} of ${maxRounds} rounds (condition not met)`;
|
|
78
81
|
}
|
|
79
82
|
|
|
80
|
-
/** The run's gates and loops ({gates, loops}), or null for a v2 run or a v3 run with none. */
|
|
81
|
-
export function controlOf(state) {
|
|
82
|
-
const program = state?.program;
|
|
83
|
-
if (program?.schemaVersion !== PROGRAM_V3_SCHEMA_VERSION) return null;
|
|
84
|
-
const gates = program.control?.gates ?? [];
|
|
85
|
-
const loops = program.control?.loops ?? [];
|
|
86
|
-
return gates.length || loops.length ? { gates, loops } : null;
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
function defaultRecord(node, type) {
|
|
90
|
-
return type === 'loop'
|
|
91
|
-
? { id: node.id, type, status: 'pending', at: null, round: 1, maxRounds: node.maxRounds }
|
|
92
|
-
: { id: node.id, type, status: 'pending', at: null };
|
|
93
|
-
}
|
|
94
|
-
|
|
95
|
-
/** Every control node's record, stored or (before the kernel's first pass) the initial one. */
|
|
96
|
-
export function controlRecords(state) {
|
|
97
|
-
const control = controlOf(state);
|
|
98
|
-
if (!control) return [];
|
|
99
|
-
const stored = new Map((state.controlNodes ?? []).map((record) => [record.id, record]));
|
|
100
|
-
return [
|
|
101
|
-
...control.loops.map((loop) => stored.get(loop.id) ?? defaultRecord(loop, 'loop')),
|
|
102
|
-
...control.gates.map((gate) => stored.get(gate.id) ?? defaultRecord(gate, 'gate')),
|
|
103
|
-
];
|
|
104
|
-
}
|
|
105
|
-
|
|
106
83
|
function ensureRecords(state) {
|
|
107
84
|
state.controlNodes = controlRecords(state);
|
|
108
85
|
return new Map(state.controlNodes.map((record) => [record.id, record]));
|
|
@@ -350,22 +327,6 @@ export function unblockControlNodes(state, { at }) {
|
|
|
350
327
|
}
|
|
351
328
|
}
|
|
352
329
|
|
|
353
|
-
/**
|
|
354
|
-
* The revision walk's edges through the run's gates and loops ([from, to]): a
|
|
355
|
-
* gate's dependencies lead to the gate, a loop's steps to the loop, so a step
|
|
356
|
-
* rerun or accepted before one reaches the steps behind it. A node that has
|
|
357
|
-
* passed is left out: the steps behind it would start before the rerun step.
|
|
358
|
-
*/
|
|
359
|
-
export function controlReachEdges(state) {
|
|
360
|
-
const control = controlOf(state);
|
|
361
|
-
if (!control) return [];
|
|
362
|
-
const status = new Map(controlRecords(state).map((record) => [record.id, record.status]));
|
|
363
|
-
const edges = [];
|
|
364
|
-
for (const loop of control.loops) if (status.get(loop.id) !== 'passed') for (const id of loop.steps) edges.push([id, loop.id]);
|
|
365
|
-
for (const gate of control.gates) if (status.get(gate.id) !== 'passed') for (const id of gate.dependsOn) edges.push([id, gate.id]);
|
|
366
|
-
return edges;
|
|
367
|
-
}
|
|
368
|
-
|
|
369
330
|
/**
|
|
370
331
|
* The kernel's control-node pass: apply the caller's continue intents, then
|
|
371
332
|
* move every gate and loop as far as its dependencies allow. Emits
|
|
@@ -39,25 +39,13 @@ import {
|
|
|
39
39
|
todayRows,
|
|
40
40
|
verifyRoundLabel,
|
|
41
41
|
} from './home-model.js';
|
|
42
|
-
import {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
planProgress,
|
|
50
|
-
planStripParts,
|
|
51
|
-
PERIOD_ITEMS,
|
|
52
|
-
pushColumns,
|
|
53
|
-
runEconomics,
|
|
54
|
-
stateStartedAt,
|
|
55
|
-
strong,
|
|
56
|
-
tint,
|
|
57
|
-
visibleLength,
|
|
58
|
-
workflowRunLabel,
|
|
59
|
-
wrapLines,
|
|
60
|
-
} from './dashboard.js';
|
|
42
|
+
import { ageText, minutesText, moneyText } from './dashboard-value-text.js';
|
|
43
|
+
import { blank, dimText, meterAnsi, strong, tint, visibleLength } from './dashboard-ansi.js';
|
|
44
|
+
import { planProgress, planStripParts, runEconomics } from './run-model.js';
|
|
45
|
+
import { PERIOD_ITEMS } from './dashboard-pages.js';
|
|
46
|
+
import { pushColumns } from './dashboard-frame.js';
|
|
47
|
+
import { stateStartedAt, workflowRunLabel } from './dashboard-run-state.js';
|
|
48
|
+
import { wrapLines } from './dashboard-agent-detail.js';
|
|
61
49
|
import { attemptMetric, tokenSourceOf } from './metrics.js';
|
|
62
50
|
import { taskAttempt } from './metrics-legacy.js';
|
|
63
51
|
import { apiMoney, formatMoney } from '../lib/usage-basis.js';
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
// The one predicate every reader uses to tell an authored-graph run from a V2
|
|
2
|
+
// one. 0.27.0 removed the authored-graph executor, so a run directory whose
|
|
3
|
+
// state.json lacks the V2 schemaVersion — or that has no state.json at all —
|
|
4
|
+
// is history: readable, listable, deletable, never driven.
|
|
5
|
+
export function isLegacyRunState(state) {
|
|
6
|
+
return state?.schemaVersion !== 'bullswarm.workflow.state.v2';
|
|
7
|
+
}
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
import { glyphs } from '../lib/glyphs.js';
|
|
11
11
|
import { countRetries, declaredEvidence, NEEDS_YOU_LABELS, roleOf } from './step-vocabulary.js';
|
|
12
12
|
import { readRunFeatures, runFeatureFlags } from './run-features.js';
|
|
13
|
-
import { formatDuration } from './
|
|
13
|
+
import { formatDuration } from './format-duration.js';
|
|
14
14
|
import { isProgramV3 } from './program-v3.js';
|
|
15
15
|
|
|
16
16
|
const LINE_CHARS = 160;
|