bullswarm 0.38.3 → 0.38.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/docs/guide/observing.md +2 -2
- package/docs/guide/routing.md +2 -2
- package/docs/guide/workflows.md +1 -0
- package/docs/reference/cli.md +1 -1
- package/docs/reference/program.md +1 -1
- package/package.json +1 -1
- package/skill/SKILL.md +12 -4
- package/skill/references/patterns.md +2 -0
- package/skill/references/program.md +5 -2
- package/skill/references/recovery.md +5 -1
- package/src/help.js +12 -5
- package/src/lib/cli-flags.js +1 -1
- package/src/workflow/cli-plan.js +12 -1
- package/src/workflow/contract-v3.js +1 -1
- package/src/workflow/plan-try-checks.js +160 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
## 0.38.5 — plan validate --try-checks (0.38.4 republished)
|
|
6
|
+
|
|
7
|
+
- release: 0.38.4 was tagged but never published to npm (a test read only the `## Unreleased` changelog section, which the release empties); 0.38.5 ships 0.38.4's changes. The docs site builds again: three wrapped lines that began with a `<placeholder>` were read as HTML.
|
|
8
|
+
|
|
9
|
+
## 0.38.4 — plan validate --try-checks
|
|
10
|
+
|
|
11
|
+
- plan validate: `--try-checks` runs each step's command check once, now,
|
|
12
|
+
against the current tree, the way a step runs it (same shell, folder,
|
|
13
|
+
environment and timeout), and prints one `try` line per check with its exit
|
|
14
|
+
and last output line (or `timed out after <n>s`) and the names of any files
|
|
15
|
+
it changed, ignored files and folders outside git included. A check that
|
|
16
|
+
exits non-zero or times out also prints the last 20 lines of its output,
|
|
17
|
+
indented under the try line, so the error is visible. A check that reads the step's output is not tried, and the results never
|
|
18
|
+
change the exit code. Without the flag validate still runs nothing; when the
|
|
19
|
+
program has command checks it now says so (`checks not run · add
|
|
20
|
+
--try-checks …`, and `checks: {tried: false, commands}` in `--json`). Use it
|
|
21
|
+
for checks that are safe to run now: a missing module or command on a `try`
|
|
22
|
+
line means the check itself is wrong.
|
|
23
|
+
- docs: `workflow step accept` on a failed step inside a loop does not end the
|
|
24
|
+
loop; to stop a loop early let it run out of rounds, then
|
|
25
|
+
`workflow continue <run> <loop>` (recorded as condition not met).
|
|
26
|
+
|
|
5
27
|
## 0.38.3 — Blind reviews and a reorganized skill
|
|
6
28
|
|
|
7
29
|
- steps: a v3 step may declare `"blindTo": ["<step id>", ...]`, naming steps
|
package/docs/guide/observing.md
CHANGED
|
@@ -264,8 +264,8 @@ A real run on grok, parked at its gate, on the Run page at 120 columns:
|
|
|
264
264
|
continued without more rounds reads `→ continued by the caller after N of
|
|
265
265
|
max rounds (condition not met)`: it never passed. Each attempt of a
|
|
266
266
|
loop step says its round (`write · round 2`).
|
|
267
|
-
- **A gate row** heads the phase of the steps behind it and reads
|
|
268
|
-
<steps>`, `waiting for you · <note>` with its `continue` command, `passed ·
|
|
267
|
+
- **A gate row** heads the phase of the steps behind it and reads
|
|
268
|
+
`waits after <steps>`, `waiting for you · <note>` with its `continue` command, `passed ·
|
|
269
269
|
continued by the caller`, or `skipped · <condition> does not hold` when its
|
|
270
270
|
`when` condition did not hold.
|
|
271
271
|
- **Answers.** `answer {…}` is the checked answer (it matched the step's schema);
|
package/docs/guide/routing.md
CHANGED
|
@@ -198,8 +198,8 @@ well: as `quota` when every reason is a usage limit, else as `unavailable`. Its
|
|
|
198
198
|
spare: pool-a at its 5-hour limit until <time>; pool-b at its weekly limit
|
|
199
199
|
until <time>`, and `back at` is the earliest known return among them. A retry
|
|
200
200
|
the step was promised (after a crash, a sign-in failure or a failed gate) that
|
|
201
|
-
finds no free pool keeps its own failure and ends its `why` with
|
|
202
|
-
<pool> <reason>; …`. When another pool that can run the step is free, routing
|
|
201
|
+
finds no free pool keeps its own failure and ends its `why` with
|
|
202
|
+
`· no retry: <pool> <reason>; …`. When another pool that can run the step is free, routing
|
|
203
203
|
picks it as usual.
|
|
204
204
|
|
|
205
205
|
A run saved by 0.37.x may end with `the workflow planner stopped on a usage
|
package/docs/guide/workflows.md
CHANGED
|
@@ -108,6 +108,7 @@ bullswarm workflow plan validate "Make the acme tests pass" --cwd=/private/tmp/v
|
|
|
108
108
|
✓ program v3 valid: 2 steps, 0 gates, 1 loop (nothing launched)
|
|
109
109
|
fix build/medium deliverable=files
|
|
110
110
|
check analyze/medium evidence=command answer after fix
|
|
111
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
111
112
|
loop until-green steps fix, check · until check's evidence passed · at most 3 rounds
|
|
112
113
|
launch bullswarm workflow goal 'Make the acme tests pass' --cwd /private/tmp/v37fix/acme --program /private/tmp/v37fix/loop.json --json
|
|
113
114
|
```
|
package/docs/reference/cli.md
CHANGED
|
@@ -255,7 +255,7 @@ Trailing `<task text...>` is mutually exclusive with `--prompt` and `--task-file
|
|
|
255
255
|
| `--avoid-provider <provider,...>` | never route to pools of these providers | unset |
|
|
256
256
|
| `--dry-run` | print the kernel's routing pick, the forecast, and the exact command that would be spawned, without spawning, recording a run, registering an assignment, or writing the decision log | off (dispatches for real) |
|
|
257
257
|
| `--no-caller` | accepted and ignored for one release: the calling agent is never a pool of its own run | removed in 0.37.0 |
|
|
258
|
-
| `--json` | print the compact machine-readable verdict; top-level `pool` and `model` name who ran the last attempt (null when nothing was dispatched; `pick` keeps the same two beside the command); its last field, details, is the command for the full record and per-attempt cost (bullswarm workflow runs result <shortId> --json) | human-readable summary ending with the run id and that command |
|
|
258
|
+
| `--json` | print the compact machine-readable verdict; top-level `pool` and `model` name who ran the last attempt (null when nothing was dispatched; `pick` keeps the same two beside the command); its last field, details, is the command for the full record and per-attempt cost (`bullswarm workflow runs result <shortId> --json`) | human-readable summary ending with the run id and that command |
|
|
259
259
|
|
|
260
260
|
A run is a one-step workflow, recorded under `workflows/<id>/` (`bullswarm workflow runs --all`). It gets the step's one automatic retry unless `--no-retry`; a usage limit exits 1 with no retry, and the pool's meter is read again at once, so a window it shows at 100% keeps the pool out of later picks until that window resets. Nothing else about a failed pool is remembered. A build or chore run must change a file (else failure kind `not-produced`). The JSON shape is in [Result envelope](/reference/result).
|
|
261
261
|
|
|
@@ -367,7 +367,7 @@ Checks receive `BULLSWARM_EVIDENCE=1`, `BULLSWARM_STEP_ID`, `BULLSWARM_STEP_OUTP
|
|
|
367
367
|
|
|
368
368
|
Checks must be read-only. Before the first item and after each one, Bullswarm hashes the deliverable: the step's `ownedFiles`, declared deliverable paths and saved response, plus every tracked file and HEAD in an isolated copy or for a build or chore step with no `ownedFiles` (it runs alone). A change fails the item with `changed the deliverable: …`, and the later items do not run. Untracked by-products there are recorded as `touched` rather than failed; declare a new file as a deliverable path when it must be protected. In an isolated copy, after the last item, Bullswarm removes the files the checks created and puts back untracked files they rewrote or deleted (up to 16 MB in total), so none of them is merged back. One it cannot put back is named in an attempt note, `check by-product not restored: <paths>`, and is left out of the ownership check and the merge-back. Elsewhere HEAD is not compared: when it moves while an item runs (another step may have committed), the item records `headMoved: true` as a fact and does not fail.
|
|
369
369
|
|
|
370
|
-
Scope a command to its step. Put a whole-suite command on a step that runs alone or last, and pass `--run` or `CI=1` yourself when a test runner watches files. Run every check once
|
|
370
|
+
Scope a command to its step. Put a whole-suite command on a step that runs alone or last, and pass `--run` or `CI=1` yourself when a test runner watches files. Run every check once before launch (`workflow plan validate --try-checks` runs each command check once in the workspace, for checks that are safe to run now) and give it a generous timeout: changing a wrong check amends the step and reruns its worker. A suite that runs longer than 600 seconds cannot be one item: split it into several items or keep it in the step's prompt. To prove finished work without rerunning it, add a `check` step with its own `evidence`; it does not rerun the work. An old kernel refuses `evidence`; pause, revise, then resume.
|
|
371
371
|
|
|
372
372
|
Each check's result is in the full `bullswarm workflow runs result <id> --json` envelope (not `--summary`) under `actions[].evidenceResults`: `status`, `exit`, `tail`, `why` and the log path. `bullswarm workflow action show <id> <step>` shows the same for each attempt under `attempts[].evidenceResults`. Each item's full output is in `evidence-<step>-attempt-<n>-<k>.log` in the run directory.
|
|
373
373
|
|
package/package.json
CHANGED
package/skill/SKILL.md
CHANGED
|
@@ -90,6 +90,7 @@ validate:
|
|
|
90
90
|
✓ program v3 valid: 2 steps, 0 gates, 0 loops (nothing launched)
|
|
91
91
|
build build/medium deliverable=files evidence=command
|
|
92
92
|
review analyze/medium answer after build route: independent of build blind to build
|
|
93
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
93
94
|
```
|
|
94
95
|
|
|
95
96
|
Keep the goal in a file and pass it as `"$(cat goal.txt)"` to both commands,
|
|
@@ -121,7 +122,11 @@ it a `--timeout` under your tool's time limit (`--until trouble --timeout 100`
|
|
|
121
122
|
for 2 minutes). A restart without `--after` attaches at the newest event and
|
|
122
123
|
skips wakes in between. Never end your turn while a run you own is still
|
|
123
124
|
running. At a gate: `bullswarm workflow continue <shortId> <gate>`; a loop out
|
|
124
|
-
of rounds takes `--rounds <1-5>`. `
|
|
125
|
+
of rounds takes `--rounds <1-5>`. `workflow step accept` on a failed step
|
|
126
|
+
inside a loop does not end the loop (the next round still starts); to stop a
|
|
127
|
+
loop early let it run out of rounds (`maxRounds`), then
|
|
128
|
+
`workflow continue <shortId> <loop>`, recorded as condition not met.
|
|
129
|
+
`watch --until trouble` also wakes on
|
|
125
130
|
`steering received` (a person left guidance: decide what it means and add
|
|
126
131
|
steps).
|
|
127
132
|
|
|
@@ -145,9 +150,12 @@ Lessons from real runs; each holds for this version.
|
|
|
145
150
|
the other. Give each writer its own files and tell it to keep other
|
|
146
151
|
workers' edits; keep a breaking rename in one step, not parallel with its
|
|
147
152
|
consumers, which would build against the old name.
|
|
148
|
-
5. **Checks are facts.** Put anything a machine can say in `evidence
|
|
149
|
-
|
|
150
|
-
|
|
153
|
+
5. **Checks are facts.** Put anything a machine can say in `evidence`. When
|
|
154
|
+
the checks are safe to run now (not ones that write, deploy, call paid
|
|
155
|
+
services or take long), run `bullswarm workflow plan validate … --try-checks`
|
|
156
|
+
before launch and read each `try` line: an error such as a missing module
|
|
157
|
+
or command means the check itself is wrong, and fixing a wrong check later
|
|
158
|
+
reruns the worker. Workers share one tree, so a later step can undo what an earlier
|
|
151
159
|
step's check proved; put the final check where nothing runs after it
|
|
152
160
|
(after integration, or on the step a gate waits behind).
|
|
153
161
|
6. **Reviews that hold.** State the contract as numbered checks, make any
|
|
@@ -122,6 +122,7 @@ validate:
|
|
|
122
122
|
✓ program v3 valid: 2 steps, 0 gates, 1 loop (nothing launched)
|
|
123
123
|
fix build/medium deliverable=files
|
|
124
124
|
check analyze/medium evidence=command answer after fix
|
|
125
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
125
126
|
loop until-green steps fix, check · until check's evidence passed · at most 3 rounds
|
|
126
127
|
```
|
|
127
128
|
|
|
@@ -258,6 +259,7 @@ validate (plan and the added steps as one program):
|
|
|
258
259
|
slice-writer build/medium deliverable=files after pick-slices
|
|
259
260
|
slice-command build/medium deliverable=files after pick-slices
|
|
260
261
|
check analyze/medium deliverable=report evidence=command after slice-writer, slice-command
|
|
262
|
+
checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)
|
|
261
263
|
gate pick-slices after plan · waits for you · Read the slices, add one build step per slice and a check, then continue
|
|
262
264
|
```
|
|
263
265
|
|
|
@@ -318,8 +318,11 @@ Rules the fields above do not show on their own:
|
|
|
318
318
|
tell it to keep other workers' edits.
|
|
319
319
|
- **Evidence: checks Bullswarm runs.** Add a command or schema check for
|
|
320
320
|
anything a machine can check (`"evidence": [{"type": "command", "cmd": "npm
|
|
321
|
-
test", "timeoutSec": 300}]`, at most 5 items).
|
|
322
|
-
launch
|
|
321
|
+
test", "timeoutSec": 300}]`, at most 5 items). Try the checks before
|
|
322
|
+
launch with `bullswarm workflow plan validate … --try-checks` when they are
|
|
323
|
+
safe to run now (it runs each command check once in the workspace and
|
|
324
|
+
prints a `try` line; a check that reads `$output` is not tried), because
|
|
325
|
+
fixing a wrong check reruns the worker.
|
|
323
326
|
- **Steps that must not repeat.** Sending or publishing is `"deliverable":
|
|
324
327
|
"outward"` with `"retry": 0`; an outward step is never retried once its
|
|
325
328
|
worker started.
|
|
@@ -236,7 +236,11 @@ finishes: the handback").
|
|
|
236
236
|
**An accept is a choice, never proof.** `step accept` lets the step's
|
|
237
237
|
dependents run, and the step reads `accepted by choice` and is counted apart
|
|
238
238
|
in the proof line (`N accepted by choice: <steps>`). Report it as your
|
|
239
|
-
decision, never as verification.
|
|
239
|
+
decision, never as verification. `step accept` on a failed step inside a
|
|
240
|
+
loop does not end the loop (the next round still starts); to stop a loop
|
|
241
|
+
early let it run out of rounds (`maxRounds`), then
|
|
242
|
+
`bullswarm workflow continue <shortId> <loop>`, which is recorded as
|
|
243
|
+
condition not met.
|
|
240
244
|
|
|
241
245
|
## Other stopping rules
|
|
242
246
|
|
package/src/help.js
CHANGED
|
@@ -1090,18 +1090,19 @@ const workflowPlanText = rich({
|
|
|
1090
1090
|
});
|
|
1091
1091
|
|
|
1092
1092
|
const workflowPlanValidateText = rich({
|
|
1093
|
-
usage: 'bullswarm workflow plan validate "<goal>" --program <file.json> [--cwd <dir>] [--summary <text>] [--json]',
|
|
1093
|
+
usage: 'bullswarm workflow plan validate "<goal>" --program <file.json> [--cwd <dir>] [--summary <text>] [--try-checks] [--json]',
|
|
1094
1094
|
purpose: 'Check a program you authored against the exact contract a launch would enforce, without '
|
|
1095
1095
|
+ 'creating a run: the same validator and the same preview state as '
|
|
1096
1096
|
+ 'workflow goal --program. Exit 0 prints the accepted actions and the launch line; exit 2 prints '
|
|
1097
1097
|
+ 'every validator issue so you can fix the file and re-run. A bullswarm.workflow.program.v2 is refused '
|
|
1098
|
-
+ '(exit 2), as a launch refuses it.',
|
|
1098
|
+
+ '(exit 2), as a launch refuses it. Command checks are not run unless you add --try-checks.',
|
|
1099
1099
|
args: [{ name: '"<goal>"', desc: 'the goal text exactly as it will be passed to workflow goal' }],
|
|
1100
1100
|
options: [
|
|
1101
1101
|
{ flag: '--program <file.json>', desc: 'a bare bullswarm.workflow.program.v3 document, or one inside a planner response envelope', default: 'required' },
|
|
1102
1102
|
{ flag: '--cwd <dir>', desc: 'working directory the goal will execute in (must exist)', default: 'current directory' },
|
|
1103
1103
|
{ flag: '--summary <text>', desc: 'one-line summary recorded for a bare program document', default: 'derived from the action purposes' },
|
|
1104
|
-
{ flag: '--
|
|
1104
|
+
{ flag: '--try-checks', desc: 'run each step\'s command check once now, in --cwd, the way a step runs it, and print a try line per check (exit, last output line when it passes, and any files it changed; a check that fails or times out also shows the last 20 lines of its output); a check that reads the step\'s output is not tried. Only for checks that are safe to run now: it may take time, touch the network or cost money, and a check that writes changes your tree (nothing is restored). The results never change the exit code', default: 'off (checks are counted, not run)' },
|
|
1105
|
+
{ flag: '--json', desc: 'print the acceptance document ({action: "plan-valid", requirements, program, checks?, next}) or the refusal ({error: "program-invalid", issues, next}) as JSON', default: 'human summary' },
|
|
1105
1106
|
{ flag: '--isolation', desc: 'validate against strict per-worker worktree isolation', default: 'off (shared workspace)' },
|
|
1106
1107
|
{ flag: '--worker-pool <pool|auto>', desc: 'pin the worker pool the preview routes with', default: 'auto (routing decides per action)' },
|
|
1107
1108
|
{ flag: '--worker-model <model|auto>', desc: 'pin the worker model the preview routes with', default: 'auto' },
|
|
@@ -1112,8 +1113,14 @@ const workflowPlanValidateText = rich({
|
|
|
1112
1113
|
{ flag: '--concurrency <n>', desc: 'execution concurrency for the previewed run', default: '4' },
|
|
1113
1114
|
{ flag: '--retry-attempts <0..3>', desc: 'automatic retries per step before it comes back to you (process failures on another pool, gate failures on the same pool with the failure attached)', default: '1' },
|
|
1114
1115
|
],
|
|
1115
|
-
safety: [
|
|
1116
|
-
|
|
1116
|
+
safety: [
|
|
1117
|
+
'read-only without --try-checks — nothing is launched, dispatched, or written; the exit code is the verdict (0 valid, 2 invalid, 1 bad cwd)',
|
|
1118
|
+
'with --try-checks it runs your command checks in --cwd (never a worker, never in the run home); whatever they do, they do: use it only for checks that are safe to run now',
|
|
1119
|
+
],
|
|
1120
|
+
examples: [
|
|
1121
|
+
{ cmd: 'bullswarm workflow plan validate "1. Fix the parser. 2. Update the docs." --cwd . --program plan.json --json' },
|
|
1122
|
+
{ cmd: 'bullswarm workflow plan validate "Fix the parser" --cwd . --program plan.json --try-checks', note: 'a missing module or command on a try line means the check itself is wrong' },
|
|
1123
|
+
],
|
|
1117
1124
|
next: 'bullswarm workflow goal "<same goal>" --cwd <dir> --program plan.json once it validates.',
|
|
1118
1125
|
});
|
|
1119
1126
|
|
package/src/lib/cli-flags.js
CHANGED
|
@@ -146,7 +146,7 @@ const TABLE = {
|
|
|
146
146
|
'concurrency', 'retry-attempts',
|
|
147
147
|
],
|
|
148
148
|
'workflow plan validate': [
|
|
149
|
-
'program', 'cwd', 'summary', 'json', 'isolation', 'worker-pool',
|
|
149
|
+
'program', 'cwd', 'summary', 'json', 'try-checks', 'isolation', 'worker-pool',
|
|
150
150
|
'worker-model', 'worker-reasoning', 'max-agents', 'max-actions',
|
|
151
151
|
'max-expansion-rounds', 'concurrency', 'retry-attempts',
|
|
152
152
|
],
|
package/src/workflow/cli-plan.js
CHANGED
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
goalNextCommands, refuseProgramInvalid, loadCallerProgram, previewValidateInitialProgram, printAdvisories,
|
|
21
21
|
ProgramV2RefusedError, refuseProgramV2,
|
|
22
22
|
} from './cli-program-checks.js';
|
|
23
|
+
import { CHECKS_NOT_RUN_LINE, commandCheckCount, tryCommandChecks, tryLine } from './plan-try-checks.js';
|
|
23
24
|
|
|
24
25
|
// Flags that only make sense on a launch have no meaning for the read-only
|
|
25
26
|
// planning commands.
|
|
@@ -149,6 +150,11 @@ async function planValidate(opts) {
|
|
|
149
150
|
}
|
|
150
151
|
if (workspaceIssues.length) return refuseProgramInvalid(goal, opts, workspaceIssues, { message: 'program invalid against the contract (nothing launched)' });
|
|
151
152
|
const next = goalNextCommands(goal, doc.intent.cwd, opts);
|
|
153
|
+
// Command checks run only when the caller asks (--try-checks): a check may
|
|
154
|
+
// be slow, cost money or touch the network. Their results never change the
|
|
155
|
+
// exit code.
|
|
156
|
+
const commands = commandCheckCount(accepted.program.actions);
|
|
157
|
+
const tried = commands && opts['try-checks'] ? await tryCommandChecks(accepted.program.actions, { cwd: doc.intent.cwd }) : null;
|
|
152
158
|
const payload = {
|
|
153
159
|
action: 'plan-valid',
|
|
154
160
|
requirements: doc.intent.requirements,
|
|
@@ -173,6 +179,7 @@ async function planValidate(opts) {
|
|
|
173
179
|
// valid program so a caller can read it without probing for the key.
|
|
174
180
|
advisories: programAdvisories(accepted.program, { requirements: null })
|
|
175
181
|
.map((item) => ({ ...item, message: v3IssueWording(item.message) })),
|
|
182
|
+
...(commands ? { checks: tried ? { tried: true, results: tried } : { tried: false, commands } } : {}),
|
|
176
183
|
next: { launch: next.launch },
|
|
177
184
|
};
|
|
178
185
|
if (opts.json) console.log(JSON.stringify(payload, null, 2));
|
|
@@ -180,7 +187,11 @@ async function planValidate(opts) {
|
|
|
180
187
|
const control = programControl(accepted.program);
|
|
181
188
|
const count = (n, word) => `${n} ${word}${n === 1 ? '' : 's'}`;
|
|
182
189
|
console.log(`✓ program v3 valid: ${count(payload.program.actions.length, 'step')}, ${count(control.gates.length, 'gate')}, ${count(control.loops.length, 'loop')} (nothing launched)`);
|
|
183
|
-
for (const action of payload.program.actions)
|
|
190
|
+
for (const action of payload.program.actions) {
|
|
191
|
+
console.log(` ${action.id.padEnd(24)} ${action.lane}/${action.effort}${action.deliverable ? ` deliverable=${action.deliverable.type}${action.deliverable.paths?.length ? `:${action.deliverable.paths.join(',')}` : ''}` : ''}${action.evidence ? ` evidence=${action.evidence.map((item) => item.type).join(',')}` : ''}${action.reasoning ? ` reasoning=${action.reasoning}` : ''}${action.answer ? ' answer' : ''}${action.dependsOn.length ? ` after ${action.dependsOn.join(', ')}` : ''}${action.route ? ` route: ${routeSummary(action.route)}` : ''}${action.blindTo?.length ? ` blind to ${action.blindTo.join(', ')}` : ''}`);
|
|
192
|
+
for (const result of tried ?? []) if (result.step === action.id) console.log(tryLine(result));
|
|
193
|
+
}
|
|
194
|
+
if (commands && !tried) console.log(CHECKS_NOT_RUN_LINE);
|
|
184
195
|
for (const line of controlSummaryLines(control)) console.log(line);
|
|
185
196
|
printAdvisories(payload.advisories);
|
|
186
197
|
console.log(` launch ${next.launch}`);
|
|
@@ -123,7 +123,7 @@ export function buildV3Contract({ goal, cwd, next, workerReasoning = null }) {
|
|
|
123
123
|
timeoutSec: { default: EVIDENCE_DEFAULT_TIMEOUT_SEC, max: EVIDENCE_MAX_TIMEOUT_SEC },
|
|
124
124
|
schemaKeywords: [...SCHEMA_ASSERTED_KEYWORDS], schemaIgnored: [...SCHEMA_IGNORED_KEYWORDS],
|
|
125
125
|
schemaFormats: ['json', 'jsonl'], outputFile: '$output', env: [...EVIDENCE_ENV_KEYS], checker: CHECKER_PATH,
|
|
126
|
-
note: 'checks are read-only: a check that changes the deliverable fails; the answer schema and a schema check accept the same keywords',
|
|
126
|
+
note: 'checks are read-only: a check that changes the deliverable fails; the answer schema and a schema check accept the same keywords; workflow plan validate --try-checks runs each command check once against the current tree before launch (only for checks safe to run now; a check that reads $output is not tried)',
|
|
127
127
|
},
|
|
128
128
|
notV3: 'purpose, affects, evidenceFor, kind, role, inputs, produces and defaults.verifyRounds belong to v2 programs; a check is an ordinary step with an answer and/or evidence',
|
|
129
129
|
},
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
// `workflow plan validate --try-checks` (0.38.4): run each step's command
|
|
2
|
+
// checks once against the current tree, the way the evidence runner runs
|
|
3
|
+
// them during a step (same shell, working folder, environment and timeout,
|
|
4
|
+
// the same tail of output), without dispatching a worker or writing into the
|
|
5
|
+
// run home. The results are information: validate's exit code never follows
|
|
6
|
+
// them, because a check that fails before the work exists is often expected.
|
|
7
|
+
|
|
8
|
+
import { lstatSync, mkdtempSync, readdirSync, readFileSync, realpathSync, rmSync } from 'node:fs';
|
|
9
|
+
import { tmpdir } from 'node:os';
|
|
10
|
+
import { join, relative, resolve, sep } from 'node:path';
|
|
11
|
+
import { evidenceEnv, evidenceItemLabel, evidenceItemTimeoutSec, runEvidenceItem, stripAnsi } from './evidence-runner.js';
|
|
12
|
+
|
|
13
|
+
const MAX_LISTED = 10;
|
|
14
|
+
const SHOW_ALL_LINES = 30;
|
|
15
|
+
// 14 first lines, not 12: for `node --test <folder>` the line with MODULE_NOT_FOUND is line 13.
|
|
16
|
+
const SHOW_HEAD_LINES = 14;
|
|
17
|
+
const SHOW_TAIL_LINES = 12;
|
|
18
|
+
const TAIL_INDENT = ' '.repeat(11);
|
|
19
|
+
|
|
20
|
+
/** A command check that reads the step's final response, which does not exist before the step runs. */
|
|
21
|
+
export const readsStepOutput = (item) => /\$output\b|BULLSWARM_STEP_OUTPUT/.test(String(item?.cmd ?? ''));
|
|
22
|
+
|
|
23
|
+
/** How many command checks the program declares. */
|
|
24
|
+
export function commandCheckCount(actions) {
|
|
25
|
+
return (actions ?? []).reduce((n, action) => n + (action.evidence ?? []).filter((item) => item?.type === 'command').length, 0);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function realPath(path) {
|
|
29
|
+
try { return realpathSync(path); } catch { return resolve(path); }
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Every file under the workspace (git's own folder aside) → size, change
|
|
33
|
+
// times and inode, ignored and untracked files included, inside or outside a
|
|
34
|
+
// git repository. A check that writes, deletes or renames a file changes its
|
|
35
|
+
// entry. Paths are relative to the workspace.
|
|
36
|
+
function fileSnapshot(root) {
|
|
37
|
+
const snapshot = new Map();
|
|
38
|
+
const walk = (directory) => {
|
|
39
|
+
let entries;
|
|
40
|
+
try { entries = readdirSync(directory, { withFileTypes: true }); } catch { return; }
|
|
41
|
+
for (const entry of entries) {
|
|
42
|
+
if (entry.name === '.git') continue;
|
|
43
|
+
const absolute = join(directory, entry.name);
|
|
44
|
+
if (entry.isDirectory()) { walk(absolute); continue; }
|
|
45
|
+
let stat;
|
|
46
|
+
try { stat = lstatSync(absolute, { bigint: true }); } catch { continue; }
|
|
47
|
+
snapshot.set(relative(root, absolute).split(sep).join('/'), `${stat.size}:${stat.mtimeNs}:${stat.ctimeNs}:${stat.ino}`);
|
|
48
|
+
}
|
|
49
|
+
};
|
|
50
|
+
walk(root);
|
|
51
|
+
return snapshot;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
function changedBetween(before, after) {
|
|
55
|
+
const paths = new Set([...before.keys(), ...after.keys()]);
|
|
56
|
+
return [...paths].filter((path) => before.get(path) !== after.get(path)).sort();
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function lastLine(tail) {
|
|
60
|
+
const lines = String(tail ?? '').split('\n').map((line) => line.trim()).filter(Boolean);
|
|
61
|
+
if (!lines.length) return null;
|
|
62
|
+
return lines[lines.length - 1];
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Run every command check of `actions` once, one after the other, in `cwd`.
|
|
67
|
+
* Resolves to `[{ step, index, cmd, exit, timedOut, tail, changedFiles }]`
|
|
68
|
+
* (index is the item's 1-based place in the step's evidence); a check that
|
|
69
|
+
* reads the step's output is listed with `notTried` and never runs.
|
|
70
|
+
*/
|
|
71
|
+
export async function tryCommandChecks(actions, { cwd, env = process.env } = {}) {
|
|
72
|
+
const root = realPath(cwd);
|
|
73
|
+
const results = [];
|
|
74
|
+
// BULLSWARM_STEP_OUTPUT and BULLSWARM_RUN_DIR point into a scratch folder
|
|
75
|
+
// outside the run home, removed afterwards.
|
|
76
|
+
const scratch = mkdtempSync(join(tmpdir(), 'bullswarm-try-'));
|
|
77
|
+
// The full output of every check is kept here, for the caller to read.
|
|
78
|
+
const logs = mkdtempSync(join(tmpdir(), 'bullswarm-try-logs-'));
|
|
79
|
+
try {
|
|
80
|
+
for (const action of actions ?? []) {
|
|
81
|
+
const items = Array.isArray(action.evidence) ? action.evidence : [];
|
|
82
|
+
for (let index = 0; index < items.length; index += 1) {
|
|
83
|
+
const item = items[index];
|
|
84
|
+
if (item?.type !== 'command') continue;
|
|
85
|
+
const base = { step: action.id, index: index + 1, cmd: item.cmd };
|
|
86
|
+
if (readsStepOutput(item)) {
|
|
87
|
+
results.push({ ...base, exit: null, timedOut: false, tail: '', changedFiles: [], notTried: 'it reads the step\'s output' });
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
const before = fileSnapshot(root);
|
|
91
|
+
const run = await runEvidenceItem(item, {
|
|
92
|
+
cwd: root,
|
|
93
|
+
env: evidenceEnv(env, { cwd: root, stepId: action.id, outFile: join(scratch, 'output.md'), runDir: scratch }),
|
|
94
|
+
outFile: join(scratch, 'output.md'),
|
|
95
|
+
logFile: join(logs, `${results.length + 1}-${action.id.replace(/[^A-Za-z0-9._-]/g, '_')}-${index + 1}.log`),
|
|
96
|
+
});
|
|
97
|
+
const changedFiles = changedBetween(before, fileSnapshot(root));
|
|
98
|
+
results.push({
|
|
99
|
+
...base,
|
|
100
|
+
exit: run.exit,
|
|
101
|
+
timedOut: Boolean(run.timedOut),
|
|
102
|
+
...(run.timedOut ? { timeoutSec: evidenceItemTimeoutSec(item) } : {}),
|
|
103
|
+
...(run.exit == null && !run.timedOut && run.why ? { why: run.why } : {}),
|
|
104
|
+
tail: run.tail,
|
|
105
|
+
...(run.log ? { log: run.log } : {}),
|
|
106
|
+
changedFiles,
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
} finally {
|
|
111
|
+
rmSync(scratch, { recursive: true, force: true });
|
|
112
|
+
}
|
|
113
|
+
return results;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// The command's own output from its log file (the file holds a header of
|
|
117
|
+
// `$ cmd`, `cwd:` and `timeout:` lines before it and a one-line footer after).
|
|
118
|
+
function logOutputLines(result) {
|
|
119
|
+
let text;
|
|
120
|
+
try { text = readFileSync(result.log, 'utf8'); } catch { return null; }
|
|
121
|
+
const marker = text.search(/\ncwd: .*\ntimeout: \d+s\n/);
|
|
122
|
+
if (marker < 0) return null;
|
|
123
|
+
let body = text.slice(text.indexOf('\n', text.indexOf('\ntimeout: ', marker) + 1) + 1);
|
|
124
|
+
const end = body.lastIndexOf('\n', body.length - 2);
|
|
125
|
+
body = end < 0 ? '' : body.slice(0, end + 1);
|
|
126
|
+
return stripAnsi(body).replace(/\s+$/, '').split('\n');
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function failedOutput(result) {
|
|
130
|
+
const lines = result.log ? logOutputLines(result) : null;
|
|
131
|
+
const all = lines ?? String(result.tail ?? '').split('\n');
|
|
132
|
+
const shown = !all.some((line) => line.trim()) ? [] : all.length <= SHOW_ALL_LINES
|
|
133
|
+
? all
|
|
134
|
+
: [...all.slice(0, SHOW_HEAD_LINES), `… ${all.length - SHOW_HEAD_LINES - SHOW_TAIL_LINES} more lines …`, ...all.slice(-SHOW_TAIL_LINES)];
|
|
135
|
+
const out = shown.map((line) => `${TAIL_INDENT}${line}`);
|
|
136
|
+
if (result.log) out.push(`${TAIL_INDENT}full output: ${result.log}`);
|
|
137
|
+
return out;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** The `try` line for one result; a check that failed or timed out is followed by its output (all of it when short, else its first and last lines) and the log file path. */
|
|
141
|
+
export function tryLine(result) {
|
|
142
|
+
const label = evidenceItemLabel({ type: 'command', cmd: result.cmd });
|
|
143
|
+
let outcome;
|
|
144
|
+
if (result.notTried) outcome = `not tried: ${result.notTried}`;
|
|
145
|
+
else if (result.timedOut) outcome = `timed out after ${result.timeoutSec}s`;
|
|
146
|
+
else if (result.exit == null) outcome = result.why ?? 'did not run';
|
|
147
|
+
else if (result.exit === 0) {
|
|
148
|
+
const line = lastLine(result.tail);
|
|
149
|
+
outcome = `exit 0${line ? ` · ${line}` : ''}`;
|
|
150
|
+
} else outcome = `exit ${result.exit}`;
|
|
151
|
+
const changed = result.changedFiles?.length
|
|
152
|
+
? ` · changed files: ${result.changedFiles.slice(0, MAX_LISTED).join(', ')}${result.changedFiles.length > MAX_LISTED ? ` (+${result.changedFiles.length - MAX_LISTED} more)` : ''}`
|
|
153
|
+
: '';
|
|
154
|
+
const head = ` try ${label} → ${outcome}${changed}`;
|
|
155
|
+
const failed = !result.notTried && (result.timedOut || (result.exit != null && result.exit !== 0));
|
|
156
|
+
const output = failed ? failedOutput(result) : [];
|
|
157
|
+
return output.length ? [head, ...output].join('\n') : head;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export const CHECKS_NOT_RUN_LINE = ' checks not run · add --try-checks to run each command check once now against the current tree (it may take time and must not change files)';
|