mandrel 2.66.0 → 2.68.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/acceptance-critic.md +2 -2
- package/.agents/agents/story-worker.md +15 -11
- package/.agents/docs/agentrc-reference.json +5 -2
- package/.agents/docs/configuration.md +38 -2
- package/.agents/docs/workflows.md +4 -2
- package/.agents/instructions.md +2 -1
- package/.agents/rules/git-conventions-reference.md +5 -5
- package/.agents/rules/git-conventions.md +1 -1
- package/.agents/schemas/agentrc.schema.json +20 -2
- package/.agents/schemas/story-deliver-terminal.schema.json +23 -1
- package/.agents/schemas/validation-evidence.schema.json +3 -1
- package/.agents/scripts/boot-sweep.js +97 -9
- package/.agents/scripts/{git-cleanup.js → clean-git.js} +2 -2
- package/.agents/scripts/clean-temp.js +54 -0
- package/.agents/scripts/clean-worktrees.js +593 -0
- package/.agents/scripts/coverage-capture.js +65 -9
- package/.agents/scripts/drain-pending-cleanup.js +5 -4
- package/.agents/scripts/evidence-gate.js +106 -8
- package/.agents/scripts/lib/baselines/coverage-refresh-scope.js +60 -0
- package/.agents/scripts/lib/baselines/crap-updater-cli.js +101 -4
- package/.agents/scripts/lib/baselines/refresh-service.js +1 -1
- package/.agents/scripts/lib/baselines/seat-missing.js +228 -0
- package/.agents/scripts/lib/child-exec.js +39 -1
- package/.agents/scripts/lib/clean-temp.js +440 -0
- package/.agents/scripts/lib/close-validation/gates.js +59 -19
- package/.agents/scripts/lib/close-validation/process.js +23 -24
- package/.agents/scripts/lib/close-validation/runner.js +71 -40
- package/.agents/scripts/lib/config/gates/coverage.schema.js +21 -0
- package/.agents/scripts/lib/config/quality.js +7 -1
- package/.agents/scripts/lib/config/temp-paths.js +15 -0
- package/.agents/scripts/lib/config-settings-schema-delivery.js +12 -3
- package/.agents/scripts/lib/coverage-baseline.js +78 -5
- package/.agents/scripts/lib/coverage-capture-affected.js +345 -0
- package/.agents/scripts/lib/coverage-capture-delta.js +180 -0
- package/.agents/scripts/lib/coverage-capture-fullscope.js +53 -32
- package/.agents/scripts/lib/coverage-capture-incremental.js +49 -26
- package/.agents/scripts/lib/coverage-capture-usage.js +1 -1
- package/.agents/scripts/lib/coverage-capture.js +121 -81
- package/.agents/scripts/lib/full-suite-lock.js +49 -46
- package/.agents/scripts/lib/full-suite-queue.js +83 -8
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/observability/source-classifier.js +4 -1
- package/.agents/scripts/lib/orchestration/code-review.js +15 -3
- package/.agents/scripts/lib/orchestration/git-cleanup/phases/cli.js +1 -1
- package/.agents/scripts/lib/orchestration/merge-poll.js +5 -0
- package/.agents/scripts/lib/orchestration/plan-runner/worktree-sweep.js +149 -97
- package/.agents/scripts/lib/orchestration/review-deposit.js +219 -0
- package/.agents/scripts/lib/orchestration/review-providers/code-review.js +11 -7
- package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +29 -10
- package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +124 -73
- package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +38 -20
- package/.agents/scripts/lib/orchestration/single-story-close/phases/lock-wait-pending.js +8 -2
- package/.agents/scripts/lib/orchestration/single-story-close/review-overlap.js +161 -0
- package/.agents/scripts/lib/orchestration/single-story-close/runner.js +47 -7
- package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +8 -16
- package/.agents/scripts/lib/process-group.js +1 -1
- package/.agents/scripts/lib/single-story-sweep.js +2 -2
- package/.agents/scripts/lib/supervised-suite.js +247 -0
- package/.agents/scripts/lib/temp-removal.js +110 -0
- package/.agents/scripts/lib/temp-retention.js +122 -73
- package/.agents/scripts/lib/wave-runner/cross-run-overlap.js +120 -0
- package/.agents/scripts/lib/wave-runner/live-probe.js +5 -1
- package/.agents/scripts/lib/worktree/canonical-path.js +34 -0
- package/.agents/scripts/lib/worktree/lifecycle/reap.js +15 -4
- package/.agents/scripts/quality-preview.js +112 -14
- package/.agents/scripts/single-story-init.js +120 -17
- package/.agents/scripts/stories-wave-tick.js +47 -0
- package/.agents/scripts/story-review-compute.js +207 -0
- package/.agents/scripts/update-coverage-baseline.js +15 -10
- package/.agents/scripts/update-crap-baseline.js +12 -2
- package/.agents/scripts/update-maintainability-baseline.js +12 -2
- package/.agents/workflows/{git-cleanup.md → clean-git.md} +10 -10
- package/.agents/workflows/clean-temp.md +67 -0
- package/.agents/workflows/clean-worktrees.md +63 -0
- package/.agents/workflows/git-deliver.md +1 -1
- package/.agents/workflows/helpers/acceptance-self-eval.md +3 -2
- package/.agents/workflows/helpers/code-review.md +7 -5
- package/.agents/workflows/helpers/deliver-digest.md +39 -36
- package/.agents/workflows/helpers/deliver-reference.md +115 -5
- package/.agents/workflows/helpers/deliver-story-reference.md +2 -2
- package/.agents/workflows/helpers/deliver-story.md +2 -1
- package/docs/CHANGELOG.md +39 -0
- package/lib/cli/registry.js +125 -18
- package/lib/migrations/steps/strip-removed-agentrc-keys.js +0 -5
- package/package.json +1 -1
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
import { createRequire } from 'node:module';
|
|
9
9
|
import path from 'node:path';
|
|
10
|
+
import { resolveUpdaterRefreshScope } from './lib/baselines/coverage-refresh-scope.js';
|
|
10
11
|
import {
|
|
11
12
|
buildCoverageUpdaterScorer,
|
|
12
13
|
resolveCoverageUpdaterScope,
|
|
@@ -40,6 +41,7 @@ const USAGE = {
|
|
|
40
41
|
],
|
|
41
42
|
notes: [
|
|
42
43
|
'Run `npm run test:coverage` first — this script never runs the suite itself.',
|
|
44
|
+
'Against an `affected`-stamped artifact only measured rows are rewritten; rows the scoped run skipped are kept.',
|
|
43
45
|
],
|
|
44
46
|
};
|
|
45
47
|
|
|
@@ -75,16 +77,19 @@ function main() {
|
|
|
75
77
|
score: scoreCoverageFinal,
|
|
76
78
|
}),
|
|
77
79
|
};
|
|
78
|
-
// No flag
|
|
79
|
-
//
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
)
|
|
87
|
-
|
|
80
|
+
// No flag and a full artifact -> the service derives the diff via
|
|
81
|
+
// `origin/main..HEAD` (its default baseRef/headRef).
|
|
82
|
+
return resolveUpdaterRefreshScope(cwd, {
|
|
83
|
+
fullScope,
|
|
84
|
+
diffScopeRef,
|
|
85
|
+
loadScope: loadC8Scope,
|
|
86
|
+
})
|
|
87
|
+
.then((scope) => refreshBaseline({ ...refreshOpts, ...scope }))
|
|
88
|
+
.then((result) => {
|
|
89
|
+
Logger.info(
|
|
90
|
+
`[Coverage] ✅ Baseline updated: ${result.envelope.rows.length} file(s) recorded at ${COVERAGE_BASELINE_PATH} (${absBaselinePath}). scope=${result.scope.mode}, wrote=${result.wrote}.`,
|
|
91
|
+
);
|
|
92
|
+
});
|
|
88
93
|
}
|
|
89
94
|
|
|
90
95
|
runAsCli(import.meta.url, main, {
|
|
@@ -4,6 +4,7 @@ import {
|
|
|
4
4
|
buildCrapUpdaterScorer,
|
|
5
5
|
parseCrapUpdaterArgs,
|
|
6
6
|
resolveCrapUpdaterOptions,
|
|
7
|
+
seatCrapBaseline,
|
|
7
8
|
} from './lib/baselines/crap-updater-cli.js';
|
|
8
9
|
import { refreshBaseline } from './lib/baselines/refresh-service.js';
|
|
9
10
|
import { runAsCli } from './lib/cli-utils.js';
|
|
@@ -31,7 +32,7 @@ import { Logger } from './lib/Logger.js';
|
|
|
31
32
|
/** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
|
|
32
33
|
const USAGE = {
|
|
33
34
|
invocation:
|
|
34
|
-
'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>]',
|
|
35
|
+
'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>] [--seat-missing]',
|
|
35
36
|
summary:
|
|
36
37
|
'Scan → score → write the CRAP baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
|
|
37
38
|
flags: [
|
|
@@ -51,6 +52,10 @@ const USAGE = {
|
|
|
51
52
|
'--diff-scope <ref>',
|
|
52
53
|
'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
|
|
53
54
|
],
|
|
55
|
+
[
|
|
56
|
+
'--seat-missing',
|
|
57
|
+
'Insert-only: write rows ONLY for methods of changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Refuses unless the coverage capture stamp is fresh and method resolution is 100%. Prints `seated: N`. Incompatible with --full-scope.',
|
|
58
|
+
],
|
|
54
59
|
],
|
|
55
60
|
notes: [
|
|
56
61
|
'Run `npm run test:coverage` first — without a coverage artifact every file is skipped.',
|
|
@@ -91,7 +96,12 @@ async function main() {
|
|
|
91
96
|
);
|
|
92
97
|
}
|
|
93
98
|
|
|
94
|
-
|
|
99
|
+
async function seat() {
|
|
100
|
+
process.exitCode = await seatCrapBaseline(process.argv.slice(2));
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
const seating = process.argv.includes('--seat-missing');
|
|
104
|
+
runAsCli(import.meta.url, seating ? seat : main, {
|
|
95
105
|
source: 'crap-baseline',
|
|
96
106
|
usage: USAGE,
|
|
97
107
|
onError: (err) => {
|
|
@@ -9,6 +9,7 @@ import './lib/runtime-deps/ensure-installed.js';
|
|
|
9
9
|
import path from 'node:path';
|
|
10
10
|
import { parseDiffScopeFlag } from './lib/baselines/diff-scope-cli.js';
|
|
11
11
|
import { refreshBaseline } from './lib/baselines/refresh-service.js';
|
|
12
|
+
import { seatMaintainabilityBaseline } from './lib/baselines/seat-missing.js';
|
|
12
13
|
import { runAsCli } from './lib/cli-utils.js';
|
|
13
14
|
import { getBaselineEpsilon } from './lib/config/quality.js';
|
|
14
15
|
import { getBaselines, resolveConfig } from './lib/config-resolver.js';
|
|
@@ -17,7 +18,7 @@ import { Logger } from './lib/Logger.js';
|
|
|
17
18
|
/** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
|
|
18
19
|
const USAGE = {
|
|
19
20
|
invocation:
|
|
20
|
-
'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>]',
|
|
21
|
+
'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>] [--seat-missing]',
|
|
21
22
|
summary:
|
|
22
23
|
'Score → write the maintainability baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
|
|
23
24
|
flags: [
|
|
@@ -29,6 +30,10 @@ const USAGE = {
|
|
|
29
30
|
'--diff-scope <ref>',
|
|
30
31
|
'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
|
|
31
32
|
],
|
|
33
|
+
[
|
|
34
|
+
'--seat-missing',
|
|
35
|
+
'Insert-only: write rows ONLY for changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Prints `seated: N`. Incompatible with --full-scope.',
|
|
36
|
+
],
|
|
32
37
|
],
|
|
33
38
|
};
|
|
34
39
|
|
|
@@ -89,7 +94,12 @@ async function main() {
|
|
|
89
94
|
);
|
|
90
95
|
}
|
|
91
96
|
|
|
92
|
-
|
|
97
|
+
async function seat() {
|
|
98
|
+
process.exitCode = await seatMaintainabilityBaseline(process.argv.slice(2));
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const seating = process.argv.includes('--seat-missing');
|
|
102
|
+
runAsCli(import.meta.url, seating ? seat : main, {
|
|
93
103
|
source: 'maintainability-baseline',
|
|
94
104
|
usage: USAGE,
|
|
95
105
|
onError: (err) => {
|
|
@@ -5,9 +5,9 @@ description: >-
|
|
|
5
5
|
`git stash` entries — each step gated by operator confirmation.
|
|
6
6
|
---
|
|
7
7
|
|
|
8
|
-
# /git
|
|
8
|
+
# /clean-git [--fast-forward-main] [--prune-remotes] [--branches] [--stashes] [--execute] [--remote] [--yes] [--include-content-merged] [--drop-stashes <ref>] [--exclude <pattern>] [--json]
|
|
9
9
|
|
|
10
|
-
`/git
|
|
10
|
+
`/clean-git` folds the four cleanup steps operators routinely run by hand
|
|
11
11
|
after a busy session into a single pipeline with per-step confirmation. It is a
|
|
12
12
|
**recovery tool**, not a routine chore: the delivering flows already reap their
|
|
13
13
|
own merged refs and fast-forward the base branch (see
|
|
@@ -22,13 +22,13 @@ Reach for it when the automated hygiene left an unusual state behind.
|
|
|
22
22
|
> skill citation.
|
|
23
23
|
|
|
24
24
|
The enumeration + reap logic lives in
|
|
25
|
-
[`git
|
|
25
|
+
[`clean-git.js`](../scripts/clean-git.js) — it computes the candidate list,
|
|
26
26
|
the skip taxonomy, the detection signals, and the JSON envelope, and prints them
|
|
27
27
|
itself. Without `--execute` the script is a **dry-run preview**; nothing is
|
|
28
28
|
mutated. When no phase flag is passed, **all four phases run** sequentially; a
|
|
29
29
|
phase flag narrows the run. A failure in one phase does not short-circuit the
|
|
30
30
|
others — each runs and reports independently. The script documents its own
|
|
31
|
-
flags: `node .agents/scripts/git
|
|
31
|
+
flags: `node .agents/scripts/clean-git.js --help`.
|
|
32
32
|
|
|
33
33
|
## Phases
|
|
34
34
|
|
|
@@ -65,24 +65,24 @@ merged-PR branch in scope unless `--exclude`d.
|
|
|
65
65
|
|
|
66
66
|
```bash
|
|
67
67
|
# Preview all four phases (no mutation).
|
|
68
|
-
node .agents/scripts/git
|
|
68
|
+
node .agents/scripts/clean-git.js
|
|
69
69
|
|
|
70
70
|
# Run everything non-interactively, including origin refs. Branches detected
|
|
71
71
|
# only by content-equivalence keep their origin ref — see the note below.
|
|
72
|
-
node .agents/scripts/git
|
|
72
|
+
node .agents/scripts/clean-git.js --execute --remote --yes
|
|
73
73
|
|
|
74
74
|
# Same, but also delete the origin refs of content-merged branches. Nobody is
|
|
75
75
|
# watching, so opting in is the whole confirmation this delete ever gets.
|
|
76
|
-
node .agents/scripts/git
|
|
76
|
+
node .agents/scripts/clean-git.js --execute --remote --yes \
|
|
77
77
|
--include-content-merged
|
|
78
78
|
|
|
79
79
|
# Only fast-forward main.
|
|
80
|
-
node .agents/scripts/git
|
|
80
|
+
node .agents/scripts/clean-git.js --fast-forward-main --execute
|
|
81
81
|
|
|
82
82
|
# Only sweep merged branches + their origin refs.
|
|
83
|
-
node .agents/scripts/git
|
|
83
|
+
node .agents/scripts/clean-git.js --branches --execute --remote
|
|
84
84
|
|
|
85
85
|
# Drop specific stashes under --yes.
|
|
86
|
-
node .agents/scripts/git
|
|
86
|
+
node .agents/scripts/clean-git.js --stashes --execute --yes \
|
|
87
87
|
--drop-stashes 'stash@{0}' --drop-stashes 'stash@{2}'
|
|
88
88
|
```
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: >-
|
|
3
|
+
Clear the temp-tree backlog the land-time purge cannot attribute: sort every
|
|
4
|
+
top-level entry under the project's tempRoot into framework, closed-issue,
|
|
5
|
+
aged and kept buckets, preview by default, and delete only confirmed buckets.
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
# /clean-temp [--execute] [--yes] [--json]
|
|
9
|
+
|
|
10
|
+
The land-time and boot-time purges only reap what they can attribute: the
|
|
11
|
+
framework's own temp layouts and `temp/scratch/`. Everything else an agent
|
|
12
|
+
dropped at the temp root is reported and left alone, so a busy consumer's temp
|
|
13
|
+
tree grows without bound. `/clean-temp` is the operator's catch-up for that
|
|
14
|
+
backlog. It classifies and deletes through the same temp-retention engine the
|
|
15
|
+
purges use — there is no second walker.
|
|
16
|
+
|
|
17
|
+
The script documents its own flags:
|
|
18
|
+
`node .agents/scripts/clean-temp.js --help`. Without `--execute` it is a
|
|
19
|
+
**dry-run preview**; nothing is deleted.
|
|
20
|
+
|
|
21
|
+
## Buckets
|
|
22
|
+
|
|
23
|
+
Every top-level entry under tempRoot lands in exactly one:
|
|
24
|
+
|
|
25
|
+
| Bucket | What it holds | Unattended (`--yes`) |
|
|
26
|
+
| --- | --- | --- |
|
|
27
|
+
| **framework** | A framework layout holding artifacts the auto-purge would take — Story-keyed on a closed Story, or past `staleDays`. | Deleted |
|
|
28
|
+
| **closed-issue** | An unrecognized entry whose basename names exactly one issue id, and that issue reads `closed`. | Deleted |
|
|
29
|
+
| **aged** | An unrecognized entry naming no id, or several, older than `staleDays`. | **Never** — an age heuristic has no attribution, so it needs a human |
|
|
30
|
+
| **kept** | Everything else, with the reason: issue open, issue read failed, too recent, reserved, or nothing spent. | Kept |
|
|
31
|
+
|
|
32
|
+
An id is a standalone run of up to seven digits in the basename. A basename
|
|
33
|
+
naming two ids is never attributed to either — it is treated as id-less.
|
|
34
|
+
|
|
35
|
+
## Constraint
|
|
36
|
+
|
|
37
|
+
> [!WARNING] `--execute` deletes files. Interactive runs confirm each bucket;
|
|
38
|
+
> `--yes` deletes the framework and closed-issue buckets only.
|
|
39
|
+
|
|
40
|
+
- Reads fail safe: an open issue, or one whose read fails, keeps its entry.
|
|
41
|
+
- `qa/`, `cache/`, `*.lock` and `signals.ndjson` at any depth are never
|
|
42
|
+
deleted.
|
|
43
|
+
- Project-scoped: the script exits 1 without deleting anything when the
|
|
44
|
+
resolved tempRoot is not inside the project root it was invoked from. It
|
|
45
|
+
never touches `$TMPDIR` or a sibling checkout.
|
|
46
|
+
|
|
47
|
+
## Steps
|
|
48
|
+
|
|
49
|
+
1. Preview and read the table (bucket, entry, size, age, reason):
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
node .agents/scripts/clean-temp.js
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
2. Delete, confirming each bucket:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
node .agents/scripts/clean-temp.js --execute
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Unattended, `--execute --yes` deletes the framework and closed-issue buckets
|
|
62
|
+
and reports the aged bucket as left for a human. `--json` emits the envelope
|
|
63
|
+
(per-bucket totals and `bytesReclaimed`) instead of the table.
|
|
64
|
+
|
|
65
|
+
Going forward, put ad-hoc scratch under `temp/scratch/story-<id>/` (or
|
|
66
|
+
`temp/scratch/` with no Story): a Story's landing reaps its scratch directory,
|
|
67
|
+
and the boot sweep age-floors the rest.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: >-
|
|
3
|
+
Reclaim disk from dead worktrees: list every worktree of this project as a
|
|
4
|
+
removal candidate (closed Story, merged branch, orphaned directory, detached
|
|
5
|
+
HEAD) or as kept with a reason, then remove candidates only on `--execute`.
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
# /clean-worktrees [--execute] [--yes] [--json]
|
|
9
|
+
|
|
10
|
+
Every Story worktree carries its own `node_modules`, so a worktree left behind
|
|
11
|
+
costs gigabytes. Workflow boot already removes `.worktrees/story-<id>` trees
|
|
12
|
+
whose Story is closed or `agent::done` (the boot sweep in
|
|
13
|
+
[`boot-sweep.js`](../scripts/boot-sweep.js)); `/clean-worktrees` is the
|
|
14
|
+
**recovery tool** for everything that sweep does not own — the backlog, trees
|
|
15
|
+
on non-Story branches, directories git no longer registers, and detached
|
|
16
|
+
trees such as the Claude Code app's `.claude/worktrees/*`.
|
|
17
|
+
|
|
18
|
+
The enumeration, classification and removal live in
|
|
19
|
+
[`clean-worktrees.js`](../scripts/clean-worktrees.js), which documents its own
|
|
20
|
+
flags: `node .agents/scripts/clean-worktrees.js --help`.
|
|
21
|
+
|
|
22
|
+
## Steps
|
|
23
|
+
|
|
24
|
+
1. **Preview.** Run `node .agents/scripts/clean-worktrees.js` (dry-run, the
|
|
25
|
+
default). It prints one row per worktree — class, path, size, branch/HEAD,
|
|
26
|
+
action — and removes nothing. Show the operator the table.
|
|
27
|
+
2. **Confirm.** Ask the operator which candidates to remove. Do not proceed
|
|
28
|
+
on your own judgment.
|
|
29
|
+
3. **Remove.** Re-run with `--execute`. In a terminal it asks per entry;
|
|
30
|
+
`--execute --yes` removes every non-`detached` candidate without asking.
|
|
31
|
+
`--json` emits the envelope, including `bytesReclaimed`.
|
|
32
|
+
|
|
33
|
+
## Classes
|
|
34
|
+
|
|
35
|
+
| Class | Candidate when |
|
|
36
|
+
| --- | --- |
|
|
37
|
+
| `closed-story` | `.worktrees/story-<id>` on `story-<id>` whose Story is closed or `agent::done`. |
|
|
38
|
+
| `merged-branch` | Any other branch whose PR is MERGED and whose HEAD is the merged head. |
|
|
39
|
+
| `orphan-dir` | A directory under `.worktrees/` that `git worktree list` does not register. |
|
|
40
|
+
| `detached` | A registered worktree with a detached HEAD. |
|
|
41
|
+
|
|
42
|
+
Everything else is **kept** with its reason: the main checkout, an open Story,
|
|
43
|
+
an unmerged branch, a dirty tree, unpushed commits, a tree in use.
|
|
44
|
+
|
|
45
|
+
## Constraint
|
|
46
|
+
|
|
47
|
+
> [!WARNING] `--execute` deletes worktree directories. Without it the script
|
|
48
|
+
> only previews.
|
|
49
|
+
|
|
50
|
+
- **Project-scoped.** Only worktrees inside the invoking checkout's project
|
|
51
|
+
root are candidates; one registered elsewhere is reported, never removed.
|
|
52
|
+
- **Unique work survives.** A dirty tree, or a HEAD no remote-tracking ref
|
|
53
|
+
contains, is never removed.
|
|
54
|
+
- **Live trees survive.** The tree this process runs from is never removed;
|
|
55
|
+
on macOS and Linux a tree any live process uses is refused outright.
|
|
56
|
+
- **`detached` needs a person.** A detached tree carries no Story label and a
|
|
57
|
+
live session may own it (a blanket prune once destroyed a live delivery's
|
|
58
|
+
worktree), so it is removed only on a per-entry interactive yes — never
|
|
59
|
+
under `--yes`.
|
|
60
|
+
- **Removal goes through the worktree removal seam** (Windows lock retry and
|
|
61
|
+
the pending-cleanup hand-off), never raw deletion.
|
|
62
|
+
- **Branches are untouched.** Deleting local or remote branches is
|
|
63
|
+
[`/clean-git`](clean-git.md)'s job.
|
|
@@ -77,7 +77,7 @@ node .agents/scripts/boot-sweep.js \
|
|
|
77
77
|
--current "$(git rev-parse --abbrev-ref HEAD)"
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
The safe subset of the `/git
|
|
80
|
+
The safe subset of the `/clean-git` phases: fast-forwards the base branch,
|
|
81
81
|
prunes stale remote-tracking refs, and reaps merged branches. It never
|
|
82
82
|
touches the stash, never reaps a candidate with unpushed work / dirty
|
|
83
83
|
worktree / open parent ticket, and always exits `0` — a failed sweep is
|
|
@@ -95,7 +95,7 @@ per-criterion, mid-delivery, and evaluates the actual work product.
|
|
|
95
95
|
without being respawned; a stale or absent stamp reports `spawn: true` and
|
|
96
96
|
the command runs for real. The credited run itself is stated once, in
|
|
97
97
|
[`deliver-digest.md`](deliver-digest.md) § 5.
|
|
98
|
-
+ Emits **one** verdict file under `temp
|
|
98
|
+
+ Emits **one** verdict file under `temp/scratch/story-<id>/` conforming to
|
|
99
99
|
[`acceptance-eval-verdict.schema.json`](../../schemas/acceptance-eval-verdict.schema.json):
|
|
100
100
|
one `{ index, criterion, verdict: met|partial|unmet, evidence,
|
|
101
101
|
verifyEvidence[] }` record per `acceptance[]` item, in acceptance-array
|
|
@@ -137,4 +137,5 @@ per-criterion, mid-delivery, and evaluates the actual work product.
|
|
|
137
137
|
(transition to `agent::blocked`) and post a `friction` comment naming the
|
|
138
138
|
unmet criteria and their evidence. Never silently proceed to close.
|
|
139
139
|
|
|
140
|
-
Write the verdict under `temp
|
|
140
|
+
Write the verdict under `temp/scratch/story-<id>/` only — a scratch artifact
|
|
141
|
+
the Story's landing reaps.
|
|
@@ -13,10 +13,12 @@ description: >-
|
|
|
13
13
|
This helper performs a comprehensive code review of a change set before it
|
|
14
14
|
is merged to `main`. The live v2 path is **Story scope only**:
|
|
15
15
|
|
|
16
|
-
- **Story scope** — reviews `
|
|
17
|
-
`
|
|
18
|
-
|
|
19
|
-
|
|
16
|
+
- **Story scope** — reviews `origin/<baseBranch>...<sha>` inside
|
|
17
|
+
`single-story-close.js`, computed alongside the close-validation gates
|
|
18
|
+
once the pre-gate self-heal commits land (`<sha>` is that HEAD) and
|
|
19
|
+
posted after the PR opens, before auto-merge; a HEAD that moved by then
|
|
20
|
+
re-reviews serially. Findings post to the PR; critical findings block
|
|
21
|
+
close (`agent::blocked`).
|
|
20
22
|
|
|
21
23
|
**Invariant — Story-scope review runs outside the maker's LLM context.**
|
|
22
24
|
The Story-scope review executes inside the `single-story-close.js` close
|
|
@@ -24,7 +26,7 @@ subprocess, **not** in the delivering child's (maker agent's) LLM context.
|
|
|
24
26
|
The close pipeline invokes it after the delivering child has exited, so
|
|
25
27
|
the change set is reviewed by a process the maker cannot influence. The
|
|
26
28
|
enforcing code path is
|
|
27
|
-
[`
|
|
29
|
+
[`computeStoryScopeReview`](../../scripts/lib/orchestration/single-story-close/phases/code-review.js)
|
|
28
30
|
→ shared
|
|
29
31
|
[`runStoryReviewCore`](../../scripts/lib/orchestration/story-close/phases/review-core.js).
|
|
30
32
|
A future refactor MUST preserve this isolation: do not move Story-scope
|
|
@@ -77,8 +77,6 @@ therefore buys a **deep review**, not a fresh acceptance critic.
|
|
|
77
77
|
> [`acceptance-self-eval.md`](acceptance-self-eval.md) — a host that cannot
|
|
78
78
|
> spawn the critic at all, noted in the friction comment if you block.
|
|
79
79
|
|
|
80
|
-
`--base <ref>` overrides `project.baseBranch`.
|
|
81
|
-
|
|
82
80
|
## 4. Acceptance self-eval (Step 1a, required)
|
|
83
81
|
|
|
84
82
|
**One verdict owner per Story** — named by `verdictOwner`: the inline
|
|
@@ -93,48 +91,55 @@ scored in **one** gate call. Bounded by `delivery.acceptanceEval.maxRounds`
|
|
|
93
91
|
--verdict <verdict-path>`
|
|
94
92
|
|
|
95
93
|
The gate reads the Story's `acceptance[]` count itself and rejects a verdict
|
|
96
|
-
whose `criteria[]` length differs **before** scoring, consuming no round
|
|
97
|
-
|
|
98
|
-
|
|
94
|
+
whose `criteria[]` length differs **before** scoring, consuming no round. A
|
|
95
|
+
second gate call in the same round spends a round for nothing and races the
|
|
96
|
+
Story-scoped ledger.
|
|
99
97
|
|
|
100
98
|
`proceed` → close. `redraft` → one more round inside the cap. `block` → **do
|
|
101
99
|
not close**: post a `friction` comment and flip `agent::blocked`.
|
|
102
100
|
Per-round mechanics: [`acceptance-self-eval.md`](acceptance-self-eval.md).
|
|
103
101
|
|
|
104
|
-
## 5. The one credited
|
|
102
|
+
## 5. The one credited run
|
|
105
103
|
|
|
106
|
-
After the self-eval loop
|
|
107
|
-
`
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
so any runner earns the credit:
|
|
104
|
+
**Preflight first — blocking.** After the self-eval loop, run the configured
|
|
105
|
+
`project.commands.lint` and `node <main-repo>/.agents/scripts/quality-preview.js
|
|
106
|
+
--changed-since origin/<baseBranch>` in the worktree; fix and commit every
|
|
107
|
+
finding. Close's gates stay authoritative
|
|
108
|
+
([`deliver-reference.md`](deliver-reference.md) § Preflight).
|
|
112
109
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
110
|
+
After the last fix commit, fetch and merge
|
|
111
|
+
`origin/<baseBranch>` into the Story branch **first**, ahead of this run and
|
|
112
|
+
the push: close's base-sync then no-ops, so neither stamp goes stale. Then run
|
|
113
|
+
**one** depositor in the worktree, picked by the predicate close registers
|
|
114
|
+
`coverage-capture` on — CRAP gate enabled **and** a `test:coverage` script:
|
|
115
|
+
|
|
116
|
+
- **Capture-active** →
|
|
117
|
+
`node <main-repo>/.agents/scripts/coverage-capture.js --cwd <workCwd>`.
|
|
118
|
+
It takes the host full-suite lock, runs `test:coverage` and writes the stamp
|
|
119
|
+
close's capture finds fresh. Signal: `Wrote content-digest capture stamp`.
|
|
120
|
+
Non-zero is a red suite (or the lock / timeout it names): fix, re-run. No
|
|
121
|
+
second `npm test`.
|
|
122
|
+
- **Otherwise** → `node <main-repo>/.agents/scripts/evidence-gate.js
|
|
123
|
+
--standalone --scope-id <storyId> --gate test --worktree <workCwd> -- npm test`.
|
|
124
|
+
Signal: `✓ test passed` — the `test` evidence close reads, keyed on the tree.
|
|
125
|
+
|
|
126
|
+
**Seat, then push.** CRAP or MI gate on → run
|
|
127
|
+
`update-crap-baseline.js --seat-missing` and its maintainability twin (signal
|
|
128
|
+
`seated: N`); commit it as `chore(baselines): baseline-refresh: …`.
|
|
129
|
+
|
|
130
|
+
A later commit voids either credit; a baseline-JSON seat keeps the stamp. Read
|
|
131
|
+
the **output**, not the exit code: a run that prints no signal deposits
|
|
132
|
+
nothing. `mandrel doctor`'s `test-credit-path` check names this project's
|
|
133
|
+
depositor. Seat order, runner shapes, background dispatch, redraft rounds:
|
|
134
|
+
[`deliver-reference.md`](deliver-reference.md) § Credited run.
|
|
133
135
|
|
|
134
136
|
`verify[]` is scoped entries **plus** this one run: an entry that is itself a
|
|
135
137
|
full-suite command is reported credited against the same record, never
|
|
136
138
|
respawned.
|
|
137
139
|
|
|
140
|
+
After the push, compute the held review and fix a CRITICAL before hand-off:
|
|
141
|
+
[`deliver-reference.md`](deliver-reference.md) § Held review.
|
|
142
|
+
|
|
138
143
|
## 6. Terminal envelope — the return contract
|
|
139
144
|
|
|
140
145
|
`single-story-close.js` emits exactly one envelope on stdout between
|
|
@@ -157,10 +162,8 @@ Required fields: `kind` (`story-deliver-terminal`), `storyId`, `status`,
|
|
|
157
162
|
reports every gate as `passed` / `failed` / `skipped` — a skipped gate is
|
|
158
163
|
reported, never omitted, so a missing gate is never read as a passing one.
|
|
159
164
|
|
|
160
|
-
|
|
161
|
-
`
|
|
162
|
-
success; a failed gate replays its tail inline. `AGENT_LOG_LEVEL=verbose`
|
|
163
|
-
restores live streaming.
|
|
165
|
+
Gate output is captured to a log, not streamed
|
|
166
|
+
([`deliver-reference.md`](deliver-reference.md) § Gate output).
|
|
164
167
|
|
|
165
168
|
## 7. When to leave this file
|
|
166
169
|
|
|
@@ -326,11 +326,15 @@ diff selects **review depth**.
|
|
|
326
326
|
|
|
327
327
|
## Async merge-confirm mode (`delivery.mergeWatch.mode: "async"`)
|
|
328
328
|
|
|
329
|
-
In async mode the close arms auto-merge
|
|
330
|
-
|
|
331
|
-
required check
|
|
332
|
-
|
|
333
|
-
|
|
329
|
+
In async mode the close arms auto-merge and probes the PR **once**. A
|
|
330
|
+
definitive probe settles as in sync mode — merged lands, closed or a red
|
|
331
|
+
required check (the head-anchored predicate) blocks, a red advisory gate
|
|
332
|
+
blocks. A probe whose checks have not started or are still running returns
|
|
333
|
+
`pending` with a `nextCommand` at once, no sleep: CI never reddens inside the
|
|
334
|
+
first minute, so a second probe could only hold the serialized slot. Two
|
|
335
|
+
shapes poll on inside the ~60s window: checks already **green** (the merge is
|
|
336
|
+
imminent — observed at the green cadence, it saves a whole confirm
|
|
337
|
+
invocation) and a red rollup still awaiting its confirming probe. When a close returns that `pending` envelope, launch its
|
|
334
338
|
`nextCommand` (`single-story-confirm-merge.js … --wait`) as a **background**
|
|
335
339
|
invocation (host background Bash — its completion re-invokes the agent) and
|
|
336
340
|
move on to the next Story; `single-story-confirm-merge.js` is idempotent and
|
|
@@ -352,3 +356,109 @@ config default stays `"sync"`. A slow-CI solo consumer may opt into `"async"`
|
|
|
352
356
|
for the same reason — a foreground wait longer than the host tool ceiling
|
|
353
357
|
expires `pending` anyway. Otherwise a one-Story run keeps `sync`: there is no
|
|
354
358
|
sibling to unblock, and the foreground wait is the cheapest path to `landed`.
|
|
359
|
+
|
|
360
|
+
## Preflight (before close) {#preflight}
|
|
361
|
+
|
|
362
|
+
Digest § 5 states the rule: before the credited suite run, the worker runs
|
|
363
|
+
the configured `project.commands.lint` (falling back to `npm run lint`) and
|
|
364
|
+
`quality-preview.js --changed-since origin/<baseBranch>` in the worktree, and
|
|
365
|
+
fixes and commits every finding. It runs **before** the credited run because a
|
|
366
|
+
fix commit afterwards would void that run's credit.
|
|
367
|
+
|
|
368
|
+
- **Why.** Close runs lint and the maintainability half of the preview
|
|
369
|
+
(`quality-preview-mi`) in its parallel phase, but a regression found there
|
|
370
|
+
still costs a close round-trip. Seconds of preflight in the worktree is
|
|
371
|
+
cheaper than any close.
|
|
372
|
+
- **The CRAP half never captures.** It scores whatever coverage artifact is
|
|
373
|
+
on disk — the worker's credited capture when one exists — and triggers no
|
|
374
|
+
capture of its own. With no artifact its methods report unscorable; a stale
|
|
375
|
+
one can invent a violation. `--only mi` runs the maintainability half alone.
|
|
376
|
+
- **Close stays authoritative.** Its `quality-preview-crap` gate scores a
|
|
377
|
+
fresh capture after `coverage-capture`; a preflight pass never skips it.
|
|
378
|
+
|
|
379
|
+
## Credited run (situational) {#credited-run}
|
|
380
|
+
|
|
381
|
+
Digest § 5 states the rule and both invocations; this is what surrounds them.
|
|
382
|
+
|
|
383
|
+
- **Why the capture.** On a capture-active project close registers
|
|
384
|
+
`coverage-capture` instead of the plain `test` gate, so an evidence-gate
|
|
385
|
+
`test` deposit buys nothing there: close logs `no credited capture stamp
|
|
386
|
+
covers this change set` and pays the whole suite on the serialized tail. The
|
|
387
|
+
worker's capture writes the content-digest stamp in the worktree, and
|
|
388
|
+
close's capture then exits on its freshness probe without spawning
|
|
389
|
+
`test:coverage`. A base-sync that merges a path under `crap.targetDirs`
|
|
390
|
+
spends the stamp, and close re-captures.
|
|
391
|
+
- **Runner shapes.** A bare `npm test` earns the `test` credit **only** where
|
|
392
|
+
the project's test script routes through mandrel's own runner, which prints
|
|
393
|
+
the outcome. On any other runner it deposits nothing and prints nothing, so
|
|
394
|
+
silence is never evidence of credit.
|
|
395
|
+
- **Background dispatch.** If the run outruns the host's sync Bash ceiling,
|
|
396
|
+
dispatch it in the **background** — its completion re-invokes you; never
|
|
397
|
+
spawn a task to poll or `sleep`-loop against it
|
|
398
|
+
([`parallel-tooling.md`](parallel-tooling.md) Rule 2).
|
|
399
|
+
- **Redraft rounds.** Run the scoped projects for the roots you changed plus
|
|
400
|
+
`verify[]`, not the whole suite; only the one run needs credit.
|
|
401
|
+
- **Seating new methods (`--seat-missing`).** Close fails a Story whose own
|
|
402
|
+
new methods have no baseline row, so after the credited run and before the
|
|
403
|
+
push the worker runs, in `<workCwd>`, for each enabled gate:
|
|
404
|
+
`node .agents/scripts/update-crap-baseline.js --seat-missing` (CRAP) and
|
|
405
|
+
`node .agents/scripts/update-maintainability-baseline.js --seat-missing`
|
|
406
|
+
(MI). Each scores the files changed since the `origin/<baseBranch>`
|
|
407
|
+
merge-base and writes **only** rows whose (path, method) key is absent —
|
|
408
|
+
every existing row stays byte-identical, including rows whose scores moved
|
|
409
|
+
(re-scoring stays close's auto-refresh). It prints `seated: N`; `seated: 0`
|
|
410
|
+
writes nothing. Commit a change as `chore(baselines): baseline-refresh: …`.
|
|
411
|
+
- **Seat refusals.** The CRAP seat exits non-zero and writes nothing unless
|
|
412
|
+
the coverage-capture stamp is fresh for the tree **and** method resolution
|
|
413
|
+
over the in-scope files is exactly 100% — a lower rate means the artifact's
|
|
414
|
+
coordinates predate the tree. The refusal names the rate, the unresolved
|
|
415
|
+
files and the fix: re-run the digest § 5 capture, then seat.
|
|
416
|
+
- **Seat order vs credit.** The capture stamp digests scorable sources only,
|
|
417
|
+
so a baseline-JSON-only seat commit leaves it fresh and the capture credit
|
|
418
|
+
stands. The evidence-gate `test` credit is keyed on the tree, so on that
|
|
419
|
+
path seat first — MI is static and needs no coverage — then run the suite.
|
|
420
|
+
|
|
421
|
+
## Held review at hand-off {#held-review}
|
|
422
|
+
|
|
423
|
+
The one home of this rule; the digest and the worker contract point here.
|
|
424
|
+
After the credited run **and** the push, the worker computes the Story-scope
|
|
425
|
+
code review itself, before handing off:
|
|
426
|
+
|
|
427
|
+
```bash
|
|
428
|
+
node <main-repo>/.agents/scripts/story-review-compute.js --story <storyId> --cwd <workCwd>
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
- **What it does.** It runs close's own review computation (the configured
|
|
432
|
+
provider chain) against `origin/<baseBranch>...story-<id>`, posts nothing,
|
|
433
|
+
takes no full-suite lock, and writes
|
|
434
|
+
`temp/orchestration/story-review-<id>.json` beside the terminal envelope,
|
|
435
|
+
keyed on the **diff digest** (sha256 of the exact three-dot diff text). It
|
|
436
|
+
exits 0 whatever the findings; non-zero means the provider threw — report
|
|
437
|
+
it in the hand-off; close computes the review itself.
|
|
438
|
+
- **A CRITICAL is the worker's to fix.** Fix, commit, re-run the credited
|
|
439
|
+
run (the fix commit voided its credit), push, and re-run the compute — the
|
|
440
|
+
acceptance loop's redraft discipline, bounded by
|
|
441
|
+
`delivery.acceptanceEval.maxRounds`. Still CRITICAL at the cap → take the
|
|
442
|
+
blocked path. Anything else goes in the hand-off as the severity tally.
|
|
443
|
+
- **Close adopts, else computes.** When the deposit's digest equals the
|
|
444
|
+
digest of the diff at close's held-review start, close starts no review and
|
|
445
|
+
posts the deposit after PR-open. A clean base-sync merge moves HEAD without
|
|
446
|
+
changing the diff, so it keeps the deposit; a commit that changes the diff
|
|
447
|
+
does not, and close reviews as it would without one. The CRITICAL halt and
|
|
448
|
+
`--override-review-block` apply to an adopted result unchanged.
|
|
449
|
+
|
|
450
|
+
## Gate output {#gate-output}
|
|
451
|
+
|
|
452
|
+
Close writes gate lines to `temp/orchestration/close-gates-<storyId>.log` and
|
|
453
|
+
reports a one-line digest on success; a failed gate replays its tail inline.
|
|
454
|
+
`AGENT_LOG_LEVEL=verbose` restores live streaming. A gate exiting `75` (its
|
|
455
|
+
full-suite lock wait expired) logs a deferred line and the close settles
|
|
456
|
+
`pending`; one exiting `124` (the suite outran its timeout) logs a timeout
|
|
457
|
+
line naming host contention — neither is reported as failing tests.
|
|
458
|
+
|
|
459
|
+
Every full-suite run ends on `⏲ suite timings: lockWaitMs=… hostWaitMs=…
|
|
460
|
+
testRunMs=…`, which close carries as the envelope's `suiteTimings`. The
|
|
461
|
+
`coverage.timeoutMs` clock starts at spawn, never in the lock queue; a suite
|
|
462
|
+
that writes `$MANDREL_SUITE_READY_FILE` when its tests start (after a
|
|
463
|
+
consumer load gate) gets a fresh bound for them, so worst-case wall is lock
|
|
464
|
+
wait + 2 × `timeoutMs`.
|
|
@@ -48,7 +48,7 @@ presence, so re-running init on a partially-initialized Story is idempotent.
|
|
|
48
48
|
### Merged-`story-*` sweep
|
|
49
49
|
|
|
50
50
|
Between the fetch and the branch seed, init runs the same primitive as
|
|
51
|
-
`<agentRoot>/scripts/git
|
|
51
|
+
`<agentRoot>/scripts/clean-git.js` scoped to `story-*` in
|
|
52
52
|
`--execute --remote` mode, excluding this run's `story-<id>`, reaping merged
|
|
53
53
|
siblings' local refs, `origin/` refs and stale tracking refs in one pass.
|
|
54
54
|
|
|
@@ -539,7 +539,7 @@ once per project to delete the conflicting bot workflows entirely.
|
|
|
539
539
|
> cause), or after a manual merge on a `--no-wait-merge` run:
|
|
540
540
|
|
|
541
541
|
```bash
|
|
542
|
-
node .agents/scripts/git
|
|
542
|
+
node .agents/scripts/clean-git.js \
|
|
543
543
|
--execute \
|
|
544
544
|
--remote \
|
|
545
545
|
--yes \
|