mandrel 2.67.0 → 2.68.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/.agents/agents/story-worker.md +15 -11
  2. package/.agents/docs/agentrc-reference.json +3 -1
  3. package/.agents/docs/configuration.md +36 -1
  4. package/.agents/schemas/agentrc.schema.json +14 -1
  5. package/.agents/schemas/story-deliver-terminal.schema.json +23 -1
  6. package/.agents/schemas/validation-evidence.schema.json +3 -1
  7. package/.agents/scripts/coverage-capture.js +65 -9
  8. package/.agents/scripts/evidence-gate.js +106 -8
  9. package/.agents/scripts/lib/baselines/coverage-refresh-scope.js +60 -0
  10. package/.agents/scripts/lib/baselines/crap-updater-cli.js +101 -4
  11. package/.agents/scripts/lib/baselines/refresh-service.js +1 -1
  12. package/.agents/scripts/lib/baselines/seat-missing.js +228 -0
  13. package/.agents/scripts/lib/child-exec.js +39 -1
  14. package/.agents/scripts/lib/close-validation/gates.js +59 -19
  15. package/.agents/scripts/lib/close-validation/process.js +23 -24
  16. package/.agents/scripts/lib/close-validation/runner.js +71 -40
  17. package/.agents/scripts/lib/config/gates/coverage.schema.js +21 -0
  18. package/.agents/scripts/lib/config/quality.js +7 -1
  19. package/.agents/scripts/lib/config/temp-paths.js +15 -0
  20. package/.agents/scripts/lib/config-settings-schema-delivery.js +1 -1
  21. package/.agents/scripts/lib/coverage-baseline.js +78 -5
  22. package/.agents/scripts/lib/coverage-capture-affected.js +345 -0
  23. package/.agents/scripts/lib/coverage-capture-delta.js +180 -0
  24. package/.agents/scripts/lib/coverage-capture-fullscope.js +53 -32
  25. package/.agents/scripts/lib/coverage-capture-incremental.js +49 -26
  26. package/.agents/scripts/lib/coverage-capture-usage.js +1 -1
  27. package/.agents/scripts/lib/coverage-capture.js +121 -81
  28. package/.agents/scripts/lib/full-suite-lock.js +49 -46
  29. package/.agents/scripts/lib/full-suite-queue.js +83 -8
  30. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  31. package/.agents/scripts/lib/observability/source-classifier.js +1 -0
  32. package/.agents/scripts/lib/orchestration/code-review.js +15 -3
  33. package/.agents/scripts/lib/orchestration/merge-poll.js +5 -0
  34. package/.agents/scripts/lib/orchestration/review-deposit.js +219 -0
  35. package/.agents/scripts/lib/orchestration/review-providers/code-review.js +11 -7
  36. package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +29 -10
  37. package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +124 -73
  38. package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +38 -20
  39. package/.agents/scripts/lib/orchestration/single-story-close/phases/lock-wait-pending.js +8 -2
  40. package/.agents/scripts/lib/orchestration/single-story-close/review-overlap.js +161 -0
  41. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +47 -7
  42. package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +8 -16
  43. package/.agents/scripts/lib/process-group.js +1 -1
  44. package/.agents/scripts/lib/supervised-suite.js +247 -0
  45. package/.agents/scripts/lib/wave-runner/cross-run-overlap.js +120 -0
  46. package/.agents/scripts/lib/wave-runner/live-probe.js +5 -1
  47. package/.agents/scripts/quality-preview.js +112 -14
  48. package/.agents/scripts/stories-wave-tick.js +47 -0
  49. package/.agents/scripts/story-review-compute.js +207 -0
  50. package/.agents/scripts/update-coverage-baseline.js +15 -10
  51. package/.agents/scripts/update-crap-baseline.js +12 -2
  52. package/.agents/scripts/update-maintainability-baseline.js +12 -2
  53. package/.agents/workflows/helpers/code-review.md +7 -5
  54. package/.agents/workflows/helpers/deliver-digest.md +39 -36
  55. package/.agents/workflows/helpers/deliver-reference.md +115 -5
  56. package/.agents/workflows/helpers/deliver-story.md +2 -1
  57. package/docs/CHANGELOG.md +26 -0
  58. package/lib/cli/registry.js +125 -18
  59. package/lib/migrations/steps/strip-removed-agentrc-keys.js +0 -5
  60. package/package.json +1 -1
@@ -0,0 +1,207 @@
1
+ #!/usr/bin/env node
2
+
3
+ /**
4
+ * story-review-compute.js — compute the Story-scope review on the worker, at
5
+ * push time, and deposit it for close to adopt.
6
+ *
7
+ * Runs close's compute step (`computeStoryScopeReview` over the configured
8
+ * provider chain, posting nothing) against `origin/<base>...story-<id>` and
9
+ * writes one held-review JSON beside the terminal envelope, keyed on the diff
10
+ * digest. Close adopts it when the digest still matches, so the serialized
11
+ * close tail only posts; a CRITICAL surfaces here, where the worker can fix
12
+ * and re-push inside its own parallel loop.
13
+ *
14
+ * Posts nothing to GitHub and takes no full-suite lock. Exit 0 whatever the
15
+ * findings; 1 only on a usage error or a provider throw.
16
+ */
17
+
18
+ import { parseArgs } from 'node:util';
19
+
20
+ import { runAsCli } from './lib/cli-utils.js';
21
+ import { resolveConfig } from './lib/config-resolver.js';
22
+ import { getStoryBranch, gitSpawn } from './lib/git-utils.js';
23
+ import { runCodeReview } from './lib/orchestration/code-review.js';
24
+ import { resolveSharedBaseRef } from './lib/orchestration/review-base-ref.js';
25
+ import {
26
+ buildReviewDeposit,
27
+ computeReviewDiffDigest,
28
+ resolveRefSha,
29
+ writeReviewDeposit,
30
+ } from './lib/orchestration/review-deposit.js';
31
+ import { computeStoryScopeReview } from './lib/orchestration/single-story-close/phases/code-review.js';
32
+
33
+ const USAGE = {
34
+ invocation:
35
+ 'node .agents/scripts/story-review-compute.js --story <id> [--cwd <workCwd>]',
36
+ summary:
37
+ 'Compute the Story-scope code review of origin/<base>...story-<id> without posting it, and write the held result (keyed on the diff digest) for close to adopt. Exits 0 whatever the findings.',
38
+ flags: [
39
+ ['--story <id>', 'Story issue number; the head ref is story-<id>.'],
40
+ [
41
+ '--cwd <path>',
42
+ 'Checkout to resolve the branch and diff in (default: the current directory).',
43
+ ],
44
+ ],
45
+ };
46
+
47
+ /**
48
+ * @param {string[]} argv
49
+ * @returns {{ storyId: number|null, cwd: string|null }}
50
+ */
51
+ export function parseArgv(argv) {
52
+ const { values } = parseArgs({
53
+ args: argv,
54
+ options: {
55
+ story: { type: 'string' },
56
+ cwd: { type: 'string' },
57
+ },
58
+ strict: false,
59
+ });
60
+ const storyId = Number.parseInt(values.story ?? '', 10);
61
+ return {
62
+ storyId: Number.isInteger(storyId) && storyId > 0 ? storyId : null,
63
+ cwd: values.cwd ?? null,
64
+ };
65
+ }
66
+
67
+ /**
68
+ * @param {object} severity
69
+ * @returns {string}
70
+ */
71
+ function formatTally(severity) {
72
+ const s = severity ?? {};
73
+ return `critical=${s.critical ?? 0} high=${s.high ?? 0} medium=${s.medium ?? 0} suggestion=${s.suggestion ?? 0}`;
74
+ }
75
+
76
+ /**
77
+ * @param {{ storyId: number, cwd: string, config: object }} input
78
+ * @param {{
79
+ * gitSpawnFn?: typeof gitSpawn,
80
+ * runCodeReviewFn?: typeof runCodeReview,
81
+ * writeDepositFn?: typeof writeReviewDeposit,
82
+ * progress?: (tag: string, msg: string) => void,
83
+ * nowIso?: () => string,
84
+ * }} [deps]
85
+ * @returns {Promise<{ written: boolean, path?: string, reason?: string,
86
+ * deposit?: object }>}
87
+ */
88
+ export async function computeStoryReviewDeposit(
89
+ { storyId, cwd, config },
90
+ deps = {},
91
+ ) {
92
+ const {
93
+ gitSpawnFn = gitSpawn,
94
+ runCodeReviewFn = runCodeReview,
95
+ writeDepositFn = writeReviewDeposit,
96
+ progress = () => {},
97
+ nowIso = () => new Date().toISOString(),
98
+ } = deps;
99
+ const storyBranch = getStoryBranch(storyId);
100
+ const baseBranch = config?.project?.baseBranch ?? 'main';
101
+ const headSha = resolveRefSha({ cwd, ref: storyBranch, gitSpawnFn });
102
+ if (!headSha) {
103
+ return { written: false, reason: `could not resolve ${storyBranch}` };
104
+ }
105
+ const base = resolveSharedBaseRef({ baseBranch, cwd, gitSpawnFn });
106
+ const diffDigest = computeReviewDiffDigest({
107
+ cwd,
108
+ baseRef: base.resolved ? base.ref : null,
109
+ headRef: headSha,
110
+ gitSpawnFn,
111
+ });
112
+ if (!diffDigest) {
113
+ return {
114
+ written: false,
115
+ reason: `could not read the ${base.remoteRef ?? baseBranch}...${storyBranch} diff`,
116
+ };
117
+ }
118
+ const computed = await computeStoryScopeReview({
119
+ cwd,
120
+ storyId,
121
+ headRef: headSha,
122
+ baseBranch,
123
+ deferPost: true,
124
+ provider: null,
125
+ runCodeReviewFn,
126
+ gitSpawnFn,
127
+ progress,
128
+ });
129
+ if (!computed.result) {
130
+ return { written: false, reason: 'the review base is unresolvable' };
131
+ }
132
+ const deposit = buildReviewDeposit({
133
+ storyId,
134
+ headSha,
135
+ baseRef: base.ref,
136
+ diffDigest,
137
+ result: computed.result,
138
+ createdAt: nowIso(),
139
+ });
140
+ const path = writeDepositFn(deposit, { config });
141
+ return { written: true, path, deposit };
142
+ }
143
+
144
+ /**
145
+ * @param {string[]} [argv]
146
+ * @param {{
147
+ * resolveConfigImpl?: typeof resolveConfig,
148
+ * stdout?: { write: (s: string) => void },
149
+ * cwd?: string,
150
+ * } & Parameters<typeof computeStoryReviewDeposit>[1]} [deps]
151
+ * @returns {Promise<object>} the outcome
152
+ */
153
+ export async function runStoryReviewComputeCli(
154
+ argv = process.argv.slice(2),
155
+ deps = {},
156
+ ) {
157
+ const {
158
+ resolveConfigImpl = resolveConfig,
159
+ stdout = process.stdout,
160
+ cwd: defaultCwd = process.cwd(),
161
+ ...computeDeps
162
+ } = deps;
163
+ const { storyId, cwd } = parseArgv(argv);
164
+ if (!storyId) {
165
+ throw new Error(
166
+ 'story-review-compute: --story <id> is required (a positive integer).',
167
+ );
168
+ }
169
+ const workCwd = cwd ?? defaultCwd;
170
+ const config = resolveConfigImpl({ cwd: workCwd });
171
+ const progress =
172
+ computeDeps.progress ??
173
+ ((tag, msg) => stdout.write(`[story-review-compute] [${tag}] ${msg}\n`));
174
+ const outcome = await computeStoryReviewDeposit(
175
+ { storyId, cwd: workCwd, config },
176
+ { ...computeDeps, progress },
177
+ );
178
+ if (!outcome.written) {
179
+ stdout.write(
180
+ `[story-review-compute] ⏭ No held review written: ${outcome.reason}. Close computes the review itself.\n`,
181
+ );
182
+ return outcome;
183
+ }
184
+ const { deposit } = outcome;
185
+ stdout.write(
186
+ `[story-review-compute] ✅ Held review for Story #${storyId} written → ${outcome.path}\n` +
187
+ `[story-review-compute] diff ${deposit.diffDigest.slice(0, 12)} @ ${deposit.headSha.slice(0, 12)} · ${formatTally(deposit.severity)}\n`,
188
+ );
189
+ if (deposit.halted) {
190
+ stdout.write(
191
+ `[story-review-compute] ❌ CRITICAL: fix, commit and re-push before hand-off. Report:\n${deposit.report}\n`,
192
+ );
193
+ }
194
+ return outcome;
195
+ }
196
+
197
+ async function main() {
198
+ await runStoryReviewComputeCli();
199
+ return 0;
200
+ }
201
+
202
+ runAsCli(import.meta.url, main, {
203
+ source: 'story-review-compute',
204
+ propagateExitCode: true,
205
+ errorPrefix: '[story-review-compute] ❌ Fatal error',
206
+ usage: USAGE,
207
+ });
@@ -7,6 +7,7 @@
7
7
 
8
8
  import { createRequire } from 'node:module';
9
9
  import path from 'node:path';
10
+ import { resolveUpdaterRefreshScope } from './lib/baselines/coverage-refresh-scope.js';
10
11
  import {
11
12
  buildCoverageUpdaterScorer,
12
13
  resolveCoverageUpdaterScope,
@@ -40,6 +41,7 @@ const USAGE = {
40
41
  ],
41
42
  notes: [
42
43
  'Run `npm run test:coverage` first — this script never runs the suite itself.',
44
+ 'Against an `affected`-stamped artifact only measured rows are rewritten; rows the scoped run skipped are kept.',
43
45
  ],
44
46
  };
45
47
 
@@ -75,16 +77,19 @@ function main() {
75
77
  score: scoreCoverageFinal,
76
78
  }),
77
79
  };
78
- // No flag -> scopeFiles=null + fullScope=false -> the service derives the
79
- // diff via `origin/main..HEAD` (its default baseRef/headRef).
80
- if (fullScope) refreshOpts.fullScope = true;
81
- else if (diffScopeRef) refreshOpts.baseRef = diffScopeRef;
82
-
83
- return refreshBaseline(refreshOpts).then((result) => {
84
- Logger.info(
85
- `[Coverage] ✅ Baseline updated: ${result.envelope.rows.length} file(s) recorded at ${COVERAGE_BASELINE_PATH} (${absBaselinePath}). scope=${result.scope.mode}, wrote=${result.wrote}.`,
86
- );
87
- });
80
+ // No flag and a full artifact -> the service derives the diff via
81
+ // `origin/main..HEAD` (its default baseRef/headRef).
82
+ return resolveUpdaterRefreshScope(cwd, {
83
+ fullScope,
84
+ diffScopeRef,
85
+ loadScope: loadC8Scope,
86
+ })
87
+ .then((scope) => refreshBaseline({ ...refreshOpts, ...scope }))
88
+ .then((result) => {
89
+ Logger.info(
90
+ `[Coverage] ✅ Baseline updated: ${result.envelope.rows.length} file(s) recorded at ${COVERAGE_BASELINE_PATH} (${absBaselinePath}). scope=${result.scope.mode}, wrote=${result.wrote}.`,
91
+ );
92
+ });
88
93
  }
89
94
 
90
95
  runAsCli(import.meta.url, main, {
@@ -4,6 +4,7 @@ import {
4
4
  buildCrapUpdaterScorer,
5
5
  parseCrapUpdaterArgs,
6
6
  resolveCrapUpdaterOptions,
7
+ seatCrapBaseline,
7
8
  } from './lib/baselines/crap-updater-cli.js';
8
9
  import { refreshBaseline } from './lib/baselines/refresh-service.js';
9
10
  import { runAsCli } from './lib/cli-utils.js';
@@ -31,7 +32,7 @@ import { Logger } from './lib/Logger.js';
31
32
  /** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
32
33
  const USAGE = {
33
34
  invocation:
34
- 'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>]',
35
+ 'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>] [--seat-missing]',
35
36
  summary:
36
37
  'Scan → score → write the CRAP baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
37
38
  flags: [
@@ -51,6 +52,10 @@ const USAGE = {
51
52
  '--diff-scope <ref>',
52
53
  'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
53
54
  ],
55
+ [
56
+ '--seat-missing',
57
+ 'Insert-only: write rows ONLY for methods of changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Refuses unless the coverage capture stamp is fresh and method resolution is 100%. Prints `seated: N`. Incompatible with --full-scope.',
58
+ ],
54
59
  ],
55
60
  notes: [
56
61
  'Run `npm run test:coverage` first — without a coverage artifact every file is skipped.',
@@ -91,7 +96,12 @@ async function main() {
91
96
  );
92
97
  }
93
98
 
94
- runAsCli(import.meta.url, main, {
99
+ async function seat() {
100
+ process.exitCode = await seatCrapBaseline(process.argv.slice(2));
101
+ }
102
+
103
+ const seating = process.argv.includes('--seat-missing');
104
+ runAsCli(import.meta.url, seating ? seat : main, {
95
105
  source: 'crap-baseline',
96
106
  usage: USAGE,
97
107
  onError: (err) => {
@@ -9,6 +9,7 @@ import './lib/runtime-deps/ensure-installed.js';
9
9
  import path from 'node:path';
10
10
  import { parseDiffScopeFlag } from './lib/baselines/diff-scope-cli.js';
11
11
  import { refreshBaseline } from './lib/baselines/refresh-service.js';
12
+ import { seatMaintainabilityBaseline } from './lib/baselines/seat-missing.js';
12
13
  import { runAsCli } from './lib/cli-utils.js';
13
14
  import { getBaselineEpsilon } from './lib/config/quality.js';
14
15
  import { getBaselines, resolveConfig } from './lib/config-resolver.js';
@@ -17,7 +18,7 @@ import { Logger } from './lib/Logger.js';
17
18
  /** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
18
19
  const USAGE = {
19
20
  invocation:
20
- 'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>]',
21
+ 'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>] [--seat-missing]',
21
22
  summary:
22
23
  'Score → write the maintainability baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
23
24
  flags: [
@@ -29,6 +30,10 @@ const USAGE = {
29
30
  '--diff-scope <ref>',
30
31
  'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
31
32
  ],
33
+ [
34
+ '--seat-missing',
35
+ 'Insert-only: write rows ONLY for changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Prints `seated: N`. Incompatible with --full-scope.',
36
+ ],
32
37
  ],
33
38
  };
34
39
 
@@ -89,7 +94,12 @@ async function main() {
89
94
  );
90
95
  }
91
96
 
92
- runAsCli(import.meta.url, main, {
97
+ async function seat() {
98
+ process.exitCode = await seatMaintainabilityBaseline(process.argv.slice(2));
99
+ }
100
+
101
+ const seating = process.argv.includes('--seat-missing');
102
+ runAsCli(import.meta.url, seating ? seat : main, {
93
103
  source: 'maintainability-baseline',
94
104
  usage: USAGE,
95
105
  onError: (err) => {
@@ -13,10 +13,12 @@ description: >-
13
13
  This helper performs a comprehensive code review of a change set before it
14
14
  is merged to `main`. The live v2 path is **Story scope only**:
15
15
 
16
- - **Story scope** — reviews `main...story-<storyId>` (or
17
- `project.baseBranch...story-<storyId>`) inside `single-story-close.js`
18
- after the PR opens and before auto-merge. Findings post to the PR;
19
- critical findings block close (`agent::blocked`).
16
+ - **Story scope** — reviews `origin/<baseBranch>...<sha>` inside
17
+ `single-story-close.js`, computed alongside the close-validation gates
18
+ once the pre-gate self-heal commits land (`<sha>` is that HEAD) and
19
+ posted after the PR opens, before auto-merge; a HEAD that moved by then
20
+ re-reviews serially. Findings post to the PR; critical findings block
21
+ close (`agent::blocked`).
20
22
 
21
23
  **Invariant — Story-scope review runs outside the maker's LLM context.**
22
24
  The Story-scope review executes inside the `single-story-close.js` close
@@ -24,7 +26,7 @@ subprocess, **not** in the delivering child's (maker agent's) LLM context.
24
26
  The close pipeline invokes it after the delivering child has exited, so
25
27
  the change set is reviewed by a process the maker cannot influence. The
26
28
  enforcing code path is
27
- [`runStoryScopeReview`](../../scripts/lib/orchestration/single-story-close/phases/code-review.js)
29
+ [`computeStoryScopeReview`](../../scripts/lib/orchestration/single-story-close/phases/code-review.js)
28
30
  → shared
29
31
  [`runStoryReviewCore`](../../scripts/lib/orchestration/story-close/phases/review-core.js).
30
32
  A future refactor MUST preserve this isolation: do not move Story-scope
@@ -77,8 +77,6 @@ therefore buys a **deep review**, not a fresh acceptance critic.
77
77
  > [`acceptance-self-eval.md`](acceptance-self-eval.md) — a host that cannot
78
78
  > spawn the critic at all, noted in the friction comment if you block.
79
79
 
80
- `--base <ref>` overrides `project.baseBranch`.
81
-
82
80
  ## 4. Acceptance self-eval (Step 1a, required)
83
81
 
84
82
  **One verdict owner per Story** — named by `verdictOwner`: the inline
@@ -93,48 +91,55 @@ scored in **one** gate call. Bounded by `delivery.acceptanceEval.maxRounds`
93
91
  --verdict <verdict-path>`
94
92
 
95
93
  The gate reads the Story's `acceptance[]` count itself and rejects a verdict
96
- whose `criteria[]` length differs **before** scoring, consuming no round;
97
- `--expected-criteria` is accepted but redundant. A second gate call in the
98
- same round spends a round for nothing and races the Story-scoped ledger.
94
+ whose `criteria[]` length differs **before** scoring, consuming no round. A
95
+ second gate call in the same round spends a round for nothing and races the
96
+ Story-scoped ledger.
99
97
 
100
98
  `proceed` → close. `redraft` → one more round inside the cap. `block` → **do
101
99
  not close**: post a `friction` comment and flip `agent::blocked`.
102
100
  Per-round mechanics: [`acceptance-self-eval.md`](acceptance-self-eval.md).
103
101
 
104
- ## 5. The one credited suite run
102
+ ## 5. The one credited run
105
103
 
106
- After the self-eval loop's last fix commit, fetch and merge
107
- `origin/<baseBranch>` into the Story branch **first**, ahead of this run and
108
- the push: close's base-sync then no-ops, so neither stamp goes stale. Then
109
- run the suite **once** in the worktree through the depositor — it spawns the
110
- project's own `npm test`, whatever that resolves to, and stamps the result,
111
- so any runner earns the credit:
104
+ **Preflight first — blocking.** After the self-eval loop, run the configured
105
+ `project.commands.lint` and `node <main-repo>/.agents/scripts/quality-preview.js
106
+ --changed-since origin/<baseBranch>` in the worktree; fix and commit every
107
+ finding. Close's gates stay authoritative
108
+ ([`deliver-reference.md`](deliver-reference.md) § Preflight).
112
109
 
113
- ```bash
114
- node <main-repo>/.agents/scripts/evidence-gate.js --standalone \
115
- --scope-id <storyId> --gate test --worktree <workCwd> -- npm test
116
- ```
117
-
118
- Green deposits the `test` evidence close reads, keyed on the tree, so close
119
- reports the gate as **credited** at unchanged HEAD — a later commit voids it.
120
- Read its **output**, not the exit code: `✓ test passed` is the signal. The
121
- CRAP gate still captures coverage itself when it needs an artifact.
122
-
123
- A bare `npm test` earns the same credit **only** where the project's test
124
- script routes through mandrel's own runner, which prints the outcome. On any
125
- other runner it deposits nothing and prints nothing, so silence is never
126
- evidence of credit; `mandrel doctor`'s `test-credit-path` check names which
127
- shape this project is. If the suite outruns the host's sync Bash ceiling,
128
- dispatch it in the **background** — its completion re-invokes you; never spawn
129
- a task to poll or `sleep`-loop against it
130
- ([`parallel-tooling.md`](parallel-tooling.md) Rule 2). Redraft rounds run the
131
- scoped projects for the roots you changed plus `verify[]`, not the whole
132
- suite; only this run needs credit.
110
+ After the last fix commit, fetch and merge
111
+ `origin/<baseBranch>` into the Story branch **first**, ahead of this run and
112
+ the push: close's base-sync then no-ops, so neither stamp goes stale. Then run
113
+ **one** depositor in the worktree, picked by the predicate close registers
114
+ `coverage-capture` on — CRAP gate enabled **and** a `test:coverage` script:
115
+
116
+ - **Capture-active** →
117
+ `node <main-repo>/.agents/scripts/coverage-capture.js --cwd <workCwd>`.
118
+ It takes the host full-suite lock, runs `test:coverage` and writes the stamp
119
+ close's capture finds fresh. Signal: `Wrote content-digest capture stamp`.
120
+ Non-zero is a red suite (or the lock / timeout it names): fix, re-run. No
121
+ second `npm test`.
122
+ - **Otherwise** → `node <main-repo>/.agents/scripts/evidence-gate.js
123
+ --standalone --scope-id <storyId> --gate test --worktree <workCwd> -- npm test`.
124
+ Signal: `✓ test passed` — the `test` evidence close reads, keyed on the tree.
125
+
126
+ **Seat, then push.** CRAP or MI gate on → run
127
+ `update-crap-baseline.js --seat-missing` and its maintainability twin (signal
128
+ `seated: N`); commit it as `chore(baselines): baseline-refresh: …`.
129
+
130
+ A later commit voids either credit; a baseline-JSON seat keeps the stamp. Read
131
+ the **output**, not the exit code: a run that prints no signal deposits
132
+ nothing. `mandrel doctor`'s `test-credit-path` check names this project's
133
+ depositor. Seat order, runner shapes, background dispatch, redraft rounds:
134
+ [`deliver-reference.md`](deliver-reference.md) § Credited run.
133
135
 
134
136
  `verify[]` is scoped entries **plus** this one run: an entry that is itself a
135
137
  full-suite command is reported credited against the same record, never
136
138
  respawned.
137
139
 
140
+ After the push, compute the held review and fix a CRITICAL before hand-off:
141
+ [`deliver-reference.md`](deliver-reference.md) § Held review.
142
+
138
143
  ## 6. Terminal envelope — the return contract
139
144
 
140
145
  `single-story-close.js` emits exactly one envelope on stdout between
@@ -157,10 +162,8 @@ Required fields: `kind` (`story-deliver-terminal`), `storyId`, `status`,
157
162
  reports every gate as `passed` / `failed` / `skipped` — a skipped gate is
158
163
  reported, never omitted, so a missing gate is never read as a passing one.
159
164
 
160
- **Gate output is captured, not streamed.** Close writes gate lines to
161
- `temp/orchestration/close-gates-<storyId>.log` and reports a one-line digest on
162
- success; a failed gate replays its tail inline. `AGENT_LOG_LEVEL=verbose`
163
- restores live streaming.
165
+ Gate output is captured to a log, not streamed
166
+ ([`deliver-reference.md`](deliver-reference.md) § Gate output).
164
167
 
165
168
  ## 7. When to leave this file
166
169
 
@@ -326,11 +326,15 @@ diff selects **review depth**.
326
326
 
327
327
  ## Async merge-confirm mode (`delivery.mergeWatch.mode: "async"`)
328
328
 
329
- In async mode the close arms auto-merge, probes for ~60s (catching an instant
330
- merge or, via the head-anchored required-check predicate, an instantly-red
331
- required check) and then returns `pending` with a `nextCommand`, instead of
332
- holding the host tool slot for a merge that lands after the wait would have
333
- expired anyway. When a close returns that `pending` envelope, launch its
329
+ In async mode the close arms auto-merge and probes the PR **once**. A
330
+ definitive probe settles as in sync mode — merged lands, closed or a red
331
+ required check (the head-anchored predicate) blocks, a red advisory gate
332
+ blocks. A probe whose checks have not started or are still running returns
333
+ `pending` with a `nextCommand` at once, no sleep: CI never reddens inside the
334
+ first minute, so a second probe could only hold the serialized slot. Two
335
+ shapes poll on inside the ~60s window: checks already **green** (the merge is
336
+ imminent — observed at the green cadence, it saves a whole confirm
337
+ invocation) and a red rollup still awaiting its confirming probe. When a close returns that `pending` envelope, launch its
334
338
  `nextCommand` (`single-story-confirm-merge.js … --wait`) as a **background**
335
339
  invocation (host background Bash — its completion re-invokes the agent) and
336
340
  move on to the next Story; `single-story-confirm-merge.js` is idempotent and
@@ -352,3 +356,109 @@ config default stays `"sync"`. A slow-CI solo consumer may opt into `"async"`
352
356
  for the same reason — a foreground wait longer than the host tool ceiling
353
357
  expires `pending` anyway. Otherwise a one-Story run keeps `sync`: there is no
354
358
  sibling to unblock, and the foreground wait is the cheapest path to `landed`.
359
+
360
+ ## Preflight (before close) {#preflight}
361
+
362
+ Digest § 5 states the rule: before the credited suite run, the worker runs
363
+ the configured `project.commands.lint` (falling back to `npm run lint`) and
364
+ `quality-preview.js --changed-since origin/<baseBranch>` in the worktree, and
365
+ fixes and commits every finding. It runs **before** the credited run because a
366
+ fix commit afterwards would void that run's credit.
367
+
368
+ - **Why.** Close runs lint and the maintainability half of the preview
369
+ (`quality-preview-mi`) in its parallel phase, but a regression found there
370
+ still costs a close round-trip. Seconds of preflight in the worktree is
371
+ cheaper than any close.
372
+ - **The CRAP half never captures.** It scores whatever coverage artifact is
373
+ on disk — the worker's credited capture when one exists — and triggers no
374
+ capture of its own. With no artifact its methods report unscorable; a stale
375
+ one can invent a violation. `--only mi` runs the maintainability half alone.
376
+ - **Close stays authoritative.** Its `quality-preview-crap` gate scores a
377
+ fresh capture after `coverage-capture`; a preflight pass never skips it.
378
+
379
+ ## Credited run (situational) {#credited-run}
380
+
381
+ Digest § 5 states the rule and both invocations; this is what surrounds them.
382
+
383
+ - **Why the capture.** On a capture-active project close registers
384
+ `coverage-capture` instead of the plain `test` gate, so an evidence-gate
385
+ `test` deposit buys nothing there: close logs `no credited capture stamp
386
+ covers this change set` and pays the whole suite on the serialized tail. The
387
+ worker's capture writes the content-digest stamp in the worktree, and
388
+ close's capture then exits on its freshness probe without spawning
389
+ `test:coverage`. A base-sync that merges a path under `crap.targetDirs`
390
+ spends the stamp, and close re-captures.
391
+ - **Runner shapes.** A bare `npm test` earns the `test` credit **only** where
392
+ the project's test script routes through mandrel's own runner, which prints
393
+ the outcome. On any other runner it deposits nothing and prints nothing, so
394
+ silence is never evidence of credit.
395
+ - **Background dispatch.** If the run outruns the host's sync Bash ceiling,
396
+ dispatch it in the **background** — its completion re-invokes you; never
397
+ spawn a task to poll or `sleep`-loop against it
398
+ ([`parallel-tooling.md`](parallel-tooling.md) Rule 2).
399
+ - **Redraft rounds.** Run the scoped projects for the roots you changed plus
400
+ `verify[]`, not the whole suite; only the one run needs credit.
401
+ - **Seating new methods (`--seat-missing`).** Close fails a Story whose own
402
+ new methods have no baseline row, so after the credited run and before the
403
+ push the worker runs, in `<workCwd>`, for each enabled gate:
404
+ `node .agents/scripts/update-crap-baseline.js --seat-missing` (CRAP) and
405
+ `node .agents/scripts/update-maintainability-baseline.js --seat-missing`
406
+ (MI). Each scores the files changed since the `origin/<baseBranch>`
407
+ merge-base and writes **only** rows whose (path, method) key is absent —
408
+ every existing row stays byte-identical, including rows whose scores moved
409
+ (re-scoring stays close's auto-refresh). It prints `seated: N`; `seated: 0`
410
+ writes nothing. Commit a change as `chore(baselines): baseline-refresh: …`.
411
+ - **Seat refusals.** The CRAP seat exits non-zero and writes nothing unless
412
+ the coverage-capture stamp is fresh for the tree **and** method resolution
413
+ over the in-scope files is exactly 100% — a lower rate means the artifact's
414
+ coordinates predate the tree. The refusal names the rate, the unresolved
415
+ files and the fix: re-run the digest § 5 capture, then seat.
416
+ - **Seat order vs credit.** The capture stamp digests scorable sources only,
417
+ so a baseline-JSON-only seat commit leaves it fresh and the capture credit
418
+ stands. The evidence-gate `test` credit is keyed on the tree, so on that
419
+ path seat first — MI is static and needs no coverage — then run the suite.
420
+
421
+ ## Held review at hand-off {#held-review}
422
+
423
+ The one home of this rule; the digest and the worker contract point here.
424
+ After the credited run **and** the push, the worker computes the Story-scope
425
+ code review itself, before handing off:
426
+
427
+ ```bash
428
+ node <main-repo>/.agents/scripts/story-review-compute.js --story <storyId> --cwd <workCwd>
429
+ ```
430
+
431
+ - **What it does.** It runs close's own review computation (the configured
432
+ provider chain) against `origin/<baseBranch>...story-<id>`, posts nothing,
433
+ takes no full-suite lock, and writes
434
+ `temp/orchestration/story-review-<id>.json` beside the terminal envelope,
435
+ keyed on the **diff digest** (sha256 of the exact three-dot diff text). It
436
+ exits 0 whatever the findings; non-zero means the provider threw — report
437
+ it in the hand-off; close computes the review itself.
438
+ - **A CRITICAL is the worker's to fix.** Fix, commit, re-run the credited
439
+ run (the fix commit voided its credit), push, and re-run the compute — the
440
+ acceptance loop's redraft discipline, bounded by
441
+ `delivery.acceptanceEval.maxRounds`. Still CRITICAL at the cap → take the
442
+ blocked path. Anything else goes in the hand-off as the severity tally.
443
+ - **Close adopts, else computes.** When the deposit's digest equals the
444
+ digest of the diff at close's held-review start, close starts no review and
445
+ posts the deposit after PR-open. A clean base-sync merge moves HEAD without
446
+ changing the diff, so it keeps the deposit; a commit that changes the diff
447
+ does not, and close reviews as it would without one. The CRITICAL halt and
448
+ `--override-review-block` apply to an adopted result unchanged.
449
+
450
+ ## Gate output {#gate-output}
451
+
452
+ Close writes gate lines to `temp/orchestration/close-gates-<storyId>.log` and
453
+ reports a one-line digest on success; a failed gate replays its tail inline.
454
+ `AGENT_LOG_LEVEL=verbose` restores live streaming. A gate exiting `75` (its
455
+ full-suite lock wait expired) logs a deferred line and the close settles
456
+ `pending`; one exiting `124` (the suite outran its timeout) logs a timeout
457
+ line naming host contention — neither is reported as failing tests.
458
+
459
+ Every full-suite run ends on `⏲ suite timings: lockWaitMs=… hostWaitMs=…
460
+ testRunMs=…`, which close carries as the envelope's `suiteTimings`. The
461
+ `coverage.timeoutMs` clock starts at spawn, never in the lock queue; a suite
462
+ that writes `$MANDREL_SUITE_READY_FILE` when its tests start (after a
463
+ consumer load gate) gets a fresh bound for them, so worst-case wall is lock
464
+ wait + 2 × `timeoutMs`.
@@ -101,7 +101,8 @@ Step 2.5's credited suite run is the sole exception.
101
101
 
102
102
  After the self-eval loop's last fix commit, run the one credited suite run
103
103
  in the worktree — **digest § 5** is its only home and carries the
104
- invocation. Red → fix, commit, re-run.
104
+ invocation. Red → fix, commit, re-run. Then, before the push, seat the
105
+ baseline rows for methods the Story added (`--seat-missing`, digest § 5).
105
106
 
106
107
  Push `story-<storyId>` to `origin`, confirming the remote ref moved. Then
107
108
  (sub-agent dispatch only) return the hand-off — Story id, `workCwd`,
package/docs/CHANGELOG.md CHANGED
@@ -15,6 +15,32 @@ All notable changes to this project will be documented in this file.
15
15
  -->
16
16
  <!-- markdownlint-disable-file MD004 MD012 MD037 -->
17
17
 
18
+ ## [2.68.0](https://github.com/dsj1984/mandrel/compare/mandrel-v2.67.0...mandrel-v2.68.0) (2026-09-26)
19
+
20
+
21
+ ### Added
22
+
23
+ * baselines: add an insert-only `--seat-missing` refresh and have the story-worker seat rows for methods its diff adds before push ([#5486](https://github.com/dsj1984/mandrel/issues/5486)) ([#5492](https://github.com/dsj1984/mandrel/issues/5492)) ([325d7bb](https://github.com/dsj1984/mandrel/commit/325d7bbbc0bb08bb6fe3f05287d959f2b42d4da2))
24
+ * close-validation: fail cheap gates before coverage capture and report lock expiry / timeout honestly ([#5471](https://github.com/dsj1984/mandrel/issues/5471)) ([#5474](https://github.com/dsj1984/mandrel/issues/5474)) ([ece115d](https://github.com/dsj1984/mandrel/commit/ece115d7961566c21de45f79f44b3957389e7623))
25
+ * coverage-capture: after a base-sync, re-measure only main's delta instead of re-running the whole suite ([#5487](https://github.com/dsj1984/mandrel/issues/5487)) ([#5496](https://github.com/dsj1984/mandrel/issues/5496)) ([03694a4](https://github.com/dsj1984/mandrel/commit/03694a4bfa970ba122c1f508a7ed4b1271ccb222))
26
+ * coverage-capture: opt-in `affected` scope that runs a consumer-owned scoped coverage script ([#5472](https://github.com/dsj1984/mandrel/issues/5472)) ([#5483](https://github.com/dsj1984/mandrel/issues/5483)) ([50bf942](https://github.com/dsj1984/mandrel/commit/50bf9424d38e776270cfaf19935dd265bb950625))
27
+ * coverage-capture: spend the suite timeout on tests, queue on the full-suite lock, and make the cap configurable ([#5485](https://github.com/dsj1984/mandrel/issues/5485)) ([#5493](https://github.com/dsj1984/mandrel/issues/5493)) ([6a5bc69](https://github.com/dsj1984/mandrel/commit/6a5bc6929138e70ca66a9aea0e4354dcf5319ed7))
28
+ * deliver: compute the Story-scope review at worker push time, keyed on the diff digest, so close posts a held result and a CRITICAL is fixed before hand-off ([#5480](https://github.com/dsj1984/mandrel/issues/5480)) ([#5490](https://github.com/dsj1984/mandrel/issues/5490)) ([c43b6bf](https://github.com/dsj1984/mandrel/commit/c43b6bf3beca4aeb0dc6350f3fe1f5c9cdcda250))
29
+ * deliver: make the worker's one credited run the coverage capture close will otherwise re-pay, and let doctor name the credit gaps ([#5477](https://github.com/dsj1984/mandrel/issues/5477)) ([#5482](https://github.com/dsj1984/mandrel/issues/5482)) ([67e8d06](https://github.com/dsj1984/mandrel/commit/67e8d0687a6254f8ff8ab9957867c69588642aad))
30
+ * deliver: warn in the beat envelope when a Story's footprint overlaps another session's in-flight Story ([#5488](https://github.com/dsj1984/mandrel/issues/5488)) ([#5491](https://github.com/dsj1984/mandrel/issues/5491)) ([be68220](https://github.com/dsj1984/mandrel/commit/be682205dc5959e12819bd2f2463c352449acd09))
31
+
32
+
33
+ ### Fixed
34
+
35
+ * close: in async merge-watch mode return `pending` after one probe instead of burning the 60s window ([#5479](https://github.com/dsj1984/mandrel/issues/5479)) ([#5484](https://github.com/dsj1984/mandrel/issues/5484)) ([440b8bf](https://github.com/dsj1984/mandrel/commit/440b8bf70ed53d973c635662a12791a1d9f51953))
36
+ * **close:** bound the full-suite lock wait by the suite's kill bound (refs [#5478](https://github.com/dsj1984/mandrel/issues/5478)) ([#5481](https://github.com/dsj1984/mandrel/issues/5481)) ([c7b7719](https://github.com/dsj1984/mandrel/commit/c7b7719bf50f7bc557c1979505ebab6a6851345b))
37
+ * coverage-capture: a red capture must never leave an artifact a later pr… ([#5494](https://github.com/dsj1984/mandrel/issues/5494)) ([#5495](https://github.com/dsj1984/mandrel/issues/5495)) ([fa72738](https://github.com/dsj1984/mandrel/commit/fa72738d486626e892b5fd89e0350f745d925cfc))
38
+
39
+
40
+ ### Performance
41
+
42
+ * close: run the Story-scope code review concurrently with the validation gates ([#5473](https://github.com/dsj1984/mandrel/issues/5473)) ([#5476](https://github.com/dsj1984/mandrel/issues/5476)) ([e098205](https://github.com/dsj1984/mandrel/commit/e09820592d70ba600df288c4e4b5ae1abd5b9aa1))
43
+
18
44
  ## [2.67.0](https://github.com/dsj1984/mandrel/compare/mandrel-v2.66.0...mandrel-v2.67.0) (2026-09-26)
19
45
 
20
46