mandrel 2.67.0 → 2.69.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/story-worker.md +15 -11
- package/.agents/docs/agentrc-reference.json +3 -1
- package/.agents/docs/configuration.md +36 -1
- package/.agents/schemas/agentrc.schema.json +14 -1
- package/.agents/schemas/story-deliver-terminal.schema.json +23 -1
- package/.agents/schemas/validation-evidence.schema.json +3 -1
- package/.agents/scripts/coverage-capture.js +65 -9
- package/.agents/scripts/evidence-gate.js +106 -8
- package/.agents/scripts/lib/baselines/coverage-refresh-scope.js +60 -0
- package/.agents/scripts/lib/baselines/crap-updater-cli.js +101 -4
- package/.agents/scripts/lib/baselines/refresh-service.js +1 -1
- package/.agents/scripts/lib/baselines/seat-missing.js +228 -0
- package/.agents/scripts/lib/child-exec.js +39 -1
- package/.agents/scripts/lib/close-validation/gates.js +59 -19
- package/.agents/scripts/lib/close-validation/process.js +23 -24
- package/.agents/scripts/lib/close-validation/runner.js +71 -40
- package/.agents/scripts/lib/config/gates/coverage.schema.js +21 -0
- package/.agents/scripts/lib/config/quality.js +7 -1
- package/.agents/scripts/lib/config/temp-paths.js +15 -0
- package/.agents/scripts/lib/config-settings-schema-delivery.js +1 -1
- package/.agents/scripts/lib/coverage-baseline.js +78 -5
- package/.agents/scripts/lib/coverage-capture-affected.js +345 -0
- package/.agents/scripts/lib/coverage-capture-delta.js +180 -0
- package/.agents/scripts/lib/coverage-capture-fullscope.js +53 -32
- package/.agents/scripts/lib/coverage-capture-incremental.js +49 -26
- package/.agents/scripts/lib/coverage-capture-usage.js +1 -1
- package/.agents/scripts/lib/coverage-capture.js +121 -81
- package/.agents/scripts/lib/full-suite-lock.js +49 -46
- package/.agents/scripts/lib/full-suite-queue.js +83 -8
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/observability/source-classifier.js +1 -0
- package/.agents/scripts/lib/orchestration/code-review.js +15 -3
- package/.agents/scripts/lib/orchestration/merge-poll.js +5 -0
- package/.agents/scripts/lib/orchestration/review-deposit.js +219 -0
- package/.agents/scripts/lib/orchestration/review-providers/code-review.js +11 -7
- package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +29 -10
- package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +124 -73
- package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +38 -20
- package/.agents/scripts/lib/orchestration/single-story-close/phases/lock-wait-pending.js +8 -2
- package/.agents/scripts/lib/orchestration/single-story-close/review-overlap.js +161 -0
- package/.agents/scripts/lib/orchestration/single-story-close/runner.js +47 -7
- package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +8 -16
- package/.agents/scripts/lib/process-group.js +1 -1
- package/.agents/scripts/lib/supervised-suite.js +247 -0
- package/.agents/scripts/lib/wave-runner/cross-run-overlap.js +120 -0
- package/.agents/scripts/lib/wave-runner/live-probe.js +5 -1
- package/.agents/scripts/quality-preview.js +112 -14
- package/.agents/scripts/stories-wave-tick.js +47 -0
- package/.agents/scripts/story-review-compute.js +207 -0
- package/.agents/scripts/update-coverage-baseline.js +15 -10
- package/.agents/scripts/update-crap-baseline.js +12 -2
- package/.agents/scripts/update-maintainability-baseline.js +12 -2
- package/.agents/workflows/audit-security.md +0 -1
- package/.agents/workflows/helpers/code-review.md +7 -5
- package/.agents/workflows/helpers/deliver-digest.md +39 -36
- package/.agents/workflows/helpers/deliver-reference.md +115 -5
- package/.agents/workflows/helpers/deliver-story.md +2 -1
- package/docs/CHANGELOG.md +33 -0
- package/lib/cli/registry.js +125 -18
- package/lib/migrations/steps/strip-removed-agentrc-keys.js +0 -5
- package/package.json +1 -1
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* story-review-compute.js — compute the Story-scope review on the worker, at
|
|
5
|
+
* push time, and deposit it for close to adopt.
|
|
6
|
+
*
|
|
7
|
+
* Runs close's compute step (`computeStoryScopeReview` over the configured
|
|
8
|
+
* provider chain, posting nothing) against `origin/<base>...story-<id>` and
|
|
9
|
+
* writes one held-review JSON beside the terminal envelope, keyed on the diff
|
|
10
|
+
* digest. Close adopts it when the digest still matches, so the serialized
|
|
11
|
+
* close tail only posts; a CRITICAL surfaces here, where the worker can fix
|
|
12
|
+
* and re-push inside its own parallel loop.
|
|
13
|
+
*
|
|
14
|
+
* Posts nothing to GitHub and takes no full-suite lock. Exit 0 whatever the
|
|
15
|
+
* findings; 1 only on a usage error or a provider throw.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { parseArgs } from 'node:util';
|
|
19
|
+
|
|
20
|
+
import { runAsCli } from './lib/cli-utils.js';
|
|
21
|
+
import { resolveConfig } from './lib/config-resolver.js';
|
|
22
|
+
import { getStoryBranch, gitSpawn } from './lib/git-utils.js';
|
|
23
|
+
import { runCodeReview } from './lib/orchestration/code-review.js';
|
|
24
|
+
import { resolveSharedBaseRef } from './lib/orchestration/review-base-ref.js';
|
|
25
|
+
import {
|
|
26
|
+
buildReviewDeposit,
|
|
27
|
+
computeReviewDiffDigest,
|
|
28
|
+
resolveRefSha,
|
|
29
|
+
writeReviewDeposit,
|
|
30
|
+
} from './lib/orchestration/review-deposit.js';
|
|
31
|
+
import { computeStoryScopeReview } from './lib/orchestration/single-story-close/phases/code-review.js';
|
|
32
|
+
|
|
33
|
+
const USAGE = {
|
|
34
|
+
invocation:
|
|
35
|
+
'node .agents/scripts/story-review-compute.js --story <id> [--cwd <workCwd>]',
|
|
36
|
+
summary:
|
|
37
|
+
'Compute the Story-scope code review of origin/<base>...story-<id> without posting it, and write the held result (keyed on the diff digest) for close to adopt. Exits 0 whatever the findings.',
|
|
38
|
+
flags: [
|
|
39
|
+
['--story <id>', 'Story issue number; the head ref is story-<id>.'],
|
|
40
|
+
[
|
|
41
|
+
'--cwd <path>',
|
|
42
|
+
'Checkout to resolve the branch and diff in (default: the current directory).',
|
|
43
|
+
],
|
|
44
|
+
],
|
|
45
|
+
};
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* @param {string[]} argv
|
|
49
|
+
* @returns {{ storyId: number|null, cwd: string|null }}
|
|
50
|
+
*/
|
|
51
|
+
export function parseArgv(argv) {
|
|
52
|
+
const { values } = parseArgs({
|
|
53
|
+
args: argv,
|
|
54
|
+
options: {
|
|
55
|
+
story: { type: 'string' },
|
|
56
|
+
cwd: { type: 'string' },
|
|
57
|
+
},
|
|
58
|
+
strict: false,
|
|
59
|
+
});
|
|
60
|
+
const storyId = Number.parseInt(values.story ?? '', 10);
|
|
61
|
+
return {
|
|
62
|
+
storyId: Number.isInteger(storyId) && storyId > 0 ? storyId : null,
|
|
63
|
+
cwd: values.cwd ?? null,
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* @param {object} severity
|
|
69
|
+
* @returns {string}
|
|
70
|
+
*/
|
|
71
|
+
function formatTally(severity) {
|
|
72
|
+
const s = severity ?? {};
|
|
73
|
+
return `critical=${s.critical ?? 0} high=${s.high ?? 0} medium=${s.medium ?? 0} suggestion=${s.suggestion ?? 0}`;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* @param {{ storyId: number, cwd: string, config: object }} input
|
|
78
|
+
* @param {{
|
|
79
|
+
* gitSpawnFn?: typeof gitSpawn,
|
|
80
|
+
* runCodeReviewFn?: typeof runCodeReview,
|
|
81
|
+
* writeDepositFn?: typeof writeReviewDeposit,
|
|
82
|
+
* progress?: (tag: string, msg: string) => void,
|
|
83
|
+
* nowIso?: () => string,
|
|
84
|
+
* }} [deps]
|
|
85
|
+
* @returns {Promise<{ written: boolean, path?: string, reason?: string,
|
|
86
|
+
* deposit?: object }>}
|
|
87
|
+
*/
|
|
88
|
+
export async function computeStoryReviewDeposit(
|
|
89
|
+
{ storyId, cwd, config },
|
|
90
|
+
deps = {},
|
|
91
|
+
) {
|
|
92
|
+
const {
|
|
93
|
+
gitSpawnFn = gitSpawn,
|
|
94
|
+
runCodeReviewFn = runCodeReview,
|
|
95
|
+
writeDepositFn = writeReviewDeposit,
|
|
96
|
+
progress = () => {},
|
|
97
|
+
nowIso = () => new Date().toISOString(),
|
|
98
|
+
} = deps;
|
|
99
|
+
const storyBranch = getStoryBranch(storyId);
|
|
100
|
+
const baseBranch = config?.project?.baseBranch ?? 'main';
|
|
101
|
+
const headSha = resolveRefSha({ cwd, ref: storyBranch, gitSpawnFn });
|
|
102
|
+
if (!headSha) {
|
|
103
|
+
return { written: false, reason: `could not resolve ${storyBranch}` };
|
|
104
|
+
}
|
|
105
|
+
const base = resolveSharedBaseRef({ baseBranch, cwd, gitSpawnFn });
|
|
106
|
+
const diffDigest = computeReviewDiffDigest({
|
|
107
|
+
cwd,
|
|
108
|
+
baseRef: base.resolved ? base.ref : null,
|
|
109
|
+
headRef: headSha,
|
|
110
|
+
gitSpawnFn,
|
|
111
|
+
});
|
|
112
|
+
if (!diffDigest) {
|
|
113
|
+
return {
|
|
114
|
+
written: false,
|
|
115
|
+
reason: `could not read the ${base.remoteRef ?? baseBranch}...${storyBranch} diff`,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
const computed = await computeStoryScopeReview({
|
|
119
|
+
cwd,
|
|
120
|
+
storyId,
|
|
121
|
+
headRef: headSha,
|
|
122
|
+
baseBranch,
|
|
123
|
+
deferPost: true,
|
|
124
|
+
provider: null,
|
|
125
|
+
runCodeReviewFn,
|
|
126
|
+
gitSpawnFn,
|
|
127
|
+
progress,
|
|
128
|
+
});
|
|
129
|
+
if (!computed.result) {
|
|
130
|
+
return { written: false, reason: 'the review base is unresolvable' };
|
|
131
|
+
}
|
|
132
|
+
const deposit = buildReviewDeposit({
|
|
133
|
+
storyId,
|
|
134
|
+
headSha,
|
|
135
|
+
baseRef: base.ref,
|
|
136
|
+
diffDigest,
|
|
137
|
+
result: computed.result,
|
|
138
|
+
createdAt: nowIso(),
|
|
139
|
+
});
|
|
140
|
+
const path = writeDepositFn(deposit, { config });
|
|
141
|
+
return { written: true, path, deposit };
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* @param {string[]} [argv]
|
|
146
|
+
* @param {{
|
|
147
|
+
* resolveConfigImpl?: typeof resolveConfig,
|
|
148
|
+
* stdout?: { write: (s: string) => void },
|
|
149
|
+
* cwd?: string,
|
|
150
|
+
* } & Parameters<typeof computeStoryReviewDeposit>[1]} [deps]
|
|
151
|
+
* @returns {Promise<object>} the outcome
|
|
152
|
+
*/
|
|
153
|
+
export async function runStoryReviewComputeCli(
|
|
154
|
+
argv = process.argv.slice(2),
|
|
155
|
+
deps = {},
|
|
156
|
+
) {
|
|
157
|
+
const {
|
|
158
|
+
resolveConfigImpl = resolveConfig,
|
|
159
|
+
stdout = process.stdout,
|
|
160
|
+
cwd: defaultCwd = process.cwd(),
|
|
161
|
+
...computeDeps
|
|
162
|
+
} = deps;
|
|
163
|
+
const { storyId, cwd } = parseArgv(argv);
|
|
164
|
+
if (!storyId) {
|
|
165
|
+
throw new Error(
|
|
166
|
+
'story-review-compute: --story <id> is required (a positive integer).',
|
|
167
|
+
);
|
|
168
|
+
}
|
|
169
|
+
const workCwd = cwd ?? defaultCwd;
|
|
170
|
+
const config = resolveConfigImpl({ cwd: workCwd });
|
|
171
|
+
const progress =
|
|
172
|
+
computeDeps.progress ??
|
|
173
|
+
((tag, msg) => stdout.write(`[story-review-compute] [${tag}] ${msg}\n`));
|
|
174
|
+
const outcome = await computeStoryReviewDeposit(
|
|
175
|
+
{ storyId, cwd: workCwd, config },
|
|
176
|
+
{ ...computeDeps, progress },
|
|
177
|
+
);
|
|
178
|
+
if (!outcome.written) {
|
|
179
|
+
stdout.write(
|
|
180
|
+
`[story-review-compute] ⏭ No held review written: ${outcome.reason}. Close computes the review itself.\n`,
|
|
181
|
+
);
|
|
182
|
+
return outcome;
|
|
183
|
+
}
|
|
184
|
+
const { deposit } = outcome;
|
|
185
|
+
stdout.write(
|
|
186
|
+
`[story-review-compute] ✅ Held review for Story #${storyId} written → ${outcome.path}\n` +
|
|
187
|
+
`[story-review-compute] diff ${deposit.diffDigest.slice(0, 12)} @ ${deposit.headSha.slice(0, 12)} · ${formatTally(deposit.severity)}\n`,
|
|
188
|
+
);
|
|
189
|
+
if (deposit.halted) {
|
|
190
|
+
stdout.write(
|
|
191
|
+
`[story-review-compute] ❌ CRITICAL: fix, commit and re-push before hand-off. Report:\n${deposit.report}\n`,
|
|
192
|
+
);
|
|
193
|
+
}
|
|
194
|
+
return outcome;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
async function main() {
|
|
198
|
+
await runStoryReviewComputeCli();
|
|
199
|
+
return 0;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
runAsCli(import.meta.url, main, {
|
|
203
|
+
source: 'story-review-compute',
|
|
204
|
+
propagateExitCode: true,
|
|
205
|
+
errorPrefix: '[story-review-compute] ❌ Fatal error',
|
|
206
|
+
usage: USAGE,
|
|
207
|
+
});
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
import { createRequire } from 'node:module';
|
|
9
9
|
import path from 'node:path';
|
|
10
|
+
import { resolveUpdaterRefreshScope } from './lib/baselines/coverage-refresh-scope.js';
|
|
10
11
|
import {
|
|
11
12
|
buildCoverageUpdaterScorer,
|
|
12
13
|
resolveCoverageUpdaterScope,
|
|
@@ -40,6 +41,7 @@ const USAGE = {
|
|
|
40
41
|
],
|
|
41
42
|
notes: [
|
|
42
43
|
'Run `npm run test:coverage` first — this script never runs the suite itself.',
|
|
44
|
+
'Against an `affected`-stamped artifact only measured rows are rewritten; rows the scoped run skipped are kept.',
|
|
43
45
|
],
|
|
44
46
|
};
|
|
45
47
|
|
|
@@ -75,16 +77,19 @@ function main() {
|
|
|
75
77
|
score: scoreCoverageFinal,
|
|
76
78
|
}),
|
|
77
79
|
};
|
|
78
|
-
// No flag
|
|
79
|
-
//
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
)
|
|
87
|
-
|
|
80
|
+
// No flag and a full artifact -> the service derives the diff via
|
|
81
|
+
// `origin/main..HEAD` (its default baseRef/headRef).
|
|
82
|
+
return resolveUpdaterRefreshScope(cwd, {
|
|
83
|
+
fullScope,
|
|
84
|
+
diffScopeRef,
|
|
85
|
+
loadScope: loadC8Scope,
|
|
86
|
+
})
|
|
87
|
+
.then((scope) => refreshBaseline({ ...refreshOpts, ...scope }))
|
|
88
|
+
.then((result) => {
|
|
89
|
+
Logger.info(
|
|
90
|
+
`[Coverage] ✅ Baseline updated: ${result.envelope.rows.length} file(s) recorded at ${COVERAGE_BASELINE_PATH} (${absBaselinePath}). scope=${result.scope.mode}, wrote=${result.wrote}.`,
|
|
91
|
+
);
|
|
92
|
+
});
|
|
88
93
|
}
|
|
89
94
|
|
|
90
95
|
runAsCli(import.meta.url, main, {
|
|
@@ -4,6 +4,7 @@ import {
|
|
|
4
4
|
buildCrapUpdaterScorer,
|
|
5
5
|
parseCrapUpdaterArgs,
|
|
6
6
|
resolveCrapUpdaterOptions,
|
|
7
|
+
seatCrapBaseline,
|
|
7
8
|
} from './lib/baselines/crap-updater-cli.js';
|
|
8
9
|
import { refreshBaseline } from './lib/baselines/refresh-service.js';
|
|
9
10
|
import { runAsCli } from './lib/cli-utils.js';
|
|
@@ -31,7 +32,7 @@ import { Logger } from './lib/Logger.js';
|
|
|
31
32
|
/** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
|
|
32
33
|
const USAGE = {
|
|
33
34
|
invocation:
|
|
34
|
-
'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>]',
|
|
35
|
+
'node .agents/scripts/update-crap-baseline.js [--baseline <path>] [--coverage <path>] [--full-scope | --diff-scope <ref>] [--seat-missing]',
|
|
35
36
|
summary:
|
|
36
37
|
'Scan → score → write the CRAP baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
|
|
37
38
|
flags: [
|
|
@@ -51,6 +52,10 @@ const USAGE = {
|
|
|
51
52
|
'--diff-scope <ref>',
|
|
52
53
|
'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
|
|
53
54
|
],
|
|
55
|
+
[
|
|
56
|
+
'--seat-missing',
|
|
57
|
+
'Insert-only: write rows ONLY for methods of changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Refuses unless the coverage capture stamp is fresh and method resolution is 100%. Prints `seated: N`. Incompatible with --full-scope.',
|
|
58
|
+
],
|
|
54
59
|
],
|
|
55
60
|
notes: [
|
|
56
61
|
'Run `npm run test:coverage` first — without a coverage artifact every file is skipped.',
|
|
@@ -91,7 +96,12 @@ async function main() {
|
|
|
91
96
|
);
|
|
92
97
|
}
|
|
93
98
|
|
|
94
|
-
|
|
99
|
+
async function seat() {
|
|
100
|
+
process.exitCode = await seatCrapBaseline(process.argv.slice(2));
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
const seating = process.argv.includes('--seat-missing');
|
|
104
|
+
runAsCli(import.meta.url, seating ? seat : main, {
|
|
95
105
|
source: 'crap-baseline',
|
|
96
106
|
usage: USAGE,
|
|
97
107
|
onError: (err) => {
|
|
@@ -9,6 +9,7 @@ import './lib/runtime-deps/ensure-installed.js';
|
|
|
9
9
|
import path from 'node:path';
|
|
10
10
|
import { parseDiffScopeFlag } from './lib/baselines/diff-scope-cli.js';
|
|
11
11
|
import { refreshBaseline } from './lib/baselines/refresh-service.js';
|
|
12
|
+
import { seatMaintainabilityBaseline } from './lib/baselines/seat-missing.js';
|
|
12
13
|
import { runAsCli } from './lib/cli-utils.js';
|
|
13
14
|
import { getBaselineEpsilon } from './lib/config/quality.js';
|
|
14
15
|
import { getBaselines, resolveConfig } from './lib/config-resolver.js';
|
|
@@ -17,7 +18,7 @@ import { Logger } from './lib/Logger.js';
|
|
|
17
18
|
/** `runAsCli` answers `--help` before `main`, so a usage probe never writes. */
|
|
18
19
|
const USAGE = {
|
|
19
20
|
invocation:
|
|
20
|
-
'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>]',
|
|
21
|
+
'node .agents/scripts/update-maintainability-baseline.js [--full-scope | --diff-scope <ref>] [--seat-missing]',
|
|
21
22
|
summary:
|
|
22
23
|
'Score → write the maintainability baseline. With no scope flag the refresh is scoped to the files changed in `origin/main..HEAD`; out-of-scope rows are preserved verbatim.',
|
|
23
24
|
flags: [
|
|
@@ -29,6 +30,10 @@ const USAGE = {
|
|
|
29
30
|
'--diff-scope <ref>',
|
|
30
31
|
'Scope the refresh to files changed between <ref> and HEAD. Incompatible with --full-scope.',
|
|
31
32
|
],
|
|
33
|
+
[
|
|
34
|
+
'--seat-missing',
|
|
35
|
+
'Insert-only: write rows ONLY for changed files (merge-base of `--diff-scope <ref>`, default `origin/<baseBranch>`) that have no baseline row; every existing row stays byte-identical. Prints `seated: N`. Incompatible with --full-scope.',
|
|
36
|
+
],
|
|
32
37
|
],
|
|
33
38
|
};
|
|
34
39
|
|
|
@@ -89,7 +94,12 @@ async function main() {
|
|
|
89
94
|
);
|
|
90
95
|
}
|
|
91
96
|
|
|
92
|
-
|
|
97
|
+
async function seat() {
|
|
98
|
+
process.exitCode = await seatMaintainabilityBaseline(process.argv.slice(2));
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const seating = process.argv.includes('--seat-missing');
|
|
102
|
+
runAsCli(import.meta.url, seating ? seat : main, {
|
|
93
103
|
source: 'maintainability-baseline',
|
|
94
104
|
usage: USAGE,
|
|
95
105
|
onError: (err) => {
|
|
@@ -13,10 +13,12 @@ description: >-
|
|
|
13
13
|
This helper performs a comprehensive code review of a change set before it
|
|
14
14
|
is merged to `main`. The live v2 path is **Story scope only**:
|
|
15
15
|
|
|
16
|
-
- **Story scope** — reviews `
|
|
17
|
-
`
|
|
18
|
-
|
|
19
|
-
|
|
16
|
+
- **Story scope** — reviews `origin/<baseBranch>...<sha>` inside
|
|
17
|
+
`single-story-close.js`, computed alongside the close-validation gates
|
|
18
|
+
once the pre-gate self-heal commits land (`<sha>` is that HEAD) and
|
|
19
|
+
posted after the PR opens, before auto-merge; a HEAD that moved by then
|
|
20
|
+
re-reviews serially. Findings post to the PR; critical findings block
|
|
21
|
+
close (`agent::blocked`).
|
|
20
22
|
|
|
21
23
|
**Invariant — Story-scope review runs outside the maker's LLM context.**
|
|
22
24
|
The Story-scope review executes inside the `single-story-close.js` close
|
|
@@ -24,7 +26,7 @@ subprocess, **not** in the delivering child's (maker agent's) LLM context.
|
|
|
24
26
|
The close pipeline invokes it after the delivering child has exited, so
|
|
25
27
|
the change set is reviewed by a process the maker cannot influence. The
|
|
26
28
|
enforcing code path is
|
|
27
|
-
[`
|
|
29
|
+
[`computeStoryScopeReview`](../../scripts/lib/orchestration/single-story-close/phases/code-review.js)
|
|
28
30
|
→ shared
|
|
29
31
|
[`runStoryReviewCore`](../../scripts/lib/orchestration/story-close/phases/review-core.js).
|
|
30
32
|
A future refactor MUST preserve this isolation: do not move Story-scope
|
|
@@ -77,8 +77,6 @@ therefore buys a **deep review**, not a fresh acceptance critic.
|
|
|
77
77
|
> [`acceptance-self-eval.md`](acceptance-self-eval.md) — a host that cannot
|
|
78
78
|
> spawn the critic at all, noted in the friction comment if you block.
|
|
79
79
|
|
|
80
|
-
`--base <ref>` overrides `project.baseBranch`.
|
|
81
|
-
|
|
82
80
|
## 4. Acceptance self-eval (Step 1a, required)
|
|
83
81
|
|
|
84
82
|
**One verdict owner per Story** — named by `verdictOwner`: the inline
|
|
@@ -93,48 +91,55 @@ scored in **one** gate call. Bounded by `delivery.acceptanceEval.maxRounds`
|
|
|
93
91
|
--verdict <verdict-path>`
|
|
94
92
|
|
|
95
93
|
The gate reads the Story's `acceptance[]` count itself and rejects a verdict
|
|
96
|
-
whose `criteria[]` length differs **before** scoring, consuming no round
|
|
97
|
-
|
|
98
|
-
|
|
94
|
+
whose `criteria[]` length differs **before** scoring, consuming no round. A
|
|
95
|
+
second gate call in the same round spends a round for nothing and races the
|
|
96
|
+
Story-scoped ledger.
|
|
99
97
|
|
|
100
98
|
`proceed` → close. `redraft` → one more round inside the cap. `block` → **do
|
|
101
99
|
not close**: post a `friction` comment and flip `agent::blocked`.
|
|
102
100
|
Per-round mechanics: [`acceptance-self-eval.md`](acceptance-self-eval.md).
|
|
103
101
|
|
|
104
|
-
## 5. The one credited
|
|
102
|
+
## 5. The one credited run
|
|
105
103
|
|
|
106
|
-
After the self-eval loop
|
|
107
|
-
`
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
so any runner earns the credit:
|
|
104
|
+
**Preflight first — blocking.** After the self-eval loop, run the configured
|
|
105
|
+
`project.commands.lint` and `node <main-repo>/.agents/scripts/quality-preview.js
|
|
106
|
+
--changed-since origin/<baseBranch>` in the worktree; fix and commit every
|
|
107
|
+
finding. Close's gates stay authoritative
|
|
108
|
+
([`deliver-reference.md`](deliver-reference.md) § Preflight).
|
|
112
109
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
110
|
+
After the last fix commit, fetch and merge
|
|
111
|
+
`origin/<baseBranch>` into the Story branch **first**, ahead of this run and
|
|
112
|
+
the push: close's base-sync then no-ops, so neither stamp goes stale. Then run
|
|
113
|
+
**one** depositor in the worktree, picked by the predicate close registers
|
|
114
|
+
`coverage-capture` on — CRAP gate enabled **and** a `test:coverage` script:
|
|
115
|
+
|
|
116
|
+
- **Capture-active** →
|
|
117
|
+
`node <main-repo>/.agents/scripts/coverage-capture.js --cwd <workCwd>`.
|
|
118
|
+
It takes the host full-suite lock, runs `test:coverage` and writes the stamp
|
|
119
|
+
close's capture finds fresh. Signal: `Wrote content-digest capture stamp`.
|
|
120
|
+
Non-zero is a red suite (or the lock / timeout it names): fix, re-run. No
|
|
121
|
+
second `npm test`.
|
|
122
|
+
- **Otherwise** → `node <main-repo>/.agents/scripts/evidence-gate.js
|
|
123
|
+
--standalone --scope-id <storyId> --gate test --worktree <workCwd> -- npm test`.
|
|
124
|
+
Signal: `✓ test passed` — the `test` evidence close reads, keyed on the tree.
|
|
125
|
+
|
|
126
|
+
**Seat, then push.** CRAP or MI gate on → run
|
|
127
|
+
`update-crap-baseline.js --seat-missing` and its maintainability twin (signal
|
|
128
|
+
`seated: N`); commit it as `chore(baselines): baseline-refresh: …`.
|
|
129
|
+
|
|
130
|
+
A later commit voids either credit; a baseline-JSON seat keeps the stamp. Read
|
|
131
|
+
the **output**, not the exit code: a run that prints no signal deposits
|
|
132
|
+
nothing. `mandrel doctor`'s `test-credit-path` check names this project's
|
|
133
|
+
depositor. Seat order, runner shapes, background dispatch, redraft rounds:
|
|
134
|
+
[`deliver-reference.md`](deliver-reference.md) § Credited run.
|
|
133
135
|
|
|
134
136
|
`verify[]` is scoped entries **plus** this one run: an entry that is itself a
|
|
135
137
|
full-suite command is reported credited against the same record, never
|
|
136
138
|
respawned.
|
|
137
139
|
|
|
140
|
+
After the push, compute the held review and fix a CRITICAL before hand-off:
|
|
141
|
+
[`deliver-reference.md`](deliver-reference.md) § Held review.
|
|
142
|
+
|
|
138
143
|
## 6. Terminal envelope — the return contract
|
|
139
144
|
|
|
140
145
|
`single-story-close.js` emits exactly one envelope on stdout between
|
|
@@ -157,10 +162,8 @@ Required fields: `kind` (`story-deliver-terminal`), `storyId`, `status`,
|
|
|
157
162
|
reports every gate as `passed` / `failed` / `skipped` — a skipped gate is
|
|
158
163
|
reported, never omitted, so a missing gate is never read as a passing one.
|
|
159
164
|
|
|
160
|
-
|
|
161
|
-
`
|
|
162
|
-
success; a failed gate replays its tail inline. `AGENT_LOG_LEVEL=verbose`
|
|
163
|
-
restores live streaming.
|
|
165
|
+
Gate output is captured to a log, not streamed
|
|
166
|
+
([`deliver-reference.md`](deliver-reference.md) § Gate output).
|
|
164
167
|
|
|
165
168
|
## 7. When to leave this file
|
|
166
169
|
|
|
@@ -326,11 +326,15 @@ diff selects **review depth**.
|
|
|
326
326
|
|
|
327
327
|
## Async merge-confirm mode (`delivery.mergeWatch.mode: "async"`)
|
|
328
328
|
|
|
329
|
-
In async mode the close arms auto-merge
|
|
330
|
-
|
|
331
|
-
required check
|
|
332
|
-
|
|
333
|
-
|
|
329
|
+
In async mode the close arms auto-merge and probes the PR **once**. A
|
|
330
|
+
definitive probe settles as in sync mode — merged lands, closed or a red
|
|
331
|
+
required check (the head-anchored predicate) blocks, a red advisory gate
|
|
332
|
+
blocks. A probe whose checks have not started or are still running returns
|
|
333
|
+
`pending` with a `nextCommand` at once, no sleep: CI never reddens inside the
|
|
334
|
+
first minute, so a second probe could only hold the serialized slot. Two
|
|
335
|
+
shapes poll on inside the ~60s window: checks already **green** (the merge is
|
|
336
|
+
imminent — observed at the green cadence, it saves a whole confirm
|
|
337
|
+
invocation) and a red rollup still awaiting its confirming probe. When a close returns that `pending` envelope, launch its
|
|
334
338
|
`nextCommand` (`single-story-confirm-merge.js … --wait`) as a **background**
|
|
335
339
|
invocation (host background Bash — its completion re-invokes the agent) and
|
|
336
340
|
move on to the next Story; `single-story-confirm-merge.js` is idempotent and
|
|
@@ -352,3 +356,109 @@ config default stays `"sync"`. A slow-CI solo consumer may opt into `"async"`
|
|
|
352
356
|
for the same reason — a foreground wait longer than the host tool ceiling
|
|
353
357
|
expires `pending` anyway. Otherwise a one-Story run keeps `sync`: there is no
|
|
354
358
|
sibling to unblock, and the foreground wait is the cheapest path to `landed`.
|
|
359
|
+
|
|
360
|
+
## Preflight (before close) {#preflight}
|
|
361
|
+
|
|
362
|
+
Digest § 5 states the rule: before the credited suite run, the worker runs
|
|
363
|
+
the configured `project.commands.lint` (falling back to `npm run lint`) and
|
|
364
|
+
`quality-preview.js --changed-since origin/<baseBranch>` in the worktree, and
|
|
365
|
+
fixes and commits every finding. It runs **before** the credited run because a
|
|
366
|
+
fix commit afterwards would void that run's credit.
|
|
367
|
+
|
|
368
|
+
- **Why.** Close runs lint and the maintainability half of the preview
|
|
369
|
+
(`quality-preview-mi`) in its parallel phase, but a regression found there
|
|
370
|
+
still costs a close round-trip. Seconds of preflight in the worktree is
|
|
371
|
+
cheaper than any close.
|
|
372
|
+
- **The CRAP half never captures.** It scores whatever coverage artifact is
|
|
373
|
+
on disk — the worker's credited capture when one exists — and triggers no
|
|
374
|
+
capture of its own. With no artifact its methods report unscorable; a stale
|
|
375
|
+
one can invent a violation. `--only mi` runs the maintainability half alone.
|
|
376
|
+
- **Close stays authoritative.** Its `quality-preview-crap` gate scores a
|
|
377
|
+
fresh capture after `coverage-capture`; a preflight pass never skips it.
|
|
378
|
+
|
|
379
|
+
## Credited run (situational) {#credited-run}
|
|
380
|
+
|
|
381
|
+
Digest § 5 states the rule and both invocations; this is what surrounds them.
|
|
382
|
+
|
|
383
|
+
- **Why the capture.** On a capture-active project close registers
|
|
384
|
+
`coverage-capture` instead of the plain `test` gate, so an evidence-gate
|
|
385
|
+
`test` deposit buys nothing there: close logs `no credited capture stamp
|
|
386
|
+
covers this change set` and pays the whole suite on the serialized tail. The
|
|
387
|
+
worker's capture writes the content-digest stamp in the worktree, and
|
|
388
|
+
close's capture then exits on its freshness probe without spawning
|
|
389
|
+
`test:coverage`. A base-sync that merges a path under `crap.targetDirs`
|
|
390
|
+
spends the stamp, and close re-captures.
|
|
391
|
+
- **Runner shapes.** A bare `npm test` earns the `test` credit **only** where
|
|
392
|
+
the project's test script routes through mandrel's own runner, which prints
|
|
393
|
+
the outcome. On any other runner it deposits nothing and prints nothing, so
|
|
394
|
+
silence is never evidence of credit.
|
|
395
|
+
- **Background dispatch.** If the run outruns the host's sync Bash ceiling,
|
|
396
|
+
dispatch it in the **background** — its completion re-invokes you; never
|
|
397
|
+
spawn a task to poll or `sleep`-loop against it
|
|
398
|
+
([`parallel-tooling.md`](parallel-tooling.md) Rule 2).
|
|
399
|
+
- **Redraft rounds.** Run the scoped projects for the roots you changed plus
|
|
400
|
+
`verify[]`, not the whole suite; only the one run needs credit.
|
|
401
|
+
- **Seating new methods (`--seat-missing`).** Close fails a Story whose own
|
|
402
|
+
new methods have no baseline row, so after the credited run and before the
|
|
403
|
+
push the worker runs, in `<workCwd>`, for each enabled gate:
|
|
404
|
+
`node .agents/scripts/update-crap-baseline.js --seat-missing` (CRAP) and
|
|
405
|
+
`node .agents/scripts/update-maintainability-baseline.js --seat-missing`
|
|
406
|
+
(MI). Each scores the files changed since the `origin/<baseBranch>`
|
|
407
|
+
merge-base and writes **only** rows whose (path, method) key is absent —
|
|
408
|
+
every existing row stays byte-identical, including rows whose scores moved
|
|
409
|
+
(re-scoring stays close's auto-refresh). It prints `seated: N`; `seated: 0`
|
|
410
|
+
writes nothing. Commit a change as `chore(baselines): baseline-refresh: …`.
|
|
411
|
+
- **Seat refusals.** The CRAP seat exits non-zero and writes nothing unless
|
|
412
|
+
the coverage-capture stamp is fresh for the tree **and** method resolution
|
|
413
|
+
over the in-scope files is exactly 100% — a lower rate means the artifact's
|
|
414
|
+
coordinates predate the tree. The refusal names the rate, the unresolved
|
|
415
|
+
files and the fix: re-run the digest § 5 capture, then seat.
|
|
416
|
+
- **Seat order vs credit.** The capture stamp digests scorable sources only,
|
|
417
|
+
so a baseline-JSON-only seat commit leaves it fresh and the capture credit
|
|
418
|
+
stands. The evidence-gate `test` credit is keyed on the tree, so on that
|
|
419
|
+
path seat first — MI is static and needs no coverage — then run the suite.
|
|
420
|
+
|
|
421
|
+
## Held review at hand-off {#held-review}
|
|
422
|
+
|
|
423
|
+
The one home of this rule; the digest and the worker contract point here.
|
|
424
|
+
After the credited run **and** the push, the worker computes the Story-scope
|
|
425
|
+
code review itself, before handing off:
|
|
426
|
+
|
|
427
|
+
```bash
|
|
428
|
+
node <main-repo>/.agents/scripts/story-review-compute.js --story <storyId> --cwd <workCwd>
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
- **What it does.** It runs close's own review computation (the configured
|
|
432
|
+
provider chain) against `origin/<baseBranch>...story-<id>`, posts nothing,
|
|
433
|
+
takes no full-suite lock, and writes
|
|
434
|
+
`temp/orchestration/story-review-<id>.json` beside the terminal envelope,
|
|
435
|
+
keyed on the **diff digest** (sha256 of the exact three-dot diff text). It
|
|
436
|
+
exits 0 whatever the findings; non-zero means the provider threw — report
|
|
437
|
+
it in the hand-off; close computes the review itself.
|
|
438
|
+
- **A CRITICAL is the worker's to fix.** Fix, commit, re-run the credited
|
|
439
|
+
run (the fix commit voided its credit), push, and re-run the compute — the
|
|
440
|
+
acceptance loop's redraft discipline, bounded by
|
|
441
|
+
`delivery.acceptanceEval.maxRounds`. Still CRITICAL at the cap → take the
|
|
442
|
+
blocked path. Anything else goes in the hand-off as the severity tally.
|
|
443
|
+
- **Close adopts, else computes.** When the deposit's digest equals the
|
|
444
|
+
digest of the diff at close's held-review start, close starts no review and
|
|
445
|
+
posts the deposit after PR-open. A clean base-sync merge moves HEAD without
|
|
446
|
+
changing the diff, so it keeps the deposit; a commit that changes the diff
|
|
447
|
+
does not, and close reviews as it would without one. The CRITICAL halt and
|
|
448
|
+
`--override-review-block` apply to an adopted result unchanged.
|
|
449
|
+
|
|
450
|
+
## Gate output {#gate-output}
|
|
451
|
+
|
|
452
|
+
Close writes gate lines to `temp/orchestration/close-gates-<storyId>.log` and
|
|
453
|
+
reports a one-line digest on success; a failed gate replays its tail inline.
|
|
454
|
+
`AGENT_LOG_LEVEL=verbose` restores live streaming. A gate exiting `75` (its
|
|
455
|
+
full-suite lock wait expired) logs a deferred line and the close settles
|
|
456
|
+
`pending`; one exiting `124` (the suite outran its timeout) logs a timeout
|
|
457
|
+
line naming host contention — neither is reported as failing tests.
|
|
458
|
+
|
|
459
|
+
Every full-suite run ends on `⏲ suite timings: lockWaitMs=… hostWaitMs=…
|
|
460
|
+
testRunMs=…`, which close carries as the envelope's `suiteTimings`. The
|
|
461
|
+
`coverage.timeoutMs` clock starts at spawn, never in the lock queue; a suite
|
|
462
|
+
that writes `$MANDREL_SUITE_READY_FILE` when its tests start (after a
|
|
463
|
+
consumer load gate) gets a fresh bound for them, so worst-case wall is lock
|
|
464
|
+
wait + 2 × `timeoutMs`.
|
|
@@ -101,7 +101,8 @@ Step 2.5's credited suite run is the sole exception.
|
|
|
101
101
|
|
|
102
102
|
After the self-eval loop's last fix commit, run the one credited suite run
|
|
103
103
|
in the worktree — **digest § 5** is its only home and carries the
|
|
104
|
-
invocation. Red → fix, commit, re-run.
|
|
104
|
+
invocation. Red → fix, commit, re-run. Then, before the push, seat the
|
|
105
|
+
baseline rows for methods the Story added (`--seat-missing`, digest § 5).
|
|
105
106
|
|
|
106
107
|
Push `story-<storyId>` to `origin`, confirming the remote ref moved. Then
|
|
107
108
|
(sub-agent dispatch only) return the hand-off — Story id, `workCwd`,
|