mandrel 2.57.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. package/.agents/agents/story-worker.md +12 -11
  2. package/.agents/scripts/evidence-gate.js +17 -1
  3. package/.agents/scripts/lib/orchestration/code-review.js +7 -3
  4. package/.agents/scripts/lib/orchestration/pinned-identifier-lint.js +137 -0
  5. package/.agents/scripts/lib/orchestration/plan-context.js +11 -3
  6. package/.agents/scripts/lib/orchestration/plan-persist/acceptance-handle-repair.js +107 -0
  7. package/.agents/scripts/lib/orchestration/plan-persist/changes-repair.js +6 -1
  8. package/.agents/scripts/lib/orchestration/plan-persist/persist-helpers.js +14 -9
  9. package/.agents/scripts/lib/orchestration/plan-persist/run-plan-persist.js +17 -7
  10. package/.agents/scripts/lib/orchestration/plan-text-hygiene.js +15 -5
  11. package/.agents/scripts/lib/orchestration/review-base-ref.js +138 -0
  12. package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +37 -5
  13. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +6 -1
  14. package/.agents/scripts/lib/story-body/story-body.js +36 -2
  15. package/.agents/scripts/lib/templates/decomposer-prompts.js +53 -4
  16. package/.agents/scripts/lib/test-run-credit.js +23 -12
  17. package/.agents/workflows/helpers/deliver-digest.md +22 -15
  18. package/.agents/workflows/helpers/deliver-story-reference.md +31 -11
  19. package/.agents/workflows/helpers/deliver-story.md +6 -5
  20. package/.agents/workflows/helpers/plan-reference.md +35 -6
  21. package/.agents/workflows/mandrel-plan.md +6 -2
  22. package/docs/CHANGELOG.md +13 -0
  23. package/lib/cli/registry.js +98 -2
  24. package/package.json +1 -1
@@ -0,0 +1,138 @@
1
+ /**
2
+ * lib/orchestration/review-base-ref.js — the base ref a Story-scope review is
3
+ * allowed to measure against (Story #5325).
4
+ *
5
+ * ## The defect this closes
6
+ *
7
+ * A close runs base-sync and the Story-scope review against what everyone
8
+ * called "the base branch" — but the two meant different refs. Base-sync
9
+ * fetches and merges `origin/<base>`; the review passed the **bare** branch
10
+ * name, which git resolves to the LOCAL `refs/heads/<base>`, i.e. to whatever
11
+ * that ref last fast-forwarded to. On a checkout whose local base is behind
12
+ * its remote, the review's `<base>...<head>` diff therefore contains every
13
+ * commit the local ref is missing — other people's landed work, scored as if
14
+ * this Story had written it. Those findings gate the land, and the operator's
15
+ * only exit is a `review-block-overridden` on blockers that were never real.
16
+ *
17
+ * ## Contract
18
+ *
19
+ * One resolution, at the review phase boundary, threaded into the change-set
20
+ * enumeration, the provider review and the local lens pass — so both arms of
21
+ * the review agree about what changed and neither can inherit local drift.
22
+ *
23
+ * Resolution **fails safe rather than falling back**. Base-sync normally
24
+ * guarantees the remote ref is present, but a `--skip-sync` close or a
25
+ * remote-less checkout can reach the review with no `origin/<base>` at all.
26
+ * Reviewing the local ref anyway is the defect, so an unresolvable base
27
+ * produces a degradation record — carried on the review's existing
28
+ * `degraded` / `degradations[]` envelope — and no findings whatsoever.
29
+ */
30
+
31
+ import { gitSpawn } from '../git-utils.js';
32
+ import { degradationEnvelope } from './review-providers/degraded-gates.js';
33
+
34
+ /**
35
+ * The remote-tracking spelling of a base branch — `main` → `origin/main`.
36
+ * Already-qualified input passes through, so a caller naming `origin/main`
37
+ * is not double-prefixed.
38
+ *
39
+ * @param {unknown} baseBranch
40
+ * @returns {string|null} the remote-tracking ref, or `null` when unnameable.
41
+ */
42
+ export function remoteBaseRef(baseBranch) {
43
+ const base = typeof baseBranch === 'string' ? baseBranch.trim() : '';
44
+ if (base.length === 0) return null;
45
+ return base.startsWith('origin/') ? base : `origin/${base}`;
46
+ }
47
+
48
+ /**
49
+ * Resolve the base ref base-sync merged from, **verified present** in this
50
+ * clone. See the module header for why there is no local-ref fallback.
51
+ *
52
+ * @param {{
53
+ * baseBranch: string,
54
+ * cwd?: string,
55
+ * gitSpawnFn?: typeof gitSpawn,
56
+ * }} args
57
+ * @returns {{ ref: string|null, resolved: boolean, remoteRef: string|null }}
58
+ * `ref` is non-null only when `resolved`; `remoteRef` is the ref that was
59
+ * probed, for the caller's degradation surface.
60
+ */
61
+ export function resolveSharedBaseRef({
62
+ baseBranch,
63
+ cwd = process.cwd(),
64
+ gitSpawnFn = gitSpawn,
65
+ } = {}) {
66
+ const remoteRef = remoteBaseRef(baseBranch);
67
+ if (remoteRef === null) {
68
+ return { ref: null, resolved: false, remoteRef: null };
69
+ }
70
+ try {
71
+ const probe = gitSpawnFn(
72
+ cwd,
73
+ 'rev-parse',
74
+ '--verify',
75
+ '--quiet',
76
+ `${remoteRef}^{commit}`,
77
+ );
78
+ if (probe?.status === 0) {
79
+ return { ref: remoteRef, resolved: true, remoteRef };
80
+ }
81
+ } catch {
82
+ // A spawn failure and a missing ref are the same answer here: the shared
83
+ // base cannot be vouched for.
84
+ }
85
+ return { ref: null, resolved: false, remoteRef };
86
+ }
87
+
88
+ /**
89
+ * The review outcome for a close that cannot establish the shared base.
90
+ *
91
+ * Shaped as the same envelope a completed review returns — an all-zero
92
+ * severity tally, nothing posted, and the `degraded` / `degradations[]` pair
93
+ * the close and the rendered comment already read — so the surfaces reading
94
+ * it cannot mistake "no findings" for "reviewed and clean". Reported, not
95
+ * blocking, matching the posture of every other degraded review gate.
96
+ *
97
+ * @param {{
98
+ * storyId: number|string,
99
+ * baseBranch: string,
100
+ * remoteRef: string|null,
101
+ * progress: (tag: string, msg: string) => void,
102
+ * progressTag?: string,
103
+ * }} args
104
+ * @returns {object}
105
+ */
106
+ export function unresolvedBaseReviewOutcome({
107
+ storyId,
108
+ baseBranch,
109
+ remoteRef,
110
+ progress,
111
+ progressTag = 'REVIEW',
112
+ }) {
113
+ const surface = remoteRef ?? `origin/${baseBranch}`;
114
+ progress(
115
+ progressTag,
116
+ `⚠️ Story-scope review for Story #${storyId} did not run: cannot resolve ` +
117
+ `${surface}, the base ref base-sync merges from. Diffing the local ` +
118
+ `${baseBranch} instead would score commits this Story never made, so no ` +
119
+ 'findings are raised. Fetch the base ref (or re-run without ' +
120
+ '--skip-sync) to restore the review.',
121
+ );
122
+ return {
123
+ halted: false,
124
+ skipped: true,
125
+ severity: { critical: 0, high: 0, medium: 0, suggestion: 0 },
126
+ posted: false,
127
+ postedCommentId: null,
128
+ ...degradationEnvelope([
129
+ {
130
+ tool: 'story-scope-review',
131
+ gate: 'base-ref-resolution',
132
+ surface,
133
+ reason: 'remote-base-ref-unresolved',
134
+ },
135
+ ]),
136
+ crossRefPosted: false,
137
+ };
138
+ }
@@ -25,9 +25,21 @@
25
25
  * shares a single invocation pattern (Story #3653). Review depth needs no
26
26
  * input here: it is derived from this Story's own diff inside `runCodeReview`
27
27
  * (Story #4542).
28
+ *
29
+ * Story #5325 — this phase boundary is where the base ref is resolved, once.
30
+ * The resolved ref threads into `runStoryReviewCore`, which hands it to both
31
+ * the change-set enumeration (and through it the provider review) and the
32
+ * local lens pass, so one resolution corrects both arms. It resolves to
33
+ * `origin/<baseBranch>` — the ref base-sync merged from — rather than the bare
34
+ * branch name git would resolve to the local `refs/heads/<baseBranch>`, whose
35
+ * drift would otherwise be scored as this Story's own change.
28
36
  */
29
37
 
30
38
  import { parsePrNumberFromUrl } from '../../../github-url.js';
39
+ import {
40
+ resolveSharedBaseRef,
41
+ unresolvedBaseReviewOutcome,
42
+ } from '../../review-base-ref.js';
31
43
  import { degradationEnvelope } from '../../review-providers/degraded-gates.js';
32
44
  import { runStoryReviewCore } from '../../story-close/phases/review-core.js';
33
45
  import { postStructuredComment } from '../../ticketing/state.js';
@@ -78,23 +90,25 @@ export function buildStoryReviewCrossRefBody({
78
90
  async function invokeStoryReviewCore({
79
91
  storyId,
80
92
  storyBranch,
81
- baseBranch,
93
+ baseRef,
82
94
  prNumber,
83
95
  provider,
84
96
  runCodeReviewFn,
85
97
  runLocalLensReviewFn,
86
98
  appendFindingsYieldFn,
99
+ gitSpawnFn,
87
100
  progress,
88
101
  }) {
89
102
  return runStoryReviewCore({
90
103
  storyId,
91
- baseRef: baseBranch,
104
+ baseRef,
92
105
  headRef: storyBranch,
93
106
  commentTargetId: prNumber,
94
107
  provider,
95
108
  progress,
96
109
  progressTag: 'REVIEW',
97
110
  runCodeReviewFn,
111
+ gitSpawnFn,
98
112
  // Forward the seams only when the caller injects them; otherwise
99
113
  // `runStoryReviewCore` uses its defaults. `undefined` deep-merges to
100
114
  // the default via the destructuring default there.
@@ -153,6 +167,9 @@ async function postStoryReviewCrossRef({
153
167
  * Failure modes:
154
168
  * - When `prNumber` is null (couldn't parse), the review is skipped
155
169
  * and the function returns `{ halted: false, skipped: true }`.
170
+ * - When `origin/<baseBranch>` cannot be resolved, the review is skipped,
171
+ * a `base-ref-resolution` degradation is recorded on the returned
172
+ * envelope, and no findings are raised (Story #5325).
156
173
  * - When the runner throws, the close fails non-zero (the throw
157
174
  * propagates) — a Story-scope review failure is not silently
158
175
  * ignored.
@@ -170,6 +187,7 @@ async function postStoryReviewCrossRef({
170
187
  * runCodeReviewFn: Function,
171
188
  * runLocalLensReviewFn?: Function,
172
189
  * appendFindingsYieldFn?: Function,
190
+ * gitSpawnFn?: Function,
173
191
  * progress: (tag: string, msg: string) => void,
174
192
  * }} args
175
193
  * @returns {Promise<{
@@ -185,7 +203,7 @@ async function postStoryReviewCrossRef({
185
203
  * }>}
186
204
  */
187
205
  export async function runStoryScopeReview({
188
- cwd: _cwd,
206
+ cwd,
189
207
  storyId,
190
208
  storyBranch,
191
209
  baseBranch,
@@ -195,6 +213,7 @@ export async function runStoryScopeReview({
195
213
  runCodeReviewFn,
196
214
  runLocalLensReviewFn,
197
215
  appendFindingsYieldFn,
216
+ gitSpawnFn,
198
217
  progress,
199
218
  }) {
200
219
  if (prNumber == null) {
@@ -205,20 +224,33 @@ export async function runStoryScopeReview({
205
224
  return { halted: false, skipped: true };
206
225
  }
207
226
 
227
+ // One resolution per close, at the phase boundary: `baseRef` threads from
228
+ // here into the change set, the provider review and the local lens pass.
229
+ const base = resolveSharedBaseRef({ baseBranch, cwd, gitSpawnFn });
230
+ if (!base.resolved) {
231
+ return unresolvedBaseReviewOutcome({
232
+ storyId,
233
+ baseBranch,
234
+ remoteRef: base.remoteRef,
235
+ progress,
236
+ });
237
+ }
238
+
208
239
  progress(
209
240
  'REVIEW',
210
- `Running Story-scope code review for Story #${storyId} (${baseBranch}...${storyBranch}) → PR #${prNumber}...`,
241
+ `Running Story-scope code review for Story #${storyId} (${base.ref}...${storyBranch}) → PR #${prNumber}...`,
211
242
  );
212
243
 
213
244
  const result = await invokeStoryReviewCore({
214
245
  storyId,
215
246
  storyBranch,
216
- baseBranch,
247
+ baseRef: base.ref,
217
248
  prNumber,
218
249
  provider,
219
250
  runCodeReviewFn,
220
251
  runLocalLensReviewFn,
221
252
  appendFindingsYieldFn,
253
+ gitSpawnFn,
222
254
  progress,
223
255
  });
224
256
 
@@ -7,7 +7,7 @@ import {
7
7
  import { runCloseValidation } from '../../close-validation/runner.js';
8
8
  import { getCiDelivery } from '../../config/ci.js';
9
9
  import { resolveConfig } from '../../config-resolver.js';
10
- import { getStoryBranch, gitSync } from '../../git-utils.js';
10
+ import { getStoryBranch, gitSpawn, gitSync } from '../../git-utils.js';
11
11
  import { Logger } from '../../Logger.js';
12
12
  import { emitTerminalFriction } from '../../observability/runtime-friction.js';
13
13
  import { emitTerseResult } from '../../observability/terse-result.js';
@@ -318,6 +318,11 @@ async function openAndReviewPr({
318
318
  prNumber,
319
319
  provider,
320
320
  runCodeReviewFn: injectedRunCodeReview ?? runCodeReviewDefault,
321
+ // The runner owns the git seam it hands its phases (same as `gitSync` to
322
+ // `pushStoryBranch`). The review needs it to resolve `origin/<base>` —
323
+ // the ref base-sync merged from — before it will score anything
324
+ // (Story #5325).
325
+ gitSpawnFn: gitSpawn,
321
326
  progress,
322
327
  });
323
328
  if (reviewOutcome.halted) {
@@ -164,7 +164,14 @@ const HUMANIZED_PATH_ENTRY_RE = /^`([^`]+)`\s+—\s+(\S+)$/;
164
164
  // AC-<n> presentation prefix on acceptance checkboxes (Story #4600). The
165
165
  // numbering is a stable 1-based human handle only — parse() strips it so the
166
166
  // top-level acceptance[] machine contract round-trips byte-identical.
167
- const AC_PREFIX_RE = /^AC-\d+:\s+/;
167
+ //
168
+ // The lettered form (`AC-14a:`) is accepted too (Story #5323). Nothing emits
169
+ // one — `serialize()` numbers from the array index — but a Story planned from
170
+ // an existing ticket can copy one out of the source issue's rendered
171
+ // checkboxes, and a body that already carries one must still parse to the
172
+ // handle-free text or the round-trip invariant breaks for it alone. Trailing
173
+ // whitespace is optional so `AC-3:text` normalises as readily as `AC-3: text`.
174
+ const AC_PREFIX_RE = /^AC-\d+[a-z]?:\s*/i;
168
175
 
169
176
  // Machine-managed marker lines a body authored before Story #5312 may still
170
177
  // carry: the `> **Wide:** <reason>` rationale line (Story #4600), the
@@ -560,6 +567,33 @@ function parseTextListSection(lines) {
560
567
  * @returns {ParseResult}
561
568
  * @throws {StoryBodyParseError} When the body is structurally unrecoverable.
562
569
  */
570
+ /**
571
+ * Strip the presentation `AC-<n>:` handle off one acceptance item.
572
+ *
573
+ * The handle belongs to {@link serialize}, which numbers every checkbox from
574
+ * its position in `acceptance[]`; an authored item that already carries one
575
+ * would render doubled (`- [ ] AC-1: AC-1: …`) and a lettered handle copied
576
+ * from a source ticket would survive into the machine contract. Both parse
577
+ * and the persist-side normalisation resolve the grammar here so the two can
578
+ * never disagree about what a handle is (Story #5323).
579
+ *
580
+ * Stacked handles are peeled in full — a body persisted while the doubling
581
+ * was live carries two, and leaving the inner one would normalise to
582
+ * something that still is not the authored text.
583
+ *
584
+ * @param {string} item
585
+ * @returns {{ text: string, stripped: boolean }} The handle-free text, and
586
+ * whether anything was removed.
587
+ */
588
+ export function stripAcceptanceHandle(item) {
589
+ const original = String(item ?? '');
590
+ let text = original;
591
+ while (AC_PREFIX_RE.test(text)) {
592
+ text = text.replace(AC_PREFIX_RE, '');
593
+ }
594
+ return { text, stripped: text !== original };
595
+ }
596
+
563
597
  export function parse(input) {
564
598
  if (input === null || input === undefined) {
565
599
  throw new StoryBodyParseError('Story body is null or undefined', {
@@ -618,7 +652,7 @@ export function parse(input) {
618
652
  // The AC-<n> checkbox prefix is presentation-only (Story #4600): strip it
619
653
  // so acceptance[] round-trips byte-identical to the authored array.
620
654
  const acceptance = parseTextListSection(sections.get('acceptance') ?? []).map(
621
- (a) => a.replace(AC_PREFIX_RE, ''),
655
+ (a) => stripAcceptanceHandle(a).text,
622
656
  );
623
657
  const verify = parseTextListSection(sections.get('verify') ?? []);
624
658
  const references = parsePathEntrySection(
@@ -28,10 +28,14 @@ import { BODY_FORMAT_LINTS } from '../story-body/body-format-lints.js';
28
28
  * partition rules that only mean anything once a draft has siblings:
29
29
  * every Story must earn its slot in the wave schedule, and every
30
30
  * acceptance criterion belongs to exactly one Story.
31
+ * - **The tickets-mode rules** ({@link ticketsModePromptField}, Story
32
+ * #5323) — what to re-derive rather than carry when the seed is an
33
+ * existing ticket whose body is already in Story shape.
31
34
  *
32
- * The envelope carries the core as `systemPrompts.story` and the split rules
33
- * as `systemPrompts.storySplitRules`; a planner reads the second only when
34
- * the default-single split policy clears.
35
+ * The envelope carries the core as `systemPrompts.story`, the split rules as
36
+ * `systemPrompts.storySplitRules` and the tickets rules as
37
+ * `systemPrompts.storyTicketsRules`; a planner reads the second only when the
38
+ * default-single split policy clears, and the third only in tickets mode.
35
39
  */
36
40
 
37
41
  /**
@@ -132,7 +136,7 @@ The **persisted** \`body\` renders these markdown sections (in order) — you au
132
136
  - **goal** (in body string): One sentence stating WHY this Story exists.
133
137
  - **spec** (optional, in body string as \`## Spec\`): The technical approach at the altitude the SPEC PROSE CONTRACT below fixes — contract and invariants, never implementation narration. Write as much as the work needs and no more; persist keeps Specs inline at any length and never writes them under \`docs/\`.
134
138
  - **slicing** (optional): Ordered intra-session checkpoints for one Story, one line each. Not a fan-out table and not a duplicate of Acceptance.
135
- - **changes** (in body string): Each entry is an object \`{ path, assumption }\` where \`assumption\` is one of \`creates | refactors-existing | deletes\`. Acceptable path shapes include explicit files (\`src/components/Foo.tsx\`), glob patterns (\`tests/e2e/*.spec.ts\`, \`**/*.astro\`), and module identifiers that resolve to files. Use \`refactors-existing\` for in-place edits to a file already on \`main\`; \`creates\` for net-new files; \`deletes\` for removals. Persist probes every path against the base branch and repairs a plain-string bullet or a trailing parenthetical into the object form for you; a \`creates\` on an existing path or a \`refactors-existing\` on an absent one is a dry-run warning, and only a \`deletes\` naming an absent path is refused.
139
+ - **changes** (in body string): Each entry is an object \`{ path, assumption }\` where \`assumption\` is one of \`creates | refactors-existing | deletes\`. **Name the files the deliverer authors, and omit generated artifacts** — quality baselines, generated test indexes, migration journals, lockfiles and the like are regenerated by the work itself, the refresh is a close-gate concern, and declaring one needlessly reserves a footprint that serializes sibling Stories at dispatch. Acceptable path shapes include explicit files (\`src/components/Foo.tsx\`), glob patterns (\`tests/e2e/*.spec.ts\`, \`**/*.astro\`), and module identifiers that resolve to files. Use \`refactors-existing\` for in-place edits to a file already on \`main\`; \`creates\` for net-new files; \`deletes\` for removals. Persist probes every path against the base branch and repairs a plain-string bullet or a trailing parenthetical into the object form for you; a \`creates\` on an existing path or a \`refactors-existing\` on an absent one is a dry-run warning, and only a \`deletes\` naming an absent path is refused.
136
140
  - **acceptance** (top-level array on the ticket object): Each item is an **outcome a PR reviewer can confirm from the diff and the verify output** — what is true of the codebase once the Story lands, stated at the altitude of the capability (a command that now exits 0 against a named input, a behavior a named test now asserts, a config that now fails validation on a retired key, a document that now records a decision). Aim for **three to six** items: fewer than three usually means the outcome is under-specified; more than six usually means acceptance is re-listing the footprint or the mechanical checks that belong in \`verify[]\`. Push grep-shaped probes, file-exists checks and exit-code tests down into \`verify[]\`; never pin an internal helper name or a private file path into an acceptance item the advisory \`changes[]\` is free to reshape. UNACCEPTABLE: "verify by reading the diff", "looks good", "matches the spec".
137
141
  - **verify** (top-level array on the ticket object): The **mechanical checks** — exact commands or test paths the deliverer runs and the acceptance critic consumes as evidence: \`node --test tests/x.test.js\`, \`npm run lint\`, \`npm run validate\`, a scoped grep. Every acceptance item should be confirmable from at least one verify entry's output plus the diff. Stories with zero verify entries fail validation.
138
142
  - **Bodies record decisions, never questions to the operator.** Never persist an open question ("Flag if…", "TBD", "confirm with the operator") into a Story body — the executing sub-agent is non-interactive and cannot answer it, and the dry-run warns on every one it finds. Triage each unknown by who can resolve it: an AFK-shaped unknown (a fact in docs, a third-party API surface, observable repo behavior) MUST be resolved by your own research before authoring — never restated as an assumption; only a HITL-shaped unknown (a genuine product or architecture call the operator owns) may be restated as a declarative Key Assumption the agent can act on, stating the default chosen (a decision-made-by-default).
@@ -221,6 +225,32 @@ You are splitting past the default-single policy, so simulate the delivery sched
221
225
  - Express ordering with \`depends_on\` (a sibling slug, or \`#<id>\` for an open Story from an earlier plan). A Story whose \`verify[]\` runs against a file a sibling creates MUST \`depends_on\` that sibling, so the file exists when verification runs.`;
222
226
  }
223
227
 
228
+ /**
229
+ * The rules that only apply when the seed is an existing ticket (Story
230
+ * #5323).
231
+ *
232
+ * A `--tickets` seed arrives already in Story shape — rendered `AC-<n>:`
233
+ * checkboxes, a `## Verify` list, a `## Changes` footprint — and an author
234
+ * reading it as a template carries that shape forward instead of re-deriving
235
+ * it. The observed failure (swarm-os #2707 / #2708, planned from #2542 under
236
+ * mandrel 2.57.0) was a Story whose acceptance list was the source's, handles
237
+ * and all, and whose verify entries carried a tier suffix retired two
238
+ * releases earlier. The source ticket is **evidence**, not a draft.
239
+ *
240
+ * @returns {string}
241
+ */
242
+ function renderStoryTicketsRules() {
243
+ return `#### TICKETS-MODE DRAFT — the source ticket is evidence, not a template:
244
+
245
+ You are planning from one or more existing tickets. Read them for **what the work is** — the problem, the constraints, the commands that verify it — and re-derive everything else. Specifically:
246
+
247
+ 1. **Re-derive \`acceptance[]\` from the goal.** Do not copy the source's \`## Acceptance\` list, and never carry its \`AC-<n>:\` handles — the body renderer numbers the checkboxes itself, so a copied handle renders doubled. A source ticket carrying fifteen criteria is telling you its acceptance was over-specified, not that yours must be: state the three to six outcomes a PR reviewer can confirm, and let the rest fall to \`verify[]\`.
248
+ 2. **A mechanical check is a \`verify[]\` command, not an acceptance item.** "Baselines refreshed", "lint exits 0", "the generated index is regenerated", "the quality gate passes" are commands the deliverer runs and the critic reads as evidence. Carrying them as acceptance items inflates the binding contract with work every close already gates.
249
+ 3. **Read the source's \`verify[]\` for the commands it names, not for its shape.** Take the test paths and scripts; drop any trailing tier suffix (\`(unit)\`, \`(contract)\`, \`(e2e)\`, \`(validate)\`) and any \`manual:<reason>\` escape — a verify entry is a bare command.
250
+ 4. **Re-derive the footprint against the tree as it is now.** The source ticket's \`## Changes\` predicted a repository that has since moved; probe the paths you cite and omit the generated artifacts it listed.
251
+ 5. **Do not carry the source's prose wholesale.** Its current-state narration and per-file walkthroughs are exactly what the SPEC PROSE CONTRACT above forbids. Restate the contract and the invariants; the deliverer reads the code for the rest.`;
252
+ }
253
+
224
254
  /**
225
255
  * Render the story-author prompt for a draft of `storyCount` Stories: the
226
256
  * N=1 core, plus the schedule and partition rules when the draft has
@@ -233,3 +263,22 @@ export function renderStoryAuthorPrompt({ storyCount = 1 } = {}) {
233
263
  const core = renderStoryAuthorCore();
234
264
  return storyCount > 1 ? `${core}\n\n${renderStorySplitRules()}` : core;
235
265
  }
266
+
267
+ /**
268
+ * The mode-conditional slice of `systemPrompts`.
269
+ *
270
+ * `storyTicketsRules` only means anything when the seed is an existing
271
+ * ticket, and an envelope carrying it in every mode teaches the author to
272
+ * look for a source ticket a `--seed` run does not have. Returning a
273
+ * spreadable object rather than a nullable string keeps the decision here,
274
+ * beside the prompt it selects, instead of as a branch in the envelope
275
+ * builder.
276
+ *
277
+ * @param {string|undefined} mode The plan-context mode.
278
+ * @returns {{ storyTicketsRules?: string }}
279
+ */
280
+ export function ticketsModePromptField(mode) {
281
+ return mode === 'tickets'
282
+ ? { storyTicketsRules: renderStoryTicketsRules() }
283
+ : {};
284
+ }
@@ -1,15 +1,23 @@
1
1
  /**
2
- * lib/test-run-credit.js — let a green bare `npm test` earn the credit close
3
- * reads (Story #5313).
2
+ * lib/test-run-credit.js — let a green `npm test` **that routes through
3
+ * mandrel's own runner** earn the credit close reads (Story #5313, scoped by
4
+ * Story #5324).
4
5
  *
5
- * Until this module the only suite run that deposited credit was the one
6
- * shaped exactly like the close gate — `coverage-capture.js --cwd <worktree>`
7
- * or `evidence-gate.js --standalone … -- npm test` — and the digest, the
8
- * worker boot context and the reference all carried prose explaining which
9
- * invocation to type. A worker that ran the project's own test runner paid
10
- * for the suite and then close paid for it again. The runner is the natural
11
- * depositor: it knows the tree it ran against, whether the run was green,
12
- * and whether it ran the whole suite.
6
+ * This is a **bonus, not the contract.** The deposit every project can rely
7
+ * on is `evidence-gate.js --standalone --scope-id <id> --gate test --worktree
8
+ * <workCwd> -- npm test`: it spawns whatever `npm test` resolves to and
9
+ * stamps what it just ran, so it is honest on any runner. What this module
10
+ * adds is that a repo whose `test` script *is* `run-tests.js` need not type
11
+ * that wrapper — the runner already knows the tree it ran against, whether
12
+ * the run was green, and whether it ran the whole suite, so it deposits on
13
+ * the way out.
14
+ *
15
+ * The reach is therefore exactly one call site: `run-tests.js`. A consumer
16
+ * whose `npm test` is `vitest run` or `jest` never loads this module, so it
17
+ * deposits nothing **and prints nothing** — silence is not a signal, and no
18
+ * delivery surface may tell an agent to confirm credit by reading for the
19
+ * line below. `mandrel doctor`'s `test-credit-path` check reports which of
20
+ * the two shapes a project is and names the wrapper as the remedy.
13
21
  *
14
22
  * On a green **full-tier** run inside a `story-<id>` checkout the runner
15
23
  * records the `test` gate's evidence in the same keyspace
@@ -143,8 +151,11 @@ export function depositTestRunCredit({
143
151
  }
144
152
 
145
153
  /**
146
- * Deposit and say so on stderr — the runner's one-line hook. The line is
147
- * the only surface a worker sees, so it names the outcome by reason.
154
+ * Deposit and say so on stderr — the runner's one-line hook, printed only
155
+ * when this runner is the one running. It names the outcome by reason, so a
156
+ * green run that deposited nothing (wrong branch, partial tier) says so
157
+ * rather than passing silently; a project on another runner prints no line
158
+ * at all, which is why absence of this line is never evidence either way.
148
159
  *
149
160
  * @param {Parameters<typeof depositTestRunCredit>[0] & { log?: (line: string) => void }} args
150
161
  * @returns {ReturnType<typeof depositTestRunCredit>}
@@ -94,26 +94,33 @@ whose `criteria[]` length differs **before** scoring, consuming no round;
94
94
  not close**: post a `friction` comment and flip `agent::blocked`.
95
95
  Per-round mechanics: [`acceptance-self-eval.md`](acceptance-self-eval.md).
96
96
 
97
- ## 5. The one full-suite run
97
+ ## 5. The one credited suite run
98
98
 
99
- After the self-eval loop's last fix commit, run the project test runner
100
- **once** in the worktree:
99
+ After the self-eval loop's last fix commit, run the suite **once** in the
100
+ worktree through the depositor — it spawns the project's own `npm test`,
101
+ whatever that resolves to, and stamps the result, so any runner earns the
102
+ credit:
101
103
 
102
104
  ```bash
103
- npm test # in <workCwd>
105
+ node <main-repo>/.agents/scripts/evidence-gate.js --standalone \
106
+ --scope-id <storyId> --gate test --worktree <workCwd> -- npm test
104
107
  ```
105
108
 
106
- A green full run on `story-<id>` deposits the `test` evidence close reads,
107
- keyed on the tree, so close reports the gate as **credited** at unchanged
108
- HEAD — a later commit voids it. The CRAP gate still captures coverage itself
109
- when it needs an artifact. If the suite outruns the host's sync Bash
110
- ceiling, dispatch it in the **background** — its completion re-invokes you;
111
- never spawn a task to poll or `sleep`-loop against it
112
- ([`parallel-tooling.md`](parallel-tooling.md) Rule 2). Read the **output**,
113
- not the exit code: the runner prints whether it deposited credit, and a run
114
- off the Story branch or of a partial tier deposits nothing and says so.
115
- Redraft rounds run the scoped projects for the roots you changed plus
116
- `verify[]`, not the whole suite; only this run needs credit.
109
+ Green deposits the `test` evidence close reads, keyed on the tree, so close
110
+ reports the gate as **credited** at unchanged HEAD — a later commit voids it.
111
+ Read its **output**, not the exit code: `✓ test passed` is the signal. The
112
+ CRAP gate still captures coverage itself when it needs an artifact.
113
+
114
+ A bare `npm test` earns the same credit **only** where the project's test
115
+ script routes through mandrel's own runner, which prints the outcome. On any
116
+ other runner it deposits nothing and prints nothing, so silence is never
117
+ evidence of credit; `mandrel doctor`'s `test-credit-path` check names which
118
+ shape this project is. If the suite outruns the host's sync Bash ceiling,
119
+ dispatch it in the **background** — its completion re-invokes you; never spawn
120
+ a task to poll or `sleep`-loop against it
121
+ ([`parallel-tooling.md`](parallel-tooling.md) Rule 2). Redraft rounds run the
122
+ scoped projects for the roots you changed plus `verify[]`, not the whole
123
+ suite; only this run needs credit.
117
124
 
118
125
  `verify[]` is scoped entries **plus** this one run: an entry that is itself a
119
126
  full-suite command is reported credited against the same record, never
@@ -237,17 +237,37 @@ the failure class that actually bounces deliveries: close-validation
237
237
  discovers them only after the whole close pipeline has run, at several times
238
238
  the cost of one full-suite run in the worktree.
239
239
 
240
- **Run it once, last, so close can credit it.** The run belongs **after** the
241
- self-eval loop's last fix commit; redraft rounds run scoped tests. A green
242
- `npm test` in the worktree on `story-<id>` deposits the `test` evidence
243
- record close reads (Story #5313 — `lib/test-run-credit.js`), keyed on HEAD
244
- and the tree fingerprint and hashed on the exact command close spawns, so
245
- close reports the gate as credited at unchanged HEAD instead of re-running
246
- the suite. The credit expires the moment it stops describing the tree: any
247
- later commit invalidates it and close re-runs the suite for real, so this
248
- never trades away the gate. The CRAP gate still runs `coverage-capture.js`
249
- itself when it needs a fresh artifact — the capture stamp is a claim about
250
- `coverage/coverage-final.json`, which a bare `npm test` does not produce.
240
+ **Run it once, last, through the depositor.** The run belongs **after** the
241
+ self-eval loop's last fix commit; redraft rounds run scoped tests. Run it in
242
+ the worktree on `story-<id>` as
243
+
244
+ ```bash
245
+ node <main-repo>/.agents/scripts/evidence-gate.js \
246
+ --standalone --scope-id <storyId> --gate test \
247
+ --worktree <workCwd> -- npm test
248
+ ```
249
+
250
+ The wrapper spawns the project's own `npm test` — whatever that resolves to —
251
+ and records the pass into the Story evidence keyspace, so the credit is
252
+ runner-agnostic by construction: it stamps only what it just ran. The record
253
+ is keyed on HEAD and the tree fingerprint and hashed on the exact command
254
+ close spawns, so close reports the gate as credited at unchanged HEAD instead
255
+ of re-running the suite. The credit expires the moment it stops describing the
256
+ tree: any later commit invalidates it and close re-runs the suite for real, so
257
+ this never trades away the gate. The CRAP gate still runs
258
+ `coverage-capture.js` itself when it needs a fresh artifact — the capture
259
+ stamp is a claim about `coverage/coverage-final.json`, which `npm test` alone
260
+ does not produce.
261
+
262
+ **A bare `npm test` is a bonus, not the contract.** It deposits the same
263
+ record only where the project's `test` script routes through mandrel's own
264
+ runner (`run-tests.js` → `lib/test-run-credit.js`, Story #5313), which prints
265
+ the outcome. A project whose `npm test` is `vitest run`, `jest` or any other
266
+ runner never reaches that code, so it prints nothing and deposits nothing —
267
+ silence is not a signal, and nothing here asks you to confirm the credit by
268
+ reading for a line that cannot appear. `mandrel doctor`'s `test-credit-path`
269
+ check reports which shape a project is and names the command above as its
270
+ remedy.
251
271
 
252
272
  **`verify[]` reuses the same credit.** A `verify[]` entry that is itself a
253
273
  full-suite command is reported **credited** against that record rather than
@@ -93,12 +93,13 @@ are reference § Step 2. Hard gates always run in Step 3 — the derived level
93
93
  never disables them; do **not** pre-run the chain here — Step 2.5's credited
94
94
  suite run is the sole exception.
95
95
 
96
- ### Step 2.5 — The one full-suite run, the push, then hand off
96
+ ### Step 2.5 — The one credited suite run, the push, then hand off
97
97
 
98
- After the self-eval loop's last fix commit, run `npm test` **once** in the
99
- worktree (**digest § 5**): a green full run deposits the `test` credit close
100
- reads, keyed on the tree, so only a *later* commit invalidates it. Red →
101
- fix, commit, re-run.
98
+ After the self-eval loop's last fix commit, run the suite **once** in the
99
+ worktree through the depositor — `evidence-gate.js … --gate test -- npm test`,
100
+ spelled out in **digest § 5**. It runs whatever `npm test` resolves to and
101
+ stamps that, so the `test` credit is earned on any runner and only a *later*
102
+ commit invalidates it. Red → fix, commit, re-run.
102
103
 
103
104
  Push `story-<storyId>` to `origin`, confirming the remote ref moved. Then
104
105
  (sub-agent dispatch only) return the hand-off — Story id, `workCwd`,