mandrel 2.43.0 → 2.44.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,327 @@
1
+ /**
2
+ * plan-run-labels/reap.js — the cohort label's end of life (Story #5189).
3
+ *
4
+ * `plan-persist` mints one `plan-run::<id>` label per run so the Stories one
5
+ * plan authored stay filterable in the GitHub UI. That is load-bearing while
6
+ * any Story in the cohort is open, and inert the moment they are all closed —
7
+ * but nothing expressed the second half, so the vocabulary grew one label per
8
+ * persist forever. A consumer repository measured 235 `plan-run::` labels out
9
+ * of 313 total; past that a paged label listing stops seeing the labels that
10
+ * sort after the pile, and every list-then-create caller starts failing.
11
+ *
12
+ * This module is the **one decision engine** behind both surfaces that act on
13
+ * that end of life: the per-Story close tail (automatic, incremental) and
14
+ * `prune-plan-run-labels.js` (manual, whole-repository). Keeping the decision
15
+ * here rather than in either caller is what stops the two from disagreeing
16
+ * about when a label is spent.
17
+ *
18
+ * **The decision.** A cohort label is reapable only when it carries at least
19
+ * one issue and every issue carrying it is closed. The "at least one" clause
20
+ * is the subtle half: a label carrying *zero* issues is indistinguishable from
21
+ * one an in-flight persist has just minted before creating its Stories, and
22
+ * deleting it would break that run. So a zero-issue label is reported under an
23
+ * `unreferenced` reason and reaped only under an explicit opt-in — a default
24
+ * sweep is safe to run concurrently with a persist.
25
+ *
26
+ * **Never load-bearing.** Both surfaces treat a reap failure as a warning.
27
+ * Label hygiene is a chore; nothing downstream reads the cohort label as an
28
+ * input (see `PLAN_RUN_LABEL_PREFIX`'s own docblock), so a failed delete costs
29
+ * a stale label and nothing else.
30
+ *
31
+ * @module lib/orchestration/plan-run-labels/reap
32
+ * @see Story #5189
33
+ */
34
+
35
+ import { PLAN_RUN_LABEL_PREFIX } from '../plan-persist/story-ops.js';
36
+
37
+ /**
38
+ * Why a cohort label was — or was not — judged reapable. One reason per
39
+ * label, so a `--json` report is auditable without re-deriving anything.
40
+ *
41
+ * - `all-closed` — carries issues, every one closed. Reapable.
42
+ * - `open-stories` — at least one issue still open. Not reapable, ever.
43
+ * - `unreferenced` — carries no issues at all. Reapable only under the
44
+ * explicit opt-in, because an in-flight persist looks exactly like this.
45
+ */
46
+ export const REAP_REASONS = Object.freeze({
47
+ ALL_CLOSED: 'all-closed',
48
+ OPEN_STORIES: 'open-stories',
49
+ UNREFERENCED: 'unreferenced',
50
+ });
51
+
52
+ /**
53
+ * Is `name` a plan-run cohort label?
54
+ *
55
+ * The prefix is imported from `plan-persist/story-ops.js` rather than
56
+ * re-declared: minting and reaping must not be able to drift onto two
57
+ * different strings, which is exactly the failure a copied literal invites.
58
+ *
59
+ * @param {unknown} name
60
+ * @returns {boolean}
61
+ */
62
+ function isPlanRunLabel(name) {
63
+ return typeof name === 'string' && name.startsWith(PLAN_RUN_LABEL_PREFIX);
64
+ }
65
+
66
+ /**
67
+ * Project an arbitrary label collection onto the sorted, de-duplicated set of
68
+ * cohort label names. Accepts either bare strings (a ticket's `labels[]`) or
69
+ * `{ name }` rows (the label listing port), so both callers hand this whatever
70
+ * their own read returned.
71
+ *
72
+ * @param {Array<string|{ name?: string }>} [names]
73
+ * @returns {string[]}
74
+ */
75
+ function selectCohortLabels(names) {
76
+ const seen = new Set();
77
+ for (const raw of Array.isArray(names) ? names : []) {
78
+ const name = typeof raw === 'string' ? raw : raw?.name;
79
+ if (isPlanRunLabel(name)) seen.add(name);
80
+ }
81
+ return [...seen].sort();
82
+ }
83
+
84
+ /**
85
+ * Decide one cohort label.
86
+ *
87
+ * Reads through `listIssuesByLabel({ state: 'all' })` — the paginating read
88
+ * port — so the verdict is never a function of how many issues fit on one API
89
+ * page. `state` is compared case-insensitively against `closed` and anything
90
+ * else counts as open: an unknown state must never be read as "safe to
91
+ * delete".
92
+ *
93
+ * @param {{ provider: object, label: string, includeUnreferenced?: boolean }} args
94
+ * @returns {Promise<{
95
+ * label: string,
96
+ * reapable: boolean,
97
+ * reason: string,
98
+ * issueCount: number,
99
+ * openIssues: number[],
100
+ * }>}
101
+ */
102
+ async function decideCohortLabel({
103
+ provider,
104
+ label,
105
+ includeUnreferenced = false,
106
+ }) {
107
+ const result = await provider.listIssuesByLabel({
108
+ state: 'all',
109
+ labels: label,
110
+ });
111
+ const issues = Array.isArray(result) ? result : [];
112
+ const open = issues.filter(
113
+ (issue) => String(issue?.state ?? '').toLowerCase() !== 'closed',
114
+ );
115
+ if (issues.length === 0) {
116
+ return {
117
+ label,
118
+ reapable: includeUnreferenced === true,
119
+ reason: REAP_REASONS.UNREFERENCED,
120
+ issueCount: 0,
121
+ openIssues: [],
122
+ };
123
+ }
124
+ if (open.length > 0) {
125
+ return {
126
+ label,
127
+ reapable: false,
128
+ reason: REAP_REASONS.OPEN_STORIES,
129
+ issueCount: issues.length,
130
+ openIssues: open
131
+ .map((issue) => issue?.number)
132
+ .filter((n) => Number.isInteger(n)),
133
+ };
134
+ }
135
+ return {
136
+ label,
137
+ reapable: true,
138
+ reason: REAP_REASONS.ALL_CLOSED,
139
+ issueCount: issues.length,
140
+ openIssues: [],
141
+ };
142
+ }
143
+
144
+ /**
145
+ * Decide a set of cohort labels.
146
+ *
147
+ * Sequential on purpose. The whole-repository sweep can face hundreds of
148
+ * labels, and a fan-out over a shared REST budget buys wall-clock at the cost
149
+ * of the one property an operator auditing a pile actually needs: a
150
+ * deterministic, label-ordered report.
151
+ *
152
+ * @param {{
153
+ * provider: object,
154
+ * labels: Array<string|{ name?: string }>,
155
+ * includeUnreferenced?: boolean,
156
+ * }} args
157
+ * @returns {Promise<Array<object>>} one decision per cohort label, name-sorted.
158
+ */
159
+ async function evaluateCohortLabels({
160
+ provider,
161
+ labels,
162
+ includeUnreferenced = false,
163
+ }) {
164
+ const decisions = [];
165
+ for (const label of selectCohortLabels(labels)) {
166
+ decisions.push(
167
+ await decideCohortLabel({ provider, label, includeUnreferenced }),
168
+ );
169
+ }
170
+ return decisions;
171
+ }
172
+
173
+ /**
174
+ * Decide, then (unless `check`) delete.
175
+ *
176
+ * A delete that throws is recorded in `failed[]` and warned about rather than
177
+ * propagated: one unreachable label must not abandon the rest of a sweep, and
178
+ * on the close path it must not touch the land. A delete the provider reports
179
+ * as a no-op (`deleted: false` — the label was already gone) is still a
180
+ * success; that is what makes a re-run idempotent.
181
+ *
182
+ * @param {{
183
+ * provider: object,
184
+ * labels: Array<string|{ name?: string }>,
185
+ * includeUnreferenced?: boolean,
186
+ * check?: boolean,
187
+ * onWarn?: ((message: string) => void)|null,
188
+ * }} args
189
+ * @returns {Promise<{
190
+ * check: boolean,
191
+ * decisions: Array<object>,
192
+ * reapable: string[],
193
+ * deleted: Array<{ label: string, existed: boolean }>,
194
+ * failed: Array<{ label: string, detail: string }>,
195
+ * }>}
196
+ */
197
+ async function reapCohortLabels({
198
+ provider,
199
+ labels,
200
+ includeUnreferenced = false,
201
+ check = false,
202
+ onWarn = null,
203
+ }) {
204
+ const decisions = await evaluateCohortLabels({
205
+ provider,
206
+ labels,
207
+ includeUnreferenced,
208
+ });
209
+ const reapable = decisions.filter((d) => d.reapable).map((d) => d.label);
210
+ const deleted = [];
211
+ const failed = [];
212
+ if (check !== true) {
213
+ for (const label of reapable) {
214
+ try {
215
+ const outcome = await provider.deleteLabel(label);
216
+ deleted.push({ label, existed: outcome?.deleted !== false });
217
+ } catch (err) {
218
+ const detail = String(err?.message ?? err);
219
+ failed.push({ label, detail });
220
+ onWarn?.(`could not delete cohort label "${label}": ${detail}`);
221
+ }
222
+ }
223
+ }
224
+ return { check: check === true, decisions, reapable, deleted, failed };
225
+ }
226
+
227
+ /**
228
+ * The automatic surface's entry point: reap the cohort labels carried by the
229
+ * Story that just closed.
230
+ *
231
+ * The Story's own label set comes from `getTicket` (labels are immutable for
232
+ * this purpose, so a cached snapshot is fine); every *state* judgment comes
233
+ * from the fresh `listIssuesByLabel` read inside {@link decideCohortLabel}, so
234
+ * a primed ticket cache cannot make a still-open sibling look closed.
235
+ *
236
+ * A Story whose own issue has not yet registered as closed — the merge webhook
237
+ * that fires `Closes #<id>` is not instantaneous — simply reports
238
+ * `open-stories` and is left alone. Deferring is the safe direction, and the
239
+ * manual sweep is the backstop that collects whatever a race leaves behind.
240
+ *
241
+ * @param {{
242
+ * storyId: number,
243
+ * provider: object,
244
+ * includeUnreferenced?: boolean,
245
+ * onWarn?: ((message: string) => void)|null,
246
+ * }} args
247
+ * @returns {Promise<object>} the {@link reapCohortLabels} envelope, plus
248
+ * `evaluated` — how many cohort labels the Story carried.
249
+ */
250
+ export async function reapPlanRunLabelsForStory({
251
+ storyId,
252
+ provider,
253
+ includeUnreferenced = false,
254
+ onWarn = null,
255
+ }) {
256
+ const ticket = await provider.getTicket(storyId);
257
+ const labels = selectCohortLabels(ticket?.labels);
258
+ if (labels.length === 0) {
259
+ return {
260
+ evaluated: 0,
261
+ check: false,
262
+ decisions: [],
263
+ reapable: [],
264
+ deleted: [],
265
+ failed: [],
266
+ };
267
+ }
268
+ const outcome = await reapCohortLabels({
269
+ provider,
270
+ labels,
271
+ includeUnreferenced,
272
+ onWarn,
273
+ });
274
+ return { evaluated: labels.length, ...outcome };
275
+ }
276
+
277
+ /**
278
+ * The manual surface's entry point: sweep every cohort label in the repository.
279
+ *
280
+ * Reads the whole label vocabulary through the paginating listing port and
281
+ * projects it onto the cohort axis here, so the sweep is bounded by what the
282
+ * repository actually holds rather than by an API page size.
283
+ *
284
+ * @param {{
285
+ * provider: object,
286
+ * includeUnreferenced?: boolean,
287
+ * check?: boolean,
288
+ * onWarn?: ((message: string) => void)|null,
289
+ * }} args
290
+ * @returns {Promise<object>} the {@link reapCohortLabels} envelope, plus
291
+ * `totalLabels` (whole vocabulary) and `evaluated` (the cohort slice).
292
+ */
293
+ export async function sweepCohortLabels({
294
+ provider,
295
+ includeUnreferenced = false,
296
+ check = false,
297
+ onWarn = null,
298
+ }) {
299
+ const all = await provider.listLabels();
300
+ const rows = Array.isArray(all) ? all : [];
301
+ const labels = selectCohortLabels(rows);
302
+ const outcome = await reapCohortLabels({
303
+ provider,
304
+ labels,
305
+ includeUnreferenced,
306
+ check,
307
+ onWarn,
308
+ });
309
+ return { totalLabels: rows.length, evaluated: labels.length, ...outcome };
310
+ }
311
+
312
+ /**
313
+ * Test-only surface. The five helpers below compose the two exported entry
314
+ * points (`reapPlanRunLabelsForStory`, `sweepCohortLabels`) and have no
315
+ * production consumer outside this module, so exporting each one individually
316
+ * would advertise five API surfaces nothing imports — and `dead-exports
317
+ * --production` correctly reports each as dead. They are still worth unit
318
+ * testing per arm, which is what this barrel is for; it follows the same
319
+ * `__testing` idiom `git-probes.js` and `source-classifier.js` use.
320
+ */
321
+ export const __testing = {
322
+ isPlanRunLabel,
323
+ selectCohortLabels,
324
+ decideCohortLabel,
325
+ evaluateCohortLabels,
326
+ reapCohortLabels,
327
+ };
@@ -45,6 +45,28 @@
45
45
  * and that intent stands). It additionally reports a `degradations[]` array
46
46
  * naming **which** surface could not run and **why**, so the review outcome can
47
47
  * say "this gate did not run" instead of silently reading clean.
48
+ *
49
+ * ## The code surface never got the same treatment (Story #5193)
50
+ *
51
+ * Story #4839 gave the *markdown* surface a disk probe and left the code
52
+ * surface spawning `npx --no biome` and classifying the exit code. On npm 11.x
53
+ * that spawn exits **0 with empty output** when biome is absent, so
54
+ * `parseLintOutput` saw no failure, produced no degradation, and the surface
55
+ * contributed `errors: 0, warnings: 0` — a *silent false clean*, which is
56
+ * strictly worse than a degradation: a degraded surface announces itself, a
57
+ * falsely-clean one is trusted. Since nothing in the framework requires either
58
+ * runner, "absent" is the default state of a consumer checkout.
59
+ *
60
+ * Fix: runner resolution is a **precondition** for every surface, not an
61
+ * outcome inferred from an exit code. Both surfaces now resolve through
62
+ * {@link resolveRunner}, and an unresolved runner is never spawned.
63
+ *
64
+ * The same measurement invalidated the `NPX_UNRESOLVABLE` sentinel: current npm
65
+ * answers an unresolvable bin with `npm error code E404`, not `could not
66
+ * determine executable to run`, so every genuinely-unresolvable runner was
67
+ * being labelled `unparseable-output`. The sentinel now recognises both shapes.
68
+ * It still earns its keep after the disk probe: the probe only sees
69
+ * `node_modules/.bin`, so a runner resolvable some other way can still fail.
48
70
  */
49
71
 
50
72
  import { spawnSync } from 'node:child_process';
@@ -54,8 +76,14 @@ import path from 'node:path';
54
76
  /** Paths these extensions land on the biome (code) runner. */
55
77
  const CODE_EXTENSIONS = /\.(js|mjs|cjs|jsx|ts|tsx|json|jsonc)$/i;
56
78
 
57
- /** npx's message when the requested bin cannot be resolved. */
58
- const NPX_UNRESOLVABLE = /could not determine executable to run/i;
79
+ /**
80
+ * npx's output when the requested bin cannot be resolved. Two shapes: the
81
+ * legacy message, and the `E404` current npm answers with instead (measured
82
+ * 2026-09-07 on npm 11.13.0). Matching the E404 *code* rather than any
83
+ * `npm error` line keeps a genuine runner error out of this classification.
84
+ */
85
+ const NPX_UNRESOLVABLE =
86
+ /could not determine executable to run|npm (?:error|ERR!)\s+code\s+E404/i;
59
87
 
60
88
  /** Biome's exit-1 message when every supplied path is config-excluded. */
61
89
  const BIOME_EMPTY_SCOPE = /No files were processed in the specified paths/i;
@@ -74,6 +102,15 @@ const MARKDOWN_RUNNERS = Object.freeze([
74
102
  }),
75
103
  ]);
76
104
 
105
+ /**
106
+ * Code runners in preference order. `@biomejs/biome` installs a bare `biome`
107
+ * bin, which is also the canonical name this surface reports itself under when
108
+ * nothing resolves.
109
+ */
110
+ const CODE_RUNNERS = Object.freeze([
111
+ Object.freeze({ bin: 'biome', extraArgs: Object.freeze([]) }),
112
+ ]);
113
+
77
114
  /** Reason codes carried on a degradation record. */
78
115
  const DEGRADATION_REASONS = Object.freeze({
79
116
  RUNNER_NOT_INSTALLED: 'runner-not-installed',
@@ -103,24 +140,30 @@ function spawnLintRunner(bin, args, cwd) {
103
140
  }
104
141
 
105
142
  /**
106
- * Pure-ish: pick the first markdown runner whose bin is actually installed
107
- * under `<cwd>/node_modules/.bin`. Returns `null` when none is — an honest
108
- * "this surface has no runner" that the caller reports rather than silently
109
- * folding into a generic parse failure.
143
+ * Pure-ish: pick the first candidate whose bin is actually installed under
144
+ * `<cwd>/node_modules/.bin`. Returns `null` when none is — an honest "this
145
+ * surface has no runner" that the caller reports rather than silently folding
146
+ * into a generic parse failure.
110
147
  *
111
148
  * The disk probe (rather than "spawn and see") is what makes the failure
112
- * *nameable*: `npx --no <missing-bin>` yields only a generic npm error, which
113
- * is precisely how the defect hid for months.
149
+ * *nameable*, and — since Story #5193 — what makes it *visible at all* on the
150
+ * code surface: `npx --no <missing-bin>` answers with a generic npm error at
151
+ * best and an empty exit 0 at worst, which is precisely how both defects hid.
152
+ *
153
+ * Probing `node_modules/.bin` only is a deliberate bound: a globally-installed
154
+ * runner reads as absent here, which degrades the gate honestly rather than
155
+ * trusting a spawn nobody resolved.
114
156
  *
115
157
  * Not exported: it is reachable — and asserted — through {@link runScopedLint},
116
158
  * whose `existsFn` seam drives every resolution branch.
117
159
  *
160
+ * @param {ReadonlyArray<{ bin: string, extraArgs: ReadonlyArray<string> }>} candidates
118
161
  * @param {string} cwd
119
- * @param {(p: string) => boolean} [existsFn] Injected for testing.
162
+ * @param {(p: string) => boolean} existsFn Injected for testing.
120
163
  * @returns {{ bin: string, extraArgs: ReadonlyArray<string> }|null}
121
164
  */
122
- function resolveMarkdownRunner(cwd, existsFn = existsSync) {
123
- for (const candidate of MARKDOWN_RUNNERS) {
165
+ function resolveRunner(candidates, cwd, existsFn) {
166
+ for (const candidate of candidates) {
124
167
  const base = path.join(cwd, 'node_modules', '.bin', candidate.bin);
125
168
  if (existsFn(base)) return candidate;
126
169
  if (
@@ -133,6 +176,49 @@ function resolveMarkdownRunner(cwd, existsFn = existsSync) {
133
176
  return null;
134
177
  }
135
178
 
179
+ /**
180
+ * Pure: the summary a surface reports when it has no runner to spawn. Counts
181
+ * are zero *and* `executionFailed` is true, so the row can never be read as a
182
+ * clean result — the invariant Story #5193 restored.
183
+ *
184
+ * @returns {ReturnType<typeof parseLintOutput>}
185
+ */
186
+ function unresolvedRunnerSummary() {
187
+ return {
188
+ errors: 0,
189
+ warnings: 0,
190
+ parsed: false,
191
+ executionFailed: true,
192
+ emptyScope: false,
193
+ reason: DEGRADATION_REASONS.RUNNER_NOT_INSTALLED,
194
+ };
195
+ }
196
+
197
+ /**
198
+ * Resolve one surface's runner and, only if it resolved, spawn and classify it.
199
+ * Shared by both surfaces so neither can drift back into spawn-and-see.
200
+ *
201
+ * @param {{
202
+ * label: string,
203
+ * candidates: ReadonlyArray<{ bin: string, extraArgs: ReadonlyArray<string> }>,
204
+ * buildArgs: (runner: { bin: string, extraArgs: ReadonlyArray<string> }) => string[],
205
+ * cwd: string,
206
+ * runnerFn: typeof spawnLintRunner,
207
+ * existsFn: (p: string) => boolean,
208
+ * }} args
209
+ * @returns {{ surface: string, summary: ReturnType<typeof parseLintOutput> }}
210
+ */
211
+ function runSurface({ label, candidates, buildArgs, cwd, runnerFn, existsFn }) {
212
+ const runner = resolveRunner(candidates, cwd, existsFn);
213
+ if (runner === null) {
214
+ return { surface: label, summary: unresolvedRunnerSummary() };
215
+ }
216
+ return {
217
+ surface: runner.bin,
218
+ summary: parseLintOutput(runnerFn(runner.bin, buildArgs(runner), cwd)),
219
+ };
220
+ }
221
+
136
222
  /**
137
223
  * Pure: split changed paths into the file lists each lint runner consumes.
138
224
  *
@@ -267,33 +353,28 @@ export function runScopedLint(
267
353
 
268
354
  const surfaces = [];
269
355
  if (code.length > 0) {
270
- surfaces.push({
271
- surface: 'biome',
272
- summary: parseLintOutput(runnerFn('biome', ['lint', ...code], cwd)),
273
- });
356
+ surfaces.push(
357
+ runSurface({
358
+ label: 'biome',
359
+ candidates: CODE_RUNNERS,
360
+ buildArgs: (runner) => ['lint', ...code, ...runner.extraArgs],
361
+ cwd,
362
+ runnerFn,
363
+ existsFn,
364
+ }),
365
+ );
274
366
  }
275
367
  if (md.length > 0) {
276
- const runner = resolveMarkdownRunner(cwd, existsFn);
277
- if (runner === null) {
278
- surfaces.push({
279
- surface: 'markdownlint',
280
- summary: {
281
- errors: 0,
282
- warnings: 0,
283
- parsed: false,
284
- executionFailed: true,
285
- emptyScope: false,
286
- reason: DEGRADATION_REASONS.RUNNER_NOT_INSTALLED,
287
- },
288
- });
289
- } else {
290
- surfaces.push({
291
- surface: runner.bin,
292
- summary: parseLintOutput(
293
- runnerFn(runner.bin, [...md, ...runner.extraArgs], cwd),
294
- ),
295
- });
296
- }
368
+ surfaces.push(
369
+ runSurface({
370
+ label: 'markdownlint',
371
+ candidates: MARKDOWN_RUNNERS,
372
+ buildArgs: (runner) => [...md, ...runner.extraArgs],
373
+ cwd,
374
+ runnerFn,
375
+ existsFn,
376
+ }),
377
+ );
297
378
  }
298
379
 
299
380
  return mergeSurfaceSummaries(surfaces);
@@ -44,6 +44,7 @@ import {
44
44
  executeFastForward as defaultExecuteFastForward,
45
45
  planFastForward as defaultPlanFastForward,
46
46
  } from '../../git-cleanup/phases/fast-forward.js';
47
+ import { reapPlanRunLabelsForStory as defaultReapPlanRunLabelsForStory } from '../../plan-run-labels/reap.js';
47
48
  import { reassertStatusColumn as defaultReassertStatusColumn } from '../../reassert-status-column.js';
48
49
  import { releaseStoryLease as defaultReleaseStoryLease } from '../../single-story-lease-guard.js';
49
50
  import { captureStoryFollowUps as defaultCaptureStoryFollowUps } from '../../story-follow-ups.js';
@@ -290,6 +291,70 @@ async function stepLeaseRelease({
290
291
  };
291
292
  }
292
293
 
294
+ /**
295
+ * The whole-repository sweep an operator runs when the automatic reap could
296
+ * not finish its job. Named in the warning itself so the next step is in the
297
+ * message rather than in a runbook nobody opens mid-incident.
298
+ */
299
+ const REAP_SWEEP_REMEDY = 'node .agents/scripts/prune-plan-run-labels.js';
300
+
301
+ /**
302
+ * Reap the cohort labels the closing Story carried (Story #5189).
303
+ *
304
+ * This seam is chosen deliberately. It is the only one that fires for both
305
+ * single- and multi-Story runs: the multi-Story run epilogue is keyed on a
306
+ * synthesized ad-hoc id, never sees the cohort label, and reports
307
+ * `applicable: false` at N=1 — which is the planning default, so wiring the
308
+ * reap there would leave the common case unreaped forever.
309
+ *
310
+ * Best-effort in the strongest sense the tail offers: the outcome is
311
+ * deliberately NOT reported in the returned `tail` envelope. A per-step
312
+ * boolean is the right shape for a step whose failure degrades the *report of
313
+ * the land* — a missed follow-up, an unresynced status column. Label
314
+ * vocabulary hygiene is not that: nothing downstream reads a cohort label as
315
+ * an input, so a failed reap costs one stale label. Surfacing it in the
316
+ * envelope would make a close whose label read flaked terminate differently
317
+ * from one where no label was reapable, for no difference an operator can act
318
+ * on. The failure is named in a warning instead, and the whole-repository
319
+ * sweep (`node .agents/scripts/prune-plan-run-labels.js`) collects whatever
320
+ * the automatic path misses.
321
+ *
322
+ * Never reaps a zero-issue label: that shape is indistinguishable from a label
323
+ * an in-flight persist has just minted, so the opt-in stays off here.
324
+ */
325
+ async function stepPlanRunLabelReap({
326
+ storyId,
327
+ provider,
328
+ progress,
329
+ reapPlanRunLabelsForStoryFn,
330
+ }) {
331
+ const warn = (message) =>
332
+ progress?.(
333
+ 'POST-LAND',
334
+ `⚠️ plan-run label reap: ${message} — sweep the pile with ` +
335
+ `"${REAP_SWEEP_REMEDY}".`,
336
+ );
337
+ const outcome = await reapPlanRunLabelsForStoryFn({
338
+ storyId,
339
+ provider,
340
+ onWarn: warn,
341
+ });
342
+ const reaped = outcome?.deleted?.length ?? 0;
343
+ if (reaped > 0) {
344
+ progress?.(
345
+ 'POST-LAND',
346
+ `🏷️ Reaped ${reaped} spent plan-run label(s): ` +
347
+ `${outcome.deleted.map((d) => d.label).join(', ')}.`,
348
+ );
349
+ }
350
+ return {
351
+ ok: (outcome?.failed?.length ?? 0) === 0,
352
+ detail: outcome?.failed?.length
353
+ ? outcome.failed.map((f) => f.label).join(', ')
354
+ : null,
355
+ };
356
+ }
357
+
293
358
  /**
294
359
  * Run the whole post-land tail. Never throws.
295
360
  *
@@ -330,6 +395,7 @@ async function stepLeaseRelease({
330
395
  * @param {Function} [args.acquireLockWithWaitFn] Test seam.
331
396
  * @param {Function} [args.purgeStoryTempArtifactsFn] Test seam.
332
397
  * @param {Function} [args.releaseStoryLeaseFn] Test seam.
398
+ * @param {Function} [args.reapPlanRunLabelsForStoryFn] Test seam.
333
399
  * @returns {Promise<{ followUps: boolean, statusResync: boolean, refCleanup: boolean, baseFastForward: boolean, tempPurge: boolean, leaseRelease: boolean, details: Record<string, string|null> }>}
334
400
  */
335
401
  export async function runPostLandTail({
@@ -350,6 +416,7 @@ export async function runPostLandTail({
350
416
  acquireLockWithWaitFn = defaultAcquireLockWithWait,
351
417
  purgeStoryTempArtifactsFn = defaultPurgeStoryTempArtifacts,
352
418
  releaseStoryLeaseFn = defaultReleaseStoryLease,
419
+ reapPlanRunLabelsForStoryFn = defaultReapPlanRunLabelsForStory,
353
420
  }) {
354
421
  progress?.('POST-LAND', `🧾 Running land tail for Story #${storyId}...`);
355
422
 
@@ -402,6 +469,20 @@ export async function runPostLandTail({
402
469
  }),
403
470
  { name: 'status-column resync', progress },
404
471
  );
472
+ // Story #5189 — the cohort label's end of life. Runs with the other
473
+ // GitHub-touching steps (outside the checkout lock) and contributes nothing
474
+ // to `tail`; see `stepPlanRunLabelReap` for why that omission is the point.
475
+ await step(
476
+ () =>
477
+ stepPlanRunLabelReap({
478
+ storyId,
479
+ provider,
480
+ progress,
481
+ reapPlanRunLabelsForStoryFn,
482
+ }),
483
+ { name: 'plan-run label reap', progress },
484
+ );
485
+
405
486
  // Local-checkout mutations: serialized behind a best-effort cross-process
406
487
  // lock (Story #4622). Acquire once, run both steps, release in `finally`.
407
488
  const lockCfg = config?.delivery?.postLandLock ?? {};
@@ -9,12 +9,17 @@
9
9
  * explicitly, and always state them (as `none` when healthy) so an absent line
10
10
  * can never be mistaken for a clean gate.
11
11
  *
12
- * A degraded gate is **reported, not blocking**: the canonical `npm run lint`
13
- * close-validation gate has already covered this diff before the review phase
14
- * runs, so failing the merge on a secondary read of an already-gated surface
15
- * would cost delivery without buying coverage. The rationale for that posture
16
- * lives with the channel itself in
12
+ * A degraded gate is **reported, not blocking**: this review is a secondary
13
+ * read, and the close does not gate the merge on it. The rationale for that
14
+ * posture lives with the channel itself in
17
15
  * [`review-providers/degraded-gates.js`](../../review-providers/degraded-gates.js).
16
+ *
17
+ * What this module must **not** do (Story #5193) is tell the operator that the
18
+ * canonical `npm run lint` close gate covered the surface instead. A stub
19
+ * `lint` script is a supported consumer shape, so that claim is unverifiable
20
+ * from here — and asserting it talks the operator out of the exact concern the
21
+ * degradation was raised to surface. State the posture; never vouch for
22
+ * coverage this module cannot see.
18
23
  */
19
24
 
20
25
  import { summarizeDegradations } from '../../review-providers/degraded-gates.js';
@@ -58,8 +63,8 @@ export function formatReviewOutcomeLines({
58
63
  if (summarizeDegradations(degradations) !== 'none') {
59
64
  lines.push(
60
65
  '⚠️ Review ran DEGRADED — the surface(s) above were not reviewed. The close ' +
61
- 'is not blocked (the canonical `npm run lint` close gate already covered ' +
62
- 'this diff), but this review does not vouch for them.',
66
+ 'is not blocked — a secondary review does not gate the merge — but ' +
67
+ 'nothing here vouches for those surfaces.',
63
68
  );
64
69
  }
65
70
  return lines;