mandrel 2.54.0 → 2.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/.agents/agents/story-worker.md +24 -23
  2. package/.agents/audit-checklists/accessibility.md +0 -3
  3. package/.agents/audit-checklists/mobile.md +0 -4
  4. package/.agents/docs/agentrc-reference.json +4 -2
  5. package/.agents/docs/configuration.md +2 -0
  6. package/.agents/schemas/agentrc.schema.json +15 -1
  7. package/.agents/schemas/lifecycle/merge.unlanded.schema.json +2 -1
  8. package/.agents/schemas/story-deliver-terminal.schema.json +1 -0
  9. package/.agents/scripts/audit-to-stories.js +158 -7
  10. package/.agents/scripts/check-audit-attribution.js +119 -62
  11. package/.agents/scripts/check-test-portability.js +512 -0
  12. package/.agents/scripts/coverage-capture.js +17 -10
  13. package/.agents/scripts/evidence-gate.js +31 -4
  14. package/.agents/scripts/generate-workflows-doc.js +65 -14
  15. package/.agents/scripts/git-cleanup.js +4 -0
  16. package/.agents/scripts/lib/ITicketingProvider.js +78 -0
  17. package/.agents/scripts/lib/audit-advisories.js +195 -0
  18. package/.agents/scripts/lib/audit-attribution.js +22 -0
  19. package/.agents/scripts/lib/audit-to-stories/dedupe-against-github.js +68 -5
  20. package/.agents/scripts/lib/audit-to-stories/issue-index.js +83 -0
  21. package/.agents/scripts/lib/audit-to-stories/ledger-commit.js +60 -114
  22. package/.agents/scripts/lib/audit-to-stories/ledger-pr.js +347 -0
  23. package/.agents/scripts/lib/audit-to-stories/parse-audit-md.js +169 -44
  24. package/.agents/scripts/lib/baselines/merge-envelopes.js +298 -32
  25. package/.agents/scripts/lib/bootstrap/baseline-merge-driver.js +180 -14
  26. package/.agents/scripts/lib/cli-args.js +26 -0
  27. package/.agents/scripts/lib/close-validation/gates.js +113 -7
  28. package/.agents/scripts/lib/close-validation/process.js +7 -3
  29. package/.agents/scripts/lib/close-validation/runner.js +62 -11
  30. package/.agents/scripts/lib/config/ci.js +28 -9
  31. package/.agents/scripts/lib/config-settings-schema-delivery.js +7 -0
  32. package/.agents/scripts/lib/config-settings-schema.js +19 -1
  33. package/.agents/scripts/lib/coverage-capture-fullscope.js +23 -11
  34. package/.agents/scripts/lib/coverage-capture-incremental.js +22 -16
  35. package/.agents/scripts/lib/coverage-capture-usage.js +5 -1
  36. package/.agents/scripts/lib/coverage-capture.js +77 -3
  37. package/.agents/scripts/lib/findings/route-finding.js +4 -2
  38. package/.agents/scripts/lib/full-suite-lock.js +232 -6
  39. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  40. package/.agents/scripts/lib/git/sync-from-base.js +130 -13
  41. package/.agents/scripts/lib/observability/source-classifier.js +1 -0
  42. package/.agents/scripts/lib/orchestration/check-baselines/phases/compare.js +10 -2
  43. package/.agents/scripts/lib/orchestration/check-baselines/phases/refresh-ack.js +75 -15
  44. package/.agents/scripts/lib/orchestration/deliver-recover.js +82 -43
  45. package/.agents/scripts/lib/orchestration/dependency-candidates.js +8 -4
  46. package/.agents/scripts/lib/orchestration/epic-candidates.js +9 -4
  47. package/.agents/scripts/lib/orchestration/epic-container.js +66 -4
  48. package/.agents/scripts/lib/orchestration/epic-rollup.js +233 -84
  49. package/.agents/scripts/lib/orchestration/file-assumptions.js +218 -16
  50. package/.agents/scripts/lib/orchestration/git-cleanup/phases/branches.js +93 -7
  51. package/.agents/scripts/lib/orchestration/git-cleanup/phases/git-probes.js +22 -6
  52. package/.agents/scripts/lib/orchestration/git-cleanup/phases/parse-args.js +26 -5
  53. package/.agents/scripts/lib/orchestration/git-cleanup/phases/phase-drivers.js +13 -2
  54. package/.agents/scripts/lib/orchestration/git-cleanup/phases/render.js +35 -5
  55. package/.agents/scripts/lib/orchestration/merge-block-class.js +18 -3
  56. package/.agents/scripts/lib/orchestration/merge-poll.js +284 -40
  57. package/.agents/scripts/lib/orchestration/plan-persist/epic-adoption.js +49 -2
  58. package/.agents/scripts/lib/orchestration/plan-persist/epic-ops.js +43 -7
  59. package/.agents/scripts/lib/orchestration/plan-persist/run-plan-persist.js +24 -1
  60. package/.agents/scripts/lib/orchestration/plan-persist/story-ops.js +5 -0
  61. package/.agents/scripts/lib/orchestration/plan-persist/summary.js +3 -0
  62. package/.agents/scripts/lib/orchestration/plan-persist/supersede-ops.js +63 -0
  63. package/.agents/scripts/lib/orchestration/plan-persist/wave-serialisation.js +110 -0
  64. package/.agents/scripts/lib/orchestration/planning/memory-pool-advisory.js +130 -40
  65. package/.agents/scripts/lib/orchestration/resolve-stories.js +44 -1
  66. package/.agents/scripts/lib/orchestration/review-providers/native.js +31 -11
  67. package/.agents/scripts/lib/orchestration/review-providers/scoped-lint.js +27 -24
  68. package/.agents/scripts/lib/orchestration/run-epilogue.js +59 -38
  69. package/.agents/scripts/lib/orchestration/single-story-close/close-note.js +81 -0
  70. package/.agents/scripts/lib/orchestration/single-story-close/failed-terminal.js +40 -51
  71. package/.agents/scripts/lib/orchestration/single-story-close/phases/auto-merge.js +10 -2
  72. package/.agents/scripts/lib/orchestration/single-story-close/phases/base-sync.js +101 -0
  73. package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +351 -28
  74. package/.agents/scripts/lib/orchestration/single-story-close/phases/options.js +27 -6
  75. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +117 -22
  76. package/.agents/scripts/lib/orchestration/story-close/baseline-upward-writeback.js +94 -12
  77. package/.agents/scripts/lib/orchestration/story-close/format-autofix.js +6 -1
  78. package/.agents/scripts/lib/orchestration/ticket-validator.js +25 -14
  79. package/.agents/scripts/lib/orchestration/ticketing/bulk.js +30 -0
  80. package/.agents/scripts/lib/orchestration/verify-credit.js +37 -0
  81. package/.agents/scripts/lib/pinned-override-notes.js +41 -53
  82. package/.agents/scripts/lib/pinned-override-resolve.js +212 -0
  83. package/.agents/scripts/lib/qa/resolve-qa-contract.js +18 -0
  84. package/.agents/scripts/lib/single-story-sweep/sweep-lock.js +173 -9
  85. package/.agents/scripts/lib/skills/walk-skill-files.js +24 -7
  86. package/.agents/scripts/lib/test-temp.js +167 -30
  87. package/.agents/scripts/lib/validation-evidence.js +37 -0
  88. package/.agents/scripts/lib/wave-runner/footprint.js +167 -14
  89. package/.agents/scripts/lib/wave-runner/live-probe.js +7 -1
  90. package/.agents/scripts/lib/wave-runner/ready-set.js +1 -1
  91. package/.agents/scripts/merge-baseline.js +175 -21
  92. package/.agents/scripts/providers/github/errors.js +22 -1
  93. package/.agents/scripts/providers/github/issues.js +106 -1
  94. package/.agents/scripts/providers/github/sub-issue-add.js +18 -1
  95. package/.agents/scripts/providers/github.js +6 -0
  96. package/.agents/scripts/resolve-stories.js +44 -34
  97. package/.agents/scripts/single-story-close.js +5 -0
  98. package/.agents/scripts/stories-wave-tick.js +37 -13
  99. package/.agents/templates/docs/audit-sweep-runbook.md +41 -7
  100. package/.agents/workflows/audit-accessibility.md +16 -31
  101. package/.agents/workflows/audit-mobile.md +20 -37
  102. package/.agents/workflows/git-cleanup.md +17 -3
  103. package/.agents/workflows/helpers/audit-lens-core.md +45 -0
  104. package/.agents/workflows/helpers/deliver-digest.md +7 -6
  105. package/.agents/workflows/helpers/deliver-reference.md +35 -14
  106. package/.agents/workflows/helpers/deliver-story-reference.md +7 -4
  107. package/.agents/workflows/helpers/deliver-story.md +15 -12
  108. package/.agents/workflows/helpers/plan-reference.md +7 -0
  109. package/.agents/workflows/mandrel-plan.md +4 -7
  110. package/.agents/workflows/memory-consolidate.md +14 -9
  111. package/docs/CHANGELOG.md +27 -0
  112. package/lib/cli/registry.js +64 -21
  113. package/lib/cli/sync.js +27 -2
  114. package/package.json +7 -4
@@ -40,6 +40,13 @@ import { resolveStoryDispatchMode } from './complexity-gate.js';
40
40
  /** Labels/state that mean a blocker no longer gates its dependents. */
41
41
  const DONE_LABEL = 'agent::done';
42
42
 
43
+ /**
44
+ * The lifecycle-label prefix a deliverable Story carries. Any `agent::*` label
45
+ * will do — the resolver is not a state machine and does not care WHICH state a
46
+ * Story is in, only that it has been through the step that assigns one.
47
+ */
48
+ const AGENT_LABEL_PREFIX = 'agent::';
49
+
43
50
  /**
44
51
  * Module-private: `toStoryRecord` and `isSatisfiedBlocker` are its only
45
52
  * callers. The ancestor exported it with no external consumer, which is how
@@ -64,9 +71,12 @@ function normalizeIssueLabels(issue) {
64
71
  *
65
72
  * @param {object} issue
66
73
  * @param {number} [requestedId] The id the operator asked for, for error text.
74
+ * @param {{ allowUnlabelled?: boolean }} [options] `allowUnlabelled` waives the
75
+ * `agent::*` guard below — the deliberate escape hatch for delivering a Story
76
+ * whose state label is absent for a reason the operator knows about.
67
77
  * @returns {{ id, title, body, url, labels, state, assignees }}
68
78
  */
69
- export function toStoryRecord(issue, requestedId) {
79
+ export function toStoryRecord(issue, requestedId, { allowUnlabelled } = {}) {
70
80
  const id = Number(issue?.number ?? issue?.id ?? requestedId);
71
81
  if (!Number.isInteger(id) || id <= 0) {
72
82
  throw new Error(
@@ -88,6 +98,9 @@ export function toStoryRecord(issue, requestedId) {
88
98
  `v2 is Story-only — re-plan it as a v2 Story or finish it on a pre-v2 checkout.`,
89
99
  );
90
100
  }
101
+ // Last, so the two shape refusals above — not a Story at all, and a v1 body —
102
+ // keep naming their own remedy rather than being masked by a missing label.
103
+ assertDispatchable(id, labels, allowUnlabelled);
91
104
  return {
92
105
  id,
93
106
  title: String(issue?.title ?? ''),
@@ -106,6 +119,36 @@ export function toStoryRecord(issue, requestedId) {
106
119
  };
107
120
  }
108
121
 
122
+ /**
123
+ * Refuse a Story that has never been through planning.
124
+ *
125
+ * The audit sweep files Stories deliberately WITHOUT an `agent::*` label: their
126
+ * bodies are audit prose — a symptom and a recommendation — not a scoped change
127
+ * with acceptance criteria a worker can verify against, and the sweep's runbook
128
+ * says so. But `/mandrel-deliver` takes ids, and nothing downstream re-checked the
129
+ * label, so naming a freshly-filed audit Story dispatched a worker at an
130
+ * unenriched body: the run then either invented its own acceptance criteria or
131
+ * blocked several minutes in, having taken the Story's lease and flipped it to
132
+ * `agent::executing` on the way.
133
+ *
134
+ * The label is the cheap, honest signal that the enrich step ran — no state
135
+ * machine is consulted, only that SOME `agent::*` label exists.
136
+ *
137
+ * @param {number} id
138
+ * @param {string[]} labels
139
+ * @param {boolean} [allowUnlabelled]
140
+ */
141
+ function assertDispatchable(id, labels, allowUnlabelled) {
142
+ if (allowUnlabelled) return;
143
+ if (labels.some((l) => l.startsWith(AGENT_LABEL_PREFIX))) return;
144
+ throw new Error(
145
+ `[resolve-stories] Issue #${id} carries no "${AGENT_LABEL_PREFIX}*" label, so it has not been ` +
146
+ `through planning — an audit sweep files Stories without one on purpose (its runbook's ` +
147
+ `"Enrich before you deliver" step). Route it through /mandrel-plan first, which applies ` +
148
+ `agent::ready once the finding is a scoped slice. Pass --allow-unlabelled to deliver it as-is.`,
149
+ );
150
+ }
151
+
109
152
  /**
110
153
  * A blocker stops gating once its issue is closed or carries `agent::done`.
111
154
  *
@@ -372,11 +372,25 @@ export async function analyzeChangedFiles(
372
372
  /**
373
373
  * Pure: turn a lint summary into Finding(s). Lint errors collapse into a
374
374
  * single high-risk finding (the structured comment shows the count); lint
375
- * warnings collapse into a single suggestion. An `executionFailed` summary
376
- * produces **zero** findings (Story #4699): a runner that could not execute
377
- * is an operational degradation, not a code finding — the provider routes it
378
- * to friction telemetry instead so severity counts reflect code findings
379
- * only.
375
+ * warnings collapse into a single suggestion.
376
+ *
377
+ * **Findings come from the parsed counts, never from the execution flag**
378
+ * (Story #5282). `executionFailed` is the OR across the biome and markdownlint
379
+ * surfaces, so gating findings on it let *one* absent runner discard the
380
+ * *other* surface's real errors — and since the code surface's disk probe
381
+ * (#5193) degrades in every checkout without `node_modules/.bin/biome`, that
382
+ * was the default state of a consumer checkout: markdownlint errors reached
383
+ * neither the findings nor the severity tally while the outcome read clean
384
+ * apart from a degradation line.
385
+ *
386
+ * Story #4699's intent is preserved exactly, because it was never about the
387
+ * flag: a degradation is still not a `Finding` — it has no counts to report,
388
+ * so a surface that could not execute contributes `parsed: false` and zero
389
+ * counts and produces nothing here, while travelling on the degradation and
390
+ * friction-telemetry channels under its own name. A summary that explicitly
391
+ * reports `parsed: false` therefore yields no findings whatever its counts
392
+ * claim; a summary omitting `parsed` (an injected or pre-#4839 shape) is
393
+ * scored on its counts as before.
380
394
  *
381
395
  * @param {{ errors: number, warnings: number, parsed?: boolean, skipped?: boolean, mode?: string, executionFailed?: boolean, evidenceSkipped?: boolean }} lintSummary
382
396
  * @returns {Finding[]}
@@ -385,7 +399,7 @@ export function buildLintFindings(lintSummary) {
385
399
  if (lintSummary.mode === 'off') return [];
386
400
  if (lintSummary.evidenceSkipped) return [];
387
401
  if (lintSummary.skipped) return [];
388
- if (lintSummary.executionFailed) return [];
402
+ if (lintSummary.parsed === false) return [];
389
403
  const findings = [];
390
404
  if (lintSummary.errors > 0) {
391
405
  findings.push({
@@ -428,6 +442,7 @@ async function runLintPhase({
428
442
  mode: 'off',
429
443
  executionFailed: false,
430
444
  degradations: [],
445
+ surfaces: [],
431
446
  };
432
447
  }
433
448
  logger?.info?.(
@@ -592,15 +607,19 @@ export function createNativeProvider(deps = {}) {
592
607
  // Story #4839 — telemetry alone left the review's own verdict unable to
593
608
  // distinguish "lint ran and found nothing" from "lint never ran", so
594
609
  // the same degradation is also recorded on the outcome channel. It is
595
- // still never a `Finding`: the friction emission below is unchanged and
596
- // severity counts remain code-findings-only.
610
+ // still never a `Finding`: the friction emission below is unchanged.
611
+ //
612
+ // Story #5282 — this branch is about the degraded surface only. A
613
+ // sibling surface that *did* run still contributes its parsed counts
614
+ // to `buildLintFindings` below, so a degradation here no longer
615
+ // suppresses the other surface's errors.
597
616
  recordedDegradations = buildLintDegradations(lintSummary);
598
617
  logger?.warn?.(
599
618
  `[native-review] Lint runner could not execute (${recordedDegradations
600
619
  .map((d) => `${d.surface}: ${d.reason}`)
601
620
  .join(
602
621
  '; ',
603
- )}) — reported as a degraded gate on the review outcome and recorded as friction telemetry; no finding emitted. Verify with the canonical \`npm run lint\` before merging.`,
622
+ )}) — reported as a degraded gate on the review outcome and recorded as friction telemetry; the degradation itself is never a finding, and any surface that did run still reports its own errors. Verify with the canonical \`npm run lint\` before merging.`,
604
623
  );
605
624
  try {
606
625
  await emitToolDegradationFn({
@@ -622,8 +641,9 @@ export function createNativeProvider(deps = {}) {
622
641
 
623
642
  // Canonical ordering: critical (maintainability) first, then high
624
643
  // (lint errors), then medium (size/volume warnings), then suggestion
625
- // (lint warnings). An execution failure contributes to none of these
626
- // tiers — it travels on the degradation channel. The renderer
644
+ // (lint warnings). An execution failure contributes no counts of its
645
+ // own to these tiers — it travels on the degradation channel — but it
646
+ // no longer suppresses a sibling surface's. The renderer
627
647
  // re-bucketizes by severity tier, so this order only matters for
628
648
  // stability of fixture outputs.
629
649
  return [
@@ -289,36 +289,38 @@ export function parseLintOutput(result) {
289
289
  * one degradation record naming itself. Merging *summaries* rather than raw
290
290
  * output is what stops one runner's failure from becoming the other's verdict.
291
291
  *
292
+ * The OR is deliberately lossy — it answers "did any surface fail?", which is
293
+ * the only question the degradation channel asks. Story #5282 added the
294
+ * `surfaces[]` rows so a consumer can ask the *other* questions the OR cannot
295
+ * answer: which surface the merged counts came from, and — via each row's
296
+ * `parsed` and `executionFailed` — whether an absent biome or a biome run that
297
+ * simply reported nothing is behind a zero. Consumers that read the flat
298
+ * counts are unaffected; the rows are additive.
299
+ *
292
300
  * @param {Array<{ surface: string, summary: ReturnType<typeof parseLintOutput> }>} surfaces
293
301
  */
294
302
  function mergeSurfaceSummaries(surfaces) {
295
- let errors = 0;
296
- let warnings = 0;
297
- let parsed = false;
298
- let executionFailed = false;
299
- const degradations = [];
300
-
301
- for (const { surface, summary } of surfaces) {
302
- errors += summary.errors;
303
- warnings += summary.warnings;
304
- if (summary.parsed) parsed = true;
305
- if (summary.executionFailed) {
306
- executionFailed = true;
307
- degradations.push({
308
- surface,
309
- reason: summary.reason ?? DEGRADATION_REASONS.UNPARSEABLE_OUTPUT,
310
- });
311
- }
312
- }
303
+ const rows = surfaces.map(({ surface, summary }) => ({
304
+ surface,
305
+ parsed: summary.parsed,
306
+ errors: summary.errors,
307
+ warnings: summary.warnings,
308
+ executionFailed: summary.executionFailed,
309
+ reason: summary.reason ?? DEGRADATION_REASONS.UNPARSEABLE_OUTPUT,
310
+ }));
311
+ const total = (field) => rows.reduce((sum, row) => sum + row[field], 0);
313
312
 
314
313
  return {
315
- errors,
316
- warnings,
317
- parsed,
318
- executionFailed,
314
+ errors: total('errors'),
315
+ warnings: total('warnings'),
316
+ parsed: rows.some((row) => row.parsed),
317
+ executionFailed: rows.some((row) => row.executionFailed),
319
318
  skipped: false,
320
319
  mode: 'changed-only',
321
- degradations,
320
+ degradations: rows
321
+ .filter((row) => row.executionFailed)
322
+ .map(({ surface, reason }) => ({ surface, reason })),
323
+ surfaces: rows.map(({ reason, ...row }) => row),
322
324
  };
323
325
  }
324
326
 
@@ -329,7 +331,7 @@ function mergeSurfaceSummaries(surfaces) {
329
331
  * @param {string} cwd
330
332
  * @param {typeof spawnLintRunner} [runnerFn]
331
333
  * @param {{ existsFn?: (p: string) => boolean }} [deps] Test seam for runner resolution.
332
- * @returns {{ errors: number, warnings: number, parsed: boolean, skipped: boolean, mode: 'changed-only'|'off', executionFailed: boolean, degradations: Array<{ surface: string, reason: string }> }}
334
+ * @returns {{ errors: number, warnings: number, parsed: boolean, skipped: boolean, mode: 'changed-only'|'off', executionFailed: boolean, degradations: Array<{ surface: string, reason: string }>, surfaces: Array<{ surface: string, parsed: boolean, errors: number, warnings: number, executionFailed: boolean }> }}
333
335
  */
334
336
  export function runScopedLint(
335
337
  changedFiles,
@@ -348,6 +350,7 @@ export function runScopedLint(
348
350
  mode: 'changed-only',
349
351
  executionFailed: false,
350
352
  degradations: [],
353
+ surfaces: [],
351
354
  };
352
355
  }
353
356
 
@@ -7,8 +7,8 @@
7
7
  * 2. Rolls up friction follow-ups across every Story in the run and
8
8
  * files/posts them on the primary Story.
9
9
  * 3. Checks sibling Spec/acceptance coherence across Story bodies.
10
- * 4. Reports the container Epics whose children all landed (Story #5139),
11
- * delegating the derivation and the close to `epic-rollup.js`.
10
+ * 4. Reports what the per-Story land tails left the run's container Epics
11
+ * in — closed, or still open (Story #5139; read-only since #5280).
12
12
  *
13
13
  * There is no inert planner-only path: `planRunEpilogue` enumerates steps
14
14
  * and `runPlanRunEpilogue` executes them. Single-Story runs skip the
@@ -21,7 +21,7 @@ import { selectAudits } from '../audit-suite/index.js';
21
21
  import { graduateRetroProposals } from '../feedback-loop/retro-proposals-graduator.js';
22
22
  import { gitSpawn } from '../git-utils.js';
23
23
  import { Logger } from '../Logger.js';
24
- import { rollUpEpicForStory } from './epic-rollup.js';
24
+ import { isEpicTicket } from './epic-container.js';
25
25
  import { composeRoutedProposals } from './retro-proposals.js';
26
26
  import {
27
27
  assessRollupOutcome,
@@ -44,54 +44,51 @@ export const RUN_EPILOGUE_STEP_KINDS = Object.freeze([
44
44
  ]);
45
45
 
46
46
  /**
47
- * Close every container Epic whose children all landed in this run.
47
+ * Report which container Epics this run's Stories left closed, and which are
48
+ * still open.
48
49
  *
49
- * Delegates to `epic-rollup.js` (Story #5205) rather than deriving anything
50
- * itself. That module owns the child→parent scan, the body-checklist-union-
51
- * native-sub-issue child read, and the one-way closure rule, and it is also
52
- * invoked from the per-Story land tail — which is what closes a container
53
- * whose last open child was a single Story, a case this epilogue never
54
- * reaches because a one-Story run reports `applicable: false`.
50
+ * **It derives nothing and writes nothing.** It used to: it walked every Story
51
+ * and re-ran the full rollup, closing containers itself. That made sense while
52
+ * the rollup only fired on the edges someone had wired, and the epilogue was
53
+ * the backstop for the ones that were missed. Every child state change is now
54
+ * an edge — init, post-land, the supersede close — so by the time the last
55
+ * Story of a run has landed, its container has already been derived from a
56
+ * complete child set by that Story's own land tail. Re-deriving here would ask
57
+ * the same question a second time and answer it identically, at the cost of a
58
+ * full re-read per Story and a second writer on the same issue.
55
59
  *
56
- * The step survives for its report: this is the surface an operator reads to
57
- * see which containers a multi-Story run closed and which are still pending.
60
+ * What survives is the report, which is why the step exists at all: one place
61
+ * an operator reads to see what a multi-Story run did to its containers. A
62
+ * pending Epic here is a real signal — it means a land tail's rollup declined
63
+ * to close, and the tail's own outcome says why.
58
64
  *
59
- * Non-fatal throughout: the epilogue is a reporting tail, and a container
60
- * left open costs tidiness, not correctness.
65
+ * Non-fatal throughout: a container it cannot read is simply not reported.
61
66
  *
62
- * @param {{ stories: string[], provider: object, config?: object }} opts
67
+ * @param {{ stories: string[], provider: object }} opts
63
68
  * @returns {Promise<{ kind: string, closed: number[], pending: number[] }>}
64
69
  */
65
- async function executeEpicClose({ stories, provider, config }) {
70
+ async function executeEpicClose({ stories, provider }) {
66
71
  const closed = new Set();
67
72
  const pending = new Set();
68
- // Siblings share a container, so an Epic resolved by one Story's rollup is
69
- // withheld from the next one's — otherwise the second Story would re-derive
70
- // and re-close what the first already closed.
73
+ // Siblings share a container: resolve each distinct Epic once, however many
74
+ // of the run's Stories point at it.
71
75
  const seen = new Set();
72
76
 
73
- // `rollUpEpicForStory` never throws and always returns the full envelope,
74
- // so its three lists are read directly — a `?? []` guard here would be an
75
- // unreachable branch asserting a contract the module already keeps.
76
77
  for (const raw of stories) {
77
78
  const storyId = Number(raw);
78
79
  if (!Number.isInteger(storyId) || storyId <= 0) continue;
79
- const outcome = await rollUpEpicForStory({
80
- storyId,
81
- provider,
82
- config,
83
- skipEpicIds: seen,
84
- });
85
- for (const epic of outcome.epics) seen.add(epic.epicId);
86
- for (const epicId of outcome.closed) closed.add(epicId);
87
- for (const epicId of outcome.pending) pending.add(epicId);
80
+ const epic = await readContainerFor({ storyId, provider });
81
+ if (!epic) continue;
82
+ const epicId = Number(epic.id);
83
+ if (!Number.isInteger(epicId) || seen.has(epicId)) continue;
84
+ seen.add(epicId);
85
+ if (String(epic.state ?? '').toLowerCase() === 'closed') {
86
+ closed.add(epicId);
87
+ } else {
88
+ pending.add(epicId);
89
+ }
88
90
  }
89
91
 
90
- // An Epic this run closed can also have been reported pending by an
91
- // earlier Story's rollup, when a sibling had not landed yet. The close is
92
- // the later, truer answer.
93
- for (const epicId of closed) pending.delete(epicId);
94
-
95
92
  return {
96
93
  kind: 'epic-close',
97
94
  closed: [...closed],
@@ -99,6 +96,30 @@ async function executeEpicClose({ stories, provider, config }) {
99
96
  };
100
97
  }
101
98
 
99
+ /**
100
+ * Read one Story's container Epic, or null.
101
+ *
102
+ * One request per Story via the declared parent port. Degrades to null on any
103
+ * failure and on a provider without the port — this is a report, and a
104
+ * container it could not read is better omitted than guessed at.
105
+ *
106
+ * @param {{ storyId: number, provider: object }} opts
107
+ * @returns {Promise<object|null>}
108
+ */
109
+ async function readContainerFor({ storyId, provider }) {
110
+ if (typeof provider?.getParentIssue !== 'function') return null;
111
+ try {
112
+ const parent = await provider.getParentIssue(storyId);
113
+ return parent && isEpicTicket(parent) ? parent : null;
114
+ } catch (err) {
115
+ Logger.warn(
116
+ `[run-epilogue] could not read the container for Story #${storyId} ` +
117
+ `(${err?.message ?? err}); omitting it from the Epic report.`,
118
+ );
119
+ return null;
120
+ }
121
+ }
122
+
102
123
  /**
103
124
  * @param {string|number|{ id?: string|number, slug?: string }} entry
104
125
  * @returns {string|null}
@@ -185,7 +206,7 @@ export function planRunEpilogue({ planRunId, stories } = {}) {
185
206
  },
186
207
  {
187
208
  kind: 'epic-close',
188
- description: `Close any container Epic whose children all landed in run ${effectiveRunId}`,
209
+ description: `Report the container Epic state the land tails of run ${effectiveRunId} left behind`,
189
210
  stories: ids,
190
211
  },
191
212
  ];
@@ -892,7 +913,7 @@ export async function runPlanRunEpilogue({
892
913
  );
893
914
  } else if (step.kind === 'epic-close') {
894
915
  results.push(
895
- await executeEpicClose({ stories: plan.stories, provider, config }),
916
+ await executeEpicClose({ stories: plan.stories, provider }),
896
917
  );
897
918
  }
898
919
  } catch (err) {
@@ -0,0 +1,81 @@
1
+ /**
2
+ * close-note.js — the human-readable `note` on close's result record
3
+ * (Story #5266).
4
+ *
5
+ * ## Why this is its own module
6
+ *
7
+ * The note used to branch on `waitedForMerge` — whether close *waited* — and
8
+ * not on `merged`. A bounded wait that expired with the PR still open
9
+ * therefore wrote "Close-and-land: PR merge confirmed … the issue closed"
10
+ * into `story-close-result-<id>.log` **beside `merged: false`**, while the
11
+ * schema-validated terminal envelope correctly reported
12
+ * `status: pending, phase: confirm-merge`. That log is what close's summary
13
+ * line points the operator at, so the contradiction is what gets read first:
14
+ * a merge that never happened, reported as confirmed.
15
+ *
16
+ * The invariant this module exists to hold:
17
+ *
18
+ * > **No note may assert a state the same object denies.**
19
+ *
20
+ * Every branch below is therefore derived from the result's OWN
21
+ * `merged` / `directMerged` / `autoMergeEnabled` / `landCompleted` fields —
22
+ * the ones the note ships next to — so no input can produce a note that
23
+ * contradicts them. Story #5279 added the fourth, because `merged: true`
24
+ * used to imply the flip, the issue close and the post-land tail: a direct
25
+ * squash-merge under `--no-wait-merge`, and a merge whose `agent::done` write
26
+ * failed, now reach it with none of the three, and reusing the confirmed
27
+ * wording for them would reintroduce the defect above one field over. The
28
+ * unmerged branches deliberately claim nothing about the Story's label state
29
+ * either: close's ending may be `pending` OR `blocked` with the same
30
+ * `merged: false`, and the terminal envelope is the authority on which. They
31
+ * also avoid the merge-completion vocabulary entirely — no `agent::done`, no
32
+ * "the merge confirms" — so that a reader skimming for those words cannot
33
+ * take a next-step instruction for a report of what happened.
34
+ */
35
+
36
+ /**
37
+ * The one line the note is not allowed to get wrong.
38
+ *
39
+ * @param {{ merged?: boolean, directMerged?: boolean,
40
+ * autoMergeEnabled?: boolean, landCompleted?: boolean }} result
41
+ * `landCompleted` — did THIS run flip `agent::done`, close the issue and
42
+ * run the post-land tail? Defaults to `merged`, the pre-#5279 equivalence.
43
+ * @returns {string}
44
+ */
45
+ export function deriveCloseNote({
46
+ merged = false,
47
+ directMerged = false,
48
+ autoMergeEnabled = false,
49
+ landCompleted = merged,
50
+ } = {}) {
51
+ if (merged) {
52
+ const how = directMerged
53
+ ? 'PR merge confirmed by a DIRECT squash-merge (native auto-merge was ' +
54
+ 'unavailable on this repository).'
55
+ : 'PR merge confirmed.';
56
+ return landCompleted
57
+ ? `Close-and-land: ${how} Story flipped agent::closing → agent::done, ` +
58
+ 'the issue closed (confirmStoryMerged), and the post-land tail ran.'
59
+ : `Close-and-land: ${how} This close did NOT finish the land: ` +
60
+ 'agent::done was not flipped and the post-land tail did not run. ' +
61
+ 'Finish it with single-story-confirm-merge.js, idempotent against an ' +
62
+ 'already-merged PR. The terminal envelope is authoritative.';
63
+ }
64
+ if (autoMergeEnabled) {
65
+ return (
66
+ 'PR open against baseBranch and NOT merged; auto-merge is armed. ' +
67
+ 'GitHub will squash-merge it once the required checks pass — resume ' +
68
+ 'with single-story-confirm-merge.js then, to finish the land and ' +
69
+ 'release the lease this close is still holding. The terminal envelope ' +
70
+ '(status / phase / blocked) is authoritative for what happened here.'
71
+ );
72
+ }
73
+ return (
74
+ 'PR open against baseBranch and NOT merged; auto-merge was not armed ' +
75
+ '(see autoMergeReason). The operator owns the land: merge via the GitHub ' +
76
+ 'UI, then resume with single-story-confirm-merge.js to finish the land ' +
77
+ 'and release the lease this close is still holding. The terminal ' +
78
+ 'envelope (status / phase / blocked) is authoritative for what happened ' +
79
+ 'here.'
80
+ );
81
+ }
@@ -50,8 +50,11 @@ const GATE_PHASES = Object.freeze([
50
50
  ]);
51
51
 
52
52
  /**
53
- * The names the split baselines gate registers under, mirrored from
54
- * `BASELINES_GATE_NAMES` in `lib/close-validation/gates.js` (Story #5172).
53
+ * The names the unified baselines gate can register under, mirrored from
54
+ * `BASELINES_GATE_NAMES` in `lib/close-validation/gates.js` (Story #5172) —
55
+ * all three, matching the projection the SUCCESS path applies in
56
+ * `runner.js#baselinesEnvelopeGates`, so the two endings of one run cannot
57
+ * key the same gate differently.
55
58
  *
56
59
  * Deliberately a local copy rather than an import: several close suites
57
60
  * replace that module wholesale via `t.mock.module`, and a named import here
@@ -61,48 +64,37 @@ const GATE_PHASES = Object.freeze([
61
64
  * against each other so the copy cannot drift.
62
65
  */
63
66
  const BASELINES_ENTRY_NAMES = Object.freeze([
67
+ 'check-baselines',
64
68
  'check-baselines-independent',
65
69
  'check-baselines-coverage',
66
70
  ]);
67
71
 
68
72
  /**
69
- * Outcome for each split baselines entry on a run that died at `phase`.
73
+ * Project the baselines entries out of the per-gate outcomes THIS run
74
+ * observed (Story #5279).
70
75
  *
71
- * The two entries sit in ONE pipeline phase, so the phase walk alone cannot
72
- * separate them — `failedGate` (tagged onto the error by the close-validation
73
- * phase) is what names the entry that actually broke. Rules, in the module's
74
- * house style of never claiming a pass it cannot prove:
75
- * - validation skipped, or the run died before reaching it → both `skipped`.
76
- * - the run cleared validation entirely → both `passed`.
77
- * - the run died IN validation on the coverage-independent entry → that one
78
- * `failed`, the coverage one `skipped` (it runs behind `coverage-capture`,
79
- * which the failure pre-empted).
80
- * - died on the coverage-consuming entry → that one `failed`, and the
81
- * independent one `passed`: it is in the parallel partition that must go
82
- * green before any serial gate starts.
83
- * - died in validation on some other gate → both `skipped`; which of them
84
- * had run is not knowable from the phase alone.
76
+ * This used to RECONSTRUCT them: it inferred, from the phase the run died in
77
+ * plus the name of the failing gate, what the two split entries "must have"
78
+ * done — and it did so over a hardcoded pair, so every failed close reported
79
+ * both split names whether or not the run had ever registered them. A repo
80
+ * whose config resolves to the unsplit `check-baselines`, or to only one half
81
+ * of the pair, got envelope keys for gates that did not exist; a run that
82
+ * died at `init` got them too, reported as `skipped`, which reads as "the
83
+ * gate was turned off" rather than "there was no such gate".
85
84
  *
86
- * @param {string} phase
87
- * @param {{ skipValidation?: boolean, failedGate?: string|null }} args
85
+ * Reporting instead of reconstructing removes the whole class: an outcome
86
+ * appears only for a gate the run actually observed, and the runner tags
87
+ * exactly that set onto the error (`err.closeGates`).
88
+ *
89
+ * @param {Record<string, string>|null|undefined} observedGates
88
90
  * @returns {Record<string, 'passed'|'failed'|'skipped'>}
89
91
  */
90
- function baselinesGatesForFailedPhase(phase, { skipValidation, failedGate }) {
91
- const [independent, coverage] = BASELINES_ENTRY_NAMES;
92
- const both = (outcome) => ({ [independent]: outcome, [coverage]: outcome });
93
- const failedAt = PHASE_ORDER.indexOf(phase);
94
- const validationAt = PHASE_ORDER.indexOf('close-validation');
95
- if (skipValidation || failedAt < 0 || failedAt < validationAt) {
96
- return both('skipped');
97
- }
98
- if (failedAt > validationAt) return both('passed');
99
- if (failedGate === independent) {
100
- return { [independent]: 'failed', [coverage]: 'skipped' };
101
- }
102
- if (failedGate === coverage) {
103
- return { [independent]: 'passed', [coverage]: 'failed' };
92
+ function baselinesGatesObserved(observedGates) {
93
+ const out = {};
94
+ for (const [name, outcome] of Object.entries(observedGates ?? {})) {
95
+ if (BASELINES_ENTRY_NAMES.includes(name)) out[name] = outcome;
104
96
  }
105
- return both('skipped');
97
+ return out;
106
98
  }
107
99
 
108
100
  /**
@@ -119,14 +111,17 @@ function baselinesGatesForFailedPhase(phase, { skipValidation, failedGate }) {
119
111
  * turned off via `--skip-validation` / `--skip-sync` is `skipped` too (it did
120
112
  * not pass — it never ran).
121
113
  *
122
- * Story #5172 — the reported set also carries the two split baselines
123
- * entries under their own names, so a failed close says WHICH half of the
124
- * baselines gate breached instead of a single generic verdict.
114
+ * Story #5172 — the reported set also carries the baselines entries under
115
+ * their own names, so a failed close says WHICH half of the baselines gate
116
+ * breached instead of a single generic verdict. Story #5279 — those names
117
+ * are REPORTED from `observedGates`, never reconstructed, so only a gate the
118
+ * run registered can appear.
125
119
  *
126
120
  * @param {string} phase The phase the run died in.
127
- * @param {{ skipValidation?: boolean, skipSync?: boolean, failedGate?: string|null }} args
128
- * Parsed CLI args, plus the gate name tagged onto the error by the
129
- * close-validation phase.
121
+ * @param {{ skipValidation?: boolean, skipSync?: boolean,
122
+ * observedGates?: Record<string, string>|null }} args
123
+ * Parsed CLI args, plus the per-gate outcomes the runner tagged onto the
124
+ * error.
130
125
  * @returns {Record<string, 'passed'|'failed'|'skipped'>}
131
126
  */
132
127
  export function gatesForFailedPhase(phase, args = {}) {
@@ -139,13 +134,7 @@ export function gatesForFailedPhase(phase, args = {}) {
139
134
  else if (failedAt < 0 || at > failedAt) gates[gate] = 'skipped';
140
135
  else gates[gate] = skipped[gate] ? 'skipped' : 'passed';
141
136
  }
142
- return {
143
- ...gates,
144
- ...baselinesGatesForFailedPhase(phase, {
145
- skipValidation: args.skipValidation,
146
- failedGate: args.failedGate ?? null,
147
- }),
148
- };
137
+ return { ...gates, ...baselinesGatesObserved(args.observedGates) };
149
138
  }
150
139
 
151
140
  /**
@@ -163,9 +152,9 @@ export function gatesForFailedPhase(phase, args = {}) {
163
152
  * holding the script had been reaped mid-run. On failure this returns null
164
153
  * and the caller rethrows the original.
165
154
  *
166
- * `err.closeGate` — tagged by the close-validation phase — names the gate that
167
- * died inside that phase, which is what lets the reported gates separate the
168
- * two split baselines entries (Story #5172).
155
+ * `err.closeGates` — tagged by the runner — carries the per-gate outcomes the
156
+ * run observed, which is what lets the reported gates name the baselines
157
+ * entries that actually ran (Story #5172 / #5279) instead of a hardcoded pair.
169
158
  *
170
159
  * @param {unknown} err
171
160
  * @param {{ storyId?: string|number, skipValidation?: boolean, skipSync?: boolean }} args
@@ -186,7 +175,7 @@ export function failedTerminalFor(err, args = {}) {
186
175
  phase,
187
176
  gates: gatesForFailedPhase(phase, {
188
177
  ...args,
189
- failedGate: err?.closeGate ?? null,
178
+ observedGates: err?.closeGates ?? null,
190
179
  }),
191
180
  failure: { reason: String(err?.message ?? err) },
192
181
  nextCommand: NEXT_COMMANDS.recover(storyId),
@@ -66,7 +66,7 @@ import { resolveAutoMergeArmCwd } from '../../auto-merge-cwd.js';
66
66
  import {
67
67
  advisoryCheckFailedBlocksArm,
68
68
  deriveRedHeadRuns,
69
- formatAdvisoryGateReason,
69
+ resolveAdvisoryGateVerdict,
70
70
  selectBlockingRedRuns,
71
71
  } from '../../merge-poll.js';
72
72
 
@@ -430,10 +430,17 @@ async function evaluateAdvisoryGate({
430
430
  probe.redHeadRuns,
431
431
  advisoryAllowlist,
432
432
  );
433
+ // Story #5266 — the class travels with the reason. This pre-arm gate
434
+ // classifies on the text the ROLLUP carried (a legacy StatusContext's
435
+ // `description`); the merge wait, which owns the common case, additionally
436
+ // reads the check-run output. Either way a run whose failure cannot be read
437
+ // as "never finished" keeps the `advisory-gate-red` verdict.
438
+ const verdict = resolveAdvisoryGateVerdict({ blockingRuns });
433
439
  return {
434
440
  blocked: true,
435
441
  blockingRuns,
436
- reason: formatAdvisoryGateReason(blockingRuns),
442
+ blockClass: verdict.blockClass,
443
+ reason: verdict.reason,
437
444
  };
438
445
  }
439
446
 
@@ -522,6 +529,7 @@ export async function runAutoMergePhase({
522
529
  autoMergeReason: 'advisory-gate-red',
523
530
  advisoryGate: {
524
531
  blockingRuns: advisory.blockingRuns,
532
+ blockClass: advisory.blockClass,
525
533
  reason: advisory.reason,
526
534
  },
527
535
  };