mandrel 2.56.0 → 2.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/.agents/agents/plan-critic.md +13 -18
  2. package/.agents/agents/story-worker.md +25 -34
  3. package/.agents/docs/agentrc-reference.json +0 -30
  4. package/.agents/docs/configuration.md +8 -28
  5. package/.agents/docs/execution-reference.md +5 -5
  6. package/.agents/docs/quality-gates.md +8 -7
  7. package/.agents/instructions.md +9 -10
  8. package/.agents/schemas/agentrc.schema.json +9 -185
  9. package/.agents/schemas/story-deliver-terminal.schema.json +1 -1
  10. package/.agents/scripts/acceptance-eval.js +107 -17
  11. package/.agents/scripts/ceremony-derive.js +191 -0
  12. package/.agents/scripts/check-context-budget.js +28 -33
  13. package/.agents/scripts/check-cyclomatic.js +4 -3
  14. package/.agents/scripts/deliver-light.js +31 -94
  15. package/.agents/scripts/lib/audit-suite/checklist-threading.js +15 -2
  16. package/.agents/scripts/lib/baselines/coverage-updater-cli.js +110 -0
  17. package/.agents/scripts/lib/baselines/crap-preview-scan.js +25 -0
  18. package/.agents/scripts/lib/baselines/crap-updater-cli.js +223 -0
  19. package/.agents/scripts/lib/bdd-scenario-budget.js +21 -3
  20. package/.agents/scripts/lib/bootstrap/quality-bootstrap.js +0 -1
  21. package/.agents/scripts/lib/close-validation/gates.js +52 -1
  22. package/.agents/scripts/lib/config/acceptance-eval.js +25 -57
  23. package/.agents/scripts/lib/config/delivery-routing.js +7 -33
  24. package/.agents/scripts/lib/config/explain.js +0 -19
  25. package/.agents/scripts/lib/config/limits.js +18 -78
  26. package/.agents/scripts/lib/config/quality.js +6 -3
  27. package/.agents/scripts/lib/config/runners.js +3 -2
  28. package/.agents/scripts/lib/config-settings-schema-delivery.js +15 -68
  29. package/.agents/scripts/lib/config-settings-schema-quality.js +0 -14
  30. package/.agents/scripts/lib/config-settings-schema.js +16 -143
  31. package/.agents/scripts/lib/crap-engine.js +35 -4
  32. package/.agents/scripts/lib/crap-utils.js +17 -1
  33. package/.agents/scripts/lib/cyclomatic-ceiling.js +19 -7
  34. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  35. package/.agents/scripts/lib/observability/runtime-friction.js +1 -1
  36. package/.agents/scripts/lib/observability/source-classifier.js +1 -0
  37. package/.agents/scripts/lib/orchestration/acceptance-eval-decision.js +5 -4
  38. package/.agents/scripts/lib/orchestration/ceremony-routing.js +19 -73
  39. package/.agents/scripts/lib/orchestration/complexity-gate.js +46 -212
  40. package/.agents/scripts/lib/orchestration/file-assumptions.js +32 -17
  41. package/.agents/scripts/lib/orchestration/light-escalation.js +3 -3
  42. package/.agents/scripts/lib/orchestration/light-suitability.js +66 -233
  43. package/.agents/scripts/lib/orchestration/plan-context.js +181 -387
  44. package/.agents/scripts/lib/orchestration/plan-critic-conditions.js +42 -153
  45. package/.agents/scripts/lib/orchestration/plan-critics-evaluate.js +14 -70
  46. package/.agents/scripts/lib/orchestration/plan-persist/changes-repair.js +300 -0
  47. package/.agents/scripts/lib/orchestration/plan-persist/persist-helpers.js +131 -168
  48. package/.agents/scripts/lib/orchestration/plan-persist/run-plan-persist.js +118 -297
  49. package/.agents/scripts/lib/orchestration/plan-persist/soft-findings.js +55 -0
  50. package/.agents/scripts/lib/orchestration/plan-persist/story-ops.js +16 -65
  51. package/.agents/scripts/lib/orchestration/plan-persist/wave-serialisation.js +22 -35
  52. package/.agents/scripts/lib/orchestration/plan-text-hygiene.js +30 -139
  53. package/.agents/scripts/lib/orchestration/planning/memory-pool-advisory.js +61 -223
  54. package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +5 -0
  55. package/.agents/scripts/lib/orchestration/single-story-close/phases/pre-gate-steps.js +46 -16
  56. package/.agents/scripts/lib/orchestration/story-close/context-budget-writeback.js +213 -0
  57. package/.agents/scripts/lib/orchestration/task-body-validator.js +10 -63
  58. package/.agents/scripts/lib/orchestration/ticket-validator-conflicts.js +33 -539
  59. package/.agents/scripts/lib/orchestration/ticket-validator-sizing.js +21 -414
  60. package/.agents/scripts/lib/orchestration/ticket-validator.js +54 -118
  61. package/.agents/scripts/lib/orchestration/verify-credit.js +69 -24
  62. package/.agents/scripts/lib/story-body/body-format-lints.js +15 -85
  63. package/.agents/scripts/lib/story-body/story-body.js +17 -237
  64. package/.agents/scripts/lib/templates/decomposer-prompts.js +84 -121
  65. package/.agents/scripts/lib/test-isolate/cli-options.js +93 -0
  66. package/.agents/scripts/lib/test-isolate/progress-log.js +45 -0
  67. package/.agents/scripts/lib/test-isolate/render-report.js +97 -0
  68. package/.agents/scripts/lib/test-isolate/run-isolate.js +87 -0
  69. package/.agents/scripts/lib/test-run-credit.js +266 -0
  70. package/.agents/scripts/lib/wave-runner/footprint.js +48 -358
  71. package/.agents/scripts/lib/wave-runner/ready-set.js +6 -5
  72. package/.agents/scripts/lib/workers/crap-worker.js +32 -41
  73. package/.agents/scripts/plan-context.js +7 -9
  74. package/.agents/scripts/plan-critics.js +28 -54
  75. package/.agents/scripts/plan-persist.js +25 -68
  76. package/.agents/scripts/quality-preview.js +51 -0
  77. package/.agents/scripts/run-tests.js +12 -0
  78. package/.agents/scripts/stories-wave-tick.js +23 -45
  79. package/.agents/scripts/test-isolate.js +13 -180
  80. package/.agents/scripts/update-coverage-baseline.js +25 -70
  81. package/.agents/scripts/update-crap-baseline.js +19 -123
  82. package/.agents/skills/core/scope-triage/SKILL.md +3 -3
  83. package/.agents/workflows/audit-clean-code.md +4 -3
  84. package/.agents/workflows/helpers/acceptance-self-eval.md +41 -41
  85. package/.agents/workflows/helpers/code-quality-guardrails.md +4 -4
  86. package/.agents/workflows/helpers/code-review.md +2 -3
  87. package/.agents/workflows/helpers/deliver-digest.md +41 -57
  88. package/.agents/workflows/helpers/deliver-light.md +40 -105
  89. package/.agents/workflows/helpers/deliver-reference.md +1 -1
  90. package/.agents/workflows/helpers/deliver-story-reference.md +37 -58
  91. package/.agents/workflows/helpers/deliver-story.md +9 -13
  92. package/.agents/workflows/helpers/plan-reference.md +132 -219
  93. package/.agents/workflows/mandrel-plan.md +27 -40
  94. package/.agents/workflows/memory-consolidate.md +9 -13
  95. package/docs/CHANGELOG.md +23 -0
  96. package/lib/migrations/index.js +4 -0
  97. package/lib/migrations/steps/2.57.0-retire-delivery-limit-knobs.js +45 -0
  98. package/lib/migrations/steps/2.57.0-retire-planning-limit-knobs.js +59 -0
  99. package/package.json +1 -1
  100. package/.agents/scripts/lib/framework-version.js +0 -39
  101. package/.agents/scripts/lib/orchestration/consolidation-precondition.js +0 -223
  102. package/.agents/scripts/lib/orchestration/plan-persist/fan-out-gate.js +0 -97
  103. package/.agents/scripts/lib/orchestration/planning/decomposer-context.js +0 -26
  104. package/.agents/scripts/lib/orchestration/spec-budget.js +0 -89
  105. package/.agents/scripts/lib/orchestration/spec-spill.js +0 -74
  106. package/.agents/scripts/lib/orchestration/verify-tier-repair.js +0 -107
@@ -1,71 +1,57 @@
1
- import { LIMITS_DEFAULTS } from '../config/limits.js';
2
1
  import {
3
2
  AUTHORING_ALTITUDE_GUIDANCE,
4
- DEFAULT_MODEL_CAPACITY,
5
3
  DELIVERABLE_GRANULARITY_GUIDANCE,
6
- resolveCapacityCeilings,
7
4
  } from '../orchestration/ticket-validator-sizing.js';
8
5
  import { BODY_FORMAT_LINTS } from '../story-body/body-format-lints.js';
9
6
 
10
7
  /**
11
- * Sole source of truth for the prompt's `maxTickets` cap is the resolved
12
- * limits block (see {@link LIMITS_DEFAULTS}). The previous standalone
13
- * `DEFAULT_MAX_TICKETS = 40` literal allowed the prompt to drift out of sync
14
- * with `planning.maxTickets` when call sites forgot to pass the
15
- * resolved value; importing it here means a fallback path (no caller-supplied
16
- * value) still tracks the framework default in `lib/config/limits.js`.
17
- *
18
- * 2-tier is the only published hierarchy after Story #4041 removed the
19
- * Feature tier: the prompt emits Stories only (direct Epic children) and
20
- * asks the planner to carry acceptance/verify as top-level ticket arrays.
8
+ * The story-author system prompt (Story #5312 — rendered from the draft's
9
+ * Story count).
21
10
  *
22
11
  * **Single source of the prompt body (Story #4162).** This module is the sole
23
- * carrier of the full decomposer system-prompt body, delivered to the host
24
- * LLM in the `systemPrompts.decompose` field of the `/mandrel-plan` context envelope
25
- * (via `lib/orchestration/planning/decomposer-context.js`), so no second
12
+ * carrier of the story-author system prompt, delivered to the host LLM in the
13
+ * `systemPrompts.story` field of the `/mandrel-plan` context envelope (via
14
+ * `lib/orchestration/plan-context.js#buildSystemPrompts`), so no second
26
15
  * verbatim copy can drift.
16
+ *
17
+ * Two layers, composed by {@link renderStoryAuthorPrompt}:
18
+ *
19
+ * - **The N=1 core** ({@link renderStoryAuthorCore}) — what every draft
20
+ * needs: the body schema, the contract-level Spec rule, the deterministic
21
+ * body-format lints, and acceptance defined as outcomes a PR reviewer can
22
+ * confirm from the diff and the verify output. It carries no delivery
23
+ * schedule, no per-file behavior paragraphs, no reviewability budget and
24
+ * no verify-tier suffix — every one of those either scored a shape the
25
+ * authoring model already judges or prescribed a proxy that became the
26
+ * goal.
27
+ * - **The N>1 rules** ({@link renderStorySplitRules}) — the schedule and
28
+ * partition rules that only mean anything once a draft has siblings:
29
+ * every Story must earn its slot in the wave schedule, and every
30
+ * acceptance criterion belongs to exactly one Story.
31
+ *
32
+ * The envelope carries the core as `systemPrompts.story` and the split rules
33
+ * as `systemPrompts.storySplitRules`; a planner reads the second only when
34
+ * the default-single split policy clears.
27
35
  */
28
- export function renderDecomposerSystemPrompt({
29
- maxTickets = LIMITS_DEFAULTS.maxTickets,
30
- } = {}) {
31
- return render2TierPrompt({ maxTickets });
32
- }
33
36
 
34
37
  /**
35
- * 2-tier prompt (Story #4041). Decomposes to Stories only — no Feature and
36
- * no Task layer. Acceptance criteria and verification commands live inline
37
- * on the Story body so the executing agent has everything it needs in one
38
- * ticket. Thematic grouping lives as prose in the Epic body / Tech Spec.
38
+ * The N=1 core of the story-author prompt.
39
+ *
40
+ * @returns {string}
39
41
  */
40
- function render2TierPrompt({ maxTickets }) {
41
- // v2 Stage 3: default-single — emit one Story unless the split policy clears.
42
- // Capacity thresholds are sourced from the single DEFAULT_MODEL_CAPACITY
43
- // constant (ticket-validator-sizing.js) so the prompt and the validator
44
- // cannot drift.
45
- const { softSessionTokens, hardSessionTokens } = resolveCapacityCeilings(
46
- DEFAULT_MODEL_CAPACITY,
47
- );
48
- // Deliverable-granularity definition + single-consumer merge rule + the
49
- // thin-dependent merge heuristic are sourced from the single
50
- // DELIVERABLE_GRANULARITY_GUIDANCE constant (ticket-validator-sizing.js) so
51
- // the prompt and the authoring SKILL cannot drift (Story #3777; the
52
- // envelope-floor sentence added by Story #4313).
42
+ export function renderStoryAuthorCore() {
53
43
  const {
54
44
  definition: granularityDefinition,
55
45
  singleConsumerRule,
56
46
  envelopeFloor,
57
47
  } = DELIVERABLE_GRANULARITY_GUIDANCE;
58
- // The binding-vs-advisory authoring altitude + the New-File Contract are
59
- // sourced from the single AUTHORING_ALTITUDE_GUIDANCE constant
60
- // (ticket-validator-sizing.js) so the prompt and the authoring SKILL cannot
61
- // drift (Story #4272).
62
48
  const {
63
49
  altitude: authoringAltitude,
64
50
  advisoryCaveat,
65
51
  newFileContract,
66
52
  } = AUTHORING_ALTITUDE_GUIDANCE;
67
53
  // The deterministic body-format lints (structured `## Changes` bullet shape,
68
- // verify-tier suffixes, …) rendered example-first from their single source
54
+ // non-empty sections) rendered example-first from their single source
69
55
  // (`lib/story-body/body-format-lints.js`) so an authored draft is lint-clean
70
56
  // by construction rather than discovered as a persist dry-run failure and
71
57
  // re-authored at resident-context prices (Story #4684).
@@ -84,10 +70,9 @@ Your job is to turn a plan seed / Tech Spec into a Story ticket array for an AI
84
70
  - Thematic grouping is prose in the Story's folded \`## Spec\` / \`## Slicing\`, never sibling tickets for coupled work.
85
71
 
86
72
  ### LABEL CONVENTIONS:
87
- - \`type::story\` is applied automatically by persist — you do not need to emit it, and no other type label is allowed (the retired Feature and Task tiers have no labels under this hierarchy).
73
+ - \`type::story\` is applied automatically by persist — you do not need to emit it, and no other type label is allowed.
88
74
  - \`labels[]\` is **optional**. Emit it only to request an *additional* label; persist sanitizes the list before applying it.
89
75
  - Do **not** emit \`agent::*\` labels — lifecycle state is runtime-owned, and persist applies \`agent::ready\` itself once every checkpoint is on the ticket.
90
- - Do **not** emit \`persona::*\` labels — the behavioral persona concept (and its label axis) was removed in v2.
91
76
 
92
77
  ### OUTPUT FORMAT:
93
78
  You MUST respond ONLY with a valid JSON array of objects. No prose, no markdown blocks.
@@ -99,8 +84,8 @@ You MUST respond ONLY with a valid JSON array of objects. No prose, no markdown
99
84
  "type": "story",
100
85
  "title": "Short descriptive title",
101
86
  "body": <string — see STORY BODY SCHEMA below>,
102
- "acceptance": ["<testable, observable criterion>", ...],
103
- "verify": ["<exact command or test path> (<tier>)", ...],
87
+ "acceptance": ["<outcome a PR reviewer can confirm>", ...],
88
+ "verify": ["<exact command or test path>", ...],
104
89
  "labels": ["<extra-label>"] (optional — type::story is applied automatically; omit this field unless you need an additional label),
105
90
  "depends_on": ["slug-of-blocking-dependency"] (optional array of Story slugs that block execution)
106
91
  }
@@ -109,7 +94,7 @@ You MUST respond ONLY with a valid JSON array of objects. No prose, no markdown
109
94
  **Slug format**: \`^[a-z0-9][a-z0-9-]*$\` — hyphen-case only. Underscores are rejected by the validator.
110
95
 
111
96
  ### STORY BODY SCHEMA (REQUIRED FOR EVERY STORY):
112
- \`body\` is either the serialized markdown **string** (the section format below) or a **structured object** carrying the same fields (\`goal\`, optional \`slicing\` / \`spec\`, \`changes\`, optional \`non_goals\` / \`wide\` / \`reason_to_exist\`) — persist parses either shape and serializes the canonical markdown itself, so you never need to read \`story-body.js\` or hand-assemble the markdown (the \`stories.template.json\` file emitted next to the plan-context envelope is a ready-to-fill structured-object skeleton). Stories are consumed by non-interactive sub-agents that must self-verify from the Story ticket alone — so the ticket must carry everything an agent needs to execute and self-verify.
97
+ \`body\` is either the serialized markdown **string** (the section format below) or a **structured object** carrying the same fields (\`goal\`, optional \`slicing\` / \`spec\`, \`changes\`, optional \`non_goals\`) — persist parses either shape and serializes the canonical markdown itself, so you never need to read \`story-body.js\` or hand-assemble the markdown (the \`stories.template.json\` file emitted next to the plan-context envelope is a ready-to-fill structured-object skeleton). Stories are consumed by non-interactive sub-agents that must self-verify from the Story ticket alone — so the ticket must carry everything an agent needs to execute and self-verify.
113
98
 
114
99
  The \`acceptance[]\` and \`verify[]\` arrays live at the **top level** of the Story ticket object — that is the machine contract the validator reads. Author each list **once, at top level**, and **omit** the \`## Acceptance\` / \`## Verify\` sections from the authored \`body\` string: persist syncs the top-level arrays into those sections so the GitHub issue stays a complete executable document. The validator resolves both fields from the top level, so an omitted section is the expected shape, not a violation.
115
100
 
@@ -124,18 +109,18 @@ The **persisted** \`body\` renders these markdown sections (in order) — you au
124
109
  <optional ordered intra-session checkpoints — not a second Spec or AC table>
125
110
 
126
111
  ## Spec
127
- <optional lean technical approach — do NOT restate Goal / Acceptance / Verify>
112
+ <optional technical approach at contract level — do NOT restate Goal / Acceptance / Verify>
128
113
 
129
114
  ## Changes
130
115
  - {"path": "<file path>", "assumption": "creates" | "refactors-existing" | "deletes"}
131
116
  - ...
132
117
 
133
118
  ## Acceptance <-- synthesized by persist from acceptance[]; do not author
134
- - [ ] <testable, observable criterion>
119
+ - [ ] <outcome a PR reviewer can confirm>
135
120
  - ...
136
121
 
137
122
  ## Verify <-- synthesized by persist from verify[]; do not author
138
- - <exact command or test path> (<tier>)
123
+ - <exact command or test path>
139
124
  - ...
140
125
 
141
126
  ## Non-Goals
@@ -145,16 +130,12 @@ The **persisted** \`body\` renders these markdown sections (in order) — you au
145
130
  #### STORY BODY RULES:
146
131
 
147
132
  - **goal** (in body string): One sentence stating WHY this Story exists.
148
- - **spec** (optional, in body string as \`## Spec\`): Lean technical approach only, at the altitude the SPEC PROSE CONTRACT below fixes — contract and invariants, never implementation narration. If the Spec is large enough to feel like its own document, the Story is probably too big — split it. Persist keeps Specs inline and rejects over-budget Specs (never writes them under \`docs/\`).
149
- - **slicing** (optional): Ordered intra-session checkpoints for one Story. Not a fan-out table and not a duplicate of Acceptance.
150
- - **changes** (in body string): Each entry is an object \`{ path, assumption }\` where \`assumption\` is one of \`creates | refactors-existing | deletes\`. Acceptable path shapes include explicit files (\`src/components/Foo.tsx\`), glob patterns (\`tests/e2e/*.spec.ts\`, \`**/*.astro\`), and module identifiers that resolve to files. Use \`refactors-existing\` for in-place edits to a file already on \`main\`; \`creates\` for net-new files; \`deletes\` for removals.
151
- - **acceptance** (top-level array on the ticket object): Items MUST be observable from outside the agent. Acceptable shapes: a specific command exits 0, a file exists at a given path, a snapshot test matches, a \`data-testid\` resolves under a given selector, a row count in a fixture matches. UNACCEPTABLE: "verify by reading the diff", "looks good", "matches the spec" — push these down into a \`verify\` command instead.
152
- - **verify** (top-level array on the ticket object): Each entry MUST name a testing tier in parentheses, drawn from \`unit\` / \`contract\` / \`e2e\` / \`validate\`. Example: \`npm run test -- src/x.test.ts (unit)\`, \`npm run validate (validate)\`. Stories with zero verify entries SHOULD fail validation; if a story is genuinely unverifiable in isolation (e.g., a copy edit auditor will eyeball), the literal entry \`manual:<reason>\` is allowed so the absence is intentional, not lazy. Manual entries without a reason are rejected.
153
- - **reason to exist** (REQUIRED, encoded as the \`reason_to_exist\` field of the \`<!-- meta: {...} -->\` comment appended to the serialized body string — NOT a top-level ticket field): One sentence stating the single coherent reason this Story exists, distinct from its broader \`## Goal\` prose. Every Story MUST carry a non-empty \`reason_to_exist\`; it is the machine-checkable form of the cohesion rule (**one Story = one coherent change with one reason to exist**) and the \`epic-plan-consolidate\` critic flags any Story whose body carries no non-empty reason to exist. Encode it as \`<!-- meta: {"reason_to_exist": "..."} -->\`.
154
- - **Observed-behavior claims open with \`Current state (verified <date>)\`.** Any Spec claim about how the codebase behaves today MUST open with that preamble (e.g. \`Current state (verified 2026-07-17): …\`) so a reader can tell a verified observation from an assumption, and can tell when the observation went stale.
155
- - **Intent-then-proxy acceptance shape.** When an acceptance item verifies through a proxy check (a grep, a file-exists probe, an exit-code test), state the intent clause before the proxy check — what outcome the check stands in for — so the proxy never becomes the goal (e.g. "the workflow names hygiene findings as re-author input: \`grep -n "textHygiene" …\` exits 0").
156
- - **Slicing checkpoints are one line each.** Each \`## Slicing\` checkpoint is a single line naming the checkpoint; implementation detail lives in \`## Spec\`, never duplicated into Slicing. A Slicing section outweighing its Spec is a defect the text-hygiene lint flags.
157
- - **Bodies record decisions, never questions to the operator.** Never persist an open question ("Flag if…", "TBD", "confirm with the operator") into a Story body — the executing sub-agent is non-interactive and cannot answer it. Triage each unknown by who can resolve it: an AFK-shaped unknown (a fact in docs, a third-party API surface, observable repo behavior) MUST be resolved by your own research before authoring — never restated as an assumption; only a HITL-shaped unknown (a genuine product or architecture call the operator owns) may be restated as a declarative Key Assumption the agent can act on, stating the default chosen (a decision-made-by-default).
133
+ - **spec** (optional, in body string as \`## Spec\`): The technical approach at the altitude the SPEC PROSE CONTRACT below fixes — contract and invariants, never implementation narration. Write as much as the work needs and no more; persist keeps Specs inline at any length and never writes them under \`docs/\`.
134
+ - **slicing** (optional): Ordered intra-session checkpoints for one Story, one line each. Not a fan-out table and not a duplicate of Acceptance.
135
+ - **changes** (in body string): Each entry is an object \`{ path, assumption }\` where \`assumption\` is one of \`creates | refactors-existing | deletes\`. Acceptable path shapes include explicit files (\`src/components/Foo.tsx\`), glob patterns (\`tests/e2e/*.spec.ts\`, \`**/*.astro\`), and module identifiers that resolve to files. Use \`refactors-existing\` for in-place edits to a file already on \`main\`; \`creates\` for net-new files; \`deletes\` for removals. Persist probes every path against the base branch and repairs a plain-string bullet or a trailing parenthetical into the object form for you; a \`creates\` on an existing path or a \`refactors-existing\` on an absent one is a dry-run warning, and only a \`deletes\` naming an absent path is refused.
136
+ - **acceptance** (top-level array on the ticket object): Each item is an **outcome a PR reviewer can confirm from the diff and the verify output** — what is true of the codebase once the Story lands, stated at the altitude of the capability (a command that now exits 0 against a named input, a behavior a named test now asserts, a config that now fails validation on a retired key, a document that now records a decision). Aim for **three to six** items: fewer than three usually means the outcome is under-specified; more than six usually means acceptance is re-listing the footprint or the mechanical checks that belong in \`verify[]\`. Push grep-shaped probes, file-exists checks and exit-code tests down into \`verify[]\`; never pin an internal helper name or a private file path into an acceptance item the advisory \`changes[]\` is free to reshape. UNACCEPTABLE: "verify by reading the diff", "looks good", "matches the spec".
137
+ - **verify** (top-level array on the ticket object): The **mechanical checks** — exact commands or test paths the deliverer runs and the acceptance critic consumes as evidence: \`node --test tests/x.test.js\`, \`npm run lint\`, \`npm run validate\`, a scoped grep. Every acceptance item should be confirmable from at least one verify entry's output plus the diff. Stories with zero verify entries fail validation.
138
+ - **Bodies record decisions, never questions to the operator.** Never persist an open question ("Flag if…", "TBD", "confirm with the operator") into a Story body — the executing sub-agent is non-interactive and cannot answer it, and the dry-run warns on every one it finds. Triage each unknown by who can resolve it: an AFK-shaped unknown (a fact in docs, a third-party API surface, observable repo behavior) MUST be resolved by your own research before authoring — never restated as an assumption; only a HITL-shaped unknown (a genuine product or architecture call the operator owns) may be restated as a declarative Key Assumption the agent can act on, stating the default chosen (a decision-made-by-default).
158
139
  - **non_goals** (OPTIONAL, in body string as the \`## Non-Goals\` section): A short list of capabilities or changes this Story explicitly does NOT deliver — an advisory negative-scope bound that fences the executing agent away from adjacent work. It is **advisory and NON-GATING**: the validator does not require, count, or reject on it, and an absent or empty section renders nothing. Use the EXACT single-word hyphenated heading spelling \`## Non-Goals\` (a space-separated heading like \`## Out of Scope\` is NOT recognized by the parser and will be dropped). Reach for it when a Story's negative boundary is non-obvious from its \`acceptance[]\` alone; omit it otherwise.
159
140
 
160
141
  #### SPEC PROSE CONTRACT — state the contract, not the implementation:
@@ -164,13 +145,13 @@ The Story is executed by a frontier-model deliverer that reads the codebase itse
164
145
  - **Spec states the contract and invariants**: interfaces, status codes, security invariants, and load-bearing constraints — each with its why. That is the whole job of the Spec.
165
146
  - **Implementation choices belong to the deliverer** unless a choice is load-bearing; a load-bearing choice is stated as a constraint (with why it binds), never as a walkthrough of how to code it.
166
147
  - **No per-file behavior paragraphs.** The \`## Changes\` list already names the footprint; do not narrate what each file will do.
167
- - **No current-state narration.** Do not describe how the codebase works today as scene-setting; the deliverer reads the code. The only current-state prose allowed is a claim a decision depends on, opening with \`Current state (verified <date>)\` per the observed-behavior rule above.
148
+ - **No current-state narration.** Do not describe how the codebase works today as scene-setting; the deliverer reads the code. State only the claims a decision depends on.
168
149
  - **Do not author a \`## References\` section.** Read-only context the deliverer needs is discoverable from the contract and the footprint.
169
150
  - **Acceptance criteria remain the binding contract** — the Spec constrains and explains; \`acceptance[]\` binds.
170
151
 
171
152
  #### DETERMINISTIC BODY-FORMAT LINTS — author lint-clean by construction:
172
153
 
173
- Persist enforces the deterministic body-format rules below and **rejects** an authored body that violates any of them. Author every Story to satisfy all of them on the FIRST draft — each rule is stated example-first so there is nothing to discover by trial-and-error. \`verify-tier-suffix\` is the one rule persist repairs for you: when the tier is unambiguously inferable from the command, persist appends it and proceeds; when it is not, the entry is still rejected and you must choose the tier. \`changes-path-entry-shape\` emits the corrected form in the dry-run failure output but is never applied for you.
154
+ Persist enforces the deterministic body-format rules below and **rejects** an authored body that violates any of them. Author every Story to satisfy all of them on the FIRST draft — each rule is stated example-first so there is nothing to discover by trial-and-error. \`changes-path-entry-shape\` is the one rule persist repairs for you: a plain-string bullet whose path can be salvaged is rewritten into the object form by probing the base branch, and the repair is reported in the dry-run output.
174
155
 
175
156
  ${bodyFormatLintChecklist}
176
157
 
@@ -182,51 +163,17 @@ ${newFileContract}
182
163
 
183
164
  ${advisoryCaveat}
184
165
 
185
- #### STORY SIZING — COHESION FIRST (session capacity is only a backstop):
166
+ #### STORY SIZING — COHESION, NOT COUNT:
186
167
 
187
168
  **Decompose at deliverable granularity, not module/task level.** ${granularityDefinition}
188
169
 
189
- The primary question is **cohesion, not count**: *is this one coherent change with one reason to exist?* File count cannot tell a trivial rename from a hard parser+caller+config change — so lead with the change's reason, not its size. Frontier models one-shot capability-sized work in a single pass — do not fragment a coherent capability into dependent slices just to stay "small."
170
+ The only sizing question is **cohesion**: *is this one coherent change with one reason to exist?* There is no ceiling on a Story's footprint, Spec length or acceptance count — a broad contract cutover is one Story when every changed site changes for the same reason. Frontier models one-shot capability-sized work in a single pass; do not fragment a coherent capability into dependent slices to stay "small", and do not pad a Story with adjacent work to look "complete".
190
171
 
191
172
  ${envelopeFloor}
192
173
 
193
- - **One Story = one coherent change with one reason to exist.** If you cannot state that reason in a sentence, the Story is probably two Stories — or two Stories that should be one. State that sentence explicitly in the Story's \`reason_to_exist\` meta field (see STORY BODY RULES) so the consolidate critic can check it.
174
+ - **One Story = one coherent change with one reason to exist.** If you cannot state that reason in a sentence, the Story is probably two Stories — or two Stories that should be one.
194
175
  - ${singleConsumerRule}
195
176
  - **Split independent, parallelizable work** into sibling Stories — but only when the pieces genuinely have separate reasons to exist.
196
- - **Declare \`wide\` with a one-line reason when a change is legitimately broad** (a cohesive cutover whose authored ticket mass is high for one reason). Declaring \`wide\` lifts the hard session-mass ceiling — see below.
197
-
198
- **Capacity backstop (validator-enforced).** Absolute authored-token ceilings from the single \`DEFAULT_MODEL_CAPACITY\` constant in \`ticket-validator-sizing.js\` — not operator-tunable. They catch Spec novels, not capability-sized Stories:
199
-
200
- - Soft advisory (**${softSessionTokens} tokens**): authored session mass above this emits a nudge to check cohesion or declare \`wide\`.
201
- - Hard ceiling (**${hardSessionTokens} tokens**): authored session mass above this is **rejected** unless the Story declares \`wide\` with a reason.
202
- - Session mass = **authored tokens only** (goal / reason / Spec / acceptance / verify / slicing / change-path text). File count and AC count do **not** inflate mass — a long binding contract or a broad file footprint is never by itself a reason to fragment one coherent capability into dependent slices.
203
-
204
- #### DELIVERY-SCHEDULE SIMULATION — the story count must earn itself:
205
-
206
- Before emitting, simulate the delivery schedule your plan implies, and judge the plan by its schedule — not by how tidy the taxonomy looks:
207
-
208
- 1. **Build the wave schedule.** A Story runs only after every \`depends_on\` completes, and two Stories that name the same file in \`changes[]\` cannot run in the same wave (the scheduler serializes file-overlapping Stories even when no \`depends_on\` edge links them).
209
- 2. **Compute the parallelism yield**: story count ÷ critical-path length in waves. A yield near 1.0 means the plan is a serial chain — N Stories that deliver no faster than one Story while paying N delivery sessions (branch, PR, review, CI).
210
- 3. **Every Story must earn its slot** by at least one of:
211
- - **(a) parallelism** — it actually runs concurrently with a sibling in the schedule you just built ("logically independent" does not count; *schedule*-independent does);
212
- - **(b) risk isolation** — it isolates a consumer-facing behavior change or high-risk cutover into its own reviewable, revertable unit;
213
- - **(c) cohesion break** — merged into its neighbor it would no longer be one coherent change with one reason to exist.
214
- 4. **A dependent link with none of those justifications merges into its consumer.** This generalizes the single-consumer merge rule from pairs to chains.
215
- 5. **Hot-file rule.** When one file appears in the \`changes[]\` of more than a third of your Stories, the slicing axis cuts across a shared seam — merge the Stories that co-edit it, or re-slice along the seam so each Story owns its files.
216
-
217
- End each Story's \`reason_to_exist\` with its justification letter and one clause, e.g. "… (a: runs in wave 1 alongside <slug>)" or "(b: isolates the auto-merge default change)". A reason that names only a topic ("config work", "docs") with no justification is a merge signal.
218
-
219
- #### \`wide\` DECLARATION (optional — for legitimately broad changes):
220
-
221
- A Story whose footprint is legitimately broad declares \`wide\` carrying a one-line human-readable reason. Encode it in the \`<!-- meta: {"wide": {"reason": "..."}} -->\` comment that \`serialize()\` appends to the body string — it is NOT a top-level ticket field:
222
-
223
- \`\`\`json
224
- "wide": { "reason": "hard contract cutover: migrate every <X> call site in one PR" }
225
- \`\`\`
226
-
227
- Declaring \`wide\` with a non-empty reason **lifts the hard session-mass rejection** — no Story is rejected for width when it states why it is broad. Omit \`wide\` for ordinary Stories; a wide footprint with no \`wide\` declaration emits only an advisory nudge (check cohesion or declare \`wide\`), never a rejection on its own.
228
-
229
- **Glob entries** in \`changes[]\` (bullets containing \`*\`) mark the Story footprint as \`unknown-width\`: the numeric ceiling cannot bound a glob, so it is skipped. A Story carrying glob changes with no \`wide\` declaration emits an advisory nudge.
230
177
 
231
178
  #### UI / TESTID INVARIANCE (per CLAUDE.md safety rule):
232
179
 
@@ -242,31 +189,47 @@ Every \`changes[]\` entry is a \`{ path, assumption }\` object — a prose bulle
242
189
 
243
190
  - Stories that touch user-visible copy, brand assets, or visual style MUST cite the relevant section of \`docs/style-guide.md\` in \`acceptance\` (e.g. \`"acceptance": ["Hero copy matches docs/style-guide.md §3 (voice & tone)"]\`). If \`docs/style-guide.md\` does not exist or has no relevant section, state that explicitly: \`"acceptance": ["docs/style-guide.md absent — copy reviewed against the inline brand brief in the plan seed"]\`. Silence on style sourcing is a smell.
244
191
 
245
- ### WAVE-0 BDD SCAFFOLD STORY (features-first; emit when your plan verifies against a scenario that does not exist yet):
246
- The plan-context envelope's \`bddScenarios\` field is the index of Gherkin scenarios that **already exist on \`main\`** — one row per scenario, carrying its \`.feature\` file path, line, scenario title and tags. It is the live signal for this rule: a \`.feature\` path + scenario your plan needs but that appears in no \`bddScenarios\` row does not exist yet. The framework is features-first: implementing Stories reference those \`.feature\` paths in their \`verify[]\` lines, so the files MUST already exist when those Stories run — otherwise verification fails mid-delivery on a missing file.
192
+ CRITICAL: Dependencies should follow execution blockers. There is no parent ticket — never emit a 'parent_slug' field.
193
+ IMPORTANT DEPENDENCY RULE: Story-to-Story dependencies are expressed via \`depends_on\` (one Story depends_on another Story's slug). Use this to express execution ordering across the plan.
194
+ **Never stop mid-array.** Always emit complete JSON — partial arrays are rejected by the validator.`;
195
+ }
247
196
 
248
- When **one or more** \`.feature\` scenarios your plan verifies against are absent from \`bddScenarios\`, you MUST emit **exactly one** dedicated wave-0 scaffold Story whose sole job is to create those \`.feature\` files with \`@skip\`-tagged scenarios BEFORE any implementation Story runs:
197
+ /**
198
+ * The rules that only apply once a draft has more than one Story: the
199
+ * delivery-schedule simulation that makes each Story earn its slot, and the
200
+ * acceptance partition persist enforces at N>1.
201
+ *
202
+ * @returns {string}
203
+ */
204
+ export function renderStorySplitRules() {
205
+ return `#### MULTI-STORY DRAFT — the story count must earn itself:
249
206
 
250
- - **goal**: contains the literal token \`bdd-scaffold\` (e.g. "bdd-scaffold: create the @skip-tagged feature files the implementation Stories verify against").
251
- - **depends_on**: EMPTY (\`[]\`) — it runs first, in wave 0.
252
- - **changes**: one entry per distinct absent \`.feature\` file, each \`{ "path": "<feature file path>", "assumption": "creates" }\`.
253
- - **acceptance**: MUST assert (a) every new \`.feature\` file exists AND (b) every new scenario within them carries an \`@skip\` tag. Keep these observable (a grep/validate command exits 0, a file exists at a path).
254
- - **verify**: a grep/validate command (tier \`validate\`), NOT an e2e runner — verifying that a file exists with the required tags needs no browser/playwright run. Example: \`grep -rL '@skip' tests/features/<area>/*.feature (validate)\` paired with an existence check.
255
- - Each implementation Story whose \`verify[]\` references one of these scaffolded \`.feature\` paths MUST \`depends_on\` the scaffold Story (so the scaffold lands in an earlier wave). Omitting the link trips the soft \`missing-bdd-scaffold\` validator finding.
207
+ You are splitting past the default-single policy, so simulate the delivery schedule your plan implies and judge the plan by its schedule — not by how tidy the taxonomy looks:
256
208
 
257
- When every scenario your plan verifies against is already present in \`bddScenarios\`, do NOT emit a scaffold Story — there is nothing to create.
209
+ 1. **Build the wave schedule.** A Story runs only after every \`depends_on\` completes, and two Stories that name the same file in \`changes[]\` cannot run in the same wave (the scheduler serializes file-overlapping Stories even when no \`depends_on\` edge links them).
210
+ 2. **Every Story must earn its slot** by at least one of:
211
+ - **(a) parallelism** — it actually runs concurrently with a sibling in the schedule you just built ("logically independent" does not count; *schedule*-independent does);
212
+ - **(b) risk isolation** — it isolates a consumer-facing behavior change or high-risk cutover into its own reviewable, revertable unit;
213
+ - **(c) cohesion break** — merged into its neighbor it would no longer be one coherent change with one reason to exist.
214
+ 3. **A dependent link with none of those justifications merges into its consumer.** This generalizes the single-consumer merge rule from pairs to chains: N Stories that deliver no faster than one Story pay N delivery sessions (branch, PR, review, CI) for nothing.
215
+ 4. **When one file appears in the \`changes[]\` of most of your Stories, the slicing axis cuts across a shared seam** — merge the Stories that co-edit it, or re-slice along the seam so each Story owns its files.
258
216
 
259
- ### SCOPE-OVERLAP FLAGGING (docs/runbook downstream of config work):
260
- When a "docs update" / "runbook" / "README" Story appears downstream of an earlier Story in the same plan whose AC already covers updating the same document (e.g. a "config + runbook" Story followed by a "docs" Story touching the same runbook), the downstream Story's deliverable may be fully absorbed by the earlier Story. Flag the risk directly in the Story's top-level \`acceptance\` array by appending an item of the form:
261
- "Scope verification note: this story's deliverable may already be satisfied by Story #<slug-or-id>'s AC — before implementing, \`git diff main -- <path>\` against the upstream Story branch and confirm whether a substantive edit is still required, or whether only a cross-reference remains."
262
- This prevents the executing agent from redoing work the upstream Story already merged.
217
+ #### ACCEPTANCE PARTITION (persist-enforced at N>1):
263
218
 
264
- CRITICAL: Dependencies should follow execution blockers. There is no parent ticket — never emit a 'parent_slug' field.
265
- IMPORTANT DEPENDENCY RULE: Story-to-Story dependencies are expressed via \`depends_on\` (one Story depends_on another Story's slug). Use this to express execution ordering across the plan.
219
+ - Every acceptance criterion of the plan belongs to **exactly one** Story — no criterion is shared, and none is dropped. Persist refuses a draft whose criteria overlap or leave a plan-level criterion unclaimed.
220
+ - Each Story carries its **own** \`## Spec\`; a shared \`techspec.md\` cannot be folded into N>1 Stories.
221
+ - Express ordering with \`depends_on\` (a sibling slug, or \`#<id>\` for an open Story from an earlier plan). A Story whose \`verify[]\` runs against a file a sibling creates MUST \`depends_on\` that sibling, so the file exists when verification runs.`;
222
+ }
266
223
 
267
- ### REVIEWABILITY BUDGET (Story #2798):
268
- \`maxTickets = ${maxTickets}\` is a **reviewability budget**, not a hard authoring cap. It marks the count of tickets a human operator can comfortably review in one planning pass; emitting more than this overflows the operator's review window. Default behaviour:
269
- - **Stay at or under the budget when possible.** Merge narrow, single-module stories into larger, cohesive capability stories before splitting; small Stories should merge back into siblings rather than spawn their own container.
270
- - **Do NOT truncate or over-compress to fit.** If the plan genuinely needs more tickets than the budget, emit the full plan anyway and add a compact \`over_budget_rationale\` note inside the FIRST Story's \`## Goal\` section explaining (a) why the plan exceeds the budget and (b) what was already merged to keep the count down. The operator will then either accept the plan by re-running the decompose with the explicit \`--allow-over-budget\` override flag, or push back and ask for a re-scope.
271
- - **Never stop mid-array.** Always emit complete JSON — partial arrays are rejected by the validator.`;
224
+ /**
225
+ * Render the story-author prompt for a draft of `storyCount` Stories: the
226
+ * N=1 core, plus the schedule and partition rules when the draft has
227
+ * siblings.
228
+ *
229
+ * @param {{ storyCount?: number }} [args]
230
+ * @returns {string}
231
+ */
232
+ export function renderStoryAuthorPrompt({ storyCount = 1 } = {}) {
233
+ const core = renderStoryAuthorCore();
234
+ return storyCount > 1 ? `${core}\n\n${renderStorySplitRules()}` : core;
272
235
  }
@@ -0,0 +1,93 @@
1
+ /**
2
+ * lib/test-isolate/cli-options.js — argv → options for the `test-isolate` CLI.
3
+ *
4
+ * Story #5316: this lived inside `.agents/scripts/test-isolate.js`, a file no
5
+ * test imports, so it scored the CRAP formula's untested maximum (210 at
6
+ * cyclomatic 14 and 0% coverage) and carried the tree's `test-isolate.js`
7
+ * cyclomatic breach row. It sits here beside `list-files.js`, `parse-tap.js`
8
+ * and `runner.js` for the same reason they do: the CLI shell is unreachable
9
+ * from a test, and everything reachable belongs under `lib/`.
10
+ *
11
+ * The long `else if` ladder it replaces was cyclomatic 14, above the repo's
12
+ * must-fix ceiling of 12. Dispatching through two flag tables is the same
13
+ * parse — same precedence, same tolerance for a value-taking flag that ends
14
+ * the argv — at roughly half the branch count.
15
+ */
16
+
17
+ /**
18
+ * Flags that consume the following argv entry as a number, keyed by the
19
+ * option each one sets.
20
+ */
21
+ const NUMERIC_FLAGS = {
22
+ '--workers': 'workers',
23
+ '--max-bisect-depth': 'maxBisectDepth',
24
+ '--max-bisect-targets': 'maxBisectTargets',
25
+ '--suite-concurrency': 'suiteConcurrency',
26
+ };
27
+
28
+ /** Flags that are their own value. */
29
+ const BOOLEAN_FLAGS = {
30
+ '--json': 'json',
31
+ '--quiet': 'quiet',
32
+ };
33
+
34
+ /**
35
+ * Defaults every parse starts from. Deliberately module-private: nothing in
36
+ * production reads it, and exporting it so a test could assert against it
37
+ * would both make that test tautological and land a test-only export on the
38
+ * `dead-exports:production` ratchet.
39
+ */
40
+ function defaultOptions() {
41
+ return {
42
+ pattern: undefined,
43
+ workers: undefined,
44
+ maxBisectDepth: 8,
45
+ maxBisectTargets: 5,
46
+ suiteConcurrency: 8,
47
+ json: false,
48
+ quiet: false,
49
+ };
50
+ }
51
+
52
+ /**
53
+ * Parse the `test-isolate` CLI's argv.
54
+ *
55
+ * Precedence, unchanged from the ladder this replaces:
56
+ * - a known numeric flag consumes the next entry, but only when there IS a
57
+ * next entry — a trailing `--workers` is ignored rather than setting NaN;
58
+ * - a known boolean flag sets its option;
59
+ * - the FIRST non-flag argument becomes the pattern, and later ones are
60
+ * ignored;
61
+ * - anything else is ignored, so an unknown `--flag` is never mistaken for
62
+ * the pattern.
63
+ *
64
+ * @param {string[]} [argv]
65
+ * @returns {{
66
+ * pattern: string|undefined,
67
+ * workers: number|undefined,
68
+ * maxBisectDepth: number,
69
+ * maxBisectTargets: number,
70
+ * suiteConcurrency: number,
71
+ * json: boolean,
72
+ * quiet: boolean,
73
+ * }}
74
+ */
75
+ export function parseIsolateArgv(argv = []) {
76
+ const options = defaultOptions();
77
+ for (let i = 0; i < argv.length; i += 1) {
78
+ const arg = argv[i];
79
+ const numericKey = NUMERIC_FLAGS[arg];
80
+ if (numericKey && argv[i + 1]) {
81
+ options[numericKey] = Number(argv[i + 1]);
82
+ i += 1;
83
+ continue;
84
+ }
85
+ const booleanKey = BOOLEAN_FLAGS[arg];
86
+ if (booleanKey) {
87
+ options[booleanKey] = true;
88
+ continue;
89
+ }
90
+ if (!arg.startsWith('--') && !options.pattern) options.pattern = arg;
91
+ }
92
+ return options;
93
+ }
@@ -0,0 +1,45 @@
1
+ /**
2
+ * lib/test-isolate/progress-log.js — the `onProgress` sink for a
3
+ * `diagnoseIsolation` run.
4
+ *
5
+ * Story #5316: this was an anonymous arrow inlined into `runTestIsolate`'s
6
+ * argument list, which is exactly why it scored CRAP 72 (cyclomatic 8 at 0%
7
+ * coverage) under a name — `<anon runTestIsolate/(stage,payload)#0>` — that no
8
+ * test could address. As a named factory it is a plain table lookup.
9
+ */
10
+
11
+ /** How many bisection suspects are named inline before eliding the rest. */
12
+ const SUSPECT_PREVIEW = 3;
13
+
14
+ /**
15
+ * Per-stage line formatters. A stage with no entry here is ignored, which is
16
+ * what the `else if` ladder this replaces did by falling off the end — so a
17
+ * new stage emitted by the runner stays silent rather than throwing.
18
+ */
19
+ const STAGE_LINES = {
20
+ 'isolated:start': (p) => `[test-isolate] isolated phase: ${p.count} file(s)`,
21
+ 'isolated:done': () => '[test-isolate] isolated phase: done',
22
+ 'suite:start': (p) => `[test-isolate] suite phase: ${p.count} file(s)`,
23
+ 'suite:done': () => '[test-isolate] suite phase: done',
24
+ 'bisect:start': (p) => `[test-isolate] bisecting flipper: ${p.target}`,
25
+ 'bisect:done': (p) => {
26
+ const list = p.suspects.slice(0, SUSPECT_PREVIEW).join(', ');
27
+ const hidden = p.suspects.length - SUSPECT_PREVIEW;
28
+ const more = hidden > 0 ? ` (+${hidden} more)` : '';
29
+ return `[test-isolate] suspects: ${list}${more}`;
30
+ },
31
+ };
32
+
33
+ /**
34
+ * Build the progress callback `diagnoseIsolation` calls as each phase starts
35
+ * and finishes.
36
+ *
37
+ * @param {(line: string) => void} onLog
38
+ * @returns {(stage: string, payload: object) => void}
39
+ */
40
+ export function createProgressLogger(onLog) {
41
+ return (stage, payload) => {
42
+ const format = STAGE_LINES[stage];
43
+ if (format) onLog(format(payload ?? {}));
44
+ };
45
+ }
@@ -0,0 +1,97 @@
1
+ /**
2
+ * lib/test-isolate/render-report.js — the human-readable `test-isolate` report.
3
+ *
4
+ * Story #5316: extracted verbatim from `.agents/scripts/test-isolate.js`,
5
+ * where no test could reach it (CRAP 72 at cyclomatic 8, 0% coverage). Pure
6
+ * string building — no I/O, no clock — so the whole surface is assertable.
7
+ *
8
+ * The three sections it renders are split into helpers so each is
9
+ * independently readable and none of them alone approaches the cyclomatic
10
+ * ceiling; `renderReport` is left as the composition.
11
+ */
12
+
13
+ /**
14
+ * The flipper section: files that passed alone and failed in the suite, plus
15
+ * the bisection suspects for each.
16
+ *
17
+ * @param {string[]} lines Accumulator, appended in place.
18
+ * @param {import('./runner.js').IsolateReport} report
19
+ */
20
+ function pushFlipperSection(lines, report) {
21
+ if (report.flippers.length === 0) {
22
+ lines.push('✓ No flippers detected — every file that passed alone');
23
+ lines.push(' also passed in the full suite run.');
24
+ return;
25
+ }
26
+ lines.push(`✗ ${report.flippers.length} flipper(s) detected:`);
27
+ for (const f of report.flippers) lines.push(` - ${f}`);
28
+ lines.push('');
29
+ if (report.bisections.length === 0) return;
30
+ lines.push('Likely polluters (bisection suspects):');
31
+ for (const b of report.bisections) {
32
+ const tag = b.inconclusive ? ' [inconclusive]' : '';
33
+ lines.push(` ${b.file}${tag}`);
34
+ for (const s of b.suspects) lines.push(` ← ${s}`);
35
+ }
36
+ }
37
+
38
+ /**
39
+ * One env-mutating file's added/removed/changed summary. Empty parts are
40
+ * omitted, so a file that only added a var reads as `added=[...]` alone.
41
+ *
42
+ * @param {{added: string[], removed: string[], changed: string[]}} envDiff
43
+ * @returns {string}
44
+ */
45
+ function formatEnvDiff(envDiff) {
46
+ const parts = [];
47
+ if (envDiff.added.length > 0)
48
+ parts.push(`added=[${envDiff.added.join(', ')}]`);
49
+ if (envDiff.removed.length > 0) {
50
+ parts.push(`removed=[${envDiff.removed.join(', ')}]`);
51
+ }
52
+ if (envDiff.changed.length > 0) {
53
+ parts.push(`changed=[${envDiff.changed.join(', ')}]`);
54
+ }
55
+ return parts.join(' ');
56
+ }
57
+
58
+ /**
59
+ * The env-leak section: files whose process exited with `process.env` still
60
+ * mutated, called out even when no failure cascade has manifested yet.
61
+ *
62
+ * @param {string[]} lines Accumulator, appended in place.
63
+ * @param {import('./runner.js').IsolateReport} report
64
+ */
65
+ function pushEnvSection(lines, report) {
66
+ if (report.envMutators.length === 0) {
67
+ lines.push('✓ No env-var leaks detected across isolated runs.');
68
+ return;
69
+ }
70
+ lines.push(
71
+ `⚠ ${report.envMutators.length} file(s) left process.env mutated:`,
72
+ );
73
+ for (const m of report.envMutators) {
74
+ lines.push(` ${m.file}`);
75
+ lines.push(` ${formatEnvDiff(m.envDiff)}`);
76
+ }
77
+ }
78
+
79
+ /**
80
+ * Render the diagnostic report as text.
81
+ *
82
+ * @param {import('./runner.js').IsolateReport} report
83
+ * @returns {string}
84
+ */
85
+ export function renderReport(report) {
86
+ const lines = [];
87
+ lines.push('');
88
+ lines.push('=== test-isolate diagnostic report ===');
89
+ lines.push(`Files scanned: ${report.files.length}`);
90
+ lines.push(`Wall duration: ${(report.durationMs / 1000).toFixed(1)}s`);
91
+ lines.push('');
92
+ pushFlipperSection(lines, report);
93
+ lines.push('');
94
+ pushEnvSection(lines, report);
95
+ lines.push('');
96
+ return lines.join('\n');
97
+ }