mandrel 2.56.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/.agents/agents/plan-critic.md +13 -18
  2. package/.agents/agents/story-worker.md +25 -33
  3. package/.agents/docs/agentrc-reference.json +0 -30
  4. package/.agents/docs/configuration.md +8 -28
  5. package/.agents/docs/execution-reference.md +5 -5
  6. package/.agents/docs/quality-gates.md +8 -7
  7. package/.agents/instructions.md +9 -10
  8. package/.agents/schemas/agentrc.schema.json +9 -185
  9. package/.agents/schemas/story-deliver-terminal.schema.json +1 -1
  10. package/.agents/scripts/acceptance-eval.js +107 -17
  11. package/.agents/scripts/ceremony-derive.js +191 -0
  12. package/.agents/scripts/check-context-budget.js +28 -33
  13. package/.agents/scripts/check-cyclomatic.js +4 -3
  14. package/.agents/scripts/deliver-light.js +31 -94
  15. package/.agents/scripts/evidence-gate.js +17 -1
  16. package/.agents/scripts/lib/audit-suite/checklist-threading.js +15 -2
  17. package/.agents/scripts/lib/baselines/coverage-updater-cli.js +110 -0
  18. package/.agents/scripts/lib/baselines/crap-preview-scan.js +25 -0
  19. package/.agents/scripts/lib/baselines/crap-updater-cli.js +223 -0
  20. package/.agents/scripts/lib/bdd-scenario-budget.js +21 -3
  21. package/.agents/scripts/lib/bootstrap/quality-bootstrap.js +0 -1
  22. package/.agents/scripts/lib/close-validation/gates.js +52 -1
  23. package/.agents/scripts/lib/config/acceptance-eval.js +25 -57
  24. package/.agents/scripts/lib/config/delivery-routing.js +7 -33
  25. package/.agents/scripts/lib/config/explain.js +0 -19
  26. package/.agents/scripts/lib/config/limits.js +18 -78
  27. package/.agents/scripts/lib/config/quality.js +6 -3
  28. package/.agents/scripts/lib/config/runners.js +3 -2
  29. package/.agents/scripts/lib/config-settings-schema-delivery.js +15 -68
  30. package/.agents/scripts/lib/config-settings-schema-quality.js +0 -14
  31. package/.agents/scripts/lib/config-settings-schema.js +16 -143
  32. package/.agents/scripts/lib/crap-engine.js +35 -4
  33. package/.agents/scripts/lib/crap-utils.js +17 -1
  34. package/.agents/scripts/lib/cyclomatic-ceiling.js +19 -7
  35. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  36. package/.agents/scripts/lib/observability/runtime-friction.js +1 -1
  37. package/.agents/scripts/lib/observability/source-classifier.js +1 -0
  38. package/.agents/scripts/lib/orchestration/acceptance-eval-decision.js +5 -4
  39. package/.agents/scripts/lib/orchestration/ceremony-routing.js +19 -73
  40. package/.agents/scripts/lib/orchestration/code-review.js +7 -3
  41. package/.agents/scripts/lib/orchestration/complexity-gate.js +46 -212
  42. package/.agents/scripts/lib/orchestration/file-assumptions.js +32 -17
  43. package/.agents/scripts/lib/orchestration/light-escalation.js +3 -3
  44. package/.agents/scripts/lib/orchestration/light-suitability.js +66 -233
  45. package/.agents/scripts/lib/orchestration/pinned-identifier-lint.js +137 -0
  46. package/.agents/scripts/lib/orchestration/plan-context.js +189 -387
  47. package/.agents/scripts/lib/orchestration/plan-critic-conditions.js +42 -153
  48. package/.agents/scripts/lib/orchestration/plan-critics-evaluate.js +14 -70
  49. package/.agents/scripts/lib/orchestration/plan-persist/acceptance-handle-repair.js +107 -0
  50. package/.agents/scripts/lib/orchestration/plan-persist/changes-repair.js +305 -0
  51. package/.agents/scripts/lib/orchestration/plan-persist/persist-helpers.js +138 -170
  52. package/.agents/scripts/lib/orchestration/plan-persist/run-plan-persist.js +128 -297
  53. package/.agents/scripts/lib/orchestration/plan-persist/soft-findings.js +55 -0
  54. package/.agents/scripts/lib/orchestration/plan-persist/story-ops.js +16 -65
  55. package/.agents/scripts/lib/orchestration/plan-persist/wave-serialisation.js +22 -35
  56. package/.agents/scripts/lib/orchestration/plan-text-hygiene.js +36 -135
  57. package/.agents/scripts/lib/orchestration/planning/memory-pool-advisory.js +61 -223
  58. package/.agents/scripts/lib/orchestration/review-base-ref.js +138 -0
  59. package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +5 -0
  60. package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +37 -5
  61. package/.agents/scripts/lib/orchestration/single-story-close/phases/pre-gate-steps.js +46 -16
  62. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +6 -1
  63. package/.agents/scripts/lib/orchestration/story-close/context-budget-writeback.js +213 -0
  64. package/.agents/scripts/lib/orchestration/task-body-validator.js +10 -63
  65. package/.agents/scripts/lib/orchestration/ticket-validator-conflicts.js +33 -539
  66. package/.agents/scripts/lib/orchestration/ticket-validator-sizing.js +21 -414
  67. package/.agents/scripts/lib/orchestration/ticket-validator.js +54 -118
  68. package/.agents/scripts/lib/orchestration/verify-credit.js +69 -24
  69. package/.agents/scripts/lib/story-body/body-format-lints.js +15 -85
  70. package/.agents/scripts/lib/story-body/story-body.js +54 -240
  71. package/.agents/scripts/lib/templates/decomposer-prompts.js +133 -121
  72. package/.agents/scripts/lib/test-isolate/cli-options.js +93 -0
  73. package/.agents/scripts/lib/test-isolate/progress-log.js +45 -0
  74. package/.agents/scripts/lib/test-isolate/render-report.js +97 -0
  75. package/.agents/scripts/lib/test-isolate/run-isolate.js +87 -0
  76. package/.agents/scripts/lib/test-run-credit.js +277 -0
  77. package/.agents/scripts/lib/wave-runner/footprint.js +48 -358
  78. package/.agents/scripts/lib/wave-runner/ready-set.js +6 -5
  79. package/.agents/scripts/lib/workers/crap-worker.js +32 -41
  80. package/.agents/scripts/plan-context.js +7 -9
  81. package/.agents/scripts/plan-critics.js +28 -54
  82. package/.agents/scripts/plan-persist.js +25 -68
  83. package/.agents/scripts/quality-preview.js +51 -0
  84. package/.agents/scripts/run-tests.js +12 -0
  85. package/.agents/scripts/stories-wave-tick.js +23 -45
  86. package/.agents/scripts/test-isolate.js +13 -180
  87. package/.agents/scripts/update-coverage-baseline.js +25 -70
  88. package/.agents/scripts/update-crap-baseline.js +19 -123
  89. package/.agents/skills/core/scope-triage/SKILL.md +3 -3
  90. package/.agents/workflows/audit-clean-code.md +4 -3
  91. package/.agents/workflows/helpers/acceptance-self-eval.md +41 -41
  92. package/.agents/workflows/helpers/code-quality-guardrails.md +4 -4
  93. package/.agents/workflows/helpers/code-review.md +2 -3
  94. package/.agents/workflows/helpers/deliver-digest.md +46 -55
  95. package/.agents/workflows/helpers/deliver-light.md +40 -105
  96. package/.agents/workflows/helpers/deliver-reference.md +1 -1
  97. package/.agents/workflows/helpers/deliver-story-reference.md +54 -55
  98. package/.agents/workflows/helpers/deliver-story.md +10 -13
  99. package/.agents/workflows/helpers/plan-reference.md +163 -221
  100. package/.agents/workflows/mandrel-plan.md +31 -40
  101. package/.agents/workflows/memory-consolidate.md +9 -13
  102. package/docs/CHANGELOG.md +36 -0
  103. package/lib/cli/registry.js +98 -2
  104. package/lib/migrations/index.js +4 -0
  105. package/lib/migrations/steps/2.57.0-retire-delivery-limit-knobs.js +45 -0
  106. package/lib/migrations/steps/2.57.0-retire-planning-limit-knobs.js +59 -0
  107. package/package.json +1 -1
  108. package/.agents/scripts/lib/framework-version.js +0 -39
  109. package/.agents/scripts/lib/orchestration/consolidation-precondition.js +0 -223
  110. package/.agents/scripts/lib/orchestration/plan-persist/fan-out-gate.js +0 -97
  111. package/.agents/scripts/lib/orchestration/planning/decomposer-context.js +0 -26
  112. package/.agents/scripts/lib/orchestration/spec-budget.js +0 -89
  113. package/.agents/scripts/lib/orchestration/spec-spill.js +0 -74
  114. package/.agents/scripts/lib/orchestration/verify-tier-repair.js +0 -107
@@ -47,15 +47,17 @@
47
47
  * Ratchet semantics (mirroring the sibling ratchets):
48
48
  * - A gated tier grows beyond `baseline.tiers.<tier>.totalBytes +
49
49
  * baseline.toleranceBytes` → exit 1, naming the tier and its delta.
50
- * - A gated tier shrinks below its baseline total → exit 1 (Story #4872).
51
- * A ratchet that only tightens in one direction lets every measured
52
- * improvement evaporate: the recorded total keeps promising headroom the
53
- * tree no longer spends, so the next growth is absorbed by stale slack
54
- * instead of being reported. Shrinkage is therefore **actionable** —
55
- * refresh the baseline down and the gain is locked in. Unlike growth this
56
- * is deliberately **zero-tolerance**: `toleranceBytes` exists to keep a
57
- * trivial addition from churning the file, and applying it downward would
58
- * silently discard every sub-tolerance gain.
50
+ * - A gated tier shrinks below its baseline total → **exit 0**, reported as
51
+ * an informational `-` line (Story #5313). This deliberately reverses
52
+ * Story #4872's "shrink fails" rule: that rule made every trim a red gate
53
+ * whose only remedy was a hand-run `--update`, so the gain was paid for
54
+ * twice. The concern it answered — a stale total silently absorbing the
55
+ * next growth — is now met by the close's write-back seam
56
+ * (`story-close/context-budget-writeback.js`), which rewrites the lower
57
+ * total into `baselines/context-budget.json` on the Story branch so the
58
+ * gain locks in without a failing gate. Shrinkage stays zero-tolerance
59
+ * in the *report* (every byte under the total is listed) so a sub-
60
+ * tolerance gain is never discarded by the write-back either.
59
61
  * - A recorded row naming a path the measured tier no longer contains →
60
62
  * exit 1. The row describes a file that has been deleted or de-listed, so
61
63
  * the bytes it contributes to the recorded total are fiction.
@@ -341,9 +343,9 @@ function absentRows(tier, files, baseTier) {
341
343
  /**
342
344
  * Pure diff: compare the current tier map against the committed baseline. A
343
345
  * gated tier with no current files is skipped; a tier absent from the baseline
344
- * is skipped. `grown`, `shrunk` and `absent` entries all fail the gate — see
345
- * the ratchet semantics in the module header for why shrinkage is actionable
346
- * rather than informational (Story #4872).
346
+ * is skipped. `grown` and `absent` entries fail the gate; `shrunk` entries are
347
+ * reported and written back by the close (Story #5313) — see the ratchet
348
+ * semantics in the module header.
347
349
  *
348
350
  * @param {{ tiers: Record<string, Array<{ path: string, bytes: number }>> }} tierMap
349
351
  * @param {{ toleranceBytes?: number, tiers?: Record<string, { totalBytes: number }> }} baseline
@@ -384,9 +386,9 @@ export function diffBudget(tierMap, baseline) {
384
386
  delta: current - baselineBytes,
385
387
  });
386
388
  } else if (current < baselineBytes) {
387
- // Deliberately zero-tolerance: `tolerance` guards against churn from a
388
- // trivial *addition*; mirroring it downward would discard every gain
389
- // smaller than the tolerance, which is the leak this branch closes.
389
+ // Deliberately zero-tolerance in the report: `tolerance` guards against
390
+ // churn from a trivial *addition*; mirroring it downward would hide a
391
+ // gain from the write-back that locks it in (Story #5313).
390
392
  shrunk.push({
391
393
  tier,
392
394
  current,
@@ -400,26 +402,24 @@ export function diffBudget(tierMap, baseline) {
400
402
  }
401
403
 
402
404
  /**
403
- * Count the drift entries that fail the gate. Every direction is actionable
404
- * (Story #4872), so this is the one place the failure set is defined and both
405
- * the summary tag and the exit code read it.
405
+ * Count the drift entries that fail the gate: growth past tolerance and a
406
+ * recorded row the tree no longer backs. Shrinkage is not in the set (Story
407
+ * #5313 — the close writes it back instead). This is the one place the
408
+ * failure set is defined and both the summary tag and the exit code read it.
406
409
  *
407
410
  * @param {ReturnType<typeof diffBudget>} diff
408
411
  * @returns {number}
409
412
  */
410
413
  export function budgetFailureCount(diff) {
411
- return (
412
- (diff?.grown?.length ?? 0) +
413
- (diff?.shrunk?.length ?? 0) +
414
- (diff?.absent?.length ?? 0)
415
- );
414
+ return (diff?.grown?.length ?? 0) + (diff?.absent?.length ?? 0);
416
415
  }
417
416
 
418
417
  /**
419
418
  * Render the human-readable diff. `+` lines are tiers that grew beyond
420
- * tolerance; `-` lines are tiers that shrank below their recorded total or
421
- * rows naming a path the tree no longer carries. All three fail the gate. A
422
- * one-line summary always follows.
419
+ * tolerance; `-` lines are tiers that shrank below their recorded total
420
+ * (informational — the close writes the lower total back) or rows naming a
421
+ * path the tree no longer carries (a gate failure). A one-line summary
422
+ * always follows.
423
423
  *
424
424
  * @param {ReturnType<typeof diffBudget>} diff
425
425
  * @returns {string}
@@ -433,7 +433,7 @@ export function renderDiff(diff) {
433
433
  }
434
434
  for (const s of diff.shrunk) {
435
435
  lines.push(
436
- `- ${s.tier}: ${s.current} bytes is under the recorded ${s.baseline} (delta -${s.delta}) — the ratchet is holding slack the tree no longer spends; refresh baselines/context-budget.json`,
436
+ `- ${s.tier}: ${s.current} bytes is under the recorded ${s.baseline} (delta -${s.delta}) — the close writes the lower total back to baselines/context-budget.json when this branch lands`,
437
437
  );
438
438
  }
439
439
  for (const a of diff.absent ?? []) {
@@ -479,7 +479,7 @@ export function renderReachable(tierMap, baseline) {
479
479
  * stderr?: { write: (s: string) => void },
480
480
  * }} [opts]
481
481
  * @returns {Promise<number>} 0 = clean / within tolerance / shrink-only / no-op;
482
- * 1 = a gated tier grew beyond tolerance
482
+ * 1 = a gated tier grew beyond tolerance or a recorded row is unbacked
483
483
  */
484
484
  /**
485
485
  * Write a fresh budget, preserving the recorded tolerance so `--update` never
@@ -621,11 +621,6 @@ function renderFailureDiagnostics({ report, stderr }) {
621
621
  `[context-budget] ❌ a documentation tier grew beyond tolerance — refresh the budget consciously with \`node .agents/scripts/check-context-budget.js --update\` once the growth is intentional\n`,
622
622
  );
623
623
  }
624
- if (diff.shrunk.length > 0) {
625
- stderr.write(
626
- `[context-budget] ❌ a documentation tier came in under its recorded total — the ratchet is holding slack the tree no longer spends, so the next growth would be absorbed silently. Lock the gain in with \`node .agents/scripts/check-context-budget.js --update\`\n`,
627
- );
628
- }
629
624
  if (diff.absent.length > 0) {
630
625
  stderr.write(
631
626
  `[context-budget] ❌ a recorded row names a path the measured tier no longer contains — its bytes inflate the recorded total against nothing. Refresh with \`node .agents/scripts/check-context-budget.js --update\`\n`,
@@ -1,6 +1,7 @@
1
1
  /**
2
2
  * CLI: ratchet on cyclomatic complexity against
3
- * `delivery.quality.codingGuardrails.cyclomaticMustFix` (Story #4923).
3
+ * the fixed cyclomatic ceiling of 12 (Story #4923; the `cyclomaticMustFix`
4
+ * config key was retired in Story #5313).
4
5
  *
5
6
  * The must-fix ceiling was documented as blocking (`code-quality-guardrails.md`
6
7
  * promises "the close-validation chain refuses the merge") while being read by
@@ -158,7 +159,7 @@ function warnAboutBaseline({ baseline, mustFix, baselinePath, stderr }) {
158
159
  }
159
160
  if (typeof baseline.ceiling === 'number' && baseline.ceiling !== mustFix) {
160
161
  stderr.write(
161
- `[cyclomatic] ⚠ baseline was recorded at ceiling c=${baseline.ceiling} but the configured cyclomaticMustFix is c=${mustFix} — re-run with --update\n`,
162
+ `[cyclomatic] ⚠ baseline was recorded at ceiling c=${baseline.ceiling} but the enforced ceiling is c=${mustFix} — re-run with --update\n`,
162
163
  );
163
164
  }
164
165
  }
@@ -265,7 +266,7 @@ runAsCli(import.meta.url, main, {
265
266
  invocation:
266
267
  'node .agents/scripts/check-cyclomatic.js [--baseline <path>] [--update] [--json]',
267
268
  summary:
268
- 'Ratchet on cyclomatic complexity: fail when a file gains a function above `delivery.quality.codingGuardrails.cyclomaticMustFix`, or when its worst function gets worse than the recorded baseline.',
269
+ 'Ratchet on cyclomatic complexity: fail when a file gains a function above the fixed ceiling of 12, or when its worst function gets worse than the recorded baseline.',
269
270
  flags: [
270
271
  [
271
272
  '--baseline <path>',
@@ -22,12 +22,12 @@
22
22
  *
23
23
  * - **gate** (default) — judge a prompt's predicted footprint. On
24
24
  * `proceed-light` it authors the receipt Story (via the plan-persist
25
- * `createStoryIssues` surface) and prints the init/close hand-off. On
26
- * over-scope it prints `ask-operator` (attended) or emits an `escalated`
27
- * terminal envelope (`--yes`), never landing silently. An attended
28
- * `ask-operator` is answerable **either** way: `--operator-proceed-light`
29
- * records the operator's proceed answer (Story #4815), which the gate
30
- * applies only to a coarse size prediction and never to a risk rule.
25
+ * `createStoryIssues` surface) and prints the init/close hand-off,
26
+ * carrying any predicted-shape `warnings[]` (Story #5313: an over-ceiling
27
+ * prediction warns, it no longer stops). An un-ledgered verdict or an
28
+ * un-waivable risk rule emits an `escalated` terminal envelope, never
29
+ * landing silently. The former attended stop-and-ask outcome and the
30
+ * flag that answered it went with the gate they answered.
31
31
  * - **backstop** (`--backstop --story <id>`) — re-check the ACTUAL diff of
32
32
  * the Story branch after implementation; exit non-zero when it exceeds the
33
33
  * light ceilings, so an over-scope diff is blocked rather than landed.
@@ -52,8 +52,7 @@
52
52
  * node .agents/scripts/deliver-light.js --backstop --story 4741
53
53
  *
54
54
  * Exit codes: 0 ok (proceed / clean backstop), 1 usage error, 2 the gate did
55
- * not proceed light (ask-operator, or an `escalated` terminal), 3 the diff
56
- * backstop blocked.
55
+ * not proceed light (an `escalated` terminal), 3 the diff backstop blocked.
57
56
  */
58
57
 
59
58
  import { parseArgs } from 'node:util';
@@ -84,15 +83,17 @@ Usage:
84
83
  deliver-light.js --prompt <text> [--creates csv] [--refactors csv]
85
84
  [--acceptance n] [--kinds csv] [--magnitude m]
86
85
  [--uncertainty u] [--route lite|full] [--reason <text>]
87
- [--amends '#id'] [--operator-proceed-light <text>] [--yes]
86
+ [--amends '#id'] [--yes]
88
87
  deliver-light.js --backstop --story <id>
89
88
 
90
89
  The thin /deliver-light entry point: suitability gate → inline receipt Story →
91
90
  the same single-story-init.js / single-story-close.js engine /mandrel-deliver uses.
92
91
 
93
92
  The gate judges EFFORT and RISK, not artifact counts: N instances of one
94
- mechanical edit is one kind at N sites. It rejects only clearly-epic work; the
95
- --backstop pass enforces size against the actual diff.
93
+ mechanical edit is one kind at N sites. A predicted shape past a light ceiling
94
+ is a WARNING on the envelope, not a refusal (Story #5313); only an un-ledgered
95
+ verdict or an un-waivable risk rule (sensitive path, migration span) escalates.
96
+ The --backstop pass enforces size against the actual diff.
96
97
 
97
98
  Gate options:
98
99
  --prompt <text> Operator prompt describing the change. Required for the gate.
@@ -108,17 +109,9 @@ Gate options:
108
109
  --route <r> Ledgered model verdict route: lite | full.
109
110
  --reason <text> Recorded reason for a lite verdict (required for lite).
110
111
  --amends <#id> Mark this as an amendment of an existing issue.
111
- --operator-proceed-light <text>
112
- Record the operator's "proceed light" answer to an
113
- ask-operator gate, with their reason. Attended-only:
114
- refused with --yes. Waives a coarse SIZE prediction
115
- (change kinds, magnitude, uncertainty, deployable span)
116
- only — sensitivity, migration span, and an unknown
117
- footprint stay non-negotiable, the ledgered --route lite
118
- verdict is still required, and the --backstop pass still
119
- bounds the actual diff. Recorded in the receipt Story.
120
- --yes Unattended: over-scope emits an escalated terminal
121
- envelope and ENDS the session (no prompt, no fallback).
112
+ --yes Unattended marker. Escalation is terminal either way;
113
+ the flag is accepted so unattended callers keep their
114
+ invocation shape.
122
115
 
123
116
  Backstop options:
124
117
  --backstop Re-check the ACTUAL diff after implementation. Bounds the
@@ -132,9 +125,6 @@ Backstop options:
132
125
  --help Show this help.
133
126
  `;
134
127
 
135
- /** Exit code when the gate did not resolve to proceed-light. */
136
- const EXIT_NOT_PROCEED = 2;
137
-
138
128
  /**
139
129
  * Split a comma-separated path list into trimmed, non-empty entries.
140
130
  *
@@ -196,14 +186,10 @@ export function synthesizeAcceptance(count) {
196
186
  * uncertainty?: string,
197
187
  * route?: string,
198
188
  * reason?: string,
199
- * operatorProceedLight?: string,
200
- * yes?: boolean,
201
189
  * injectedRules?: object,
202
190
  * }} args `kinds` / `magnitude` / `uncertainty` are the declared effort-and-risk
203
191
  * axes the gate judges (Story #4764); omitting them declares no signal, not a
204
- * small one — an unrecognized bucket fails closed. `operatorProceedLight`
205
- * carries the operator's recorded answer to an `ask-operator` outcome
206
- * (Story #4815) and is adjudicated inside the gate, never applied here.
192
+ * small one — an unrecognized bucket is reported as a warning (Story #5313).
207
193
  * @returns {{ action: string, suitability: object, outcome: object }}
208
194
  */
209
195
  export function runLightGate({
@@ -215,8 +201,6 @@ export function runLightGate({
215
201
  uncertainty,
216
202
  route,
217
203
  reason,
218
- operatorProceedLight,
219
- yes = false,
220
204
  injectedRules,
221
205
  } = {}) {
222
206
  const predictedChanges = buildPredictedChanges({ creates, refactors });
@@ -229,11 +213,7 @@ export function runLightGate({
229
213
  verdict: { route, reason },
230
214
  injectedRules,
231
215
  });
232
- const outcome = resolveLightGateOutcome({
233
- suitability,
234
- yes,
235
- operatorOverride: operatorProceedLight,
236
- });
216
+ const outcome = resolveLightGateOutcome({ suitability });
237
217
  return { action: outcome.action, suitability, outcome };
238
218
  }
239
219
 
@@ -246,11 +226,9 @@ export function runLightGate({
246
226
  * prompt: string,
247
227
  * changedFiles?: string[],
248
228
  * amends?: string|number|null,
249
- * override?: object|null,
250
229
  * assembleFn?: typeof assemblePlanStories,
251
230
  * createFn?: typeof createStoryIssues,
252
- * }} args `override` is the applied operator scope override (Story #4815),
253
- * recorded in the receipt body so the decision is auditable from the ticket.
231
+ * }} args
254
232
  * @returns {Promise<{ storyId: number, url: string|undefined, title: string }>}
255
233
  */
256
234
  export async function createLightReceipt({
@@ -258,7 +236,6 @@ export async function createLightReceipt({
258
236
  prompt,
259
237
  changedFiles = [],
260
238
  amends = null,
261
- override = null,
262
239
  assembleFn = assemblePlanStories,
263
240
  createFn = createStoryIssues,
264
241
  } = {}) {
@@ -266,7 +243,6 @@ export async function createLightReceipt({
266
243
  prompt,
267
244
  changedFiles,
268
245
  amends,
269
- override,
270
246
  });
271
247
  const { stories } = assembleFn([ticket]);
272
248
  const { created } = await createFn({ provider, stories });
@@ -294,19 +270,6 @@ export function buildNextCommands(storyId) {
294
270
  };
295
271
  }
296
272
 
297
- /**
298
- * Was a non-blank `--operator-proceed-light` supplied? The gate core decides
299
- * whether it *applies*; this only asks whether the operator typed one, so the
300
- * attended-only refusal can fire before any adjudication.
301
- *
302
- * @param {{ 'operator-proceed-light'?: unknown }} values Parsed CLI values.
303
- * @returns {boolean}
304
- */
305
- export function hasOperatorOverride(values = {}) {
306
- const raw = values['operator-proceed-light'];
307
- return typeof raw === 'string' && raw.trim() !== '';
308
- }
309
-
310
273
  /**
311
274
  * Emit a JSON envelope on stdout (the machine surface) so a headless caller can
312
275
  * branch on it. Human-readable log lines stay on stderr.
@@ -355,12 +318,12 @@ async function runBackstopMode(values, deps = {}) {
355
318
  * envelope** and stops (Story #4746). It is placed **first**, above every
356
319
  * creation call site, so "nothing was started" is a property of the
357
320
  * control flow rather than a claim the envelope makes about itself.
358
- * - **`ask-operator`** is unchanged: the plain gate envelope and exit 2. It
359
- * is not terminal — the operator has a choice to make, and manufacturing a
360
- * terminal for it would end a session that is supposed to be waiting. The
361
- * operator's proceed answer comes back as `--operator-proceed-light`.
362
321
  * - **`proceed-light`** authors the receipt Story and prints the hand-off,
363
- * carrying any applied `override` into both the receipt and the envelope.
322
+ * carrying the predicted-shape `warnings[]` (Story #5313) on the envelope
323
+ * and on stderr, so an over-ceiling prediction is stated rather than
324
+ * silently waved through. The former attended stop-and-ask outcome is
325
+ * gone: the gate never had a question a human could answer that the diff
326
+ * backstop does not answer better.
364
327
  *
365
328
  * The injectable seams exist so the no-side-effect guarantee is testable
366
329
  * without a network: a test asserts the escalate path never reaches them.
@@ -390,17 +353,6 @@ export async function runGateMode(values, deps = {}) {
390
353
  throw new Error('[deliver-light] --prompt <text> is required for the gate');
391
354
  }
392
355
 
393
- // Attended-only, enforced loudly (Story #4815). Silently ignoring the flag
394
- // under --yes would let an automated caller pass it as a hopeful no-op and
395
- // read the resulting escalation as a bug; a usage error says which of the
396
- // two the caller has to give up.
397
- if (values.yes === true && hasOperatorOverride(values)) {
398
- process.stderr.write(HELP);
399
- throw new Error(
400
- '[deliver-light] --operator-proceed-light is attended-only and cannot be combined with --yes: an unattended run has no operator whose answer this is, and over-scope must fail closed to /mandrel-plan',
401
- );
402
- }
403
-
404
356
  const gate = runLightGate({
405
357
  creates: parseCsvPaths(values.creates),
406
358
  refactors: parseCsvPaths(values.refactors),
@@ -412,35 +364,24 @@ export async function runGateMode(values, deps = {}) {
412
364
  uncertainty: values.uncertainty,
413
365
  route: values.route,
414
366
  reason: values.reason,
415
- operatorProceedLight: values['operator-proceed-light'],
416
- yes: values.yes === true,
417
367
  });
418
368
 
419
- if (gate.action === 'escalate-plan') {
369
+ if (gate.action !== 'proceed-light') {
420
370
  const envelope = buildEscalationTerminal({
421
371
  prompt: String(values.prompt),
422
372
  reasons: gate.outcome.reasons,
423
373
  });
424
374
  emitTerminalFn(envelope);
375
+ // The refusal is still telemetered (Story #4856): the ceilings stay
376
+ // recalibratable from evidence even now that only risk rules refuse.
377
+ await recordRefusalFn({ gate, amends: values.amends });
425
378
  Logger.warn(
426
379
  `[deliver-light] ESCALATED to /mandrel-plan — this session ENDS here; run ${envelope.nextCommand} in a FRESH session: ${gate.outcome.reasons.join('; ')}`,
427
380
  );
428
381
  return exitCodeForTerminal(envelope);
429
382
  }
430
383
 
431
- if (gate.action !== 'proceed-light') {
432
- emitFn(
433
- { mode: 'gate', action: gate.action, outcome: gate.outcome },
434
- values.pretty,
435
- );
436
- await recordRefusalFn({ gate, amends: values.amends });
437
- Logger.warn(
438
- `[deliver-light] gate did not proceed light (${gate.action}): ${gate.outcome.reasons.join('; ')}`,
439
- );
440
- return EXIT_NOT_PROCEED;
441
- }
442
-
443
- const override = gate.outcome.override ?? null;
384
+ const warnings = gate.outcome.warnings ?? [];
444
385
  const provider = createProviderFn(resolveConfigFn());
445
386
  const receipt = await createReceiptFn({
446
387
  provider,
@@ -450,7 +391,6 @@ export async function runGateMode(values, deps = {}) {
450
391
  ...parseCsvPaths(values.refactors),
451
392
  ],
452
393
  amends: values.amends ?? null,
453
- override,
454
394
  });
455
395
  emitFn(
456
396
  {
@@ -458,16 +398,14 @@ export async function runGateMode(values, deps = {}) {
458
398
  action: 'proceed-light',
459
399
  storyId: receipt.storyId,
460
400
  url: receipt.url,
461
- ...(override === null ? {} : { override }),
401
+ warnings,
462
402
  nextCommands: buildNextCommands(receipt.storyId),
463
403
  outcome: gate.outcome,
464
404
  },
465
405
  values.pretty,
466
406
  );
467
- if (override !== null) {
468
- Logger.warn(
469
- `[deliver-light] operator scope override recorded on Story #${receipt.storyId}: waived "${override.overriddenCode}" — ${override.recordedReason}`,
470
- );
407
+ for (const warning of warnings) {
408
+ Logger.warn(`[deliver-light] ⚠ ${warning}`);
471
409
  }
472
410
  Logger.info(
473
411
  `[deliver-light] receipt Story #${receipt.storyId} created — hand off to single-story-init.js.`,
@@ -488,7 +426,6 @@ async function main() {
488
426
  route: { type: 'string' },
489
427
  reason: { type: 'string' },
490
428
  amends: { type: 'string' },
491
- 'operator-proceed-light': { type: 'string' },
492
429
  yes: { type: 'boolean', default: false },
493
430
  backstop: { type: 'boolean', default: false },
494
431
  story: { type: 'string' },
@@ -269,7 +269,7 @@ runAsCli(import.meta.url, main, {
269
269
  source: 'evidence-gate',
270
270
  usage: {
271
271
  invocation:
272
- 'node .agents/scripts/evidence-gate.js --scope-id <id> --gate <name> [--standalone] [--no-evidence] [--cwd <path>] [--worktree <path>]',
272
+ 'node .agents/scripts/evidence-gate.js --scope-id <id> --gate <name> [--standalone] [--no-evidence] [--cwd <path>] [--worktree <path>] -- <cmd> [args...]',
273
273
  summary:
274
274
  'Run one named gate, reusing a prior evidence stamp for the same HEAD instead of re-running it.',
275
275
  flags: [
@@ -282,6 +282,22 @@ runAsCli(import.meta.url, main, {
282
282
  ],
283
283
  ['--cwd <path>', 'Repository root (default: project root).'],
284
284
  ['--worktree <path>', 'Worktree the gate runs in.'],
285
+ [
286
+ '-- <cmd> [args...]',
287
+ 'Required. The command this gate runs, passed through verbatim (never via a shell).',
288
+ ],
289
+ ],
290
+ notes: [
291
+ [
292
+ 'Everything after the first `--` is the gate. The stamp therefore describes',
293
+ 'what actually ran, whatever that is — which is how a project on any test',
294
+ 'runner earns the close `test` credit:',
295
+ '',
296
+ ' node .agents/scripts/evidence-gate.js --standalone --scope-id 4250 \\',
297
+ ' --gate lint --worktree .worktrees/story-4250 -- npm run lint',
298
+ ' node .agents/scripts/evidence-gate.js --standalone --scope-id 4250 \\',
299
+ ' --gate test --worktree .worktrees/story-4250 -- npm test',
300
+ ].join('\n'),
285
301
  ],
286
302
  },
287
303
  });
@@ -41,16 +41,29 @@ import path from 'node:path';
41
41
  import { AUDIT_LENSES } from '../audit-to-stories/audit-lenses.js';
42
42
  import { getPaths, PROJECT_ROOT, resolveConfig } from '../config-resolver.js';
43
43
  import { Logger } from '../Logger.js';
44
- import { estimateTokens } from '../orchestration/spec-spill.js';
45
44
  import {
46
45
  changeSetLacksSiblingTest,
47
46
  matchesAnyFilePattern,
48
47
  resolveLensTier,
49
48
  } from './selector.js';
50
49
 
50
+ /**
51
+ * Rough token estimate: ~4 characters per token. Deliberately cheap and
52
+ * deterministic — this budget is the one surviving ceiling that speaks in
53
+ * tokens (Story #5312 deleted the plan-time sizing ceilings that shared the
54
+ * estimator), so the approximation matters far less than the payload and
55
+ * the cap agreeing on one number.
56
+ *
57
+ * @param {string} text
58
+ * @returns {number}
59
+ */
60
+ function estimateTokens(text) {
61
+ return Math.ceil(String(text ?? '').length / 4);
62
+ }
63
+
51
64
  /**
52
65
  * Hard cap on the assembled checklist payload, in the ≈4-char/token estimate
53
- * shared with the rest of the hydrator ({@link estimateTokens}). Generous
66
+ * above ({@link estimateTokens}). Generous
54
67
  * relative to the real checklist sizes (each distilled lens checklist is
55
68
  * ~130–190 tokens, and at most the seven local lenses can match), so a normal
56
69
  * Story is never truncated — the cap is a safety ceiling against a pathological
@@ -0,0 +1,110 @@
1
+ /**
2
+ * lib/baselines/coverage-updater-cli.js — the `update-coverage-baseline` CLI's
3
+ * scope-flag reconciliation and its bespoke scorer.
4
+ *
5
+ * Story #5316: both lived inside `update-coverage-baseline.js#main`, which no
6
+ * test imports, so `main` scored CRAP 30 and the inlined scorer another 30 —
7
+ * two of the ten methods Story #5311's honest re-anchor made visible, both at
8
+ * 0% coverage.
9
+ *
10
+ * **Why this is NOT the `refresh-service.js` default scorer.** Story #4293's
11
+ * idiom — drop the bespoke scorer, let `refreshBaseline` resolve the canonical
12
+ * default — was the obvious move here, and `buildDefaultCoverageScorer` is
13
+ * behaviour-equivalent line for line. It was rejected for one reason: the
14
+ * default is silent where this one speaks. A missing or unreadable coverage
15
+ * artifact makes any scorer return `[]`, and `refresh-service.js` has no
16
+ * empty-rows guard, so a full-scope refresh then writes an emptied baseline at
17
+ * exit 0. This scorer's `[Coverage] ❌` line is currently the only signal an
18
+ * operator gets that it happened. Converging would have traded a CRAP row for
19
+ * a quieter failure. (The fail-open itself is real and wants its own Story —
20
+ * it changes `refreshBaseline` semantics for every caller.)
21
+ */
22
+
23
+ import { Logger } from '../Logger.js';
24
+ import { parseDiffScopeFlag } from './diff-scope-cli.js';
25
+
26
+ /**
27
+ * Reconcile the two scope flags.
28
+ *
29
+ * Throws when both are present: they describe incompatible scopes, and
30
+ * silently preferring one would write a baseline the operator did not ask for.
31
+ *
32
+ * @param {string[]} [argv]
33
+ * @returns {{fullScope: boolean, diffScopeRef: string|null}}
34
+ */
35
+ export function resolveCoverageUpdaterScope(argv = []) {
36
+ const diffScopeRef = parseDiffScopeFlag(argv);
37
+ const fullScope = argv.includes('--full-scope');
38
+ if (fullScope && diffScopeRef !== null) {
39
+ throw new Error(
40
+ '[Coverage] --full-scope is incompatible with --diff-scope; pick one',
41
+ );
42
+ }
43
+ return { fullScope, diffScopeRef };
44
+ }
45
+
46
+ /**
47
+ * Build the scorer `refreshBaseline` invokes.
48
+ *
49
+ * Reads `coverage-final.json`, narrows to the c8 include/exclude scope so the
50
+ * baseline records exactly the files coverage is measured over, and — in diff
51
+ * mode — narrows again to the service-resolved in-scope list so untouched rows
52
+ * are preserved rather than re-scored.
53
+ *
54
+ * The three collaborators are named seams (`rules/test-seams.md`) so a test
55
+ * drives the whole scorer with no coverage artifact and no `.c8rc.cjs` on
56
+ * disk.
57
+ *
58
+ * @param {string} cwd
59
+ * @param {{readCoverage: Function, loadScope: Function, buildScope: Function,
60
+ * score: Function, logger?: object}} deps
61
+ * @returns {(files: string[], opts: object) => object[]}
62
+ */
63
+ export function buildCoverageUpdaterScorer(
64
+ cwd,
65
+ { readCoverage, loadScope, buildScope, score, logger = Logger } = {},
66
+ ) {
67
+ return (files, opts) => {
68
+ const effectiveCwd = opts?.cwd ?? cwd;
69
+ let raw;
70
+ try {
71
+ raw = readCoverage(effectiveCwd);
72
+ } catch (err) {
73
+ // The only operator-facing signal that the refresh is about to record
74
+ // nothing — see this module's header.
75
+ logger.error(`[Coverage] ❌ ${err.message}`);
76
+ return [];
77
+ }
78
+
79
+ const c8Config = loadScope(effectiveCwd);
80
+ const scores = score({
81
+ raw,
82
+ cwd: effectiveCwd,
83
+ scope: buildScope({
84
+ include: c8Config.include ?? [],
85
+ exclude: c8Config.exclude ?? [],
86
+ }),
87
+ });
88
+
89
+ // In diff mode, further narrow to the service-resolved in-scope file list.
90
+ const inScope =
91
+ !opts?.fullScope && Array.isArray(files) && files.length > 0
92
+ ? new Set(files)
93
+ : null;
94
+
95
+ const rows = Object.entries(scores)
96
+ .filter(([relPath]) => inScope === null || inScope.has(relPath))
97
+ .map(([relPath, s]) => ({
98
+ path: relPath,
99
+ lines: s?.lines ?? 0,
100
+ branches: s?.branches ?? 0,
101
+ functions: s?.functions ?? 0,
102
+ }));
103
+
104
+ const fileCount = Object.keys(scores).length;
105
+ logger.info(
106
+ `[Coverage] Scored ${fileCount} file(s)${inScope ? ` (${rows.length} in scope)` : ''}.`,
107
+ );
108
+ return rows;
109
+ };
110
+ }
@@ -16,6 +16,7 @@ import {
16
16
  resolveEscomplexVersion,
17
17
  scanAndScore,
18
18
  } from '../crap-utils.js';
19
+ import { CYCLOMATIC_CEILING } from '../cyclomatic-ceiling.js';
19
20
  import { resolveCrapPreviewIncremental } from './crap-preview-incremental.js';
20
21
  import { resolveCrapEnvOverrides } from './env-overrides.js';
21
22
  import {
@@ -66,6 +67,26 @@ function hasCrapRegressions(result) {
66
67
  * }} opts
67
68
  * @returns {Promise<{ exitCode: number, envelope: object }>}
68
69
  */
70
+ /**
71
+ * The scanned methods at or over the fixed cyclomatic ceiling (Story #5313),
72
+ * as advisories: `quality-preview.js` lists them and exits 0 on them, so the
73
+ * reading reaches the author without the preview ever refusing a commit on
74
+ * complexity alone. Pure.
75
+ *
76
+ * @param {Array<{ file: string, method: string, startLine: number, cyclomatic: number }>} rows
77
+ * @returns {Array<{ file: string, method: string, startLine: number, cyclomatic: number }>}
78
+ */
79
+ export function listCyclomaticAdvisories(rows) {
80
+ return (Array.isArray(rows) ? rows : [])
81
+ .filter((r) => Number(r?.cyclomatic) >= CYCLOMATIC_CEILING)
82
+ .map(({ file, method, startLine, cyclomatic }) => ({
83
+ file,
84
+ method,
85
+ startLine,
86
+ cyclomatic,
87
+ }));
88
+ }
89
+
69
90
  export async function computeCrapPreviewScan({
70
91
  crap,
71
92
  cwd,
@@ -120,6 +141,10 @@ export async function computeCrapPreviewScan({
120
141
  newMethodCeiling,
121
142
  scopeInfo: { scope, diffRef },
122
143
  });
144
+ // Story #5313: a method at or over the cyclomatic ceiling is an ADVISORY
145
+ // on the preview — reported, never a verdict. The ratchet in
146
+ // `check-cyclomatic.js` owns enforcement.
147
+ envelope.cyclomaticAdvisories = listCyclomaticAdvisories(scan.rows);
123
148
  // Story #4866 (AC-5): above the drifted-row ratio the basis is self-
124
149
  // evidently unsound and every per-method verdict below it is an artefact of
125
150
  // a mis-keyed join. Say so once, by name, and fail open.