mandrel 2.65.0 → 2.67.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/.agents/agents/acceptance-critic.md +7 -7
  2. package/.agents/agents/auditor.md +17 -18
  3. package/.agents/agents/plan-critic.md +5 -5
  4. package/.agents/agents/story-worker.md +5 -5
  5. package/.agents/docs/agentrc-reference.json +2 -1
  6. package/.agents/docs/configuration.md +2 -1
  7. package/.agents/docs/execution-reference.md +27 -5
  8. package/.agents/docs/workflows.md +4 -2
  9. package/.agents/instructions.md +12 -13
  10. package/.agents/rules/ci-remediation.md +3 -3
  11. package/.agents/rules/gherkin-standards.md +3 -2
  12. package/.agents/rules/git-conventions-reference.md +17 -8
  13. package/.agents/rules/git-conventions.md +10 -8
  14. package/.agents/rules/testing-standards.md +8 -7
  15. package/.agents/runtime-deps.json +1 -1
  16. package/.agents/schemas/agentrc.schema.json +6 -1
  17. package/.agents/scripts/boot-sweep.js +97 -9
  18. package/.agents/scripts/bootstrap.js +94 -89
  19. package/.agents/scripts/{git-cleanup.js → clean-git.js} +2 -2
  20. package/.agents/scripts/clean-temp.js +54 -0
  21. package/.agents/scripts/clean-worktrees.js +593 -0
  22. package/.agents/scripts/drain-pending-cleanup.js +5 -4
  23. package/.agents/scripts/lib/baselines/duplication-scanner.js +17 -7
  24. package/.agents/scripts/lib/bootstrap/project-bootstrap.js +78 -78
  25. package/.agents/scripts/lib/clean-temp.js +440 -0
  26. package/.agents/scripts/lib/cli/standard-args.js +60 -76
  27. package/.agents/scripts/lib/cli-args.js +26 -0
  28. package/.agents/scripts/lib/config/gates/shared.js +3 -3
  29. package/.agents/scripts/lib/config-settings-schema-delivery.js +11 -2
  30. package/.agents/scripts/lib/feedback-loop/graduate-steps.js +205 -0
  31. package/.agents/scripts/lib/feedback-loop/graduator-core.js +47 -782
  32. package/.agents/scripts/lib/feedback-loop/graduator-gh.js +449 -0
  33. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  34. package/.agents/scripts/lib/observability/close-telemetry.js +330 -0
  35. package/.agents/scripts/lib/observability/runtime-friction.js +2 -0
  36. package/.agents/scripts/lib/observability/signal-validator.js +17 -5
  37. package/.agents/scripts/lib/observability/source-classifier.js +3 -1
  38. package/.agents/scripts/lib/orchestration/code-review.js +22 -0
  39. package/.agents/scripts/lib/orchestration/git-cleanup/phases/cli.js +1 -1
  40. package/.agents/scripts/lib/orchestration/plan-metrics.js +76 -63
  41. package/.agents/scripts/lib/orchestration/plan-runner/worktree-sweep.js +149 -97
  42. package/.agents/scripts/lib/orchestration/review-providers/review-provider-factory.js +23 -0
  43. package/.agents/scripts/lib/orchestration/run-epilogue.js +6 -0
  44. package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +2 -0
  45. package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +349 -263
  46. package/.agents/scripts/lib/orchestration/single-story-close/phases/options.js +21 -7
  47. package/.agents/scripts/lib/orchestration/single-story-close/phases/review-override.js +4 -0
  48. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +327 -314
  49. package/.agents/scripts/lib/orchestration/ticket-validator.js +19 -36
  50. package/.agents/scripts/lib/signals/detectors/common.js +63 -51
  51. package/.agents/scripts/lib/single-story-sweep.js +2 -2
  52. package/.agents/scripts/lib/temp-removal.js +110 -0
  53. package/.agents/scripts/lib/temp-retention.js +122 -73
  54. package/.agents/scripts/lib/transpile.js +28 -3
  55. package/.agents/scripts/lib/worktree/canonical-path.js +34 -0
  56. package/.agents/scripts/lib/worktree/lifecycle/reap.js +15 -4
  57. package/.agents/scripts/single-story-close.js +10 -2
  58. package/.agents/scripts/single-story-confirm-merge.js +267 -238
  59. package/.agents/scripts/single-story-init.js +120 -17
  60. package/.agents/skills/core/idea-refinement/SKILL.md +6 -6
  61. package/.agents/skills/stack/qa/qa-harness/SKILL.md +1 -2
  62. package/.agents/workflows/audit-architecture.md +5 -4
  63. package/.agents/workflows/audit-documentation.md +5 -5
  64. package/.agents/workflows/audit-performance.md +10 -10
  65. package/.agents/workflows/{git-cleanup.md → clean-git.md} +10 -10
  66. package/.agents/workflows/clean-temp.md +67 -0
  67. package/.agents/workflows/clean-worktrees.md +63 -0
  68. package/.agents/workflows/git-deliver.md +1 -1
  69. package/.agents/workflows/helpers/acceptance-self-eval.md +11 -10
  70. package/.agents/workflows/helpers/audit-lens-core.md +30 -57
  71. package/.agents/workflows/helpers/deliver-digest.md +2 -2
  72. package/.agents/workflows/helpers/deliver-reference.md +3 -1
  73. package/.agents/workflows/helpers/deliver-story-reference.md +2 -2
  74. package/.agents/workflows/helpers/deliver-story.md +6 -1
  75. package/.agents/workflows/helpers/parallel-tooling.md +16 -18
  76. package/.agents/workflows/mandrel-deliver.md +1 -1
  77. package/.agents/workflows/mandrel-plan.md +6 -5
  78. package/docs/CHANGELOG.md +39 -0
  79. package/lib/cli/guarded-sync.js +87 -0
  80. package/lib/cli/sync-agents.js +9 -92
  81. package/lib/cli/sync-commands.js +9 -101
  82. package/package.json +2 -2
@@ -12,8 +12,8 @@ effort: medium
12
12
 
13
13
  <!--
14
14
  Shared common core — byte-identical across every `.agents/agents/*.md` role
15
- context, ordered FIRST so all role boots share one prompt-cache prefix
16
- (prompt-cache is keyed on the exact byte prefix; the role delta comes last).
15
+ context, ordered FIRST so every role binds the same baseline rules (the
16
+ roles pin different effort levels and share no cache prefix; delta last).
17
17
  Edit it in every role file at once —
18
18
  tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
19
19
  security-baseline stays inviolable and single-sourced — @-import it, never
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
31
31
  the step-by-step. This shared core binds every role:
32
32
 
33
33
  - **Non-interactive.** You have no input channel mid-run. Never ask
34
- clarifying questions — pick the narrowest reasonable interpretation of
35
- your charter, and when you cannot proceed, take your role's
36
- blocked/failure path instead of stalling.
34
+ clarifying questions — take the reading your charter most directly
35
+ supports, name it in your return, and when you cannot proceed, take
36
+ your role's blocked/failure path instead of stalling.
37
37
  - **Absolute paths only.** Your shell's working directory is not guaranteed
38
38
  to persist between calls; pass absolute paths for every file and script.
39
39
  - **Anti-thrashing.** When the same error class recurs despite the same fix,
@@ -98,8 +98,8 @@ For each acceptance item:
98
98
 
99
99
  ## Verdict schema (MUST)
100
100
 
101
- Write **one** verdict file under `temp/` (e.g.
102
- `temp/acceptance-verdict-<storyId>-r<round>.json`) conforming to
101
+ Write **one** verdict file under `temp/scratch/story-<storyId>/` (e.g.
102
+ `temp/scratch/story-<storyId>/acceptance-verdict-r<round>.json`) conforming to
103
103
  [`acceptance-eval-verdict.schema.json`](../schemas/acceptance-eval-verdict.schema.json):
104
104
  one `criteria[]` record per acceptance item, in acceptance-array order, with
105
105
  `index` being the criterion's position in that array.
@@ -12,8 +12,8 @@ effort: medium
12
12
 
13
13
  <!--
14
14
  Shared common core — byte-identical across every `.agents/agents/*.md` role
15
- context, ordered FIRST so all role boots share one prompt-cache prefix
16
- (prompt-cache is keyed on the exact byte prefix; the role delta comes last).
15
+ context, ordered FIRST so every role binds the same baseline rules (the
16
+ roles pin different effort levels and share no cache prefix; delta last).
17
17
  Edit it in every role file at once —
18
18
  tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
19
19
  security-baseline stays inviolable and single-sourced — @-import it, never
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
31
31
  the step-by-step. This shared core binds every role:
32
32
 
33
33
  - **Non-interactive.** You have no input channel mid-run. Never ask
34
- clarifying questions — pick the narrowest reasonable interpretation of
35
- your charter, and when you cannot proceed, take your role's
36
- blocked/failure path instead of stalling.
34
+ clarifying questions — take the reading your charter most directly
35
+ supports, name it in your return, and when you cannot proceed, take
36
+ your role's blocked/failure path instead of stalling.
37
37
  - **Absolute paths only.** Your shell's working directory is not guaranteed
38
38
  to persist between calls; pass absolute paths for every file and script.
39
39
  - **Anti-thrashing.** When the same error class recurs despite the same fix,
@@ -116,13 +116,12 @@ and a surviving **Critical** halts the delivery gate:
116
116
  - **Info** — the floor: a grounded observation asking for no scheduled work
117
117
  (accepts `Informational`). Never a home for findings that fail the bar below.
118
118
 
119
- ## Self-cross-check bar (mandatory before you write the report)
119
+ ## Self-cross-check bar
120
120
 
121
- You are your own adversarial reviewer. After drafting the Detailed Findings and
122
- **before** writing the artifact, re-open every finding and keep it only when
123
- **all** hold: a **grounded** `path:line` you actually read; **reproducible
124
- evidence** (a tool reading, a quoted snippet, or a specific standard it
125
- violates) — never "this looks wrong"; **in-scope** under the scope filter; and
121
+ A finding goes in the report only when **all** hold: a **grounded**
122
+ `path:line` you actually read; **reproducible evidence** (a tool reading, a
123
+ quoted snippet, or a specific standard it violates) — never "this looks
124
+ wrong"; **in-scope** under the scope filter; and
126
125
  an **actionable** recommendation. Drop anything resting on a sanctioned test
127
126
  seam, an entry point / public API surface, dynamic/framework reachability, an
128
127
  intentional documented deviation, or a formatter-governed style nit.
@@ -136,14 +135,14 @@ Beside it, carry one machine-readable tally of the findings you kept —
136
135
  included, `Info` never counted. `audit-to-stories` cross-checks that line
137
136
  against its parse and refuses a report whose tally is missing or wrong.
138
137
 
139
- ## Fan-out (heavyweight lenses)
138
+ ## Fan-out (operator-requested only)
140
139
 
141
- When your caller dispatches you for a single dimension of a heavyweight lens
142
- (`audit-architecture`, `audit-performance`, `audit-documentation`), audit only
143
- that dimension and return its findings; the parent merges the per-dimension
144
- results under this self-cross-check bar. Within the supported nesting-depth
145
- budget you may apply `parallel-tooling.md` Rule 3 to your own independent
146
- sub-units.
140
+ By default you audit the whole lens. When your caller dispatches you for a
141
+ single dimension — which it does only on an explicit operator request for
142
+ per-dimension fan-out — audit only that dimension and return its findings;
143
+ the caller merges the per-dimension results under this self-cross-check bar.
144
+ You never dispatch sub-agents of your own: no nested fan-out, whatever the
145
+ lens's size.
147
146
 
148
147
  ## Return contract
149
148
 
@@ -12,8 +12,8 @@ effort: medium
12
12
 
13
13
  <!--
14
14
  Shared common core — byte-identical across every `.agents/agents/*.md` role
15
- context, ordered FIRST so all role boots share one prompt-cache prefix
16
- (prompt-cache is keyed on the exact byte prefix; the role delta comes last).
15
+ context, ordered FIRST so every role binds the same baseline rules (the
16
+ roles pin different effort levels and share no cache prefix; delta last).
17
17
  Edit it in every role file at once —
18
18
  tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
19
19
  security-baseline stays inviolable and single-sourced — @-import it, never
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
31
31
  the step-by-step. This shared core binds every role:
32
32
 
33
33
  - **Non-interactive.** You have no input channel mid-run. Never ask
34
- clarifying questions — pick the narrowest reasonable interpretation of
35
- your charter, and when you cannot proceed, take your role's
36
- blocked/failure path instead of stalling.
34
+ clarifying questions — take the reading your charter most directly
35
+ supports, name it in your return, and when you cannot proceed, take
36
+ your role's blocked/failure path instead of stalling.
37
37
  - **Absolute paths only.** Your shell's working directory is not guaranteed
38
38
  to persist between calls; pass absolute paths for every file and script.
39
39
  - **Anti-thrashing.** When the same error class recurs despite the same fix,
@@ -9,8 +9,8 @@ description: >-
9
9
 
10
10
  <!--
11
11
  Shared common core — byte-identical across every `.agents/agents/*.md` role
12
- context, ordered FIRST so all role boots share one prompt-cache prefix
13
- (prompt-cache is keyed on the exact byte prefix; the role delta comes last).
12
+ context, ordered FIRST so every role binds the same baseline rules (the
13
+ roles pin different effort levels and share no cache prefix; delta last).
14
14
  Edit it in every role file at once —
15
15
  tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
16
16
  security-baseline stays inviolable and single-sourced — @-import it, never
@@ -28,9 +28,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
28
28
  the step-by-step. This shared core binds every role:
29
29
 
30
30
  - **Non-interactive.** You have no input channel mid-run. Never ask
31
- clarifying questions — pick the narrowest reasonable interpretation of
32
- your charter, and when you cannot proceed, take your role's
33
- blocked/failure path instead of stalling.
31
+ clarifying questions — take the reading your charter most directly
32
+ supports, name it in your return, and when you cannot proceed, take
33
+ your role's blocked/failure path instead of stalling.
34
34
  - **Absolute paths only.** Your shell's working directory is not guaranteed
35
35
  to persist between calls; pass absolute paths for every file and script.
36
36
  - **Anti-thrashing.** When the same error class recurs despite the same fix,
@@ -87,7 +87,8 @@
87
87
  "orchestrationLogs": true,
88
88
  "validationEvidence": true,
89
89
  "auditResults": true,
90
- "planDirs": true
90
+ "planDirs": true,
91
+ "scratch": true
91
92
  }
92
93
  },
93
94
  "deliverRunner": {
@@ -139,13 +139,14 @@ Everything `/mandrel-deliver` and `single-story-close` consume: worktree isolati
139
139
  | `execution.fullSuiteLock` | No | `boolean` | `true` | Serialize full-suite spawns (`npm test` / `npm run test:coverage`) behind a host-level advisory lock, so two concurrent deliveries on one checkout do not run two suites against the same cores. Best-effort: a wait that expires spawns anyway, so the lock can never fail a delivery. Set false — or export `MANDREL_FULL_SUITE_LOCK=0` for one invocation — to disable. |
140
140
  | `docsFreshness` | No | `object` | — | Documentation-freshness scope: the files a change of consequence is expected to touch. Read by the audit-documentation lens to seed its target set; no delivery gate enforces it. |
141
141
  | `docsFreshness.paths` | No | `array<string>` | `["README.md"]` | Repo-relative documentation paths the audit-documentation lens adds to its target set. |
142
- | `tempRetention` | No | `object` | — | Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so operator scratch files under tempRoot are reported with their size and left alone. signals.ndjson is never purged by any path. |
142
+ | `tempRetention` | No | `object` | — | Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so unrecognized files under tempRoot are reported with their size and left alone (`/clean-temp` is the operator path for them). signals.ndjson is never purged by any path. |
143
143
  | `tempRetention.enabled` | No | `boolean` | `true` | Master switch. Default true — reclaiming a landed Story's gate transcripts and validation evidence is the behaviour, and this knob turns it off. When false every purge path is a reported no-op. |
144
144
  | `tempRetention.classes` | No | `object` | — | Per-class opt-out. Each defaults to true; set one false to keep that family while the rest are purged. |
145
145
  | `tempRetention.classes.orchestrationLogs` | No | `boolean` | `true` | <tempRoot>/orchestration/*.log — close gate transcripts and terse-result detail dumps. |
146
146
  | `tempRetention.classes.validationEvidence` | No | `boolean` | `true` | Per-Story validation-evidence.json, lifecycle.ndjson, and manifest.md under the standalone and per-run story trees. |
147
147
  | `tempRetention.classes.auditResults` | No | `boolean` | `true` | <tempRoot>/audits/ — audit lens reports. |
148
148
  | `tempRetention.classes.planDirs` | No | `boolean` | `true` | <tempRoot>/plan-<slug>/ — abandoned plan authoring dirs. Age-floored only; the current run is always excluded. |
149
+ | `tempRetention.classes.scratch` | No | `boolean` | `true` | <tempRoot>/scratch/ — agent-authored scratch. `scratch/story-<id>/` is purged when that Story lands; any other `scratch/` entry is age-floored. |
149
150
  | `deliverRunner` | No | `object` | — | Bounded-concurrency knob for the /mandrel-deliver fan-out. |
150
151
  | `deliverRunner.concurrencyCap` | No | `integer` | `3` | Maximum ready Stories dispatched by /mandrel-deliver at once. Default 3. Moderate by design — keeps host-quota consumption predictable while allowing a small ready-set fan-out. Set 1 for strictly sequential delivery; raise further on hosts with adequate parallel-agent quota. See deliver.md for the sequencing model and throughput tradeoff. |
151
152
  | `deliverRunner.footprintGuard` | No | `"enforce"` \| `"advisory"` | `"enforce"` | How a file-footprint collision affects dispatch. 'enforce' (default, and the behaviour to keep unless you have a reason) withholds a Story whose footprint races a peer admitted this beat or one still in flight — the guard encodes delivery-time-only knowledge (open implementation windows, foreign leases, ground that moved since planning) that no depends_on edge can carry. 'advisory' still DETECTS every collision and reports each would-be withhold in the tick envelope, but lets dispatch follow the declared depends_on edges alone — a deliberate throughput trade for a run whose ordering is fully declared. See stories-wave-tick.js and helpers/deliver-reference.md. |
@@ -72,11 +72,12 @@ and schema mechanics are in [§ Friction telemetry](#friction-telemetry) above.
72
72
  ## FinOps & token budgeting (economic guardrails)
73
73
 
74
74
  Mandrel does **not** enforce live LLM spend from response metadata. It bounds
75
- two things, both **fixed framework constants** rather than operator knobs, and
76
- both **fail closed**: the assembled `/mandrel-plan` context envelope, and plan-time
77
- Story sizing. Your host runtime (editor / CLI) owns session quota and hard
78
- stops. Consult this section when reasoning about why `/mandrel-plan` refused an
79
- over-ceiling envelope or an over-budget Story count.
75
+ one thing, a **fixed framework constant** rather than an operator knob that
76
+ **fails closed**: the assembled `/mandrel-plan` context envelope. Plan-time
77
+ Story sizing is not bounded (retired by Story #5312). Your host runtime
78
+ (editor / CLI) owns session quota and hard stops. Consult this section when
79
+ reasoning about why `/mandrel-plan` refused an over-ceiling envelope, or when
80
+ choosing the session effort and model for a command.
80
81
 
81
82
  > **There is no configurable context budget.** `planning.context.maxBytes` /
82
83
  > `summaryMode` were removed outright in Story #4541, along with the
@@ -126,3 +127,24 @@ over-ceiling envelope or an over-budget Story count.
126
127
  nothing at plan time scores its authored mass.
127
128
  - **Host runtime**: session billing, quota exhaustion, and operator overrides
128
129
  are enforced by your provider (e.g. Claude Code), not by Mandrel scripts.
130
+
131
+ ### Session effort and model
132
+
133
+ Effort is the operator's dial, set once per session. Pick it deliberately up
134
+ front, because changing effort mid-session invalidates the prompt cache for
135
+ everything that follows.
136
+
137
+ - **Recommended session effort.** `medium` for `/mandrel-deliver`: Stories
138
+ are well-scoped by construction, so delivery rarely needs more. `high` for
139
+ `/mandrel-plan`: a Spec defect costs a redraft or a blocked Story
140
+ downstream, which is dearer than the planning turn.
141
+ - **Role agents.** `story-worker` declares no effort and inherits the
142
+ session's (Story #5426). The evaluator roles (`acceptance-critic`,
143
+ `plan-critic`, `auditor`) pin `medium`.
144
+ - **Where pins live.** Effort and model pins belong only on role agents under
145
+ `.agents/agents/`, never in workflow or command frontmatter: a command that
146
+ pinned its own effort would change effort mid-session and break the prompt
147
+ cache.
148
+ - **Escalation is an operator step.** The Agent tool takes no per-call
149
+ effort, so no workflow can escalate on its own. Before resuming a blocked
150
+ Story, raise session effort one step; raise effort before switching models.
@@ -32,7 +32,7 @@ by `node .agents/scripts/generate-workflows-doc.js`; `npm run docs:check`
32
32
  fails when it drifts from the on-disk workflow set. To change a command’s
33
33
  description, edit the workflow file’s front-matter and regenerate.
34
34
 
35
- ## Commands (29)
35
+ ## Commands (31)
36
36
 
37
37
  | Command | Description |
38
38
  | --- | --- |
@@ -55,7 +55,9 @@ description, edit the workflow file’s front-matter and regenerate.
55
55
  | `/audit-sre` | "Audit production-readiness for a release candidate: SLOs, observability, runbooks, error budgets, and rollback paths." |
56
56
  | `/audit-to-stories` | Convert findings produced by the audit-\* workflows into actionable GitHub Stories. Reads temp/audits/audit-\*-results.md, groups findings cross-audit, deduplicates against existing Issues by fingerprint, and either chains into /mandrel-plan --seed-file or opens standalone Stories. |
57
57
  | `/audit-ux-ui` | Audit UX/UI consistency and design system adherence |
58
- | `/git-cleanup` | Tidy the local checkout in four phases: fast-forward `main`, prune stale remote-tracking refs, sweep merged branches (squash-aware), and triage `git stash` entries — each step gated by operator confirmation. |
58
+ | `/clean-git` | Tidy the local checkout in four phases: fast-forward `main`, prune stale remote-tracking refs, sweep merged branches (squash-aware), and triage `git stash` entries — each step gated by operator confirmation. |
59
+ | `/clean-temp` | Clear the temp-tree backlog the land-time purge cannot attribute: sort every top-level entry under the project's tempRoot into framework, closed-issue, aged and kept buckets, preview by default, and delete only confirmed buckets. |
60
+ | `/clean-worktrees` | Reclaim disk from dead worktrees: list every worktree of this project as a removal candidate (closed Story, merged branch, orphaned directory, detached HEAD) or as kept with a reason, then remove candidates only on `--execute`. |
59
61
  | `/git-deliver` | Single ad-hoc delivery command for working-tree changes. Detects the git setup and escalates to the right terminal step — commit only, commit + push, or commit + push + open a PR with native auto-merge — picking the default from observable state and letting flags pin any level explicitly. Replaces the retired git-commit-all, git-push, and git-pr-all trio. |
60
62
  | `/mandrel-deliver` | Unified delivery entry point. Takes Story ids or a plain-language prompt, derives which path the work belongs on, and lands it via the single deliver-story engine — story-<id> → PR → main. |
61
63
  | `/mandrel-plan` | Unified planning entry point. Interrogate → author → persist. Emits one Story by default; splits into N>1 only under the default-single split policy. |
@@ -118,7 +118,7 @@ truncates with a note naming what was cut:
118
118
 
119
119
  ## 3. Core Philosophy
120
120
 
121
- 1. **Context First.** **Digest-first reading (Story #4433):** never
121
+ 1. **Context First.** **Digest-first reading:** never
122
122
  ingest the whole `project.docsContextFiles` set up front — read the
123
123
  docs digest and pull files on demand at the section it names. No
124
124
  digest (ad hoc task, `docsContextFiles` unset, null `docsDigestPath`)
@@ -127,8 +127,10 @@ truncates with a note naming what was cut:
127
127
  present. Always read the current Story's body (`## Spec` +
128
128
  `acceptance[]` / `verify[]`); prefer targeted retrieval over broad
129
129
  reads.
130
- 2. **Plan First.** For non-trivial tasks (3+ steps or architectural
131
- decisions), update the Story's `## Spec` via `/mandrel-plan` before code.
130
+ 2. **Plan First.** Planned work carries its plan in the Story's
131
+ `## Spec`, authored via `/mandrel-plan`; an unplanned prompt takes
132
+ `/mandrel-deliver`'s light path, which escalates to `/mandrel-plan`
133
+ when its gate trips.
132
134
  3. **Artifacts over Chat.** Write test/build/debug output to log
133
135
  files, not into chat.
134
136
  4. **Idempotency.** Scripts must be safe to run repeatedly.
@@ -137,16 +139,13 @@ truncates with a note naming what was cut:
137
139
 
138
140
  ## 4. Execution & Quality Discipline
139
141
 
140
- - **Re-Plan on Failure.** If a strategy fails, STOP and re-plan.
141
142
  - **Subagent Strategy.** Each spawn re-pays the full always-loaded
142
143
  context — a cost decision. Prefer inline search for small lookups;
143
144
  spawn only when the work justifies replicating context. One objective
144
145
  per subagent; depth compounds the cost (every nested level re-pays).
145
- - **Anti-Laziness / No Dead Code.** NEVER use placeholder comments like
146
- `// ... existing code ...`; every edit must leave complete, runnable code.
147
- Remove unused imports, commented-out code, and dead branches before
146
+ - **No Dead Code.** Every edit leaves complete, runnable code. Remove
147
+ unused imports, commented-out code, and dead branches before
148
148
  finalizing.
149
- - **Verification.** Include explicit verification steps in every plan.
150
149
 
151
150
  ---
152
151
 
@@ -182,7 +181,8 @@ never delivered): `/mandrel-plan` offers one above 2 Stories and
182
181
 
183
182
  All temporary files, scratch scripts, and intermediate outputs MUST
184
183
  live in the gitignored workspace-root `/temp/` directory — do NOT commit
185
- anything under it.
184
+ anything under it. Put ad-hoc scratch in `temp/scratch/story-<id>/` (or
185
+ `temp/scratch/` with no Story), the layout the temp purge reaps.
186
186
 
187
187
  ---
188
188
 
@@ -191,7 +191,6 @@ anything under it.
191
191
  `/mandrel-plan` sizes each Story as a **capability slice a frontier model
192
192
  delivers and self-verifies in one pass** — a broad footprint is normal
193
193
  when the change is cohesive, and no plan-time ceiling scores it; do not
194
- re-slice it into per-module fragments. On an out-of-scope task: **plan
195
- first** in numbered cohesive sub-steps, **commit incrementally** per
196
- sub-step, and **fail fast** — STOP and report if any sub-step fails
197
- validation.
194
+ re-slice it into per-module fragments. On an out-of-scope task,
195
+ **commit incrementally** per cohesive sub-step, and when a sub-step
196
+ fails validation, stop and apply § 1.I (re-plan or yield).
@@ -137,9 +137,9 @@ run — a blind fix to a suite nobody exercised is how the gap compounds.
137
137
 
138
138
  `capacity` and `unreproducible-tier` name failures that are proven properties
139
139
  of the **environment**: no commit on the branch can move the head SHA to clear
140
- them, so the no-rerun rule used to strand a correct delivery until a human
141
- cleared it by hand. Those two verdicts — and only those two — now buy exactly
142
- one rerun:
140
+ them, so without an allowance the no-rerun rule would strand a correct
141
+ delivery until a human cleared it by hand. Those two verdicts — and only those
142
+ two — buy exactly one rerun:
143
143
 
144
144
  1. Reach the verdict with its required readings (§ above). A green on re-run is
145
145
  never one of those readings.
@@ -143,8 +143,9 @@ authored:
143
143
  widen the regex, updating every call site in the same PR.
144
144
  4. **Add a new definition only when no reasonable match exists**, in the
145
145
  correct domain directory. Never copy-paste a step implementation to support
146
- a paraphrased scenario, and never author new step definitions during
147
- scenario authoring — record the missing step as a named gap instead.
146
+ a paraphrased scenario. A prose-only authoring pass (one whose scope
147
+ excludes step-definition code) records the missing step as a named gap
148
+ instead of writing it.
148
149
 
149
150
  When a step is superseded, mark it deprecated and migrate every call site in
150
151
  the same PR; do not leave two near-identical steps live.
@@ -15,7 +15,9 @@ Mandrel ships as the `mandrel` npm package, whose consumers pin an
15
15
  exact lockfile version; they opt into breaks at upgrade time. Operator policy
16
16
  for any contract change (config shape, baseline shape, schema, lifecycle
17
17
  payload, ticket label, dispatch artifact, public API of a script) is
18
- therefore:
18
+ therefore as follows. It governs Mandrel's own framework contracts; a
19
+ consumer's product API keeps the expand–contract rule in
20
+ [`api-conventions.md`](api-conventions.md).
19
21
 
20
22
  1. **Hard cutovers only.** Contract changes ship as a single in-tree
21
23
  migration of every producer and consumer. There is no parallel
@@ -94,13 +96,13 @@ signature is worth naming:
94
96
 
95
97
  **Invariant (stated in the core): the delivering flow owns tidying the local
96
98
  checkout — reaping its own merged refs and fast-forwarding the base branch.
97
- `/git-cleanup` is a recovery tool, not a routine chore.** The outcome every
99
+ `/clean-git` is a recovery tool, not a routine chore.** The outcome every
98
100
  delivering flow (`/mandrel-deliver`, `/git-deliver`) guarantees, with the mechanics
99
- owned by `boot-sweep.js` / `git-cleanup.js`:
101
+ owned by `boot-sweep.js` / `clean-git.js`:
100
102
 
101
103
  - **`main` is fast-forwarded** by the flow itself in its cleanup phase, so the
102
104
  next init seeds from a current base. No workflow ends by telling the operator
103
- to run `/git-cleanup` to catch up.
105
+ to run `/clean-git` to catch up.
104
106
  - **Merged local refs are reaped** at the next workflow boot's protected sweep
105
107
  (`boot-sweep.js`) — every local branch whose PR is already merged, skipping
106
108
  any candidate with unpushed work, a dirty worktree, or a still-open parent
@@ -110,9 +112,9 @@ owned by `boot-sweep.js` / `git-cleanup.js`:
110
112
  weaker content-equivalence signal (`detectedBy: 'content-merged'` — content
111
113
  already landed in the base by another route, with no merged PR or git
112
114
  ancestry of its own) is **never** reaped by the boot sweep; it is surfaced
113
- under `contentMerged` for the operator to send to `/git-cleanup` for a
115
+ under `contentMerged` for the operator to send to `/clean-git` for a
114
116
  confirmed, eyeballed reap.
115
- - **`/git-cleanup` is recovery, not routine.** Run it by hand only for a state
117
+ - **`/clean-git` is recovery, not routine.** Run it by hand only for a state
116
118
  the automated hygiene does not cover — triaging stashes, reaping across
117
119
  non-standard namespaces, or `--remote` pruning after a force-push. Reaching
118
120
  for it after every routine delivery signals the owning flow's hygiene step
@@ -154,10 +156,10 @@ different hazards:
154
156
 
155
157
  ## Meta Labels (Retrospective Signal Routing)
156
158
 
157
- Two `meta::*` labels route retrospective signals into durable substrates so
159
+ Three `meta::*` labels route retrospective signals into durable substrates so
158
160
  the `/mandrel-plan` Phase 0 fetcher (see
159
161
  [`prior-feedback-fetcher.js`](../scripts/lib/feedback-loop/prior-feedback-fetcher.js))
160
- can surface open feedback issues to the planner. Both labels live in
162
+ can surface open feedback issues to the planner. All three live in
161
163
  [`label-constants.js`](../scripts/lib/label-constants.js) under the
162
164
  `META_LABELS` export — reference them by symbol from scripts rather than
163
165
  hard-coding the string.
@@ -180,3 +182,10 @@ project-local automation). The work is scoped to the consumer's
180
182
  framework changes. Issues that span both axes should carry both labels —
181
183
  `fetchPriorFeedback` dedupes by issue number so a dual-labeled issue
182
184
  appears exactly once in the planner context.
185
+
186
+ ### `meta::platform-gap`
187
+
188
+ Apply this label to a GitHub issue whose fault lies in a shared base
189
+ config, runner fleet, or cross-repo toolchain that neither the framework
190
+ nor the consumer owns — the `--owner platform` bucket of
191
+ [`ci-remediation.md`](ci-remediation.md).
@@ -16,9 +16,8 @@ Every Story lands on a dedicated **Story branch** named
16
16
  `story-<storyId>`, seeded from `project.baseBranch` (`main` by default),
17
17
  isolated in its own worktree at `.worktrees/story-<id>/`. The runtime
18
18
  owns both via `single-story-init.js`; agents commit there only. Close
19
- opens a PR against `main` (squash + required checks). No `epic/<id>`
20
- integration branch, no `--no-ff` wave merge, no child tickets: commits
21
- land on `story-<storyId>` directly, the
19
+ opens a PR against `main` (squash + required checks). Commits land on
20
+ `story-<storyId>` directly, the
22
21
  subject referencing the Story via `(refs #<storyId>)` — see
23
22
  [`.agents/instructions.md` § 5.B](../instructions.md).
24
23
 
@@ -38,15 +37,18 @@ subject referencing the Story via `(refs #<storyId>)` — see
38
37
 
39
38
  ## Push Validation & Reliability (MUSTs)
40
39
 
41
- 1. Run the configured validation commands locally **before** `git push`.
40
+ 1. Validate locally **before** `git push`. On a Story branch that is the
41
+ one credited suite run (`deliver-digest.md` § 5); close runs every
42
+ other gate, so do not pre-run them. Elsewhere, run the configured
43
+ validation commands.
42
44
  2. Do NOT assume a push succeeded unless the output confirms the remote
43
45
  ref was updated (`[new branch]`, `[up to date]`, `... -> ...`).
44
46
  3. If a `pre-push` hook rejects, fix the cause and create a NEW follow-up
45
47
  commit — never amend the rejected commit.
46
48
  4. **Never bypass hooks** (`--no-verify`, `--no-gpg-sign`, …) without
47
- explicit operator authorization. The one recognized exception — a
48
- Biome zero-match failure under a harness-managed worktree path — is a
49
- consumer-tooling gap, **not** authorization; see
49
+ explicit operator authorization. A Biome zero-match failure under a
50
+ harness-managed worktree path is a known false negative — a
51
+ consumer-tooling gap, **not** authorization to bypass; see
50
52
  [`git-conventions-reference.md` § Push Validation](git-conventions-reference.md).
51
53
 
52
54
  ## Local checkout hygiene
@@ -54,7 +56,7 @@ subject referencing the Story via `(refs #<storyId>)` — see
54
56
  **The delivering flow owns tidying the local checkout** — it
55
57
  fast-forwards the base branch itself and reaps its own merged refs on
56
58
  the next workflow boot (the `boot-sweep.js` protected sweep).
57
- `/git-cleanup` is a recovery tool, not a routine chore — never end a
59
+ `/clean-git` is a recovery tool, not a routine chore — never end a
58
60
  workflow by telling the operator to run it. Scope rules and the
59
61
  shared-checkout contention guard:
60
62
  [`git-conventions-reference.md` § Local checkout hygiene](git-conventions-reference.md).
@@ -156,13 +156,14 @@ behaviour (`sets status to completed`, not `works`).
156
156
 
157
157
  A file that passes alone but fails inside the full `npm test` suite is **test
158
158
  pollution** — one test leaks shared state (env vars, temp files, the
159
- mock-module registry, global singletons) and a later test trips on it. Reach
160
- for `npm run test:isolate` before manually bisecting: it runs every matching
161
- file individually under `--test-concurrency=1`, then all together, flags files
162
- that pass alone but fail in the suite (**flippers**), binary-bisects the
163
- smallest reproducing subset, and reports any file that exited with leftover
164
- `process.env` mutations. The fix is almost always missing teardown — wrap the
165
- mutation in a `t.before` / `t.after` pair, or restore the prior value in
159
+ mock-module registry, global singletons) and a later test trips on it. Run
160
+ every matching file alone, then all together; the files that pass alone but
161
+ fail in the suite (**flippers**) bound the search, and bisecting them finds
162
+ the smallest reproducing subset. Mandrel's own repository automates this as
163
+ `npm run test:isolate`, which also reports leftover `process.env`
164
+ mutations; a consumer uses its runner's equivalent. The fix is almost
165
+ always missing teardown — wrap the mutation in a `t.before` / `t.after`
166
+ pair, or restore the prior value in
166
167
  `try` / `finally`.
167
168
 
168
169
  For browser-based changes, pair the cycle with runtime verification via Chrome
@@ -18,6 +18,6 @@
18
18
  "chokidar": "^5.0.0",
19
19
  "jscpd": "^4.0.0",
20
20
  "knip": "^6.17.1",
21
- "typescript": ">=5.0.0"
21
+ "typescript": ">=5.0.0 <7"
22
22
  }
23
23
  }
@@ -386,7 +386,7 @@
386
386
  },
387
387
  "tempRetention": {
388
388
  "type": "object",
389
- "description": "Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so operator scratch files under tempRoot are reported with their size and left alone. signals.ndjson is never purged by any path.",
389
+ "description": "Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so unrecognized files under tempRoot are reported with their size and left alone (`/clean-temp` is the operator path for them). signals.ndjson is never purged by any path.",
390
390
  "properties": {
391
391
  "enabled": {
392
392
  "type": "boolean",
@@ -416,6 +416,11 @@
416
416
  "type": "boolean",
417
417
  "description": "<tempRoot>/plan-<slug>/ — abandoned plan authoring dirs. Age-floored only; the current run is always excluded.",
418
418
  "default": true
419
+ },
420
+ "scratch": {
421
+ "type": "boolean",
422
+ "description": "<tempRoot>/scratch/ — agent-authored scratch. `scratch/story-<id>/` is purged when that Story lands; any other `scratch/` entry is age-floored.",
423
+ "default": true
419
424
  }
420
425
  },
421
426
  "additionalProperties": false