mandrel 2.65.0 → 2.67.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/acceptance-critic.md +7 -7
- package/.agents/agents/auditor.md +17 -18
- package/.agents/agents/plan-critic.md +5 -5
- package/.agents/agents/story-worker.md +5 -5
- package/.agents/docs/agentrc-reference.json +2 -1
- package/.agents/docs/configuration.md +2 -1
- package/.agents/docs/execution-reference.md +27 -5
- package/.agents/docs/workflows.md +4 -2
- package/.agents/instructions.md +12 -13
- package/.agents/rules/ci-remediation.md +3 -3
- package/.agents/rules/gherkin-standards.md +3 -2
- package/.agents/rules/git-conventions-reference.md +17 -8
- package/.agents/rules/git-conventions.md +10 -8
- package/.agents/rules/testing-standards.md +8 -7
- package/.agents/runtime-deps.json +1 -1
- package/.agents/schemas/agentrc.schema.json +6 -1
- package/.agents/scripts/boot-sweep.js +97 -9
- package/.agents/scripts/bootstrap.js +94 -89
- package/.agents/scripts/{git-cleanup.js → clean-git.js} +2 -2
- package/.agents/scripts/clean-temp.js +54 -0
- package/.agents/scripts/clean-worktrees.js +593 -0
- package/.agents/scripts/drain-pending-cleanup.js +5 -4
- package/.agents/scripts/lib/baselines/duplication-scanner.js +17 -7
- package/.agents/scripts/lib/bootstrap/project-bootstrap.js +78 -78
- package/.agents/scripts/lib/clean-temp.js +440 -0
- package/.agents/scripts/lib/cli/standard-args.js +60 -76
- package/.agents/scripts/lib/cli-args.js +26 -0
- package/.agents/scripts/lib/config/gates/shared.js +3 -3
- package/.agents/scripts/lib/config-settings-schema-delivery.js +11 -2
- package/.agents/scripts/lib/feedback-loop/graduate-steps.js +205 -0
- package/.agents/scripts/lib/feedback-loop/graduator-core.js +47 -782
- package/.agents/scripts/lib/feedback-loop/graduator-gh.js +449 -0
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/observability/close-telemetry.js +330 -0
- package/.agents/scripts/lib/observability/runtime-friction.js +2 -0
- package/.agents/scripts/lib/observability/signal-validator.js +17 -5
- package/.agents/scripts/lib/observability/source-classifier.js +3 -1
- package/.agents/scripts/lib/orchestration/code-review.js +22 -0
- package/.agents/scripts/lib/orchestration/git-cleanup/phases/cli.js +1 -1
- package/.agents/scripts/lib/orchestration/plan-metrics.js +76 -63
- package/.agents/scripts/lib/orchestration/plan-runner/worktree-sweep.js +149 -97
- package/.agents/scripts/lib/orchestration/review-providers/review-provider-factory.js +23 -0
- package/.agents/scripts/lib/orchestration/run-epilogue.js +6 -0
- package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +2 -0
- package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +349 -263
- package/.agents/scripts/lib/orchestration/single-story-close/phases/options.js +21 -7
- package/.agents/scripts/lib/orchestration/single-story-close/phases/review-override.js +4 -0
- package/.agents/scripts/lib/orchestration/single-story-close/runner.js +327 -314
- package/.agents/scripts/lib/orchestration/ticket-validator.js +19 -36
- package/.agents/scripts/lib/signals/detectors/common.js +63 -51
- package/.agents/scripts/lib/single-story-sweep.js +2 -2
- package/.agents/scripts/lib/temp-removal.js +110 -0
- package/.agents/scripts/lib/temp-retention.js +122 -73
- package/.agents/scripts/lib/transpile.js +28 -3
- package/.agents/scripts/lib/worktree/canonical-path.js +34 -0
- package/.agents/scripts/lib/worktree/lifecycle/reap.js +15 -4
- package/.agents/scripts/single-story-close.js +10 -2
- package/.agents/scripts/single-story-confirm-merge.js +267 -238
- package/.agents/scripts/single-story-init.js +120 -17
- package/.agents/skills/core/idea-refinement/SKILL.md +6 -6
- package/.agents/skills/stack/qa/qa-harness/SKILL.md +1 -2
- package/.agents/workflows/audit-architecture.md +5 -4
- package/.agents/workflows/audit-documentation.md +5 -5
- package/.agents/workflows/audit-performance.md +10 -10
- package/.agents/workflows/{git-cleanup.md → clean-git.md} +10 -10
- package/.agents/workflows/clean-temp.md +67 -0
- package/.agents/workflows/clean-worktrees.md +63 -0
- package/.agents/workflows/git-deliver.md +1 -1
- package/.agents/workflows/helpers/acceptance-self-eval.md +11 -10
- package/.agents/workflows/helpers/audit-lens-core.md +30 -57
- package/.agents/workflows/helpers/deliver-digest.md +2 -2
- package/.agents/workflows/helpers/deliver-reference.md +3 -1
- package/.agents/workflows/helpers/deliver-story-reference.md +2 -2
- package/.agents/workflows/helpers/deliver-story.md +6 -1
- package/.agents/workflows/helpers/parallel-tooling.md +16 -18
- package/.agents/workflows/mandrel-deliver.md +1 -1
- package/.agents/workflows/mandrel-plan.md +6 -5
- package/docs/CHANGELOG.md +39 -0
- package/lib/cli/guarded-sync.js +87 -0
- package/lib/cli/sync-agents.js +9 -92
- package/lib/cli/sync-commands.js +9 -101
- package/package.json +2 -2
|
@@ -12,8 +12,8 @@ effort: medium
|
|
|
12
12
|
|
|
13
13
|
<!--
|
|
14
14
|
Shared common core — byte-identical across every `.agents/agents/*.md` role
|
|
15
|
-
context, ordered FIRST so
|
|
16
|
-
|
|
15
|
+
context, ordered FIRST so every role binds the same baseline rules (the
|
|
16
|
+
roles pin different effort levels and share no cache prefix; delta last).
|
|
17
17
|
Edit it in every role file at once —
|
|
18
18
|
tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
|
|
19
19
|
security-baseline stays inviolable and single-sourced — @-import it, never
|
|
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
|
|
|
31
31
|
the step-by-step. This shared core binds every role:
|
|
32
32
|
|
|
33
33
|
- **Non-interactive.** You have no input channel mid-run. Never ask
|
|
34
|
-
clarifying questions —
|
|
35
|
-
your
|
|
36
|
-
blocked/failure path instead of stalling.
|
|
34
|
+
clarifying questions — take the reading your charter most directly
|
|
35
|
+
supports, name it in your return, and when you cannot proceed, take
|
|
36
|
+
your role's blocked/failure path instead of stalling.
|
|
37
37
|
- **Absolute paths only.** Your shell's working directory is not guaranteed
|
|
38
38
|
to persist between calls; pass absolute paths for every file and script.
|
|
39
39
|
- **Anti-thrashing.** When the same error class recurs despite the same fix,
|
|
@@ -98,8 +98,8 @@ For each acceptance item:
|
|
|
98
98
|
|
|
99
99
|
## Verdict schema (MUST)
|
|
100
100
|
|
|
101
|
-
Write **one** verdict file under `temp
|
|
102
|
-
`temp/acceptance-verdict
|
|
101
|
+
Write **one** verdict file under `temp/scratch/story-<storyId>/` (e.g.
|
|
102
|
+
`temp/scratch/story-<storyId>/acceptance-verdict-r<round>.json`) conforming to
|
|
103
103
|
[`acceptance-eval-verdict.schema.json`](../schemas/acceptance-eval-verdict.schema.json):
|
|
104
104
|
one `criteria[]` record per acceptance item, in acceptance-array order, with
|
|
105
105
|
`index` being the criterion's position in that array.
|
|
@@ -12,8 +12,8 @@ effort: medium
|
|
|
12
12
|
|
|
13
13
|
<!--
|
|
14
14
|
Shared common core — byte-identical across every `.agents/agents/*.md` role
|
|
15
|
-
context, ordered FIRST so
|
|
16
|
-
|
|
15
|
+
context, ordered FIRST so every role binds the same baseline rules (the
|
|
16
|
+
roles pin different effort levels and share no cache prefix; delta last).
|
|
17
17
|
Edit it in every role file at once —
|
|
18
18
|
tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
|
|
19
19
|
security-baseline stays inviolable and single-sourced — @-import it, never
|
|
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
|
|
|
31
31
|
the step-by-step. This shared core binds every role:
|
|
32
32
|
|
|
33
33
|
- **Non-interactive.** You have no input channel mid-run. Never ask
|
|
34
|
-
clarifying questions —
|
|
35
|
-
your
|
|
36
|
-
blocked/failure path instead of stalling.
|
|
34
|
+
clarifying questions — take the reading your charter most directly
|
|
35
|
+
supports, name it in your return, and when you cannot proceed, take
|
|
36
|
+
your role's blocked/failure path instead of stalling.
|
|
37
37
|
- **Absolute paths only.** Your shell's working directory is not guaranteed
|
|
38
38
|
to persist between calls; pass absolute paths for every file and script.
|
|
39
39
|
- **Anti-thrashing.** When the same error class recurs despite the same fix,
|
|
@@ -116,13 +116,12 @@ and a surviving **Critical** halts the delivery gate:
|
|
|
116
116
|
- **Info** — the floor: a grounded observation asking for no scheduled work
|
|
117
117
|
(accepts `Informational`). Never a home for findings that fail the bar below.
|
|
118
118
|
|
|
119
|
-
## Self-cross-check bar
|
|
119
|
+
## Self-cross-check bar
|
|
120
120
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
violates) — never "this looks wrong"; **in-scope** under the scope filter; and
|
|
121
|
+
A finding goes in the report only when **all** hold: a **grounded**
|
|
122
|
+
`path:line` you actually read; **reproducible evidence** (a tool reading, a
|
|
123
|
+
quoted snippet, or a specific standard it violates) — never "this looks
|
|
124
|
+
wrong"; **in-scope** under the scope filter; and
|
|
126
125
|
an **actionable** recommendation. Drop anything resting on a sanctioned test
|
|
127
126
|
seam, an entry point / public API surface, dynamic/framework reachability, an
|
|
128
127
|
intentional documented deviation, or a formatter-governed style nit.
|
|
@@ -136,14 +135,14 @@ Beside it, carry one machine-readable tally of the findings you kept —
|
|
|
136
135
|
included, `Info` never counted. `audit-to-stories` cross-checks that line
|
|
137
136
|
against its parse and refuses a report whose tally is missing or wrong.
|
|
138
137
|
|
|
139
|
-
## Fan-out (
|
|
138
|
+
## Fan-out (operator-requested only)
|
|
140
139
|
|
|
141
|
-
When your caller dispatches you for a
|
|
142
|
-
|
|
143
|
-
that dimension and return its findings;
|
|
144
|
-
results under this self-cross-check bar.
|
|
145
|
-
|
|
146
|
-
|
|
140
|
+
By default you audit the whole lens. When your caller dispatches you for a
|
|
141
|
+
single dimension — which it does only on an explicit operator request for
|
|
142
|
+
per-dimension fan-out — audit only that dimension and return its findings;
|
|
143
|
+
the caller merges the per-dimension results under this self-cross-check bar.
|
|
144
|
+
You never dispatch sub-agents of your own: no nested fan-out, whatever the
|
|
145
|
+
lens's size.
|
|
147
146
|
|
|
148
147
|
## Return contract
|
|
149
148
|
|
|
@@ -12,8 +12,8 @@ effort: medium
|
|
|
12
12
|
|
|
13
13
|
<!--
|
|
14
14
|
Shared common core — byte-identical across every `.agents/agents/*.md` role
|
|
15
|
-
context, ordered FIRST so
|
|
16
|
-
|
|
15
|
+
context, ordered FIRST so every role binds the same baseline rules (the
|
|
16
|
+
roles pin different effort levels and share no cache prefix; delta last).
|
|
17
17
|
Edit it in every role file at once —
|
|
18
18
|
tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
|
|
19
19
|
security-baseline stays inviolable and single-sourced — @-import it, never
|
|
@@ -31,9 +31,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
|
|
|
31
31
|
the step-by-step. This shared core binds every role:
|
|
32
32
|
|
|
33
33
|
- **Non-interactive.** You have no input channel mid-run. Never ask
|
|
34
|
-
clarifying questions —
|
|
35
|
-
your
|
|
36
|
-
blocked/failure path instead of stalling.
|
|
34
|
+
clarifying questions — take the reading your charter most directly
|
|
35
|
+
supports, name it in your return, and when you cannot proceed, take
|
|
36
|
+
your role's blocked/failure path instead of stalling.
|
|
37
37
|
- **Absolute paths only.** Your shell's working directory is not guaranteed
|
|
38
38
|
to persist between calls; pass absolute paths for every file and script.
|
|
39
39
|
- **Anti-thrashing.** When the same error class recurs despite the same fix,
|
|
@@ -9,8 +9,8 @@ description: >-
|
|
|
9
9
|
|
|
10
10
|
<!--
|
|
11
11
|
Shared common core — byte-identical across every `.agents/agents/*.md` role
|
|
12
|
-
context, ordered FIRST so
|
|
13
|
-
|
|
12
|
+
context, ordered FIRST so every role binds the same baseline rules (the
|
|
13
|
+
roles pin different effort levels and share no cache prefix; delta last).
|
|
14
14
|
Edit it in every role file at once —
|
|
15
15
|
tests/bootstrap/agent-shared-prefix.test.js fails on any divergence.
|
|
16
16
|
security-baseline stays inviolable and single-sourced — @-import it, never
|
|
@@ -28,9 +28,9 @@ role-delta marker below; the workflow prose your caller hands you supplies
|
|
|
28
28
|
the step-by-step. This shared core binds every role:
|
|
29
29
|
|
|
30
30
|
- **Non-interactive.** You have no input channel mid-run. Never ask
|
|
31
|
-
clarifying questions —
|
|
32
|
-
your
|
|
33
|
-
blocked/failure path instead of stalling.
|
|
31
|
+
clarifying questions — take the reading your charter most directly
|
|
32
|
+
supports, name it in your return, and when you cannot proceed, take
|
|
33
|
+
your role's blocked/failure path instead of stalling.
|
|
34
34
|
- **Absolute paths only.** Your shell's working directory is not guaranteed
|
|
35
35
|
to persist between calls; pass absolute paths for every file and script.
|
|
36
36
|
- **Anti-thrashing.** When the same error class recurs despite the same fix,
|
|
@@ -139,13 +139,14 @@ Everything `/mandrel-deliver` and `single-story-close` consume: worktree isolati
|
|
|
139
139
|
| `execution.fullSuiteLock` | No | `boolean` | `true` | Serialize full-suite spawns (`npm test` / `npm run test:coverage`) behind a host-level advisory lock, so two concurrent deliveries on one checkout do not run two suites against the same cores. Best-effort: a wait that expires spawns anyway, so the lock can never fail a delivery. Set false — or export `MANDREL_FULL_SUITE_LOCK=0` for one invocation — to disable. |
|
|
140
140
|
| `docsFreshness` | No | `object` | — | Documentation-freshness scope: the files a change of consequence is expected to touch. Read by the audit-documentation lens to seed its target set; no delivery gate enforces it. |
|
|
141
141
|
| `docsFreshness.paths` | No | `array<string>` | `["README.md"]` | Repo-relative documentation paths the audit-documentation lens adds to its target set. |
|
|
142
|
-
| `tempRetention` | No | `object` | — | Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so
|
|
142
|
+
| `tempRetention` | No | `object` | — | Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so unrecognized files under tempRoot are reported with their size and left alone (`/clean-temp` is the operator path for them). signals.ndjson is never purged by any path. |
|
|
143
143
|
| `tempRetention.enabled` | No | `boolean` | `true` | Master switch. Default true — reclaiming a landed Story's gate transcripts and validation evidence is the behaviour, and this knob turns it off. When false every purge path is a reported no-op. |
|
|
144
144
|
| `tempRetention.classes` | No | `object` | — | Per-class opt-out. Each defaults to true; set one false to keep that family while the rest are purged. |
|
|
145
145
|
| `tempRetention.classes.orchestrationLogs` | No | `boolean` | `true` | <tempRoot>/orchestration/*.log — close gate transcripts and terse-result detail dumps. |
|
|
146
146
|
| `tempRetention.classes.validationEvidence` | No | `boolean` | `true` | Per-Story validation-evidence.json, lifecycle.ndjson, and manifest.md under the standalone and per-run story trees. |
|
|
147
147
|
| `tempRetention.classes.auditResults` | No | `boolean` | `true` | <tempRoot>/audits/ — audit lens reports. |
|
|
148
148
|
| `tempRetention.classes.planDirs` | No | `boolean` | `true` | <tempRoot>/plan-<slug>/ — abandoned plan authoring dirs. Age-floored only; the current run is always excluded. |
|
|
149
|
+
| `tempRetention.classes.scratch` | No | `boolean` | `true` | <tempRoot>/scratch/ — agent-authored scratch. `scratch/story-<id>/` is purged when that Story lands; any other `scratch/` entry is age-floored. |
|
|
149
150
|
| `deliverRunner` | No | `object` | — | Bounded-concurrency knob for the /mandrel-deliver fan-out. |
|
|
150
151
|
| `deliverRunner.concurrencyCap` | No | `integer` | `3` | Maximum ready Stories dispatched by /mandrel-deliver at once. Default 3. Moderate by design — keeps host-quota consumption predictable while allowing a small ready-set fan-out. Set 1 for strictly sequential delivery; raise further on hosts with adequate parallel-agent quota. See deliver.md for the sequencing model and throughput tradeoff. |
|
|
151
152
|
| `deliverRunner.footprintGuard` | No | `"enforce"` \| `"advisory"` | `"enforce"` | How a file-footprint collision affects dispatch. 'enforce' (default, and the behaviour to keep unless you have a reason) withholds a Story whose footprint races a peer admitted this beat or one still in flight — the guard encodes delivery-time-only knowledge (open implementation windows, foreign leases, ground that moved since planning) that no depends_on edge can carry. 'advisory' still DETECTS every collision and reports each would-be withhold in the tick envelope, but lets dispatch follow the declared depends_on edges alone — a deliberate throughput trade for a run whose ordering is fully declared. See stories-wave-tick.js and helpers/deliver-reference.md. |
|
|
@@ -72,11 +72,12 @@ and schema mechanics are in [§ Friction telemetry](#friction-telemetry) above.
|
|
|
72
72
|
## FinOps & token budgeting (economic guardrails)
|
|
73
73
|
|
|
74
74
|
Mandrel does **not** enforce live LLM spend from response metadata. It bounds
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
Story sizing
|
|
78
|
-
stops. Consult this section when
|
|
79
|
-
|
|
75
|
+
one thing, a **fixed framework constant** rather than an operator knob that
|
|
76
|
+
**fails closed**: the assembled `/mandrel-plan` context envelope. Plan-time
|
|
77
|
+
Story sizing is not bounded (retired by Story #5312). Your host runtime
|
|
78
|
+
(editor / CLI) owns session quota and hard stops. Consult this section when
|
|
79
|
+
reasoning about why `/mandrel-plan` refused an over-ceiling envelope, or when
|
|
80
|
+
choosing the session effort and model for a command.
|
|
80
81
|
|
|
81
82
|
> **There is no configurable context budget.** `planning.context.maxBytes` /
|
|
82
83
|
> `summaryMode` were removed outright in Story #4541, along with the
|
|
@@ -126,3 +127,24 @@ over-ceiling envelope or an over-budget Story count.
|
|
|
126
127
|
nothing at plan time scores its authored mass.
|
|
127
128
|
- **Host runtime**: session billing, quota exhaustion, and operator overrides
|
|
128
129
|
are enforced by your provider (e.g. Claude Code), not by Mandrel scripts.
|
|
130
|
+
|
|
131
|
+
### Session effort and model
|
|
132
|
+
|
|
133
|
+
Effort is the operator's dial, set once per session. Pick it deliberately up
|
|
134
|
+
front, because changing effort mid-session invalidates the prompt cache for
|
|
135
|
+
everything that follows.
|
|
136
|
+
|
|
137
|
+
- **Recommended session effort.** `medium` for `/mandrel-deliver`: Stories
|
|
138
|
+
are well-scoped by construction, so delivery rarely needs more. `high` for
|
|
139
|
+
`/mandrel-plan`: a Spec defect costs a redraft or a blocked Story
|
|
140
|
+
downstream, which is dearer than the planning turn.
|
|
141
|
+
- **Role agents.** `story-worker` declares no effort and inherits the
|
|
142
|
+
session's (Story #5426). The evaluator roles (`acceptance-critic`,
|
|
143
|
+
`plan-critic`, `auditor`) pin `medium`.
|
|
144
|
+
- **Where pins live.** Effort and model pins belong only on role agents under
|
|
145
|
+
`.agents/agents/`, never in workflow or command frontmatter: a command that
|
|
146
|
+
pinned its own effort would change effort mid-session and break the prompt
|
|
147
|
+
cache.
|
|
148
|
+
- **Escalation is an operator step.** The Agent tool takes no per-call
|
|
149
|
+
effort, so no workflow can escalate on its own. Before resuming a blocked
|
|
150
|
+
Story, raise session effort one step; raise effort before switching models.
|
|
@@ -32,7 +32,7 @@ by `node .agents/scripts/generate-workflows-doc.js`; `npm run docs:check`
|
|
|
32
32
|
fails when it drifts from the on-disk workflow set. To change a command’s
|
|
33
33
|
description, edit the workflow file’s front-matter and regenerate.
|
|
34
34
|
|
|
35
|
-
## Commands (
|
|
35
|
+
## Commands (31)
|
|
36
36
|
|
|
37
37
|
| Command | Description |
|
|
38
38
|
| --- | --- |
|
|
@@ -55,7 +55,9 @@ description, edit the workflow file’s front-matter and regenerate.
|
|
|
55
55
|
| `/audit-sre` | "Audit production-readiness for a release candidate: SLOs, observability, runbooks, error budgets, and rollback paths." |
|
|
56
56
|
| `/audit-to-stories` | Convert findings produced by the audit-\* workflows into actionable GitHub Stories. Reads temp/audits/audit-\*-results.md, groups findings cross-audit, deduplicates against existing Issues by fingerprint, and either chains into /mandrel-plan --seed-file or opens standalone Stories. |
|
|
57
57
|
| `/audit-ux-ui` | Audit UX/UI consistency and design system adherence |
|
|
58
|
-
| `/git
|
|
58
|
+
| `/clean-git` | Tidy the local checkout in four phases: fast-forward `main`, prune stale remote-tracking refs, sweep merged branches (squash-aware), and triage `git stash` entries — each step gated by operator confirmation. |
|
|
59
|
+
| `/clean-temp` | Clear the temp-tree backlog the land-time purge cannot attribute: sort every top-level entry under the project's tempRoot into framework, closed-issue, aged and kept buckets, preview by default, and delete only confirmed buckets. |
|
|
60
|
+
| `/clean-worktrees` | Reclaim disk from dead worktrees: list every worktree of this project as a removal candidate (closed Story, merged branch, orphaned directory, detached HEAD) or as kept with a reason, then remove candidates only on `--execute`. |
|
|
59
61
|
| `/git-deliver` | Single ad-hoc delivery command for working-tree changes. Detects the git setup and escalates to the right terminal step — commit only, commit + push, or commit + push + open a PR with native auto-merge — picking the default from observable state and letting flags pin any level explicitly. Replaces the retired git-commit-all, git-push, and git-pr-all trio. |
|
|
60
62
|
| `/mandrel-deliver` | Unified delivery entry point. Takes Story ids or a plain-language prompt, derives which path the work belongs on, and lands it via the single deliver-story engine — story-<id> → PR → main. |
|
|
61
63
|
| `/mandrel-plan` | Unified planning entry point. Interrogate → author → persist. Emits one Story by default; splits into N>1 only under the default-single split policy. |
|
package/.agents/instructions.md
CHANGED
|
@@ -118,7 +118,7 @@ truncates with a note naming what was cut:
|
|
|
118
118
|
|
|
119
119
|
## 3. Core Philosophy
|
|
120
120
|
|
|
121
|
-
1. **Context First.** **Digest-first reading
|
|
121
|
+
1. **Context First.** **Digest-first reading:** never
|
|
122
122
|
ingest the whole `project.docsContextFiles` set up front — read the
|
|
123
123
|
docs digest and pull files on demand at the section it names. No
|
|
124
124
|
digest (ad hoc task, `docsContextFiles` unset, null `docsDigestPath`)
|
|
@@ -127,8 +127,10 @@ truncates with a note naming what was cut:
|
|
|
127
127
|
present. Always read the current Story's body (`## Spec` +
|
|
128
128
|
`acceptance[]` / `verify[]`); prefer targeted retrieval over broad
|
|
129
129
|
reads.
|
|
130
|
-
2. **Plan First.**
|
|
131
|
-
|
|
130
|
+
2. **Plan First.** Planned work carries its plan in the Story's
|
|
131
|
+
`## Spec`, authored via `/mandrel-plan`; an unplanned prompt takes
|
|
132
|
+
`/mandrel-deliver`'s light path, which escalates to `/mandrel-plan`
|
|
133
|
+
when its gate trips.
|
|
132
134
|
3. **Artifacts over Chat.** Write test/build/debug output to log
|
|
133
135
|
files, not into chat.
|
|
134
136
|
4. **Idempotency.** Scripts must be safe to run repeatedly.
|
|
@@ -137,16 +139,13 @@ truncates with a note naming what was cut:
|
|
|
137
139
|
|
|
138
140
|
## 4. Execution & Quality Discipline
|
|
139
141
|
|
|
140
|
-
- **Re-Plan on Failure.** If a strategy fails, STOP and re-plan.
|
|
141
142
|
- **Subagent Strategy.** Each spawn re-pays the full always-loaded
|
|
142
143
|
context — a cost decision. Prefer inline search for small lookups;
|
|
143
144
|
spawn only when the work justifies replicating context. One objective
|
|
144
145
|
per subagent; depth compounds the cost (every nested level re-pays).
|
|
145
|
-
- **
|
|
146
|
-
|
|
147
|
-
Remove unused imports, commented-out code, and dead branches before
|
|
146
|
+
- **No Dead Code.** Every edit leaves complete, runnable code. Remove
|
|
147
|
+
unused imports, commented-out code, and dead branches before
|
|
148
148
|
finalizing.
|
|
149
|
-
- **Verification.** Include explicit verification steps in every plan.
|
|
150
149
|
|
|
151
150
|
---
|
|
152
151
|
|
|
@@ -182,7 +181,8 @@ never delivered): `/mandrel-plan` offers one above 2 Stories and
|
|
|
182
181
|
|
|
183
182
|
All temporary files, scratch scripts, and intermediate outputs MUST
|
|
184
183
|
live in the gitignored workspace-root `/temp/` directory — do NOT commit
|
|
185
|
-
anything under it.
|
|
184
|
+
anything under it. Put ad-hoc scratch in `temp/scratch/story-<id>/` (or
|
|
185
|
+
`temp/scratch/` with no Story), the layout the temp purge reaps.
|
|
186
186
|
|
|
187
187
|
---
|
|
188
188
|
|
|
@@ -191,7 +191,6 @@ anything under it.
|
|
|
191
191
|
`/mandrel-plan` sizes each Story as a **capability slice a frontier model
|
|
192
192
|
delivers and self-verifies in one pass** — a broad footprint is normal
|
|
193
193
|
when the change is cohesive, and no plan-time ceiling scores it; do not
|
|
194
|
-
re-slice it into per-module fragments. On an out-of-scope task
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
validation.
|
|
194
|
+
re-slice it into per-module fragments. On an out-of-scope task,
|
|
195
|
+
**commit incrementally** per cohesive sub-step, and when a sub-step
|
|
196
|
+
fails validation, stop and apply § 1.I (re-plan or yield).
|
|
@@ -137,9 +137,9 @@ run — a blind fix to a suite nobody exercised is how the gap compounds.
|
|
|
137
137
|
|
|
138
138
|
`capacity` and `unreproducible-tier` name failures that are proven properties
|
|
139
139
|
of the **environment**: no commit on the branch can move the head SHA to clear
|
|
140
|
-
them, so the no-rerun rule
|
|
141
|
-
cleared it by hand. Those two verdicts — and only those
|
|
142
|
-
one rerun:
|
|
140
|
+
them, so without an allowance the no-rerun rule would strand a correct
|
|
141
|
+
delivery until a human cleared it by hand. Those two verdicts — and only those
|
|
142
|
+
two — buy exactly one rerun:
|
|
143
143
|
|
|
144
144
|
1. Reach the verdict with its required readings (§ above). A green on re-run is
|
|
145
145
|
never one of those readings.
|
|
@@ -143,8 +143,9 @@ authored:
|
|
|
143
143
|
widen the regex, updating every call site in the same PR.
|
|
144
144
|
4. **Add a new definition only when no reasonable match exists**, in the
|
|
145
145
|
correct domain directory. Never copy-paste a step implementation to support
|
|
146
|
-
a paraphrased scenario
|
|
147
|
-
|
|
146
|
+
a paraphrased scenario. A prose-only authoring pass (one whose scope
|
|
147
|
+
excludes step-definition code) records the missing step as a named gap
|
|
148
|
+
instead of writing it.
|
|
148
149
|
|
|
149
150
|
When a step is superseded, mark it deprecated and migrate every call site in
|
|
150
151
|
the same PR; do not leave two near-identical steps live.
|
|
@@ -15,7 +15,9 @@ Mandrel ships as the `mandrel` npm package, whose consumers pin an
|
|
|
15
15
|
exact lockfile version; they opt into breaks at upgrade time. Operator policy
|
|
16
16
|
for any contract change (config shape, baseline shape, schema, lifecycle
|
|
17
17
|
payload, ticket label, dispatch artifact, public API of a script) is
|
|
18
|
-
therefore
|
|
18
|
+
therefore as follows. It governs Mandrel's own framework contracts; a
|
|
19
|
+
consumer's product API keeps the expand–contract rule in
|
|
20
|
+
[`api-conventions.md`](api-conventions.md).
|
|
19
21
|
|
|
20
22
|
1. **Hard cutovers only.** Contract changes ship as a single in-tree
|
|
21
23
|
migration of every producer and consumer. There is no parallel
|
|
@@ -94,13 +96,13 @@ signature is worth naming:
|
|
|
94
96
|
|
|
95
97
|
**Invariant (stated in the core): the delivering flow owns tidying the local
|
|
96
98
|
checkout — reaping its own merged refs and fast-forwarding the base branch.
|
|
97
|
-
`/git
|
|
99
|
+
`/clean-git` is a recovery tool, not a routine chore.** The outcome every
|
|
98
100
|
delivering flow (`/mandrel-deliver`, `/git-deliver`) guarantees, with the mechanics
|
|
99
|
-
owned by `boot-sweep.js` / `git
|
|
101
|
+
owned by `boot-sweep.js` / `clean-git.js`:
|
|
100
102
|
|
|
101
103
|
- **`main` is fast-forwarded** by the flow itself in its cleanup phase, so the
|
|
102
104
|
next init seeds from a current base. No workflow ends by telling the operator
|
|
103
|
-
to run `/git
|
|
105
|
+
to run `/clean-git` to catch up.
|
|
104
106
|
- **Merged local refs are reaped** at the next workflow boot's protected sweep
|
|
105
107
|
(`boot-sweep.js`) — every local branch whose PR is already merged, skipping
|
|
106
108
|
any candidate with unpushed work, a dirty worktree, or a still-open parent
|
|
@@ -110,9 +112,9 @@ owned by `boot-sweep.js` / `git-cleanup.js`:
|
|
|
110
112
|
weaker content-equivalence signal (`detectedBy: 'content-merged'` — content
|
|
111
113
|
already landed in the base by another route, with no merged PR or git
|
|
112
114
|
ancestry of its own) is **never** reaped by the boot sweep; it is surfaced
|
|
113
|
-
under `contentMerged` for the operator to send to `/git
|
|
115
|
+
under `contentMerged` for the operator to send to `/clean-git` for a
|
|
114
116
|
confirmed, eyeballed reap.
|
|
115
|
-
- **`/git
|
|
117
|
+
- **`/clean-git` is recovery, not routine.** Run it by hand only for a state
|
|
116
118
|
the automated hygiene does not cover — triaging stashes, reaping across
|
|
117
119
|
non-standard namespaces, or `--remote` pruning after a force-push. Reaching
|
|
118
120
|
for it after every routine delivery signals the owning flow's hygiene step
|
|
@@ -154,10 +156,10 @@ different hazards:
|
|
|
154
156
|
|
|
155
157
|
## Meta Labels (Retrospective Signal Routing)
|
|
156
158
|
|
|
157
|
-
|
|
159
|
+
Three `meta::*` labels route retrospective signals into durable substrates so
|
|
158
160
|
the `/mandrel-plan` Phase 0 fetcher (see
|
|
159
161
|
[`prior-feedback-fetcher.js`](../scripts/lib/feedback-loop/prior-feedback-fetcher.js))
|
|
160
|
-
can surface open feedback issues to the planner.
|
|
162
|
+
can surface open feedback issues to the planner. All three live in
|
|
161
163
|
[`label-constants.js`](../scripts/lib/label-constants.js) under the
|
|
162
164
|
`META_LABELS` export — reference them by symbol from scripts rather than
|
|
163
165
|
hard-coding the string.
|
|
@@ -180,3 +182,10 @@ project-local automation). The work is scoped to the consumer's
|
|
|
180
182
|
framework changes. Issues that span both axes should carry both labels —
|
|
181
183
|
`fetchPriorFeedback` dedupes by issue number so a dual-labeled issue
|
|
182
184
|
appears exactly once in the planner context.
|
|
185
|
+
|
|
186
|
+
### `meta::platform-gap`
|
|
187
|
+
|
|
188
|
+
Apply this label to a GitHub issue whose fault lies in a shared base
|
|
189
|
+
config, runner fleet, or cross-repo toolchain that neither the framework
|
|
190
|
+
nor the consumer owns — the `--owner platform` bucket of
|
|
191
|
+
[`ci-remediation.md`](ci-remediation.md).
|
|
@@ -16,9 +16,8 @@ Every Story lands on a dedicated **Story branch** named
|
|
|
16
16
|
`story-<storyId>`, seeded from `project.baseBranch` (`main` by default),
|
|
17
17
|
isolated in its own worktree at `.worktrees/story-<id>/`. The runtime
|
|
18
18
|
owns both via `single-story-init.js`; agents commit there only. Close
|
|
19
|
-
opens a PR against `main` (squash + required checks).
|
|
20
|
-
|
|
21
|
-
land on `story-<storyId>` directly, the
|
|
19
|
+
opens a PR against `main` (squash + required checks). Commits land on
|
|
20
|
+
`story-<storyId>` directly, the
|
|
22
21
|
subject referencing the Story via `(refs #<storyId>)` — see
|
|
23
22
|
[`.agents/instructions.md` § 5.B](../instructions.md).
|
|
24
23
|
|
|
@@ -38,15 +37,18 @@ subject referencing the Story via `(refs #<storyId>)` — see
|
|
|
38
37
|
|
|
39
38
|
## Push Validation & Reliability (MUSTs)
|
|
40
39
|
|
|
41
|
-
1.
|
|
40
|
+
1. Validate locally **before** `git push`. On a Story branch that is the
|
|
41
|
+
one credited suite run (`deliver-digest.md` § 5); close runs every
|
|
42
|
+
other gate, so do not pre-run them. Elsewhere, run the configured
|
|
43
|
+
validation commands.
|
|
42
44
|
2. Do NOT assume a push succeeded unless the output confirms the remote
|
|
43
45
|
ref was updated (`[new branch]`, `[up to date]`, `... -> ...`).
|
|
44
46
|
3. If a `pre-push` hook rejects, fix the cause and create a NEW follow-up
|
|
45
47
|
commit — never amend the rejected commit.
|
|
46
48
|
4. **Never bypass hooks** (`--no-verify`, `--no-gpg-sign`, …) without
|
|
47
|
-
explicit operator authorization.
|
|
48
|
-
|
|
49
|
-
consumer-tooling gap, **not** authorization; see
|
|
49
|
+
explicit operator authorization. A Biome zero-match failure under a
|
|
50
|
+
harness-managed worktree path is a known false negative — a
|
|
51
|
+
consumer-tooling gap, **not** authorization to bypass; see
|
|
50
52
|
[`git-conventions-reference.md` § Push Validation](git-conventions-reference.md).
|
|
51
53
|
|
|
52
54
|
## Local checkout hygiene
|
|
@@ -54,7 +56,7 @@ subject referencing the Story via `(refs #<storyId>)` — see
|
|
|
54
56
|
**The delivering flow owns tidying the local checkout** — it
|
|
55
57
|
fast-forwards the base branch itself and reaps its own merged refs on
|
|
56
58
|
the next workflow boot (the `boot-sweep.js` protected sweep).
|
|
57
|
-
`/git
|
|
59
|
+
`/clean-git` is a recovery tool, not a routine chore — never end a
|
|
58
60
|
workflow by telling the operator to run it. Scope rules and the
|
|
59
61
|
shared-checkout contention guard:
|
|
60
62
|
[`git-conventions-reference.md` § Local checkout hygiene](git-conventions-reference.md).
|
|
@@ -156,13 +156,14 @@ behaviour (`sets status to completed`, not `works`).
|
|
|
156
156
|
|
|
157
157
|
A file that passes alone but fails inside the full `npm test` suite is **test
|
|
158
158
|
pollution** — one test leaks shared state (env vars, temp files, the
|
|
159
|
-
mock-module registry, global singletons) and a later test trips on it.
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
mutation in a `t.before` / `t.after`
|
|
159
|
+
mock-module registry, global singletons) and a later test trips on it. Run
|
|
160
|
+
every matching file alone, then all together; the files that pass alone but
|
|
161
|
+
fail in the suite (**flippers**) bound the search, and bisecting them finds
|
|
162
|
+
the smallest reproducing subset. Mandrel's own repository automates this as
|
|
163
|
+
`npm run test:isolate`, which also reports leftover `process.env`
|
|
164
|
+
mutations; a consumer uses its runner's equivalent. The fix is almost
|
|
165
|
+
always missing teardown — wrap the mutation in a `t.before` / `t.after`
|
|
166
|
+
pair, or restore the prior value in
|
|
166
167
|
`try` / `finally`.
|
|
167
168
|
|
|
168
169
|
For browser-based changes, pair the cycle with runtime verification via Chrome
|
|
@@ -386,7 +386,7 @@
|
|
|
386
386
|
},
|
|
387
387
|
"tempRetention": {
|
|
388
388
|
"type": "object",
|
|
389
|
-
"description": "Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so
|
|
389
|
+
"description": "Story #4794. Auto-purge of spent temp artifacts once their Story lands. Classification is an allowlist: only the declared classes below are ever deleted, so unrecognized files under tempRoot are reported with their size and left alone (`/clean-temp` is the operator path for them). signals.ndjson is never purged by any path.",
|
|
390
390
|
"properties": {
|
|
391
391
|
"enabled": {
|
|
392
392
|
"type": "boolean",
|
|
@@ -416,6 +416,11 @@
|
|
|
416
416
|
"type": "boolean",
|
|
417
417
|
"description": "<tempRoot>/plan-<slug>/ — abandoned plan authoring dirs. Age-floored only; the current run is always excluded.",
|
|
418
418
|
"default": true
|
|
419
|
+
},
|
|
420
|
+
"scratch": {
|
|
421
|
+
"type": "boolean",
|
|
422
|
+
"description": "<tempRoot>/scratch/ — agent-authored scratch. `scratch/story-<id>/` is purged when that Story lands; any other `scratch/` entry is age-floored.",
|
|
423
|
+
"default": true
|
|
419
424
|
}
|
|
420
425
|
},
|
|
421
426
|
"additionalProperties": false
|