wdi-method 0.6.19 → 0.6.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -51,6 +51,7 @@ CHECK_ORDER = (
51
51
  "memlog-home",
52
52
  "spec-names-release-prd",
53
53
  "ticket-status-one-home",
54
+ "archived-spec-closed",
54
55
  "defect-root-cause",
55
56
  "entity-one-writer",
56
57
  "spec-after-g4",
@@ -172,7 +173,7 @@ def git(root: Path, *args: str) -> str | None:
172
173
  try:
173
174
  out = subprocess.run(
174
175
  ["git", "-C", str(root), *args],
175
- capture_output=True, text=True, timeout=30, check=False,
176
+ capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, check=False,
176
177
  )
177
178
  except (OSError, subprocess.SubprocessError):
178
179
  return None
@@ -656,7 +657,8 @@ def lc_registered(c: Corpus, r: Result) -> None: # was V12
656
657
  if area not in areas:
657
658
  r.fail("lc-registered", str(ticket.get("id")),
658
659
  f"its spec is already closed, but `{area}` is not registered as an `area` "
659
- f"in components.yaml")
660
+ f"in components.yaml. For corpus- or documentation-only tickets with no "
661
+ f"application code changes, use `touches: []` instead of inventing an area name.")
660
662
  pid = str(ticket.get("component") or "")
661
663
  row = pc_by_id.get(pid)
662
664
  if row is None or (str(spec.get("id")), pid) in seen:
@@ -942,6 +944,35 @@ def ticket_status_one_home(c: Corpus, r: Result) -> None: # was V18
942
944
  "`status:` in frontmatter")
943
945
 
944
946
 
947
+ def archived_spec_closed(c: Corpus, r: Result) -> None:
948
+ """A spec whose `spec_folder` points to `.archive/` MUST have status `closed`.
949
+
950
+ Archiving is reserved for completed specs. Active work belongs in `.scratch/` where tickets
951
+ are materialized into git worktrees for implementation.
952
+ """
953
+ for spec in c.spec_list:
954
+ sid = str(spec.get("id") or "")
955
+ folder = str(spec.get("spec_folder") or "").strip()
956
+ if not folder:
957
+ for t in spec.get("tickets") or []:
958
+ if isinstance(t, dict) and str(t.get("spec_folder") or "").strip():
959
+ folder = str(t.get("spec_folder")).strip()
960
+ break
961
+ clean = folder.replace("\\", "/").strip()
962
+ while clean.startswith("./"):
963
+ clean = clean[2:]
964
+ if clean.startswith("/"):
965
+ clean = clean.lstrip("/")
966
+ norm = os.path.normpath(clean).replace("\\", "/") if clean else ""
967
+ if (clean == ".archive" or clean.startswith(".archive/") or
968
+ norm == ".archive" or norm.startswith(".archive/")):
969
+ status = str(spec.get("status") or "").strip()
970
+ if status != "closed":
971
+ r.fail("archived-spec-closed", sid,
972
+ f"points to `{folder}` under `.archive/` but its status is `{status or 'unspecified'}` — "
973
+ "only closed specs may be archived")
974
+
975
+
945
976
  PLATFORM = "_platform"
946
977
  CROSS_CUTTING = ".how/_platform/cross-cutting.md"
947
978
  # The section heading entity-one-writer looks for. A heading a SCRIPT matches is a machine-facing key, and
@@ -1288,6 +1319,7 @@ PAST_RECORD = (
1288
1319
  ".control/decisions/",
1289
1320
  ".control/questions/answered.md",
1290
1321
  ".control/reports/",
1322
+ ".archive/",
1291
1323
  )
1292
1324
  # Corpus that §25 freezes as-is. Its citation of a now-retired prototype is authorized by DEC-016.
1293
1325
  FROZEN = (".what/",)
@@ -1655,7 +1687,7 @@ def custom_room_declared(c: Corpus, r: Result) -> None: # was V27
1655
1687
  # The two rendered trees are DELIBERATELY absent from this list. They are regenerated by this
1656
1688
  # script, so a product that declines to commit derived output is making a choice the method allows.
1657
1689
  COMMITTED_DIRS = (".constitution", ".control", ".what", ".how", "_bmad-output", ".work",
1658
- ".scratch")
1690
+ ".scratch", ".archive")
1659
1691
 
1660
1692
  # Probed inside each directory above, and named so that no honest pattern would ever mean to match
1661
1693
  # it. The distinction this draws is the entire point of the check: `.work/upstream/` or
@@ -1678,7 +1710,7 @@ def _ignore_rule(root: Path, rel: str) -> tuple[bool, str]:
1678
1710
  try:
1679
1711
  out = subprocess.run(
1680
1712
  ["git", "-C", str(root), "check-ignore", "-v", "--no-index", rel],
1681
- capture_output=True, text=True, timeout=30, check=False,
1713
+ capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, check=False,
1682
1714
  )
1683
1715
  except (OSError, subprocess.SubprocessError):
1684
1716
  return (False, "")
@@ -1712,6 +1744,20 @@ def corpus_in_git(c: Corpus, r: Result) -> None:
1712
1744
  f"is excluded from git by `{rule}` — the method commits this folder, "
1713
1745
  f"so no clone has what is in it")
1714
1746
 
1747
+ dispatch_file = c.root / ".control" / "custom-dispatch.yaml"
1748
+ if dispatch_file.exists():
1749
+ try:
1750
+ ls_res = subprocess.run(
1751
+ ["git", "-C", str(c.root), "ls-files", ".control/custom-dispatch.yaml"],
1752
+ capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, check=False,
1753
+ )
1754
+ if ls_res.returncode == 0 and ls_res.stdout.strip():
1755
+ r.fail("custom-dispatch-untracked", ".control/custom-dispatch.yaml",
1756
+ "is tracked in git — local runner configuration must never be committed; "
1757
+ "add it to .gitignore and run `git rm --cached .control/custom-dispatch.yaml`")
1758
+ except (OSError, subprocess.SubprocessError):
1759
+ pass
1760
+
1715
1761
 
1716
1762
  ENGINE_HOMES = (".claude", ".agents", ".agent", ".cursor", ".codex")
1717
1763
  ENGINE_FLAGGED = ("to-spec", "to-tickets", "implement")
@@ -1858,7 +1904,7 @@ def run_checks(c: Corpus, asof: dt.date) -> Result:
1858
1904
  # no two copies left to compare.
1859
1905
  # V19 is REPEALED. It checked one line item — an `RTR-` file in .control/reports/ — and the
1860
1906
  # retrospective it archived was the only thing spec size `L` ever decided. Both went together.
1861
- for fn in (goal_has_fr, fr_has_uc, uc_scheduled, ticket_has_test, nfr_has_enforcer, refs_resolve, no_cycles, applied_dec_touches, locked_gate_passed, parallel_tickets_blocked, lc_registered, review_trace, chain_links, memlog_home, spec_names_release_prd, ticket_status_one_home, defect_root_cause, entity_one_writer, spec_after_g4, high_risk_named, mandate_accept, cites_resolve, container_built, custom_room_declared, corpus_in_git, engines_invocable, withdrawn_recorded, id_allocated_once):
1907
+ for fn in (goal_has_fr, fr_has_uc, uc_scheduled, ticket_has_test, nfr_has_enforcer, refs_resolve, no_cycles, applied_dec_touches, locked_gate_passed, parallel_tickets_blocked, lc_registered, review_trace, chain_links, memlog_home, spec_names_release_prd, ticket_status_one_home, archived_spec_closed, defect_root_cause, entity_one_writer, spec_after_g4, high_risk_named, mandate_accept, cites_resolve, container_built, custom_room_declared, corpus_in_git, engines_invocable, withdrawn_recorded, id_allocated_once):
1862
1908
  fn(c, r)
1863
1909
  plan_dates(c, r, asof)
1864
1910
  return r
@@ -2048,7 +2094,57 @@ def gen_rtm(c: Corpus) -> dict:
2048
2094
  return {"rtm": lines}
2049
2095
 
2050
2096
 
2051
- def gen_status(c: Corpus, rtm: dict, result: Result) -> dict:
2097
+ def _active_mandates(c: Corpus, asof: dt.date | None = None) -> dict:
2098
+ """Project active mandate status into status.yaml: resolution (one | none | ambiguous),
2099
+ active_ids list, and active_mandate dict.
2100
+
2101
+ An active accepted mandate MUST have type: mandate, status: accepted, an unexpired
2102
+ expires date (today <= expires), and no superseded_by or status: superseded.
2103
+ """
2104
+ today = asof or dt.date.today()
2105
+ actives: list[dict] = []
2106
+ for dec in c.decs:
2107
+ if str(dec.get("type") or "") != "mandate":
2108
+ continue
2109
+ status = str(dec.get("status") or "")
2110
+ if status != "accepted":
2111
+ continue
2112
+ did = str(dec.get("id") or "")
2113
+ ref = str(dec.get("superseded_by") or _dec_fm(c, dec).get("superseded_by") or "").strip()
2114
+ if ref or status == "superseded":
2115
+ continue
2116
+ params = dec.get("mandate") if isinstance(dec.get("mandate"), dict) else {}
2117
+ expires = _dec_date(c, {"date": params.get("expires")})
2118
+ if expires is None or expires < today:
2119
+ continue
2120
+ actives.append({
2121
+ "id": did,
2122
+ "status": status,
2123
+ "expires": expires.isoformat(),
2124
+ "scope": params.get("scope", "all"),
2125
+ })
2126
+ actives.sort(key=lambda a: a["id"])
2127
+ if len(actives) == 1:
2128
+ return {
2129
+ "resolution": "one",
2130
+ "active_ids": [actives[0]["id"]],
2131
+ "active_mandate": actives[0],
2132
+ }
2133
+ elif len(actives) > 1:
2134
+ return {
2135
+ "resolution": "ambiguous",
2136
+ "active_ids": [a["id"] for a in actives],
2137
+ "active_mandate": None,
2138
+ }
2139
+ else:
2140
+ return {
2141
+ "resolution": "none",
2142
+ "active_ids": [],
2143
+ "active_mandate": None,
2144
+ }
2145
+
2146
+
2147
+ def gen_status(c: Corpus, rtm: dict, result: Result, asof: dt.date | None = None) -> dict:
2052
2148
  lines = rtm.get("rtm") or []
2053
2149
  counted = [line for line in lines if not line.get("exempt")]
2054
2150
  exempt = len(lines) - len(counted)
@@ -2071,6 +2167,7 @@ def gen_status(c: Corpus, rtm: dict, result: Result) -> dict:
2071
2167
  "validators_red": result.red,
2072
2168
  "validators_skipped": dict(sorted(result.skipped.items())),
2073
2169
  "open_questions": _question_budget(c),
2170
+ "mandates": _active_mandates(c, asof),
2074
2171
  }
2075
2172
 
2076
2173
 
@@ -2870,7 +2967,7 @@ def page_sdd(c: Corpus, pid: str) -> str:
2870
2967
  return "\n".join(parts) + "\n"
2871
2968
 
2872
2969
 
2873
- def generate(c: Corpus, result: Result) -> list[Path]:
2970
+ def generate(c: Corpus, result: Result, asof: dt.date | None = None) -> list[Path]:
2874
2971
  """Machine tables into `.control/generated/`; every page a human reads into the two rendered
2875
2972
  trees, at the mirror path of the working document it projects."""
2876
2973
  out_dir = c.root / ".control" / "generated"
@@ -2881,7 +2978,7 @@ def generate(c: Corpus, result: Result) -> list[Path]:
2881
2978
  "risks": gen_risks(c),
2882
2979
  "dag": gen_dag(c),
2883
2980
  "rtm": rtm,
2884
- "status": gen_status(c, rtm, result),
2981
+ "status": gen_status(c, rtm, result, asof),
2885
2982
  }
2886
2983
  written = []
2887
2984
  for name in GENERATED_ORDER:
@@ -2938,6 +3035,9 @@ def main(argv: list[str] | None = None) -> int:
2938
3035
  parser.add_argument("--asof", default=None,
2939
3036
  help="reference date for plan-dates, format YYYY-MM-DD (default: today). "
2940
3037
  "Stated explicitly so a run can be repeated exactly")
3038
+ parser.add_argument("--baseline", nargs="?", const=".github/validate-baseline.txt", default=None,
3039
+ help="path to baseline findings file (default: .github/validate-baseline.txt); "
3040
+ "exits 0 if current findings match baseline exactly")
2941
3041
  args = parser.parse_args(argv)
2942
3042
 
2943
3043
  if not args.check and not args.generate:
@@ -2953,10 +3053,27 @@ def main(argv: list[str] | None = None) -> int:
2953
3053
  result = run_checks(corpus, asof)
2954
3054
 
2955
3055
  if args.generate:
2956
- for path in generate(corpus, result):
3056
+ for path in generate(corpus, result, asof):
2957
3057
  print(f" wrote {path.relative_to(root).as_posix()}")
2958
3058
 
2959
3059
  if result.findings:
3060
+ if args.baseline:
3061
+ base_p = Path(args.baseline)
3062
+ if not base_p.is_absolute():
3063
+ base_p = root / base_p
3064
+ if base_p.is_file():
3065
+ current_fmt = [f" {f.vid:<26} {f.subject}: {f.message}".rstrip()
3066
+ for f in sorted(result.findings, key=lambda f: f.sort_key)]
3067
+ base_lines = [l.rstrip() for l in base_p.read_text(encoding="utf-8").splitlines() if l.strip()]
3068
+ if sorted(current_fmt) == sorted(base_lines):
3069
+ print(f"\nGREEN (baseline match) — {len(result.findings)} finding(s) match baseline `{base_p.relative_to(root).as_posix()}`")
3070
+ if result.skipped:
3071
+ print("\nSkipped:")
3072
+ for vid, why in sorted(result.skipped.items()):
3073
+ print(f" {vid:<26} {why}")
3074
+ print(f"\nV14 reference date: {asof.isoformat()}")
3075
+ return 0
3076
+
2960
3077
  print(f"\nRED — {len(result.findings)} findings across {len(result.red)} validators\n")
2961
3078
  for finding in sorted(result.findings, key=lambda f: f.sort_key):
2962
3079
  print(f" {finding.vid:<26} {finding.subject}: {finding.message}")
@@ -14,15 +14,18 @@ hold, the same panel reviews the code. What changes is who answers when a skill
14
14
 
15
15
  | Door | When | Does | Asks |
16
16
  |---|---|---|---|
17
- | **Preflight** | **No mandate row exists at all**, and the owner asked for one in this turn | Checks everything, prints one page, waits for the owner's confirmation, writes the mandate, starts the loop | Yes — this is the only place this skill MAY ask |
17
+ | **Preflight** | **No active accepted mandate exists**, and the owner asked for one in this turn | Checks everything, prints one page, waits for the owner's confirmation, writes the mandate, starts the loop | Yes — this is the only place this skill MAY ask |
18
18
  | **Iteration** | A mandate at `accepted` whose `expires` has not passed | Reads the registry and the ledger's `## Resume`, works from where the last iteration stopped for as long as it safely can, records, returns only at one of three stops | **Never** |
19
19
  | **Finish, lapsed** | A mandate at `accepted` whose `expires` **has** passed | Goes straight to § Finish, marks the run lapsed, cancels the loop | **Never** |
20
+ | **Terminal / No-op** | A mandate at `applied` or `superseded` (or unattended invocation with no unworked mandate) | Clean no-op: reports mandate already completed/lapsed, cancels any active loop task, and halts immediately | **Never** |
20
21
 
21
22
  **A run MUST NOT write itself a mandate.** Preflight is reachable only when the owner asked for it in the
22
23
  turn that is running; a loop firing MUST NOT open it, whatever the mandate's state. Without that, an expired
23
24
  mandate would put the next firing back at preflight — where the defaults are already filled in and nobody is
24
25
  awake to refuse them — and the run would renew its own authority. The lapsed door exists precisely so the
25
- expiry ends the run instead of restarting it.
26
+ expiry ends the run instead of restarting it. If an unattended iteration fires after a mandate has already
27
+ reached `status: applied` or `superseded`, the skill MUST NOT attempt to re-open Preflight or restart work;
28
+ it MUST perform a clean, single-line no-op and cancel any lingering loop scheduler.
26
29
 
27
30
  Typing `/wdi-autopilot` while a mandate is active opens the iteration door, not the preflight. To change a
28
31
  setting, the owner supersedes the mandate with a new one — `wdi-decision` owns supersession. A superseded
@@ -44,12 +47,12 @@ NOT start the loop while any row in the first two groups is red.
44
47
  | | This skill and `wdi-build` are themselves invocable — no `skillOverrides` entry in `settings.json` set to `off` or `user-invocable-only` | Either is overridden. Nothing else can start the loop, and the override is silent |
45
48
  | | The tracker the engines publish to is configured — `docs/agents/issue-tracker.md`, written once by `/setup-matt-pocock-skills` | Missing. `to-tickets` would stop to ask for it, and this skill never asks; the owner runs the setup before confirming |
46
49
  | | Reviewers separate from the builder can be dispatched | The session cannot spawn a second agent and any touched component is `risk_accepted: low` — Step 3 of `wdi-build` would block |
47
- | | `.constitution/project/codebase-stack-guide.md` names build and test commands, **and the test command exits 0 here** | Absent or failing. Every ticket's "full suite green once" and the smoke test read it. Found at minute one, not at hour six |
48
- | | The remote accepts the run branch — `git push --dry-run` — and `main` is reachable as a PR base | Auth or remote failure. The first real push is at the first spec close, hours in |
50
+ | | `.constitution/project/codebase-stack-guide.md` names build and test commands, **and the test command exits 0 here** (prefer quiet output flags, e.g. `-- --quiet`, to keep context compact; on non-zero exit, surface the failure details and abort preflight) | Absent or failing. Every ticket's "full suite green once" and the smoke test read it. Found at minute one, not at hour six |
51
+ | | The remote accepts the run branch — `git push --dry-run` — and `development_branch` (`policy.development_branch`, default `main`) is reachable as a PR base | Auth or remote failure. Fail-closed: if the configured `development_branch` does not exist locally or on remote (`refs/heads/<branch>` or `refs/remotes/origin/<branch>`), stop immediately and report to maintainer; MUST NOT guess or silently fall back to `main` (`.constitution/method/branch-guide.md`). The first real push is at the first spec close, hours in |
49
52
  | | A CI workflow is configured | None. § Finish would wait for checks that never arrive; say so and read the local suite as the evidence instead |
50
53
  | | **No workflow fires on an intermediate push** — the run branch is pushed dozens of times and a metered runner MUST NOT start on any of them. `ci-guide.md` § Trigger shape is the check: `workflow_dispatch` present, the automatic trigger `pull_request` `types: [ready_for_review]`, no bare `on: push` | A workflow triggers on every push. Fix it before the mandate is written — one autopilot run over fifteen tickets has spent most of a month's allowance in two days — or, where the workflow is not this repo's to change, the run holds every intermediate push and the preflight page says so |
51
- | **Position** | `gates_passed` in `index.yaml`, `g4_passed` per component, validators green (`validate.py`) | A red validator. Name it; autopilot MUST NOT start on a corpus already red |
52
- | | An isolated worktree | A shared checkout. `wdi-build` refuses one, so this skill refuses earlier |
54
+ | **Position** | `gates_passed` in `index.yaml`, `g4_passed` per component, validators green (`uv run .constitution/method/scripts/validate.py --baseline`) | A red validator. Name it; autopilot MUST NOT start on a corpus already red |
55
+ | | An isolated working tree | A shared or dirty checkout. Permitted: an isolated linked worktree (`git worktree add`), or an exclusive primary working tree checked out to `autopilot/<mandate-id>` with a clean working tree (`git status --porcelain` empty) dedicated to this run (`.constitution/method/branch-guide.md`). `wdi-build` refuses un-isolated checkouts, so this skill refuses earlier |
53
56
  | | `from_gate` — the first gate the run will hold itself | Below the last passed gate. Default: the gate after the last one passed |
54
57
  | **Settings** | `scope` — the `FR` ids to deliver, or `all` | — (default `all` open `FR`) |
55
58
  | | `parked` — what stops for the owner instead of being decided: any of `promise` · `ad-n` · `sensitive` | — (default **`ad-n`**, and nothing else. `decision-guide.md` says narrowing an invariant MUST NOT be softened further, so removing it is the owner's to say out loud — not a default they never saw) |
@@ -59,6 +62,7 @@ NOT start the loop while any row in the first two groups is red.
59
62
  | | Where the ledger and the final report will be written | — |
60
63
  | | The **run branch** — `autopilot/<mandate-id>`, using the next free `DEC-` id from `decisions.yaml`, which the mandate then takes — and that the run will open **one** PR from it | The branch already exists with commits nobody can account for |
61
64
  | **Runtime** | The session runs with permission prompts bypassed | Cannot be verified from inside the session. Printed as a line the owner confirms |
65
+ | | Session survivability: on Linux/remote SSH, run inside `tmux` or `screen`; on Windows, in a dedicated persistent Windows Terminal window | Ephemeral terminal that aborts the loop on disconnect |
62
66
 
63
67
  **Every row arrives with its default already in it**, and the owner changes only what they want changed —
64
68
  the same rule the installer follows. A preflight that asks fourteen questions one at a time has failed.
@@ -102,11 +106,15 @@ On the owner's confirmation, and not before:
102
106
  The interval is the **pause between** iterations, not the length of one. An iteration that outlives it
103
107
  finishes first; the next firing waits.
104
108
 
109
+ **Session survivability across environments:**
110
+ - **Linux / Remote SSH:** Run inside a session manager such as `tmux` (`tmux new -s autopilot`) or `screen` before starting the loop. Disconnecting SSH or closing the terminal then leaves the autonomous loop running unharmed.
111
+ - **Windows (PowerShell / Windows Terminal):** `tmux` is not native to Windows PowerShell. Run the session in a dedicated persistent Windows Terminal tab or window left active, or via background subagent tools (`run_in_background`). Do not invoke or require `tmux` on Windows environments.
112
+
105
113
  ## Door 2 — One iteration
106
114
 
107
115
  Open with three reads, in this order:
108
116
 
109
- 1. `validate.py --generate`. `.control/generated/` is written by that flag and nothing else, so without it
117
+ 1. `uv run .constitution/method/scripts/validate.py --generate --baseline`. `.control/generated/` is written by that flag and nothing else, so without it
110
118
  every iteration reads a status file from before the run and re-holds gates that already passed. It sweeps
111
119
  the validators for free at the same time.
112
120
  2. **Reconcile `## Resume` against git.** Compare the run branch HEAD with the commit Resume names. A
@@ -195,7 +203,7 @@ the method's, and none of them relaxes here:
195
203
 
196
204
  ### One run, one branch, one PR
197
205
 
198
- A mandate is **one unit of work**, and it reaches `main` through **one door**: a single PR from the run
206
+ A mandate is **one unit of work**, and it reaches the active development branch (`policy.development_branch`, default `main`) through **one door**: a single PR from the run
199
207
  branch, which the **owner** merges after the final review. This is what makes the result reviewable as a
200
208
  whole instead of as a stream of PRs nobody read.
201
209
 
@@ -209,9 +217,9 @@ branch, one PR, nothing else on the remote.
209
217
  |---|---|
210
218
  | Step 4 pushes a ticket branch and opens a PR per ticket | The ticket is committed to the run branch — directly, or merged in from its own worktree by the coordinator. The ticket-closing checklist is still answered first. **No PR per ticket** |
211
219
  | Step 5 watches CI per PR | The coordinator pushes the run branch **at every spec close** — and that push starts **no cloud run**; the first push opens the one PR as a **draft**. CI runs **once**, at § Finish, and is judged exactly as Step 5 says on the pushed head SHA |
212
- | `MUST NOT merge` | Holds harder. The run never merges to `main`; the owner does, once, after § Finish |
220
+ | `MUST NOT merge` | Holds harder. The run never merges to `primary_branch` or `development_branch`; the owner does, once, after § Finish |
213
221
 
214
- A second PR is a red flag. Where a change cannot ride the run branch — a hotfix `main` needs today — it is
222
+ A second PR is a red flag. Where a change cannot ride the run branch — an urgent fix the target branch needs today — it is
215
223
  reported for the owner, not opened by the run.
216
224
 
217
225
  ### Cycle-end CI — the cloud runner fires once
@@ -283,6 +291,8 @@ what it decided while running* — and this is exactly one. `memlog-home` holds
283
291
 
284
292
  Frontmatter `artifact:` names the mandate's `DEC-` file — `memlog-home` demands it of every memlog.
285
293
 
294
+ The mandate ledger file (`.control/memlog/autopilot-<mandate-id>.md` and its companion directory `.control/memlog/autopilot-<mandate-id>/`) is **permanent** and MUST NOT be deleted or removed during spec pruning or housekeeping — it preserves the provenance of autonomous runs.
295
+
286
296
  **It has two readers who want opposite things, and that is what shapes it.** The next iteration needs a
287
297
  resume point: where the last one stopped and what to do now. The owner needs every decision, with what it
288
298
  cost. Serving both from one flat table is what made a real ledger reach 41 KB by its twenty-second
@@ -351,11 +361,16 @@ applies it.
351
361
 
352
362
  When § The work table reaches § Finish:
353
363
 
354
- 1. **Smoke test.** At `smoke_test: agent`: run the application with the commands
364
+ 1. **Smoke test.** Standardized smoke test locations:
365
+ - **Automated spec test:** `.scratch/<spec-id>-<slug>/smoke/` — spec-level smoke artifacts that archive alongside the spec when closed.
366
+ - **Human interactive test:** `.work/smoke/<target>.md` — ephemeral test scripts and physical run notes, cleaned up once the task finishes.
367
+ - **Mandate ledger:** `.control/memlog/autopilot-<mandate-id>.md` — permanent record of pass/fail results per `FR`, never pruned.
368
+
369
+ At `smoke_test: agent`: run the application with the commands
355
370
  `.constitution/project/codebase-stack-guide.md` names, exercise every closed `FR`'s proof of done from
356
371
  the PRD, record pass or fail per `FR` in the ledger. At `owner`: run nothing; the test script below is
357
372
  the whole deliverable.
358
- 2. `validate.py --generate`, then `wdi-report` intent `progress`.
373
+ 2. `uv run .constitution/method/scripts/validate.py --generate --baseline`, then `wdi-report` intent `progress`.
359
374
  3. Raise the mandate to `applied`, `touches` naming the ledger, and rewrite `## Resume` one last time so it
360
375
  reads as the run's end state rather than a step that never came.
361
376
  4. **Leave the run branch in a state the owner can merge.** A ticket still in flight is either finished or
@@ -381,8 +396,14 @@ When § The work table reaches § Finish:
381
396
  Green marks the one PR **ready for review**. Red keeps it a **draft** and is reported red: a PR marked
382
397
  ready is an invitation to merge, and the run MUST NOT extend one over a red branch, nor patch to turn it
383
398
  green at the door.
384
- 5. Cancel the loop: in Claude Code, the `loop` skill's cancel; elsewhere, tell the owner the loop has nothing
385
- left to do.
399
+
400
+ **Signed CI override:** If the mandate decision block in `decisions.yaml` records a signed `ci_override`
401
+ (e.g. `ci_override: local-only-approved-by: "<Person, Date>"` per `.constitution/method/ci-guide.md`),
402
+ the run MUST NOT mark the draft PR ready (`gh pr ready`). The PR MUST remain as a Draft, no cloud runner
403
+ is awaited, and the final output report explicitly states: *"locally verified; cloud verification intentionally deferred by mandate"*.
404
+ 5. **Cancel the loop cleanly:** in Claude Code, inspect scheduled jobs via `CronList`, identify the job firing
405
+ `/wdi-autopilot`, and call `CronDelete` on its task ID to eliminate zombie loop firings; elsewhere, call the platform's
406
+ loop cancellation mechanism or inform the owner that the loop has completed its mandate and has nothing left to do.
386
407
  6. Write the final report as the Output below. The owner merges; the run never does.
387
408
 
388
409
  ## Red Flags — STOP
@@ -409,7 +430,7 @@ When § The work table reaches § Finish:
409
430
  - Spinning until `expires` on work that is not runnable, instead of finishing and naming the blockers
410
431
  - Returning after one step while work remains and none of the three stops applies
411
432
  - A second PR, any branch but the run branch pushed, a working branch left alive at Finish, or any merge
412
- into `main` by the run — working branches and worktrees during the run are fine; surviving ones are not
433
+ into `primary_branch` or `development_branch` by the run — working branches and worktrees during the run are fine; surviving ones are not
413
434
  - Merging a red ticket into the run branch, or patching the branch forward instead of reverting the merge
414
435
  - Marking the PR ready over red CI, or handing over a run branch with a ticket half-applied
415
436
  - Parallel builders sharing a worktree, or a registry written by anyone but the coordinator
@@ -39,7 +39,7 @@ that names the engine's `SKILL.md` path and carries out its process — or from
39
39
  names which; the seams, the `to-tickets` quiz, and § When the code turns out to be right are decided by the
40
40
  coordinator and written to the ledger, one row each; and whatever the mandate lists as `parked` still stops,
41
41
  reported for the owner rather than decided. One thing changes **shape** rather than owner: a mandate is one
42
- unit of work and reaches `main` through **one PR**, so Step 4 commits the ticket to the run branch instead of
42
+ unit of work and reaches the active development branch (`policy.development_branch`, default `main`) through **one PR**, so Step 4 commits the ticket to the run branch instead of
43
43
  opening a PR per ticket, and Step 5 splits: the coordinator pushes the run branch at every spec close, but
44
44
  **the cloud run happens once, at `wdi-autopilot` § Finish** — every intermediate push starts nothing, and
45
45
  what a spec close is judged on until then is the run branch's own full suite, run locally. The checklist,
@@ -65,10 +65,11 @@ that cannot provide them is **blocked**, not excused.
65
65
  | Check | When it fails |
66
66
  |---|---|
67
67
  | Every component this spec touches has passed G4, **or** sits at `mode: catalog` | Route to `wdi-component`. `spec-after-g4` checks it, and `catalog` skipping G4 is by design, not an exception |
68
- | An isolated worktree | Isolate first. MUST NOT run in a shared checkout |
68
+ | An isolated working tree | Isolate first. MUST NOT run in a shared or dirty checkout. Permitted isolation models: an isolated linked worktree (`git worktree add`), or an exclusive primary working tree checked out to the task branch with a clean status dedicated to this task (`.constitution/method/branch-guide.md`). Standing exception: Phase 1 (Open the spec) and Phase 2 (The contract and the tickets) authoring runs directly on `development_branch` (`policy.development_branch`, default `main`) with no code changes; the isolated working tree requirement binds Phase 3 (Ship each ticket) onward |
69
69
  | Every `prd` slug names a real `.what/_prd/<initiative>/` folder | A spec without a promise covering it is a spec nobody agreed to (`spec-names-release-prd`) |
70
+ | Development branch verified locally or on remote (`refs/heads/<branch>` or `refs/remotes/origin/<branch>`) | Branch missing. Stop immediately and report to maintainer; MUST NOT guess or silently fall back to `main` (`.constitution/method/branch-guide.md`) |
70
71
 
71
- The repo commits straight to `main` and opens a PR only when asked. **Invoking this skill is that ask**, for
72
+ The repo commits straight to the active development branch (`policy.development_branch`, default `main`) during Phase 1 & 2 authoring and opens a PR only when asked. **Invoking this skill is that ask**, for
72
73
  this spec only; it MUST NOT be read as standing permission for the next change.
73
74
 
74
75
  ## Phase 1 — Open the spec
@@ -302,7 +303,7 @@ The five items that left this list moved to Phase 4, where the information actua
302
303
 
303
304
  - MUST run the repository's commit/push audit before `git push`: refuse the forbidden paths, run the guard test,
304
305
  fix content on failure. A failing guard is a finding about the content — MUST NOT weaken the guard or the test.
305
- - MUST NOT push to `main`/`master`, MUST NOT force-push, MUST NOT merge.
306
+ - MUST NOT push directly to protected branches (`primary_branch` or `development_branch` from `policy:`, default `main`), MUST NOT force-push, MUST NOT merge.
306
307
  - The coordinator MUST be the hand that pushes and opens the PR.
307
308
 
308
309
  ### Step 5 — watch CI, then judge
@@ -349,6 +350,11 @@ out.
349
350
  5. **RTM green.** Every traceability row for this spec is closed. New risks are in the risk register with an
350
351
  owner.
351
352
  6. Mark the spec `status: closed` in `specs.yaml`.
353
+ 7. **Housekeeping (Archive or Prune).** Once marked closed, offer the maintainer the choice to clean up
354
+ the completed spec directory from `.scratch/<spec-id>-<slug>/` using `/wdi-prune-or-archive`:
355
+ - **Archive:** `/wdi-prune-or-archive --spec <spec-id> --archive` (moves the directory to `.archive/specs/<spec-id>-<slug>/` and updates `spec_folder` in `specs.yaml`).
356
+ - **Prune:** `/wdi-prune-or-archive --spec <spec-id> --prune` (removes the completed spec directory from git and disk while preserving RTM history in `specs.yaml`).
357
+ If the maintainer declines or defers, leave the directory in `.scratch/`.
352
358
 
353
359
  The retrospective step is **repealed**, and `RTR-` with it. It was the only thing size `L` decided, and the
354
360
  only thing `V19` checked.
@@ -0,0 +1,138 @@
1
+ ---
2
+ name: wdi-daily-autopilot
3
+ description: Compose and launch the autonomous daily loop routine (default 10m interval) with self code-review and peer-review runners resolved from local configuration. Invoke as `/wdi-daily-autopilot [in-session] [peer] [interval] [--skip-peer-review]`.
4
+ disable-model-invocation: true
5
+ ---
6
+
7
+ # WDI Daily Autopilot Launch
8
+
9
+ Composes the autonomous daily engineering routine, verifies or initiates the owner-accepted mandate
10
+ required by `wdi-autopilot`, resolves coordinator self code-review and independent peer-review dispatch
11
+ from local configuration or agent rules, and launches the execution via `/loop <interval>` (default
12
+ `10m`).
13
+
14
+ `/wdi-daily-autopilot [self-review] [peer] [interval] [--skip-peer-review|--no-review]`:
15
+ - `[self-review]` — coordinator self code-review pass model (`default` or explicit model slug; alias `[in-session]`).
16
+ `default` indicates the current coordinating session executes self code-review directly.
17
+ - `[peer]` — independent peer reviewer runner or model identifier.
18
+ - `[interval]` — optional loop interval matching `^\d+[smhd]$` (e.g., `5m`, `10m`, `15m`). Defaults
19
+ to `10m` when omitted.
20
+ - `--skip-peer-review` / `--no-review` — bypasses the secondary peer review pass. MUST NOT disable
21
+ coordinator self code-review, TDD cycles, or automated test suites.
22
+
23
+ ## 0. Precondition
24
+
25
+ Confirm `.control/registry/index.yaml` exists in the repo root. If it does not, this is not a WDI
26
+ Method product repo — report that and stop.
27
+
28
+ ## 1. Parse Arguments and Flags
29
+
30
+ Parse inputs unambiguously using these rules:
31
+ 1. Check for review bypass flags: `--skip-peer-review` or `--no-review`. If present, mark peer review
32
+ as bypassed.
33
+ 2. Check for an interval token matching `^\d+[smhd]$`. If found, assign it to `<interval>`; otherwise
34
+ default `<interval>` to `10m`.
35
+ 3. For remaining positional arguments:
36
+ - If 1 argument remains: assign to `<peer>`, and default `<self-review>` to `default`.
37
+ - If 2 arguments remain: assign first to `<self-review>` and second to `<peer>`.
38
+ - If 0 arguments remain: read defaults from `.control/custom-dispatch.yaml` if present, else default
39
+ both to `default`.
40
+
41
+ ## 2. Resolve Runner Configuration
42
+
43
+ Inspect the repository for `.control/custom-dispatch.yaml` (if not found in the current working directory and running inside a linked git worktree, resolve it from the main repository root via `(git rev-parse --git-common-dir)/..`):
44
+ - **If `.control/custom-dispatch.yaml` exists**:
45
+ Read `runners:`, `roles:`, and `review_policy:`.
46
+ - If `review_policy.peer_review` is explicitly `false`, or if `roles.reviewer` is set to `none`, mark peer review as bypassed.
47
+ - If `<peer>` was not explicitly specified on the command line, use `roles.reviewer`. If resolved `<peer>` is `none`,
48
+ mark peer review as bypassed (coordinator self-review only).
49
+ - If `roles.deep_analyst` is `none`, document review and architecture analysis are handled by the main reviewer or
50
+ coordinator directly, without dispatching a separate deep analyst process.
51
+ - `roles.builder` is fixed to `coordinator`: the coordinating session implements code directly inside the active run worktree (ensuring tight TDD cycles, direct verification, and eliminating delegation/handoff hallucinations). Coding delegation (whether `in-session` subagents or external builder runners) is prohibited in the daily routine.
52
+ Guardrail: coordinator direct implementation MUST NOT eliminate independent peer review for components whose `risk_accepted` is not `low`; `roles.reviewer` MUST NOT be set to `none` in such cases.
53
+ - Resolve runner dispatch by `type:` for reviewer and deep analyst:
54
+ - `auto`: Evaluates whether the runner's target model is reachable in-session from the active
55
+ session profile (per the caller's global agent collaboration rules). Dispatches in-session via `Agent`
56
+ if reachable; falls back to shell-out using `command` if unreachable in-session.
57
+ - `in-session`: Dispatches strictly via in-session `Agent` subagent.
58
+ - `shell-out`: Executes the external shell-out `command` (single-string command). If `command` is absent
59
+ or empty, stop and report immediately (fail-closed).
60
+ - **If `.control/custom-dispatch.yaml` does not exist**:
61
+ Resolve reviewer dispatch through the caller's active CLI environment.
62
+
63
+ ## 3. Mandate Verification & Preflight Requirement
64
+
65
+ Per `wdi-autopilot` § Preflight, unattended loop iterations **require an active accepted mandate** in
66
+ `.control/registry/decisions.yaml` whose expiry date has not lapsed. A loop MUST NOT self-authorize
67
+ its own mandate.
68
+
69
+ 1. Check if an active accepted mandate exists:
70
+ - An active mandate MUST have BOTH `type: mandate` and `status: accepted` with an unexpired `expires:` date
71
+ (`today <= expires`) and no `superseded_by:` or `status: superseded`.
72
+ - **Primary lookup ($O(1)$):** Read `.control/generated/status.yaml` and inspect `mandates:`:
73
+ - If `mandates.resolution: one`: the active mandate is `mandates.active_mandate.id` (with its verified `expires` and `status`). Proceed directly to step 3.
74
+ - If `mandates.resolution: ambiguous`: stop and report immediately to the maintainer naming all `mandates.active_ids` (fail-closed; MUST NOT guess or choose between them).
75
+ - If `mandates.resolution: none`: proceed to step 2 (Preflight).
76
+ - **Fallback lookup (when `status.yaml` does not exist or lacks `mandates:`):**
77
+ - MUST NOT run a broad search for `type:\s*mandate` across `decisions.yaml` (which matches dozens of historical
78
+ `applied` mandates and overflows the tool output limit with hundreds of lines).
79
+ - Query specifically for an active mandate entry using multiline search (e.g. `Grep` with
80
+ `pattern: "type:\s*mandate[\s\S]{1,100}?status:\s*accepted|status:\s*accepted[\s\S]{1,100}?type:\s*mandate", multiline: true`).
81
+ Once a candidate match is found, verify its individual decision block to confirm `expires:` is present,
82
+ valid, and unexpired.
83
+ - **Fail-closed on multiple active mandates:** If more than 1 active accepted mandate is found, stop and
84
+ report immediately to the maintainer (fail-closed; MUST NOT guess or choose between them).
85
+ - MUST NOT call `Read` on the entire 1000+ line `decisions.yaml` file. When reading the latest decision ID
86
+ to determine the next candidate `DEC-` ID, read only the tail (e.g. the last 50-80 lines) of `decisions.yaml`.
87
+ 2. **If no active accepted mandate exists:**
88
+ - Execute `wdi-autopilot` Door 1 (Preflight) in this interactive turn.
89
+ - When checking open work to include in the mandate, query `specs.yaml` selectively (e.g. `Grep` for
90
+ `status:\s*(open|ready-for-dev)`) instead of reading all historical closed specs into context.
91
+ Inspect the candidate spec's folder using the `spec_folder:` path from `specs.yaml` (MUST NOT run
92
+ broad recursive searches on all of `.scratch/`).
93
+ - Run validator preflight via `uv run .constitution/method/scripts/validate.py --check --baseline`
94
+ (or `--generate --baseline`). If `.github/validate-baseline.txt` exists in the repo, the `--baseline` flag
95
+ guarantees that accepted repository baseline findings are recognized as green.
96
+ - When verifying the preflight test suite from `codebase-stack-guide.md`, run tests with quiet output flags
97
+ if supported (e.g. `-- --quiet` or standard concise runner flags) to avoid flooding the context window
98
+ with hundreds of verbose pass lines.
99
+ - **Fail-closed on test failure:** If the test suite command exits non-zero, immediately surface the failure
100
+ summary (the failing test names and error excerpt) and abort preflight; a mandate MUST NOT be opened on a
101
+ failing test suite.
102
+ - Present the one-page preflight summary and wait for the owner's explicit confirmation.
103
+ - Once confirmed, write the accepted mandate row into `decisions.yaml` and initialize its ledger.
104
+ 3. **If an active accepted mandate already exists:** Proceed directly to compose and launch the loop.
105
+
106
+ ## 4. Compose the Routine Mandate
107
+
108
+ Fill the standing five-point routine template:
109
+
110
+ ```markdown
111
+ Execute all FR/Tickets/Specs to completion under the active mandate:
112
+ 1. Coding & Implementation: author code directly as coordinator in the active worktree. No coding delegation — coordinator alone implements application and test changes following TDD red-to-green cycles. Boundary: coordinator edits application and test files only during ticket coding, and defers commits to step 5. Builder MUST NOT commit, push, merge, alter git branches, or write to .control/registry/ or .control/memlog/.
113
+ 2. Code review: perform code review — self code-review by coordinator, and peer review: <resolved peer command | "none due to config">.
114
+ 3. Testing & Verification: coordinator alone runs the authoritative test suite (from codebase-stack-guide.md) directly on the active worktree after the coding pass and verifies red-to-green evidence before staging or committing.
115
+ 4. Reviews and peer analysis: delegate document and architecture review to <resolved peer command | "coordinator self-review (peer review: none due to config)"> — follow wdi-review criteria for any touched architecture or spec documents.
116
+ 5. Worktree isolation & Integration: coordinator alone isolates ticket implementation into the run worktree, writes ledger decisions in .control/memlog/, stages/commits, and merges per wdi-autopilot. Peer review: <"peer reviewer inspects worktree without interfering with active build processes" | "none due to config">.
117
+ ```
118
+
119
+ If peer review was bypassed (via `roles.reviewer: none`, `review_policy.peer_review: false`, or `--skip-peer-review`),
120
+ explicitly record `peer review: none due to config` in items 2, 4, and 5, instructing the coordinator to perform direct
121
+ self-review and self-verification without external peer dispatch.
122
+
123
+ ## 5. Launch
124
+
125
+ Invoke the `loop` skill with `<resolved interval> /wdi-autopilot <composed text from step 4>`. This
126
+ starts the loop execution.
127
+
128
+ ## 6. Verify Immediate Execution
129
+
130
+ Before finishing, verify that `/wdi-autopilot` was invoked in this same turn for the first iteration
131
+ rather than remaining idle until the first cron interval tick. If it did not run immediately, invoke
132
+ `/wdi-autopilot` now to start the first iteration.
133
+
134
+ ## 7. Report and Stop
135
+
136
+ Report the resolved configuration (in-session mechanism, peer review status, active mandate ID, and
137
+ loop interval), confirm that the loop is active, and stop. MUST NOT intervene in or micromanage
138
+ subsequent loop iterations.