@garygentry/feature-forge 0.2.14 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. package/README.md +6 -3
  2. package/adapters/GENERATION-REPORT.md +20 -0
  3. package/adapters/claude/.feature-forge-bundle.json +1 -1
  4. package/adapters/claude/references/forge-config-schema.json +2 -2
  5. package/adapters/claude/scripts/forge-root.sh +47 -3
  6. package/adapters/claude/scripts/forge-session.py +30 -8
  7. package/adapters/claude/skills/forge-4-backlog/references/forge-config-schema.json +2 -2
  8. package/adapters/claude/skills/forge-5-loop/references/forge-config-schema.json +2 -2
  9. package/adapters/claude/skills/forge-guide/references/forge-config-schema.json +2 -2
  10. package/adapters/codex/.feature-forge-bundle.json +1 -1
  11. package/adapters/codex/references/forge-config-schema.json +2 -2
  12. package/adapters/codex/scripts/forge-root.sh +47 -3
  13. package/adapters/codex/scripts/forge-session.py +30 -8
  14. package/adapters/codex/skills/forge-4-backlog/references/forge-config-schema.json +2 -2
  15. package/adapters/codex/skills/forge-5-loop/references/forge-config-schema.json +2 -2
  16. package/adapters/codex/skills/forge-guide/references/forge-config-schema.json +2 -2
  17. package/adapters/copilot/.feature-forge-bundle.json +1 -1
  18. package/adapters/copilot/references/forge-config-schema.json +2 -2
  19. package/adapters/copilot/scripts/forge-root.sh +47 -3
  20. package/adapters/copilot/scripts/forge-session.py +30 -8
  21. package/adapters/copilot/skills/forge-4-backlog/references/forge-config-schema.json +2 -2
  22. package/adapters/copilot/skills/forge-5-loop/references/forge-config-schema.json +2 -2
  23. package/adapters/copilot/skills/forge-guide/references/forge-config-schema.json +2 -2
  24. package/adapters/cursor/.feature-forge-bundle.json +1 -1
  25. package/adapters/cursor/references/forge-config-schema.json +2 -2
  26. package/adapters/cursor/scripts/forge-root.sh +47 -3
  27. package/adapters/cursor/scripts/forge-session.py +30 -8
  28. package/adapters/cursor/skills/forge-4-backlog/references/forge-config-schema.json +2 -2
  29. package/adapters/cursor/skills/forge-5-loop/references/forge-config-schema.json +2 -2
  30. package/adapters/cursor/skills/forge-guide/references/forge-config-schema.json +2 -2
  31. package/adapters/gemini/.feature-forge-bundle.json +1 -1
  32. package/adapters/gemini/gemini-extension.json +1 -1
  33. package/adapters/gemini/references/forge-config-schema.json +2 -2
  34. package/adapters/gemini/scripts/forge-root.sh +47 -3
  35. package/adapters/gemini/scripts/forge-session.py +30 -8
  36. package/adapters/gemini/skills/forge-4-backlog/references/forge-config-schema.json +2 -2
  37. package/adapters/gemini/skills/forge-5-loop/references/forge-config-schema.json +2 -2
  38. package/adapters/gemini/skills/forge-guide/references/forge-config-schema.json +2 -2
  39. package/adapters/pi/.feature-forge-bundle.json +6 -0
  40. package/adapters/pi/agents/forge-researcher.md +139 -0
  41. package/adapters/pi/agents/forge-spec-writer.md +116 -0
  42. package/adapters/pi/agents/forge-verifier.md +126 -0
  43. package/adapters/pi/extensions/ask-user-question/LICENSE +21 -0
  44. package/adapters/pi/extensions/ask-user-question/README.md +91 -0
  45. package/adapters/pi/extensions/ask-user-question/ask-user-question.ts +298 -0
  46. package/adapters/pi/extensions/ask-user-question/config.ts +78 -0
  47. package/adapters/pi/extensions/ask-user-question/events.ts +57 -0
  48. package/adapters/pi/extensions/ask-user-question/index.ts +61 -0
  49. package/adapters/pi/extensions/ask-user-question/locales/de.json +27 -0
  50. package/adapters/pi/extensions/ask-user-question/locales/en.json +27 -0
  51. package/adapters/pi/extensions/ask-user-question/locales/es.json +27 -0
  52. package/adapters/pi/extensions/ask-user-question/locales/fr.json +27 -0
  53. package/adapters/pi/extensions/ask-user-question/locales/pt-BR.json +27 -0
  54. package/adapters/pi/extensions/ask-user-question/locales/pt.json +27 -0
  55. package/adapters/pi/extensions/ask-user-question/locales/ru.json +27 -0
  56. package/adapters/pi/extensions/ask-user-question/locales/uk.json +27 -0
  57. package/adapters/pi/extensions/ask-user-question/locales/zh.json +29 -0
  58. package/adapters/pi/extensions/ask-user-question/reconcile.ts +49 -0
  59. package/adapters/pi/extensions/ask-user-question/rpc-fallback.ts +168 -0
  60. package/adapters/pi/extensions/ask-user-question/state/build-questionnaire.ts +302 -0
  61. package/adapters/pi/extensions/ask-user-question/state/i18n-bridge.ts +53 -0
  62. package/adapters/pi/extensions/ask-user-question/state/key-router.ts +277 -0
  63. package/adapters/pi/extensions/ask-user-question/state/questionnaire-session.ts +234 -0
  64. package/adapters/pi/extensions/ask-user-question/state/row-intent.ts +145 -0
  65. package/adapters/pi/extensions/ask-user-question/state/selectors/contract.ts +26 -0
  66. package/adapters/pi/extensions/ask-user-question/state/selectors/derivations.ts +42 -0
  67. package/adapters/pi/extensions/ask-user-question/state/selectors/focus.ts +19 -0
  68. package/adapters/pi/extensions/ask-user-question/state/selectors/projections.ts +101 -0
  69. package/adapters/pi/extensions/ask-user-question/state/state-reducer.ts +292 -0
  70. package/adapters/pi/extensions/ask-user-question/state/state.ts +55 -0
  71. package/adapters/pi/extensions/ask-user-question/tool/format-answer.ts +31 -0
  72. package/adapters/pi/extensions/ask-user-question/tool/response-envelope.ts +49 -0
  73. package/adapters/pi/extensions/ask-user-question/tool/types.ts +147 -0
  74. package/adapters/pi/extensions/ask-user-question/tool/validate-questionnaire.ts +58 -0
  75. package/adapters/pi/extensions/ask-user-question/vendor-config-shim.ts +65 -0
  76. package/adapters/pi/extensions/ask-user-question/view/component-binding.ts +47 -0
  77. package/adapters/pi/extensions/ask-user-question/view/components/inline-input.ts +98 -0
  78. package/adapters/pi/extensions/ask-user-question/view/components/multi-select-view.ts +193 -0
  79. package/adapters/pi/extensions/ask-user-question/view/components/option-list-view.ts +70 -0
  80. package/adapters/pi/extensions/ask-user-question/view/components/preview/markdown-content-cache.ts +79 -0
  81. package/adapters/pi/extensions/ask-user-question/view/components/preview/preview-block-renderer.ts +111 -0
  82. package/adapters/pi/extensions/ask-user-question/view/components/preview/preview-box-renderer.ts +88 -0
  83. package/adapters/pi/extensions/ask-user-question/view/components/preview/preview-layout-decider.ts +202 -0
  84. package/adapters/pi/extensions/ask-user-question/view/components/preview/preview-pane.ts +228 -0
  85. package/adapters/pi/extensions/ask-user-question/view/components/submit-picker.ts +67 -0
  86. package/adapters/pi/extensions/ask-user-question/view/components/tab-bar.ts +59 -0
  87. package/adapters/pi/extensions/ask-user-question/view/components/wrapping-select.ts +293 -0
  88. package/adapters/pi/extensions/ask-user-question/view/dialog-builder.ts +224 -0
  89. package/adapters/pi/extensions/ask-user-question/view/props-adapter.ts +125 -0
  90. package/adapters/pi/extensions/ask-user-question/view/stateful-view.ts +26 -0
  91. package/adapters/pi/extensions/ask-user-question/view/tab-components.ts +18 -0
  92. package/adapters/pi/extensions/ask-user-question/view/tab-content-strategy.ts +252 -0
  93. package/adapters/pi/package.json +26 -0
  94. package/adapters/pi/references/epic-manifest-schema.json +125 -0
  95. package/adapters/pi/references/forge-config-schema.json +236 -0
  96. package/adapters/pi/references/pipeline-state-schema.json +191 -0
  97. package/adapters/pi/references/portable-root.md +71 -0
  98. package/adapters/pi/references/process-overview.md +143 -0
  99. package/adapters/pi/references/ralph-loop-contract.md +221 -0
  100. package/adapters/pi/references/shared-conventions.md +295 -0
  101. package/adapters/pi/references/skill-frontmatter.schema.json +17 -0
  102. package/adapters/pi/references/stack-resolution.md +54 -0
  103. package/adapters/pi/references/stacks/_generic.md +111 -0
  104. package/adapters/pi/references/stacks/go.md +157 -0
  105. package/adapters/pi/references/stacks/python.md +184 -0
  106. package/adapters/pi/references/stacks/rust.md +170 -0
  107. package/adapters/pi/references/stacks/typescript.md +134 -0
  108. package/adapters/pi/references/stage-exit-protocol.md +258 -0
  109. package/adapters/pi/references/templates/specs-hygiene/AGENTS.md +32 -0
  110. package/adapters/pi/references/templates/specs-hygiene/CLAUDE.md +31 -0
  111. package/adapters/pi/references/vendor-construct-inventory.md +50 -0
  112. package/adapters/pi/scripts/epic-manifest.py +1694 -0
  113. package/adapters/pi/scripts/forge-bootstrap.py +1070 -0
  114. package/adapters/pi/scripts/forge-init.sh +58 -0
  115. package/adapters/pi/scripts/forge-root.sh +179 -0
  116. package/adapters/pi/scripts/forge-session.py +1888 -0
  117. package/adapters/pi/scripts/validate-traceability.py +150 -0
  118. package/adapters/pi/skills/forge/SKILL.md +243 -0
  119. package/adapters/pi/skills/forge/references/pipeline-state-schema.json +191 -0
  120. package/adapters/pi/skills/forge/references/process-overview.md +143 -0
  121. package/adapters/pi/skills/forge/references/shared-conventions.md +295 -0
  122. package/adapters/pi/skills/forge/references/stage-exit-protocol.md +258 -0
  123. package/adapters/pi/skills/forge-0-epic/SKILL.md +308 -0
  124. package/adapters/pi/skills/forge-0-epic/references/edit-mode.md +266 -0
  125. package/adapters/pi/skills/forge-0-epic/references/epic-manifest-subcommands.md +75 -0
  126. package/adapters/pi/skills/forge-0-epic/references/pipeline-state-schema.json +191 -0
  127. package/adapters/pi/skills/forge-0-epic/references/portable-root.md +71 -0
  128. package/adapters/pi/skills/forge-0-epic/references/shared-conventions.md +295 -0
  129. package/adapters/pi/skills/forge-0-epic/references/stage-exit-protocol.md +258 -0
  130. package/adapters/pi/skills/forge-1-prd/SKILL.md +164 -0
  131. package/adapters/pi/skills/forge-1-prd/references/pipeline-state-schema.json +191 -0
  132. package/adapters/pi/skills/forge-1-prd/references/prd-template.md +106 -0
  133. package/adapters/pi/skills/forge-1-prd/references/shared-conventions.md +295 -0
  134. package/adapters/pi/skills/forge-1-prd/references/stage-exit-protocol.md +258 -0
  135. package/adapters/pi/skills/forge-2-tech/SKILL.md +225 -0
  136. package/adapters/pi/skills/forge-2-tech/references/pipeline-state-schema.json +191 -0
  137. package/adapters/pi/skills/forge-2-tech/references/shared-conventions.md +295 -0
  138. package/adapters/pi/skills/forge-2-tech/references/stack-discovery-checklist.md +95 -0
  139. package/adapters/pi/skills/forge-2-tech/references/stack-resolution.md +54 -0
  140. package/adapters/pi/skills/forge-2-tech/references/stacks/_generic.md +111 -0
  141. package/adapters/pi/skills/forge-2-tech/references/stacks/go.md +157 -0
  142. package/adapters/pi/skills/forge-2-tech/references/stacks/python.md +184 -0
  143. package/adapters/pi/skills/forge-2-tech/references/stacks/rust.md +170 -0
  144. package/adapters/pi/skills/forge-2-tech/references/stacks/typescript.md +134 -0
  145. package/adapters/pi/skills/forge-2-tech/references/stage-exit-protocol.md +258 -0
  146. package/adapters/pi/skills/forge-3-specs/SKILL.md +178 -0
  147. package/adapters/pi/skills/forge-3-specs/references/pipeline-state-schema.json +191 -0
  148. package/adapters/pi/skills/forge-3-specs/references/shared-conventions.md +295 -0
  149. package/adapters/pi/skills/forge-3-specs/references/spec-archetypes.md +106 -0
  150. package/adapters/pi/skills/forge-3-specs/references/spec-examples.md +71 -0
  151. package/adapters/pi/skills/forge-3-specs/references/stacks/_generic.md +111 -0
  152. package/adapters/pi/skills/forge-3-specs/references/stacks/go.md +157 -0
  153. package/adapters/pi/skills/forge-3-specs/references/stacks/python.md +184 -0
  154. package/adapters/pi/skills/forge-3-specs/references/stacks/rust.md +170 -0
  155. package/adapters/pi/skills/forge-3-specs/references/stacks/typescript.md +134 -0
  156. package/adapters/pi/skills/forge-3-specs/references/stage-exit-protocol.md +258 -0
  157. package/adapters/pi/skills/forge-4-backlog/SKILL.md +175 -0
  158. package/adapters/pi/skills/forge-4-backlog/references/forge-config-schema.json +236 -0
  159. package/adapters/pi/skills/forge-4-backlog/references/pipeline-state-schema.json +191 -0
  160. package/adapters/pi/skills/forge-4-backlog/references/shared-conventions.md +295 -0
  161. package/adapters/pi/skills/forge-4-backlog/references/stage-exit-protocol.md +258 -0
  162. package/adapters/pi/skills/forge-5-loop/SKILL.md +314 -0
  163. package/adapters/pi/skills/forge-5-loop/references/forge-config-schema.json +236 -0
  164. package/adapters/pi/skills/forge-5-loop/references/ralph-loop-contract.md +221 -0
  165. package/adapters/pi/skills/forge-5-loop/references/result-reporting.md +85 -0
  166. package/adapters/pi/skills/forge-5-loop/references/runner-contract.md +341 -0
  167. package/adapters/pi/skills/forge-5-loop/references/shared-conventions.md +295 -0
  168. package/adapters/pi/skills/forge-5-loop/references/stage-exit-protocol.md +258 -0
  169. package/adapters/pi/skills/forge-6-docs/SKILL.md +202 -0
  170. package/adapters/pi/skills/forge-6-docs/references/doc-conventions.md +126 -0
  171. package/adapters/pi/skills/forge-6-docs/references/pipeline-state-schema.json +191 -0
  172. package/adapters/pi/skills/forge-6-docs/references/shared-conventions.md +295 -0
  173. package/adapters/pi/skills/forge-bootstrap/SKILL.md +250 -0
  174. package/adapters/pi/skills/forge-bootstrap/references/templates/ci/github-actions.yml +12 -0
  175. package/adapters/pi/skills/forge-bootstrap/references/templates/generic/run.sh +3 -0
  176. package/adapters/pi/skills/forge-bootstrap/references/templates/generic/test.sh +13 -0
  177. package/adapters/pi/skills/forge-bootstrap/references/templates/go/go.mod +3 -0
  178. package/adapters/pi/skills/forge-bootstrap/references/templates/go/main.go +12 -0
  179. package/adapters/pi/skills/forge-bootstrap/references/templates/go/main_test.go +11 -0
  180. package/adapters/pi/skills/forge-bootstrap/references/templates/hygiene/AGENTS.md +35 -0
  181. package/adapters/pi/skills/forge-bootstrap/references/templates/hygiene/CLAUDE.md +36 -0
  182. package/adapters/pi/skills/forge-bootstrap/references/templates/hygiene/README.md +11 -0
  183. package/adapters/pi/skills/forge-bootstrap/references/templates/licenses/Apache-2.0/LICENSE +198 -0
  184. package/adapters/pi/skills/forge-bootstrap/references/templates/licenses/MIT/LICENSE +21 -0
  185. package/adapters/pi/skills/forge-bootstrap/references/templates/python/pyproject.toml +24 -0
  186. package/adapters/pi/skills/forge-bootstrap/references/templates/python/src/{{PKG}}/__init__.py +5 -0
  187. package/adapters/pi/skills/forge-bootstrap/references/templates/python/src/{{PKG}}/main.py +13 -0
  188. package/adapters/pi/skills/forge-bootstrap/references/templates/python/tests/test_smoke.py +8 -0
  189. package/adapters/pi/skills/forge-bootstrap/references/templates/rust/Cargo.toml +15 -0
  190. package/adapters/pi/skills/forge-bootstrap/references/templates/rust/src/lib.rs +7 -0
  191. package/adapters/pi/skills/forge-bootstrap/references/templates/rust/src/main.rs +5 -0
  192. package/adapters/pi/skills/forge-bootstrap/references/templates/rust/tests/smoke.rs +6 -0
  193. package/adapters/pi/skills/forge-bootstrap/references/templates/typescript/package.json +15 -0
  194. package/adapters/pi/skills/forge-bootstrap/references/templates/typescript/src/index.ts +4 -0
  195. package/adapters/pi/skills/forge-bootstrap/references/templates/typescript/test/smoke.test.ts +6 -0
  196. package/adapters/pi/skills/forge-bootstrap/references/templates/typescript/tsconfig.json +14 -0
  197. package/adapters/pi/skills/forge-fix/SKILL.md +98 -0
  198. package/adapters/pi/skills/forge-fix/references/shared-conventions.md +295 -0
  199. package/adapters/pi/skills/forge-fix/references/stage-exit-protocol.md +258 -0
  200. package/adapters/pi/skills/forge-guide/SKILL.md +192 -0
  201. package/adapters/pi/skills/forge-guide/references/forge-config-schema.json +236 -0
  202. package/adapters/pi/skills/forge-guide/references/process-overview.md +143 -0
  203. package/adapters/pi/skills/forge-guide/references/ralph-loop-contract.md +221 -0
  204. package/adapters/pi/skills/forge-guide/references/shared-conventions.md +295 -0
  205. package/adapters/pi/skills/forge-guide/references/stack-resolution.md +54 -0
  206. package/adapters/pi/skills/forge-guide/references/stacks/_generic.md +111 -0
  207. package/adapters/pi/skills/forge-guide/references/stacks/go.md +157 -0
  208. package/adapters/pi/skills/forge-guide/references/stacks/python.md +184 -0
  209. package/adapters/pi/skills/forge-guide/references/stacks/rust.md +170 -0
  210. package/adapters/pi/skills/forge-guide/references/stacks/typescript.md +134 -0
  211. package/adapters/pi/skills/forge-init/SKILL.md +72 -0
  212. package/adapters/pi/skills/forge-verify/SKILL.md +273 -0
  213. package/adapters/pi/skills/forge-verify/references/pipeline-state-schema.json +191 -0
  214. package/adapters/pi/skills/forge-verify/references/shared-conventions.md +295 -0
  215. package/adapters/pi/skills/forge-verify/references/verification-checklists.md +477 -0
  216. package/dist/agent-targets.d.ts +1 -1
  217. package/dist/agent-targets.js +23 -3
  218. package/dist/detect.d.ts +1 -1
  219. package/dist/detect.js +2 -1
  220. package/dist/manifest.d.ts +1 -1
  221. package/dist/manifest.js +2 -2
  222. package/dist/placements.js +5 -1
  223. package/dist/rauf.d.ts +4 -4
  224. package/dist/rauf.js +3 -3
  225. package/dist/types.d.ts +31 -6
  226. package/dist/types.js +6 -3
  227. package/package.json +14 -3
@@ -0,0 +1,221 @@
1
+ # The Loop-Runner Contract (consumer side)
2
+
3
+ feature-forge's pipeline ends by handing a `backlog.json` to an autonomous
4
+ **loop runner** that implements each item. feature-forge does not embed any
5
+ runner's internals — it talks to the runner through one indirection point: the
6
+ **`loopRunner`** block in `forge.config.json`.
7
+
8
+ ## The seam
9
+
10
+ Every command feature-forge runs against the runner is a template in
11
+ `loopRunner`, with `{bin}`, `{backlogDir}`, `{specsDir}`, and `{iterations}`
12
+ substituted at call time. `forge-5-loop` (execution), `forge-4-backlog`
13
+ (validation), and `forge-verify` (backlog validation) all render their commands
14
+ from this block — there are no hardcoded `rauf …` commands in the skills, and
15
+ even the human log filename is tokenized via `{loopRunner.logFile}`.
16
+
17
+ When `forge.config.json` has no `loopRunner` block, feature-forge uses the
18
+ built-in defaults (see `references/forge-config-schema.json`) and announces
19
+ "defaulting to the rauf loop runner."
20
+
21
+ ## The contract a runner MUST satisfy
22
+
23
+ A conforming runner MUST implement the **backlog-tool / loop-runner contract**
24
+ defined authoritatively in rauf's
25
+ [`SPEC-BACKLOG-TOOL-CONTRACT.md`](https://github.com/garygentry/rauf/blob/main/docs/SPEC-BACKLOG-TOOL-CONTRACT.md)
26
+ (Part A). In summary, it must provide:
27
+
28
+ - **A backlog schema** with the published `$id` and an optional `schemaVersion`,
29
+ whose `type`/`status` vocabularies match the contract
30
+ (`type ∈ bug|bugfix|refactor|feature|chore|test`,
31
+ `status ∈ pending|in_progress|done|blocked`).
32
+ - **A `validate` verb** with exit codes `0` (valid) / `1` (findings) / `2`
33
+ (usage/IO) that emits `{ valid, findings[] }` under `--json`. This is the
34
+ single check feature-forge trusts — it never re-implements validation.
35
+ - **The signal protocol** (`RAUF_DONE` / `RAUF_BLOCKED` / `RAUF_NEEDS_HUMAN` /
36
+ `RAUF_REVIEW`). These are emitted by the *coding agent* into its stdout and
37
+ parsed by the runner — they are not runner-authored log lines, so consumers key
38
+ off the runner's parsed events (below), never the raw `RAUF_*` tokens (which can
39
+ also appear inside an agent's prose and produce false matches).
40
+ - **A machine-readable event stream** for live supervision (`loopRunner.eventStreamCommand`,
41
+ rauf: `loop run … --ndjson`): one JSON event per line with a stable `type`
42
+ vocabulary — `item_completed` / `item_blocked` / `needs_human` / `signal_parsed`
43
+ / `loop_completed` / `loop_error` / `loop_cancelled` / `llm_stuck_warning` (a
44
+ circuit-breaker halt surfaces as `loop_error`) — plus a
45
+ derived-status JSON (`loopRunner.statusJsonCommand`, rauf: `status … --json`) and
46
+ per-iteration telemetry with a `stuckWarning` flag (`loopRunner.watchCommand`,
47
+ rauf: `status … --json` — the `loop watch` verb was removed in v0.5.0). `forge-5-loop` supervises the run through these,
48
+ **not** by parsing the human log. `followCommand` / `logCommand` are
49
+ human-formatted streams for a person watching in a terminal, not machine surfaces.
50
+ - **The state-dir layout** (per-`--backlog` isolation under `loopRunner.stateDir`).
51
+ - **The CLI verbs** mapped by `loopRunner`: run (+ event-stream) / validate /
52
+ status (+ `--json`) / list / follow / log / version.
53
+ - **A `version` verb** (`{bin} version --json` → `{ version: <semver> }`) so
54
+ feature-forge can enforce `loopRunner.minRunnerVersion` before running.
55
+
56
+ > **The runner does not pause for human input.** When the coding agent signals
57
+ > `RAUF_NEEDS_HUMAN`/`RAUF_BLOCKED`/`RAUF_REVIEW`, a conforming runner sets that
58
+ > item aside and **keeps working other runnable items to completion** (rauf:
59
+ > `runner.ts` needs_human handler). So a supervising session can surface those
60
+ > events live (visibility) and cancel early, but it cannot inject an answer and
61
+ > resume the set-aside item mid-run — resolution is a follow-up retry pass. A
62
+ > first-class pause/resume-with-answer capability is a desirable runner
63
+ > enhancement (see `plans/rauf-enhancement-recommendations.md`).
64
+
65
+ ## rauf is the default and reference implementation
66
+
67
+ rauf owns the contract spec and is the runner feature-forge defaults to. The
68
+ authoring craft itself (how to decompose specs into well-scoped, verifiable
69
+ items) lives in rauf's **`author-backlog`** skill, which `forge-4-backlog`
70
+ delegates to — so the only thing binding feature-forge to a particular runner is
71
+ the schema + `validate` verb that this contract formalizes. A future runner swap
72
+ supplies its own `loopRunner` block (its own `bin`, schema, and `validate`
73
+ command) without touching any pipeline skill.
74
+
75
+ The coding-agent dimension this contract adds (below) is **additive and
76
+ presence-gated**: it exists only when the `loopRunner` block advertises it via
77
+ `agentArgument`. A runner that omits that field has no agent dimension at all —
78
+ the seam degrades to exactly today's behavior with no error, so default-to-rauf
79
+ and pluggability are unchanged. See `## Agent selection`.
80
+
81
+ ## Agent selection
82
+
83
+ This section is **contract-level**: it states what a conforming runner exposes
84
+ and what forge does with it, and defers every algorithm to the owning specs
85
+ (`02`/`03`/`04`/`05`).
86
+
87
+ ### What a conforming runner exposes (the consumed surface)
88
+
89
+ A runner that carries a coding-agent dimension exposes, in its `loopRunner`
90
+ block:
91
+
92
+ - an **`agentArgument`** template (rauf default `--agent {agent}`) — the
93
+ tokenized launch-time flag; its **presence** advertises the agent surface;
94
+ - an **`agentsProbeCommand`** (rauf default `{bin} agents --json`) emitting
95
+ `{ agents: [{ id, displayName, available, detail? }] }` and **always exiting
96
+ 0**;
97
+ - an optional **`defaultAgent`** project-default id.
98
+
99
+ These three fields are specified in full in `references/forge-config-schema.json`
100
+ and are the schema half of this contract. forge consumes rauf's existing
101
+ `--agent <id>` flag, `rauf agents` probe, `BacklogItem.provider`, and 5-layer
102
+ precedence — it conforms to them, it does not redesign them.
103
+
104
+ ### Precedence and the run-layer mapping
105
+
106
+ The agent-selection precedence is **`item > run > project > default`**,
107
+ deliberately parallel to the model-selection precedence (`item.model >
108
+ --model/options > project default > provider default`). It is realized as:
109
+
110
+ - **item** — `BacklogItem.provider`, applied by **rauf** from the backlog. forge
111
+ **never reads, writes, or overrides** it (pass-through), so a deliberate
112
+ per-item agent always wins.
113
+ - **run** — forge's per-run selector (`forge-5-loop` Step 2d).
114
+ - **project** — forge's `loopRunner.defaultAgent`.
115
+ - **default** — the runner's own default (`claude-cli` for rauf) when forge sends
116
+ nothing.
117
+
118
+ forge owns **only** the run and project layers. It collapses run-over-project
119
+ *inside itself* into the **single** `--agent {agent}` value it emits at the
120
+ **run layer**, and lets rauf apply the item override *above* that. forge **never
121
+ re-implements rauf's resolver**. The resolution algorithm itself lives in
122
+ `03-selection-resolution-observability.md`.
123
+
124
+ ### Availability probe + unknown/unavailable disambiguation
125
+
126
+ When the resolved agent is a **non-default** id, forge runs `agentsProbeCommand`
127
+ **once (no retries)** before any loop side-effect, parses the advertised
128
+ `agents[]`, and builds the advertised id set. Because the probe **always exits
129
+ 0**, an unknown id is distinguished from a known-but-unavailable one **only by
130
+ set membership**, not by exit code:
131
+
132
+ - **Unknown id** (not in the advertised set — a typo or unsupported agent):
133
+ **hard-reject before launch**, listing the valid ids. No proceed-anyway path.
134
+ No value is interpolated into `{agent}`.
135
+ - **Known but unavailable** (`available: false`): **warn** (showing the probe's
136
+ `detail`) and let the user **proceed-anyway or choose another** — never
137
+ silently abort, never silently proceed.
138
+ - **Available**: proceed.
139
+
140
+ The advertised id set is also the **allow-list**: the only value ever
141
+ interpolated into `{agent}` is a validated, advertised id. The **default /
142
+ claude path never reaches the probe** — it incurs no extra cost. The
143
+ `classify(...)` algorithm and the rejection-error text live in
144
+ `04-availability-precheck.md`.
145
+
146
+ ### Capability gate + version floor
147
+
148
+ Agent selection is **capability-gated** on the runner advertising
149
+ `agentArgument`: a runner whose `loopRunner` omits (or empties) that field
150
+ exposes no agent surface, so the per-run selector, the probe, and any `{agent}`
151
+ substitution **vanish entirely** and no agent argument is sent — byte-identical
152
+ to today. Degradation is **silent, not an error**, keeping alternate (non-rauf)
153
+ runners first-class. The gate condition is owned by
154
+ `02-config-schema-and-gating.md`.
155
+
156
+ Independently, the **version gate** floors at the runner version that ships the
157
+ agent surface. For rauf that is **0.6.0** (`loopRunner.minRunnerVersion`): the
158
+ `--agent` flag, the `agents` probe, and the preset agent registry are present in
159
+ rauf source at 0.6.0. A successful gate therefore guarantees those surfaces
160
+ exist before any run. See `## Version gating` and
161
+ `05-runner-discovery-version-gate.md`.
162
+
163
+ **This document — the `## Agent selection` section, the `## Per-stage agent
164
+ applicability` table, and the `## validate is agent-agnostic` note — together
165
+ with the augmented `loopRunner` schema block in
166
+ `references/forge-config-schema.json` constitute the `forge-loop-runner-contract`
167
+ expose, consumed by the `packaging-docs-ci` capstone as documentation input.**
168
+
169
+ ## Per-stage agent applicability
170
+
171
+ Every forge stage that invokes the loop runner is classified here. Only
172
+ `forge-5-loop` (execution) carries the coding-agent dimension; the two
173
+ validation-only stages are agent-agnostic.
174
+
175
+ | Stage | Runner verbs | Agent dimension |
176
+ |-------|-------------|-----------------|
177
+ | `forge-5-loop` | run / eventStream / status / version | **Full** — selector, probe, `--agent` |
178
+ | `forge-4-backlog` | `validate` | **None** — agent-agnostic |
179
+ | `forge-verify` | `validate` | **None** — agent-agnostic |
180
+
181
+ - **`forge-5-loop`** is the executor: it drives the run, so it renders the run /
182
+ event-stream / status / version verbs (and `list`) and carries the full agent
183
+ surface — the Step 2d selector, the availability probe, and the rendered
184
+ `agentArgument`.
185
+ - **`forge-4-backlog`** authors and then *validates* the backlog; its only runner
186
+ call is `validateCommand`. It also reads `versionCommand` for a graceful-degrade
187
+ check, but **passes no agent** and never runs `loop run`.
188
+ - **`forge-verify`** (backlog mode) re-runs the same `validateCommand` to surface
189
+ validation findings. It carries no agent dimension. This is **contract
190
+ coverage** of forge-verify (it is classified here), with an explicit agnostic
191
+ note (below) — **not** a new agent-driven run in forge-verify.
192
+
193
+ ### `validate` is agent-agnostic
194
+
195
+ The `validate` verb (`loopRunner.validateCommand`) checks a `backlog.json`
196
+ against the backlog schema and spec references. **It does not run a coding agent
197
+ and has no agent dimension.** No agent argument — `--agent`, the `{agent}` token,
198
+ or any agent id — may **ever** be passed to backlog validation, in **any** stage
199
+ (`forge-4-backlog`, `forge-verify`, or any future caller). Backlog validation is
200
+ a pure, deterministic check; the coding agent is irrelevant to it. A contributor
201
+ who later adds agent selection to a *new* stage MUST confirm that stage runs the
202
+ *execution* surface (like `forge-5-loop`), not `validate` — agent selection
203
+ belongs to execution only. If you find yourself adding `--agent` near a
204
+ `validateCommand` render, that is a bug.
205
+
206
+ ## Version gating
207
+
208
+ feature-forge requires a runner exposing `backlog validate` + backlog
209
+ `schemaVersion`, and the unified exit-code/status contract it reads. The floor is
210
+ now the **agent-surface floor**: the runner version that ships the `--agent`
211
+ flag, the `agents` probe, and the preset agent registry. For rauf that is
212
+ **0.6.0** (`loopRunner.minRunnerVersion`).
213
+ `forge-5-loop` runs `{bin} version --json`, semver-compares the reported version
214
+ against `minRunnerVersion`, and on a missing-or-too-old runner stops with
215
+ `loopRunner.installHint` (the CLI install/upgrade command) — **before** invoking
216
+ the loop. See `COMPATIBILITY.md` for the version matrix.
217
+
218
+ > **Two different "installs."** `installHint` obtains/upgrades the runner **CLI
219
+ > binary** (e.g. rauf's `install-binary.sh`). `setupHint` installs the runner's
220
+ > **per-project artifacts** (rauf: `rauf install .`). A failing version gate is
221
+ > always the former, never the latter.
@@ -0,0 +1,85 @@
1
+ # forge-5-loop — Step 4b Result-Report Templates
2
+
3
+ These are the five verbatim result-report output templates for **Step 4b** of
4
+ `forge-5-loop/SKILL.md`. Pick **every** branch that applies (a run can be both
5
+ blocked and needs-human) and render its report.
6
+
7
+ **All items done.** Print the completion summary, then close with the **warm-acceptable
8
+ variant** of the Stage Exit Protocol (single-sourced in
9
+ `references/stage-exit-protocol.md`) — the `forge-5-loop → forge-6-docs` boundary is the
10
+ one place where clearing before the next stage is optional:
11
+ ```
12
+ Loop completed for {feature}. All {N} items implemented successfully.
13
+ ```
14
+
15
+ **The loop is complete — this is the one boundary where clearing before the next stage is optional.**
16
+
17
+ 1. **Verify is already offered above.** Impl-verify is offered interactively right after this report (Step 5b for a standalone feature, Step 6.1 for an epic member) — run it there rather than as a second gate. It runs clean-room, so it needs no fresh session.
18
+ 2. **Clearing is optional here — warm is fine.** `forge-6-docs` benefits from the still-warm context of what the loop actually did, so continuing in this same session is the easy default. A cold start also works — every artifact is on disk — but there is no need to force it.
19
+ 3. **Then run the next command** — in this warm session, or a fresh one if you prefer:
20
+
21
+ ```
22
+ /skill:forge-6-docs {feature}
23
+ ```
24
+
25
+ **Runner review pass.** A review flag (e.g. rauf's `--review`) makes the runner run
26
+ a post-loop review that **auto-creates and implements fix items** rather than handing
27
+ findings to the user — distinct from `forge-verify impl` (a clean-context audit that
28
+ writes a findings doc). When Step 4a captured a `review_completed` event, add a line
29
+ **above** "Next steps" so the pass's effect is visible and not mistaken for "nothing
30
+ happened":
31
+ ```
32
+ Runner review pass: {itemsCreated} fix item(s) created and implemented.
33
+ {summary}
34
+ ```
35
+ Omit this line when no `review_completed` event was emitted (no review flag passed).
36
+ The created items are already counted in the totals above.
37
+
38
+ **Some items need a human:**
39
+ ```
40
+ Loop completed for {feature}.
41
+ Completed: {done}/{total}
42
+ Needs human: {needsHuman} items (set aside during the run)
43
+
44
+ These items asked a question the loop couldn't answer:
45
+ - {id}: {title} — {reason}
46
+
47
+ Resolve, then retry:
48
+ - Answer the question(s) above, then re-run `/skill:forge-5-loop {feature}`
49
+ (add --retry-blocked to pick the set-aside items back up).
50
+ ```
51
+
52
+ **Some items blocked:**
53
+ ```
54
+ Loop completed for {feature}.
55
+ Completed: {done}/{total}
56
+ Blocked: {blocked} items
57
+
58
+ Blocked items:
59
+ - {id}: {title}
60
+ - {id}: {title}
61
+
62
+ Options:
63
+ - Re-run with --retry-blocked to retry blocked items
64
+ - Review blocked items manually: {bin} backlog show . {id} --backlog {backlogDir}
65
+ - Continue to docs if blocking items are non-critical
66
+ ```
67
+
68
+ **Some items deferred (runner gave up after retries — "false blocks"):**
69
+ ```
70
+ Loop completed for {feature}.
71
+ Completed: {done}/{total}
72
+ Deferred: {deferred} items (no signal after retries — likely just need another pass)
73
+
74
+ Re-run `/skill:forge-5-loop {feature}` to retry deferred items.
75
+ ```
76
+
77
+ **Some items still pending (iteration limit reached):**
78
+ ```
79
+ Loop completed for {feature}.
80
+ Completed: {done}/{total}
81
+ Pending: {pending} items (iteration limit reached)
82
+ Blocked: {blocked} items
83
+
84
+ Re-run `/skill:forge-5-loop {feature}` to continue with remaining items.
85
+ ```
@@ -0,0 +1,341 @@
1
+ # forge-5-loop — Loop-Runner Contract (launch, supervision, model precedence)
2
+
3
+ This file holds the detailed loop-runner contract relocated out of
4
+ `forge-5-loop/SKILL.md`: the event-stream vs. log-fallback **launch** detail
5
+ (Steps 3b/3d/3e), the structured-surface **monitoring** caveats, the **model
6
+ precedence** rule, and the **optional-flags catalog** referenced from Step 2d.
7
+ Every command below is rendered from `loopRunner` with token substitution, as in
8
+ the skill body.
9
+
10
+ ## Model selection precedence (Step 2d)
11
+
12
+ The runner picks the per-iteration model by precedence (highest wins):
13
+
14
+ ```
15
+ item.model > --model / options > project default > provider default
16
+ ```
17
+
18
+ So a backlog item's own `model` field overrides a `--model` flag passed to the
19
+ run, which overrides the project's configured default, which overrides the
20
+ runner/provider default. Pass `--model <model>` (optional flag below) to override
21
+ the project default for the whole run.
22
+
23
+ ## Agent selection (Step 2d)
24
+
25
+ This section is **parallel** to `## Model selection precedence` above: it governs
26
+ which **coding agent** rauf drives for the run. The entire surface is
27
+ **presence-gated** on `loopRunner.agentArgument` — when that field is absent or
28
+ empty, there is no selector, no probe, and no `{agent}` substitution, and Step 2d /
29
+ Step 3c are byte-identical to today (capability gate;
30
+ `02-config-schema-and-gating.md`, REQ-PLUG-02). The rest assumes the gate is on.
31
+
32
+ **Precedence (highest wins):**
33
+
34
+ ```
35
+ item.provider > --agent (run selection) > loopRunner.defaultAgent (project) > runner default (claude-cli)
36
+ ```
37
+
38
+ **Run-layer mapping — why forge never re-implements rauf's resolver.** forge owns
39
+ **only** its run and project layers and collapses them into **one** value
40
+ (`resolve()`: `run_selection or defaultAgent or none`), which it emits as a single
41
+ `--agent {agent}` occupying rauf's **run layer only**. rauf alone resolves
42
+ item-vs-run via its own 5-layer resolver, sitting the per-item `BacklogItem.provider`
43
+ **above** forge's run layer — so a run selection can never clobber a deliberate
44
+ per-item agent. forge **never reads, writes, or overrides** `BacklogItem.provider`
45
+ (REQ-AGENT-05). When forge sends nothing (the default path), rauf applies its own
46
+ default `claude-cli`, byte-identical to today. Empty/whitespace selections are
47
+ treated as unset, and an explicit pick of the runner default id collapses to the
48
+ default path (append nothing, run no probe). See
49
+ `03-selection-resolution-observability.md §3–§4`.
50
+
51
+ **Availability pre-check + disambiguation.** For a **non-default** resolved id only,
52
+ forge runs `loopRunner.agentsProbeCommand` **once** (no retries) and classifies the
53
+ id by **membership** in the advertised set (`{ row.id for row in agents }`), then the
54
+ matching row's `available` flag — **never** by exit code, because `rauf agents
55
+ --json` always exits 0 (an unknown id is simply absent; a known-unavailable one is
56
+ present with `available: false`):
57
+
58
+ - **UNKNOWN** (`∉` advertised set): hard-reject **before any loop side-effect**,
59
+ listing the sorted valid ids; **no proceed-anyway**; the value never interpolates
60
+ into `{agent}` (the advertised set IS the allow-list — REQ-SEC-01).
61
+ - **UNAVAILABLE** (member, `available == False`): warn with the row's `detail`, then
62
+ offer **proceed-anyway OR choose-another** — never silent.
63
+ - **AVAILABLE** (member, `available == True`): proceed; the validated id fills
64
+ `{agent}`.
65
+ - **Probe failure** (non-zero exit / unparseable / wrong shape / empty `agents[]` /
66
+ row missing `id`): surface it and offer **choose-another OR abort**; never launch
67
+ the non-default agent unvalidated, never silently fall back to the default.
68
+
69
+ The default / `claude-cli` path runs **no** probe (zero extra cost). See
70
+ `04-availability-precheck.md` for the full pre-check, classification, and allow-list,
71
+ and `02-config-schema-and-gating.md` for the capability gate.
72
+
73
+ > **Probe false-negative for Claude Code installs (advisory).** `rauf agents` may
74
+ > report `claude-cli` **unavailable** (e.g. *"credentials file not found:
75
+ > ~/.config/claude-code/credentials.json"*) even when a working `claude` CLI
76
+ > authenticates elsewhere — the probe's credential heuristic doesn't cover every
77
+ > install. This is a rauf probe concern, not something forge-5-loop fixes. The
78
+ > **default-agent path skips the probe entirely**, so an ordinary default run is
79
+ > unaffected; only an **explicit** `--agent claude-cli` would be flagged UNAVAILABLE,
80
+ > and the existing **proceed-anyway** path (above) covers it. Do not attempt to
81
+ > patch rauf's probe from here.
82
+
83
+ ### Claude-only model-alias guard (Step 2d, sub-step d-model)
84
+
85
+ When the resolved agent is **non-default** (not the default / `claude-cli` path),
86
+ forge must guard against a backlog whose items pin **Claude-specific** model aliases.
87
+ forge-4-backlog (via the rauf author-backlog skill) writes Claude tier aliases
88
+ (`opus` / `sonnet`) into each item's `model`. Because rauf's precedence puts
89
+ `item.model` **above** `--agent`, the alias is forwarded verbatim to the selected
90
+ agent; a non-Claude agent (e.g. codex) then 400s — *"The 'sonnet' model is not
91
+ supported when using Codex with a ChatGPT account."* — so **every** spawn exits 1 and
92
+ rauf reports *"Circuit breaker: 3 consecutive infra failures — halting"* with no hint
93
+ of the real cause. forge-5-loop therefore detects Claude-specific `model` aliases in
94
+ the backlog (tier aliases `opus`/`sonnet`/`haiku` or `claude-*` ids) and, before
95
+ launch, **warns** and offers (via `AskUserQuestion`) to **strip `model` for this run**
96
+ (remove the key from each affected item so each spawn uses the agent's own default) or
97
+ **proceed as-is**. forge only ever touches the `model` field — never `provider`. The
98
+ default / `claude-cli` path skips this guard (the aliases are valid there).
99
+
100
+ > **Follow-up (out of scope here — rauf repo).** The durable fix would be for the
101
+ > rauf `author-backlog` skill to keep `model` **provider-neutral** by default (or to
102
+ > document that writing a tier alias binds the backlog to Claude agents). That lives
103
+ > in the separate rauf plugin/repo, not feature-forge; tracked as a follow-up.
104
+ >
105
+ > **Follow-up (out of scope here — rauf repo).** The durable fix for the root/sandbox
106
+ > refusal (see "Root/sandbox env guard" under Step 3b) is for **rauf itself** to honor
107
+ > `IS_SANDBOX` when it launches `claude --dangerously-skip-permissions` as root (or to
108
+ > detect root+flag-refused and emit a clear error instead of an opaque circuit-break).
109
+ > feature-forge's launch-time export is the mitigation; the upstream fix lives in the
110
+ > rauf plugin/repo. Track as a follow-up.
111
+
112
+ ## Run mode (Step 2d, rauf)
113
+
114
+ **Applies only when `loopRunner.name == "rauf"`.** rauf's `--review` runs a review
115
+ pass after all iterations complete (an extra agent session that re-examines the
116
+ finished work and can file follow-up backlog items). feature-forge treats **running
117
+ with review as the recommended default** — a review pass is cheap relative to the
118
+ loop it audits, and catches gaps before the pipeline moves on to docs. So Step 2d
119
+ adds a **"Run mode"** question to the confirmation's `AskUserQuestion` surface with a
120
+ **fixed, non-improvised option order** (determinism is the point — the option set
121
+ must not vary run-to-run):
122
+
123
+ ```
124
+ Run mode:
125
+ 1. Run with review pass (recommended) → append `--review` [DEFAULT]
126
+ After all iterations, a review agent re-examines the finished work and may
127
+ file follow-up items. Recommended for every forge run.
128
+ 2. Run without review → bare rendered command
129
+ Skip the review pass — iterations only, no post-run audit.
130
+ 3. Review + retry blocked → append `--review --retry-blocked`
131
+ ONLY offered when Step 2a counted one or more `blocked` items. Runs the
132
+ review pass and also unblocks/retries the previously blocked items.
133
+ ```
134
+
135
+ Notes:
136
+
137
+ - **Option 1 is the default** and the confirmation's rendered command line shows
138
+ `--review` appended. On any pick, append the option's flags to the rendered run
139
+ command before Step 3 (launch).
140
+ - **`AskUserQuestion`'s built-in "Other"** already lets the user type ad-hoc flags
141
+ (`--model <model>`, `--timeout <min>`, or any combination) — do **not** add a
142
+ separate open-ended option for that.
143
+ - **Option 3 is conditional.** Include it only when the Step 2a tally has `blocked
144
+ > 0`; otherwise present options 1 and 2 only.
145
+ - **Version floor.** rauf's explicit `review` signal ships in 0.5.0, below the
146
+ `minRunnerVersion` floor (0.6.0) enforced at gate 1c — so `--review` is always
147
+ available once the loop is cleared to launch. No extra version check is needed.
148
+ - **Non-rauf runners.** When `loopRunner.name != "rauf"`, add **no** Run-mode
149
+ question — present the bare rendered command and let the user adjust via "Other",
150
+ byte-identical to the pre-review-default behavior. `--review` is a rauf-specific
151
+ flag; a swapped-in runner conforming to the contract need not support it.
152
+
153
+ ## Optional flags catalog (Step 2d, rauf)
154
+
155
+ These are the optional flags the user may add to the rendered run command. If the
156
+ user requests additional flags, append them to the rendered run command.
157
+
158
+ ```
159
+ --agent <id> Coding agent rauf drives this run (see Agent selection below).
160
+ Only the runner's advertised ids are valid; an unknown id is
161
+ rejected before launch. Shown only when the runner advertises
162
+ an agent surface (loopRunner.agentArgument present).
163
+ --review Run a review pass after all iterations (extra agent session)
164
+ --model <model> Override the model (see precedence above)
165
+ --timeout <min> Per-session timeout in minutes (default: 60)
166
+ --retry-blocked Unblock and retry previously blocked items
167
+ ```
168
+
169
+ ## Launch detail (Step 3b — background process)
170
+
171
+ Launch the loop **backgrounded** so it survives session end and does not block the
172
+ session, then supervise it live via the runner's structured event file.
173
+
174
+ > **Clean-tree precondition.** rauf refuses to run with uncommitted changes
175
+ > (*"Refusing to run the loop with uncommitted changes… pass --force"*). Step 3a's
176
+ > in-progress `.pipeline-state.json` write is itself an uncommitted change, so it
177
+ > **must be committed before launch** (Step 3a) — otherwise the first launch on an
178
+ > otherwise-clean repo always fails. If the tree still has unrelated uncommitted
179
+ > changes after that commit, surface it and let the user commit/stash or pass
180
+ > `--force`; never auto-pass `--force`.
181
+
182
+ > **Root/sandbox env guard.** On a hosted remote (e.g. Claude.ai) the loop often runs
183
+ > **as root**. rauf's default Claude launch is `claude -p --dangerously-skip-permissions
184
+ > …`, which the Claude CLI **refuses under root unless `IS_SANDBOX` is set** — the remote
185
+ > container is a legitimate ephemeral sandbox, but without the flag every spawn exits and
186
+ > rauf circuit-breaks (*"3 consecutive infra failures — halting"*) with no hint of the
187
+ > cause. So when — and only when — the launcher is root (`[ "$(id -u)" = 0 ]`), export
188
+ > `IS_SANDBOX="${IS_SANDBOX:-1}"` in front of the launch (an explicitly-set value is
189
+ > honored; the `:-1` only supplies a default). Non-root/local runs are unaffected — the
190
+ > guard is a no-op. **Surface a one-line note** when you set it — e.g. *"running as root →
191
+ > setting IS_SANDBOX=1 so the sandboxed runner can use --dangerously-skip-permissions"* —
192
+ > so the behavior is never silent. `forge-session.py doctor` also reports this condition.
193
+ > Both launch commands below already carry the guard.
194
+
195
+ **Do NOT redirect the run's stdout into `{loopRunner.stateDir}`.** rauf **persists
196
+ its own** `{stateDir}/events.ndjson` (structured) and `{stateDir}/{logFile}` (human)
197
+ natively, and **rotates** them at the start of every run (the prior run's files are
198
+ renamed into `{stateDir}/archive/`). A redirect like `… --ndjson >
199
+ {stateDir}/events.ndjson` therefore (a) is **redundant** — the runner writes that
200
+ file regardless — and (b) **collides** with the runner's own writer: the shell holds
201
+ a descriptor on the file the runner immediately rotates away, so the redirected
202
+ `--ndjson` stdout is orphaned into a bogus `archive/` file while the live
203
+ `events.ndjson` is the runner's native stream. It only *looks* clean by accident of
204
+ rotation timing. So:
205
+
206
+ - **Self-persisting runner (default — rauf writes `{stateDir}/events.ndjson`):**
207
+ launch the **plain `runCommand`** with `run_in_background: true` and **no
208
+ redirect** — the Bash tool already captures the run's stdout/stderr to the
209
+ background task's output file (use it to diagnose a launch refusal). Supervise by
210
+ arming the Monitor on the runner's **native** `{backlogDir}/{stateDir}/events.ndjson`
211
+ (Step 3d). Guard the very first run with the state dir:
212
+
213
+ ```
214
+ mkdir -p {backlogDir}/{loopRunner.stateDir} && { [ "$(id -u)" = 0 ] && export IS_SANDBOX="${IS_SANDBOX:-1}" || true; } && {rendered runCommand}
215
+ ```
216
+
217
+ (Note: the `--ndjson` stdout stream and `loopRunner.eventStreamCommand` are **not**
218
+ used on this path — the native file already carries the same structured records.)
219
+ - **Stdout-only runner (no native event file):** render `eventStreamCommand` (it adds
220
+ `--ndjson`) and redirect its stdout to a file **outside `{stateDir}`** so it cannot
221
+ collide with any native file or be swept into `archive/`, then Monitor that file:
222
+
223
+ ```
224
+ mkdir -p {backlogDir}/{loopRunner.stateDir} && { [ "$(id -u)" = 0 ] && export IS_SANDBOX="${IS_SANDBOX:-1}" || true; } && {rendered eventStreamCommand} > {backlogDir}/forge-events.ndjson 2>&1
225
+ ```
226
+
227
+ The background task's exit notification remains the single authoritative terminal
228
+ signal (Step 4). Loop runs can take significant time (minutes to hours depending on
229
+ backlog size).
230
+
231
+ ## Arm a Monitor on the event stream (Step 3d)
232
+
233
+ Arm the **`Monitor` tool** on the structured event stream so events flow back into
234
+ this session as they happen. Use **`persistent: true`** — runs can exceed `Monitor`'s
235
+ maximum `timeout_ms` (1 hour), and a bounded timeout would silently stop watching a
236
+ still-running loop.
237
+
238
+ **Coverage-complete filter (silence is not success).** The filter MUST match every
239
+ terminal and exception state, not just the happy path — otherwise a crash or hang
240
+ looks identical to "still running." Monitor command (NDJSON path):
241
+
242
+ ```
243
+ tail -n +1 -F {backlogDir}/{loopRunner.stateDir}/events.ndjson 2>/dev/null \
244
+ | jq -rc --unbuffered 'select(.type | test("item_completed|item_blocked|needs_human|signal_parsed|loop_completed|loop_error|loop_cancelled|llm_stuck_warning"))'
245
+ ```
246
+
247
+ > **Use `tail -F` (follow by name), not `-f` (follow by descriptor).** The runner
248
+ > **rotates** `events.ndjson` at the start of each run (renames the prior file into
249
+ > `archive/`, creates a fresh one). A Monitor that attaches with `-f` during that
250
+ > brief rotation window would follow the **archived** inode and then see silence —
251
+ > indistinguishable from a healthy quiet loop. `-F` re-opens the live file by name,
252
+ > so it always tracks the runner's current native stream. (Send `tail`'s own
253
+ > rotation chatter to `/dev/null` so it can't reach the `jq` filter.)
254
+
255
+ - **Fallback (log tail, no NDJSON):** match the runner's **structured prose
256
+ prefixes**, never the `RAUF_*` tokens (those leak inside agent output and
257
+ false-match). For rauf:
258
+
259
+ ```
260
+ tail -n +1 -F {backlogDir}/{loopRunner.stateDir}/{loopRunner.logFile} 2>/dev/null \
261
+ | grep -E --line-buffered 'Item [^ ]+ (completed|blocked):|Item [^ ]+ needs human input|Loop completed|Loop error:|Circuit breaker:'
262
+ ```
263
+
264
+ (Match `needs human input` **without** a trailing colon — the runner writes
265
+ `needs human input (set aside):`.)
266
+
267
+ If the Monitor is ever auto-stopped for event volume, re-arm with a tighter filter
268
+ (drop `item_completed`, keep the exception/terminal events).
269
+
270
+ ## React to events as they land (Step 3e)
271
+
272
+ Each Monitor event arrives as a message. React per type — but keep the user signal
273
+ high and the noise low:
274
+
275
+ - **`item_completed`** → increment a running tally. These land minutes apart, so they
276
+ won't trip the volume auto-stop; still, surface a coalesced milestone ("12/30 done")
277
+ rather than echoing every line. For an exact breakdown, run the one-shot
278
+ `{rendered statusJsonCommand}` and report `done/total` from `backlogSummary`.
279
+ - **`needs_human`** (or `signal_parsed` with `signal: "needs_human"`) → **surface
280
+ immediately** and send a **`PushNotification`** (an hours-long run means the user has
281
+ likely stepped away). **Important — the loop is NOT paused:** the runner has set that
282
+ item aside and kept working other items. So report *what* needs a human and *which*
283
+ item, then either (a) collect the user's answer via `AskUserQuestion` to **stage a
284
+ post-run retry**, or (b) offer to **cancel the run early** if the answer changes the
285
+ whole plan. Do not tell the user the loop is waiting on their reply — it isn't.
286
+ - **`item_blocked`** → surface the blocked item + reason now (visibility) and
287
+ accumulate for the final summary. Use `{rendered statusJsonCommand}` to distinguish a
288
+ genuine `blocked` from a runner-`deferred` "false block" (`backlogSummary.deferred`).
289
+ - **`loop_error`** → a real failure (this is also what a circuit-breaker halt — too many
290
+ consecutive infra failures — emits). Surface now and `PushNotification`. Offer
291
+ inspection / `--force` / re-run as appropriate.
292
+ - **Stall detection** → rauf emits an **`llm_stuck_warning`** event when an iteration
293
+ stops making progress; the filter above includes it, so surface it live (a hang
294
+ warning, not yet a failure) and offer `--force` if it persists. If you instead want to
295
+ probe on quiet, run `{rendered watchCommand}` (or read
296
+ `{backlogDir}/{loopRunner.stateDir}/iteration-status.json`) and key off its
297
+ `stuckWarning` flag. Do **not** infer a stall from `state.json.updatedAt` alone — it is
298
+ not a liveness proof.
299
+
300
+ ## Inform-user output template (Step 3c)
301
+
302
+ This is the verbatim "Loop started…" output the session shows the user after
303
+ launch. Commands are the rendered `loopRunner` monitoring commands.
304
+
305
+ When the agent surface is gated on (`loopRunner.agentArgument` present), add the
306
+ `Coding agent:` line shown below immediately after the opening `Loop started …`
307
+ line, using the same `sourceLabel` mapping as the Step 2d confirmation
308
+ (`RUN` → `"per-run selection"`, `PROJECT` →
309
+ `"project default (loopRunner.defaultAgent)"`, `DEFAULT` →
310
+ `"runner default — claude-cli"`). When the gate is off, the line is **absent** and
311
+ the template is byte-identical to today (REQ-PLUG-02). When the launch proceeded via
312
+ the UNAVAILABLE *proceed-anyway* path, use the audit variant instead:
313
+
314
+ ```
315
+ Coding agent: {resolved.agent or claude-cli} (source: {sourceLabel}).
316
+ Coding agent: {resolved.agent} (source: {sourceLabel}; proceeded despite unavailability warning).
317
+ ```
318
+
319
+ (The two lines above are alternatives — the first is the normal line; the second
320
+ replaces it only on the proceed-anyway path. This is session-side prose only; it
321
+ introduces no new event type, so the Step 3d Monitor filter is unchanged.)
322
+
323
+ ```
324
+ Loop started for {feature} ({N} items to process).
325
+ Coding agent: {resolved.agent or claude-cli} (source: {sourceLabel}). # only when the agent surface is gated on
326
+ This session is now monitoring it live — I'll report milestones and stop you in if
327
+ the loop needs a human. The loop also runs detached and survives this session ending.
328
+ Each item gets a fresh agent session with full context from the backlog and specs.
329
+
330
+ Watch directly if you like (another terminal or `!` prefix):
331
+ {rendered statusCommand} # one-shot status
332
+ {rendered followCommand} # stream live events (human)
333
+ {rendered logCommand} # tail log file
334
+ {rendered listCommand} # check item statuses
335
+
336
+ State files are at: {backlogDir}/{loopRunner.stateDir}/
337
+ - state.json (loop state)
338
+ - events.ndjson (structured event stream this session is watching)
339
+ - {loopRunner.logFile} (human event log)
340
+ - iteration-status.json (live activity, incl. stuckWarning)
341
+ ```