codex-orchestrator 2.0.10 → 2.0.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +29 -30
  3. package/dist/src/index.d.ts +4 -9
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +2 -5
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +53 -25
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +189 -197
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +7 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +44 -3
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/adapters/gh-pull-request-adapter.js +1 -0
  20. package/dist/src/v2/adapters/gh-pull-request-adapter.js.map +1 -1
  21. package/dist/src/v2/adapters/pull-requests.d.ts +1 -0
  22. package/dist/src/v2/adapters/pull-requests.d.ts.map +1 -1
  23. package/dist/src/v2/adapters/pull-requests.js +1 -0
  24. package/dist/src/v2/adapters/pull-requests.js.map +1 -1
  25. package/dist/src/v2/adapters/worktree.js +3 -3
  26. package/dist/src/v2/adapters/worktree.js.map +1 -1
  27. package/dist/src/v2/atomic-store.d.ts +15 -0
  28. package/dist/src/v2/atomic-store.d.ts.map +1 -1
  29. package/dist/src/v2/atomic-store.js +55 -8
  30. package/dist/src/v2/atomic-store.js.map +1 -1
  31. package/dist/src/v2/candidate.d.ts +136 -0
  32. package/dist/src/v2/candidate.d.ts.map +1 -0
  33. package/dist/src/v2/candidate.js +107 -0
  34. package/dist/src/v2/candidate.js.map +1 -0
  35. package/dist/src/v2/checked-change.d.ts +39 -7
  36. package/dist/src/v2/checked-change.d.ts.map +1 -1
  37. package/dist/src/v2/checked-change.js +73 -1
  38. package/dist/src/v2/checked-change.js.map +1 -1
  39. package/dist/src/v2/cli-contract.d.ts +1 -1
  40. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  41. package/dist/src/v2/cli-contract.js +4 -6
  42. package/dist/src/v2/cli-contract.js.map +1 -1
  43. package/dist/src/v2/cli.d.ts +14 -0
  44. package/dist/src/v2/cli.d.ts.map +1 -1
  45. package/dist/src/v2/cli.js +70 -21
  46. package/dist/src/v2/cli.js.map +1 -1
  47. package/dist/src/v2/code-review-report.d.ts +10 -18
  48. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  49. package/dist/src/v2/code-review-report.js +63 -60
  50. package/dist/src/v2/code-review-report.js.map +1 -1
  51. package/dist/src/v2/codex-process.d.ts +6 -1
  52. package/dist/src/v2/codex-process.d.ts.map +1 -1
  53. package/dist/src/v2/codex-process.js +36 -9
  54. package/dist/src/v2/codex-process.js.map +1 -1
  55. package/dist/src/v2/config.d.ts +0 -2
  56. package/dist/src/v2/config.d.ts.map +1 -1
  57. package/dist/src/v2/config.js +3 -6
  58. package/dist/src/v2/config.js.map +1 -1
  59. package/dist/src/v2/contained-report-operation.d.ts +17 -6
  60. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  61. package/dist/src/v2/contained-report-operation.js +11 -24
  62. package/dist/src/v2/contained-report-operation.js.map +1 -1
  63. package/dist/src/v2/containment.d.ts +1 -0
  64. package/dist/src/v2/containment.d.ts.map +1 -1
  65. package/dist/src/v2/containment.js +12 -2
  66. package/dist/src/v2/containment.js.map +1 -1
  67. package/dist/src/v2/delivery-authority.d.ts +26 -0
  68. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  69. package/dist/src/v2/delivery-authority.js +44 -0
  70. package/dist/src/v2/delivery-authority.js.map +1 -0
  71. package/dist/src/v2/direct-delivery.d.ts +19 -57
  72. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  73. package/dist/src/v2/direct-delivery.js +154 -211
  74. package/dist/src/v2/direct-delivery.js.map +1 -1
  75. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  76. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  77. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  78. package/dist/src/v2/implementation-report.d.ts +3 -1
  79. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  80. package/dist/src/v2/implementation-report.js +17 -4
  81. package/dist/src/v2/implementation-report.js.map +1 -1
  82. package/dist/src/v2/implementation-reviewer.d.ts +22 -13
  83. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  84. package/dist/src/v2/implementation-reviewer.js +96 -30
  85. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  86. package/dist/src/v2/owner-control-lock.d.ts +13 -1
  87. package/dist/src/v2/owner-control-lock.d.ts.map +1 -1
  88. package/dist/src/v2/owner-control-lock.js +66 -7
  89. package/dist/src/v2/owner-control-lock.js.map +1 -1
  90. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  91. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  92. package/dist/src/v2/pending-effect-settlement.js +69 -0
  93. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  94. package/dist/src/v2/process-identity.d.ts +45 -0
  95. package/dist/src/v2/process-identity.d.ts.map +1 -0
  96. package/dist/src/v2/process-identity.js +118 -0
  97. package/dist/src/v2/process-identity.js.map +1 -0
  98. package/dist/src/v2/proof-report.d.ts +4 -2
  99. package/dist/src/v2/proof-report.d.ts.map +1 -1
  100. package/dist/src/v2/proof-report.js +16 -44
  101. package/dist/src/v2/proof-report.js.map +1 -1
  102. package/dist/src/v2/review-feedback-coordinator.d.ts +8 -2
  103. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  104. package/dist/src/v2/review-feedback-coordinator.js +98 -71
  105. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  106. package/dist/src/v2/review-feedback.d.ts +14 -38
  107. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  108. package/dist/src/v2/review-feedback.js +48 -115
  109. package/dist/src/v2/review-feedback.js.map +1 -1
  110. package/dist/src/v2/run-issue.d.ts +139 -60
  111. package/dist/src/v2/run-issue.d.ts.map +1 -1
  112. package/dist/src/v2/run-issue.js +2236 -1974
  113. package/dist/src/v2/run-issue.js.map +1 -1
  114. package/dist/src/v2/run-state-projections.d.ts +84 -0
  115. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  116. package/dist/src/v2/run-state-projections.js +142 -0
  117. package/dist/src/v2/run-state-projections.js.map +1 -0
  118. package/dist/src/v2/run-store.d.ts +114 -82
  119. package/dist/src/v2/run-store.d.ts.map +1 -1
  120. package/dist/src/v2/run-store.js +308 -224
  121. package/dist/src/v2/run-store.js.map +1 -1
  122. package/dist/src/v2/runtime-assets.d.ts +3 -0
  123. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  124. package/dist/src/v2/runtime-assets.js +104 -0
  125. package/dist/src/v2/runtime-assets.js.map +1 -1
  126. package/dist/src/v2/runtime.d.ts +52 -15
  127. package/dist/src/v2/runtime.d.ts.map +1 -1
  128. package/dist/src/v2/runtime.js +818 -348
  129. package/dist/src/v2/runtime.js.map +1 -1
  130. package/dist/src/v2/setup.js +0 -2
  131. package/dist/src/v2/setup.js.map +1 -1
  132. package/dist/src/v2/validation-progression.d.ts +70 -0
  133. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  134. package/dist/src/v2/validation-progression.js +247 -0
  135. package/dist/src/v2/validation-progression.js.map +1 -0
  136. package/dist/src/v2/workflow-assets.d.ts +9 -3
  137. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  138. package/dist/src/v2/workflow-assets.js +256 -43
  139. package/dist/src/v2/workflow-assets.js.map +1 -1
  140. package/docs/deep-dive.md +52 -14
  141. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  142. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  143. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  144. package/internal-workflow/manifest.json +1 -1
  145. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  146. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  147. package/internal-workflow/profiles/implementer.toml +9 -0
  148. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  149. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  150. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  151. package/internal-workflow/schemas/code-review-v1.json +1 -1
  152. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  153. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  154. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  155. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  156. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  157. package/internal-workflow/skills/code-review/SKILL.md +84 -288
  158. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  159. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  160. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  161. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  162. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  163. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  164. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  165. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  166. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  167. package/internal-workflow/skills/implement/SKILL.md +135 -0
  168. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  169. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  170. package/internal-workflow/skills/plan/SKILL.md +59 -0
  171. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  172. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  173. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  174. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  175. package/internal-workflow/skills/prototype/UI.md +157 -0
  176. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  178. package/internal-workflow/skills/research/SKILL.md +110 -0
  179. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/research/evals/evals.json +49 -0
  181. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  182. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  183. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  184. package/internal-workflow/skills/tdd/mocking.md +48 -1
  185. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  186. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  187. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  188. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  189. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  190. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  191. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  192. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  193. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  194. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  195. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  196. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  197. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  198. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  199. package/package.json +1 -1
  200. package/dist/src/v2/proof-store.d.ts +0 -42
  201. package/dist/src/v2/proof-store.d.ts.map +0 -1
  202. package/dist/src/v2/proof-store.js +0 -180
  203. package/dist/src/v2/proof-store.js.map +0 -1
  204. package/dist/src/v2/route-continuations.d.ts +0 -32
  205. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  206. package/dist/src/v2/route-continuations.js +0 -2
  207. package/dist/src/v2/route-continuations.js.map +0 -1
  208. package/dist/src/v2/route-coordinator.d.ts +0 -77
  209. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  210. package/dist/src/v2/route-coordinator.js +0 -370
  211. package/dist/src/v2/route-coordinator.js.map +0 -1
  212. package/dist/src/v2/route-decision.d.ts +0 -129
  213. package/dist/src/v2/route-decision.d.ts.map +0 -1
  214. package/dist/src/v2/route-decision.js +0 -400
  215. package/dist/src/v2/route-decision.js.map +0 -1
  216. package/dist/src/v2/spec-coordinator.d.ts +0 -85
  217. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  218. package/dist/src/v2/spec-coordinator.js +0 -88
  219. package/dist/src/v2/spec-coordinator.js.map +0 -1
  220. package/dist/src/v2/spec-delivery.d.ts +0 -143
  221. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  222. package/dist/src/v2/spec-delivery.js +0 -401
  223. package/dist/src/v2/spec-delivery.js.map +0 -1
  224. package/dist/src/v2/triage-route.d.ts +0 -68
  225. package/dist/src/v2/triage-route.d.ts.map +0 -1
  226. package/dist/src/v2/triage-route.js +0 -223
  227. package/dist/src/v2/triage-route.js.map +0 -1
  228. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  229. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  230. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  231. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  232. package/dist/src/v2/waiting-human.d.ts +0 -143
  233. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  234. package/dist/src/v2/waiting-human.js +0 -408
  235. package/dist/src/v2/waiting-human.js.map +0 -1
  236. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  237. package/internal-workflow/docs/agents/review-gates.md +0 -42
  238. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  239. package/internal-workflow/evals/coding-skill-evals.json +0 -332
  240. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  241. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  242. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  243. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  244. package/internal-workflow/operations/triage/SKILL.md +0 -12
  245. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  246. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  247. package/internal-workflow/profiles/proof_agent.toml +0 -8
  248. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  249. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  250. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  251. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  252. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  253. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  254. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  255. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  256. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  257. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  258. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  259. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  260. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  261. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  262. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  263. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  264. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  265. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  266. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  267. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  268. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  269. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  270. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  271. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  272. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -126
  273. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  274. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  275. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  276. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  277. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  278. package/internal-workflow/skills/triage/SKILL.md +0 -134
  279. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
package/docs/deep-dive.md CHANGED
@@ -120,7 +120,9 @@ After claim and worktree creation, the Runner invokes `triage` against the froze
120
120
 
121
121
  The route report records inspected evidence, explicit assumptions, and route-specific details. The Runner hashes the report and persists a route receipt bound to the workflow generation.
122
122
 
123
- A malformed triage report has one report repair budget. A clean transport failure has one separate transport retry. These retries do not become implementation cycles.
123
+ A malformed triage report consumes the route-owned report-repair budget. Infrastructure failures before a recoverable report do not consume that semantic budget and are deferred to a later Runner tick; routing does not own a transport retry counter.
124
+
125
+ `triage`, `ambiguity-review`, `spec-author`, `spec-review`, and `code-review` share one canonical durable report invocation mechanism. It durably binds the exact attempt, immutable workflow generation, an opaque hash of the phase-owned prompt facts, report path, worktree baseline, and host/boot/PID/process-start/process-group fence. Report-only operations run against an attempt-owned read view; `spec-author` instead receives only its attempt-owned target-state directory and may create the correlated immutable revision there. Recovery observes the exact attempt-owned report before considering relaunch, distinguishes a positively absent report from an uncertain read, and adopts output only after process and process-group absence are positive. Prompt-fact drift, uncertain report reads, or uncertain process identity retain the fence and fail closed. Attempt-owned read views are removed before adoption or abandonment; cleanup failure is infrastructure deferral and does not consume a phase semantic budget.
124
126
 
125
127
  An `awaiting-user` proposal is privileged because it pauses autonomous work. It must describe at least two materially different observable product outcomes and prove that repository authority does not select between them. A separate `ambiguity-review` worker receives the candidate and either approves or rejects it. Only one candidate review is allowed. A rejected candidate can use the single triage repair path; an approved candidate becomes the durable route receipt.
126
128
 
@@ -152,7 +154,7 @@ Complexity, cross-cutting behavior, or insufficient executable detail makes dire
152
154
  4. The same reviewer session performs closure review against the affected defects.
153
155
  5. Only an approved revision with resolved blockers is frozen.
154
156
 
155
- Prepared and launched invocation records are persisted before and after process launch. Recovery proves process-group absence before replacing an uncertain invocation. Malformed reports and transport retries are bounded; exhaustion becomes a typed blocker rather than an unbounded author/reviewer loop.
157
+ Both `spec-author` and `spec-review` use the canonical invocation mechanism above. The specification state owns author and reviewer sessions, reviewer independence, revision and target/Closure correlation, malformed-report repair, defect ledger, repair-cycle budget, provenance, and terminal mapping. Each phase session is persisted atomically with canonical prepare, so restart reconstructs byte-identical prompt facts and adopts the exact attempt-owned report and revision before any relaunch. Infrastructure failure before recoverable output consumes no semantic budget; recovered malformed author or review output consumes the existing matching report-repair budget exactly once.
156
158
 
157
159
  The terminal result for this route is `spec-frozen` with an immutable `FrozenSpecReceipt`. The current runtime does not silently continue from a newly authored specification into implementation. That boundary keeps specification approval separately auditable.
158
160
 
@@ -192,7 +194,9 @@ An `implementation` worker receives the current cycle and any findings from prio
192
194
  - inventories tracked, staged, unstaged, untracked, and denied-path state;
193
195
  - requires the reported changed-file list to equal the observed change set.
194
196
 
195
- A clean implementation transport failure receives one separate retry only if the complete Git freshness baseline is unchanged. Any unexplained mutation converts that retry into a safety block.
197
+ Qualification repair, implementation/rework, and trusted review-feedback implementation share one canonical mutable-worktree invocation mechanism. It durably owns attempt identity, prepared/launched/adopted state, host/boot/PID/process-start/process-group fencing, the launch baseline, the exact attempt-owned report, and the adopted result snapshot. Recovery performs one bounded observation per Runner tick and adopts the exact report and worktree result before any relaunch. A positively absent report and positively absent process/group permit abandonment only when the worktree still equals the launch baseline; uncertain ownership, report reads, or effects retain the fence and yield issue-locally without publication.
198
+
199
+ Each caller supplies an opaque correlation over its authoritative prompt facts. Qualification keeps scoped failures and denied-path correlation; implementation keeps claim/issue/head, frozen criteria, cycle, mutable provenance, and candidate binding; review feedback keeps its trusted batch, source authorization, round, and target. Correlation drift never adopts stale output or authorizes a replacement launch. Infrastructure failure before recoverable output consumes no semantic budget. Once an adopted output is classified as malformed, rejected, or needing work, the owning phase clears the invocation and spends exactly its existing semantic budget in the same state transition, so replay cannot spend twice.
196
200
 
197
201
  Before configured checks, a separate `code-review` worker reviews a fingerprint of the complete implementation target. The fingerprint binds Git freshness, changed files, route decision, workflow generation, cycle, and frozen criteria. Full review covers at least acceptance criteria, correctness, and test quality; cleanup is a lens within this final review.
198
202
 
@@ -200,35 +204,69 @@ Review maintains an append-preserving defect ledger. If review returns `needs-wo
200
204
 
201
205
  Coverage text is descriptive, not an identity contract: Closure may paraphrase or omit it. Stable defect and repair-finding IDs, target revision, target fingerprint, and Closure hash carry correlation. Each target revision receives up to four report-only format repairs. Starting a new Closure revision resets that local budget. An eligible legacy terminal malformed-review report can resume only when its evidence ID proves that cause, its per-revision budget remains, and issue authorization, trusted claim, worktree identity, head, changed files, and target fingerprint still match. Current exhausted revisions remain terminal and replay without new effects.
202
206
 
203
- Review invocation intent, process IDs, report hashes, transport retries, report repairs, target revisions, and target fingerprints are durable. A crash after launch cannot cause a replacement review until process absence is proven. Review target drift after approval is a safety failure.
207
+ Canonical review invocation identity, process fencing, prompt correlation, and attempt-owned report recovery are durable. Direct-review state separately owns report repairs, target revisions, target fingerprints, reviewer independence, defect meaning, and terminal mapping; it has no review transport retry counter. A crash after launch cannot cause a replacement review until process and process-group absence are proven and its read view is settled. Review target or authoritative prompt-fact drift is a safety failure.
208
+
209
+ The run-store performs one bounded canonicalization for active states written by the replaced route/direct-review/spec-author/spec-review lifecycle owners. Safely convertible ready/completed or prepared/no-effect states are rewritten once with the removed fields absent and exact source bytes backed up. An old launched or otherwise ambiguous owner that lacks PID-reuse-resistant process identity is converted to an issue-local non-resumable safety outcome; it is never silently deleted or relaunched. Canonical validators remain exact and do not retain a legacy reader or dual-write path.
204
210
 
205
211
  ## 8. Issue-scoped checks and `CheckedChange`
206
212
 
207
213
  For a new direct run, the Runner resolves its finite check policy from the frozen issue body. A command-only `Verification:` or `## Verification:` section replaces repository-wide checks with deterministic `issue-verification-NNN` entries. Scoped entries are limited to `npm [--prefix <repository-relative-path>] test ...` and `npm [--prefix <repository-relative-path>] run <script> ...`; the Runner parses them to argv and executes them without a shell. Interpreter eval, package-exec, nested-shell, shell-composition, malformed, duplicate, mixed-validity, and ambiguous sections fail closed before any check runs. The triage worker's free-text verification output is never command authority. `config.checks` is used only when the frozen issue has no Verification section.
208
214
 
209
- Before the issue implementation starts, the Runner executes the resolved policy as a qualification gate. If any command fails, a separate sealed qualification-repair operation receives the scoped failures but not the issue acceptance criteria, and may repair the files needed to make the policy green. The Runner reruns the complete policy after each repair. Qualification repairs have their own bounded launched-attempt counter and do not consume the issue's implementation-cycle budget. Their cumulative files remain in the worktree and are included in the later implementation report, independent review, final checks, proof, and PR.
215
+ Before the issue implementation starts, the Runner executes the resolved policy as a qualification gate. If any command fails, a separate sealed qualification-repair operation receives the scoped failures but not the issue acceptance criteria, and may repair the files needed to make the policy green. The Runner reruns the complete policy after each adopted repair. Qualification repairs have their own bounded semantic repair budget and do not consume the issue's implementation-cycle budget. Their cumulative files remain in the worktree and are included in the later implementation report, independent review, final checks, proof, and PR.
210
216
 
211
- A qualification process that cannot start returns a resumable transport outcome without consuming either budget. Once launched, its existing prepared/launched receipt reserves one repair attempt; restart proves process absence and recovers its report, or requires an unchanged launch baseline before retrying. A dirty resumed worktree does not skip qualification. The main implementation cycle advances only after its own launch is durably recorded. Timeout and cancellation retain ownership until the complete process group is proven absent. Unprovable process-group quiescence or an unreported changed worktree fails closed. Invalid scoped policy remains a resumable no-effect outcome so a package-side policy correction can continue the existing run instead of replaying a terminal failure.
217
+ A qualification process that cannot start returns a resumable transport outcome without consuming either budget. Launch does not reserve or spend a semantic repair. Restart reconstructs the persisted scoped failures and correlation, observes the canonical invocation, and either adopts its exact report/result or retains the fence. The main implementation cycle likewise advances only for phase-owned semantic rework, never for transport. Timeout and cancellation retain ownership until the complete process group is proven absent; an unreported changed worktree remains fenced without relaunch. Invalid scoped policy remains a resumable no-effect outcome so a package-side policy correction can continue the existing run instead of replaying a terminal failure.
212
218
 
213
219
  Once qualification is green, the issue implementation starts. After independent review clears, the Runner executes the same policy against the complete change. Every final check must pass; a nonzero result becomes a durable task-owned repair finding and starts another implementation cycle if the five-cycle budget remains. There is no output-hash attribution and no `unchanged-failure` success state. Passed final results are reused on a safe resume, while any new repair cycle clears stale final-check and proof bindings.
214
220
 
215
- Once no task-owned check failure remains, the Runner fingerprints the reviewed files and content, stages the complete validated change, and verifies that staging did not change either binding. It then mints a `CheckedChange` capability containing:
221
+ Once no task-owned check failure remains, the Runner captures the reviewed bytes
222
+ twice with a private index initialized from expected HEAD. Tracked deletions,
223
+ mode changes, symlinks, and non-ignored untracked files enter the candidate;
224
+ ignored untracked files and only the configured untracked proof root do not. The
225
+ two trees and canonical path sets must match. The shared index is unchanged.
226
+
227
+ The stable tree is pinned by a synthetic single-parent commit under a
228
+ package-owned candidate ref. Direct review, each qualification/final check, and
229
+ Acceptance Proof run in a fresh detached worktree at that commit. A durable lease
230
+ records preparation and, before child execution proceeds, PID/process-group
231
+ ownership. Results are accepted only after HEAD, tree, and non-proof changes are
232
+ still exact.
233
+
234
+ Legacy `CheckedChange` V1 keeps its original mutable snapshot semantics. New
235
+ runs mint V2 with:
216
236
 
217
237
  - canonical repository, run ID, issue number, and cycle;
218
- - base and head SHA;
219
- - index tree SHA;
220
- - tracked and untracked content hashes;
221
- - worktree identity;
238
+ - base SHA and the complete candidate binding (binding ID, expected HEAD,
239
+ candidate commit/tree, candidate ref, changed files, and worktree identity);
222
240
  - exact changed files;
223
241
  - passed check records and check-policy hash;
224
242
  - package and proof schema versions.
225
243
 
226
- Any later change to Git, the index, tracked or untracked content, worktree identity, changed files, or check policy invalidates the capability.
244
+ V2 freshness compares candidate binding/tree and check policy, so unrelated
245
+ shared-index edits cannot invalidate or authorize the proof input.
227
246
 
228
247
  ## 9. Acceptance Proof
229
248
 
230
249
  Acceptance Proof is independent from both implementation and code review. The `acceptance-proof` worker receives the frozen issue criteria and a nominal checked-change capability. It runs in a separate contained process and may write only below its proof-owned artifact root.
231
250
 
251
+ Its proof record embeds the canonical durable invocation fence: exact attempt,
252
+ workflow generation, prompt facts, host/boot/PID/process-start/PGID identity,
253
+ report path, and proof-safe worktree baseline. Restart performs one report and
254
+ process observation per tick, adopts only the exact attempt-owned report and
255
+ hash-bound artifacts, and never relaunches while ownership is live or unknown.
256
+ Infrastructure uncertainty spends no proof repair budget. Runner-classified
257
+ malformed output spends the single existing report-repair bit exactly once and
258
+ binds the complete pre-repair artifact inventory so report-only repair cannot
259
+ modify or manufacture evidence. Immutable runner-owned iOS helper, lease,
260
+ owner, xcrun, runtime, and device-type facts are part of the same invocation
261
+ correlation and proof binding.
262
+ Proof Reports cannot claim completed checks; only the nominal `CheckedChange`
263
+ receipts can satisfy check evidence references.
264
+
265
+ For V2, proof runs against its own candidate materialization. Publishable
266
+ artifacts are hash-validated and copied back idempotently without following
267
+ symlinks; product bytes remain immutable. The same candidate tree then becomes
268
+ the durable commit intent and the observed publication commit tree.
269
+
232
270
  The Runner validates:
233
271
 
234
272
  - the exact Proof Report schema and status semantics;
@@ -247,7 +285,7 @@ Local command output and static-inspection evidence may contain machine paths be
247
285
 
248
286
  Browser proof validates current workflow evidence rather than accepting an isolated screenshot. For a configured Android surface, the trusted Runner durably reserves preparation before starting the configured AVD on an unused port with a clean ephemeral data directory. The lease records both PID and process-start identity so replay cleanup never kills a foreign emulator after port reuse, and ownership is rechecked throughout boot, install, navigation, and capture. The Runner removes any prior APK target, executes only a bounded, cancellable, process-group-owned `flutter build apk` recipe, snapshots the fresh no-symlink result outside the worker-writable tree, and installs only that exact digest-bound snapshot. It launches the application, retries exact accessibility-label navigation within configured bounds, and captures proof-bound screenshot, validated UI hierarchy, PID-scoped log, and lease. The contained proof worker can inspect immutable-digest-bound worktree artifacts but cannot invoke `adb`, the emulator, Flutter, or an Android lease helper. Terminal or exceptional proof settlement performs replay-safe lease cleanup, stops only the same Runner-created process, and removes only its validated temporary data directory. Existing physical devices, user emulators, IDE sessions, and Flutter processes are observed but never taken over. Android infrastructure or startup failure is recorded as an unfinished-UI-proof warning and does not alone block delivery; successful Android proof remains strict and cannot be claimed without the complete validated artifact set.
249
287
 
250
- `needs-rework` findings return to the same implementation loop and consume another cycle. External, safety, malformed, quiescence, or exhausted outcomes are mapped to typed run results. Only `passed` proof produces a proof receipt and permits publication.
288
+ `needs-rework` findings return to the same implementation loop and consume another cycle. External, safety, malformed, quiescence, or exhausted outcomes are mapped to typed run results. A passed receipt is persisted before candidate cleanup and remains monotonic while cleanup's existing effect/postcondition settlement retries; publication remains forbidden until cleanup is proven settled.
251
289
 
252
290
  ## 10. Runner-owned publication
253
291
 
@@ -378,6 +416,6 @@ npm pack --dry-run --json
378
416
 
379
417
  The build deletes `dist` before TypeScript compilation so removed modules cannot survive in tests or the tarball. `prepack` verifies the committed workflow and rebuilds from a clean output directory.
380
418
 
381
- `npm run smoke:live` packs and installs the exact package bytes into a temporary consumer and mutates only the configured scratch GitHub repository. The default `core-release` profile proves package installation through real model-backed operations, browser evidence, and a safety-negative path. Cleanup verifies that run-owned issues, PRs, branches, labels, worktrees, and temporary directories are absent.
419
+ `npm run smoke:live` packs and installs the exact package bytes into a temporary consumer and mutates only the configured scratch GitHub repository. The default `core-release` profile proves package installation through real model-backed operations, browser evidence, and a safety-negative path. The supplemental `authoritative-candidate-publication` scenario injects a stale shared-index entry and proves that V3 candidate-bound checks, the exact published tree, pin release, and immutable execution cleanup all converge on final worktree bytes. Cleanup verifies that run-owned issues, PRs, branches, labels, worktrees, and temporary directories are absent.
382
420
 
383
421
  Live smoke is not a normal local test and must run only with explicit authorization. Release publication is owned by the GitHub release workflow after the release commit reaches `main`.
@@ -4,21 +4,23 @@ Use the bug skills by user intent and feedback-loop quality:
4
4
 
5
5
  - `bug-root-cause-explainer`: diagnosis-only. Use when the user asks why, wants root cause, options, or no edits. Stop before implementation.
6
6
  - `diagnosing-bugs`: feedback-loop builder. Use when the bug is hard, flaky, unclear, performance-related, or cannot be proven with a tight red/green signal.
7
- - `code-debugger`: implementation. Use when the user asks to fix, chose a fix path, or the bug is obvious and reproducible.
7
+ - `implement`: implementation. Use when the user asks to fix, chose a fix path, or the bug is obvious and reproducible.
8
8
 
9
9
  Handoffs:
10
10
 
11
11
  - Explainer -> Diagnosing Bugs when root cause cannot be proven without a reliable loop.
12
- - Explainer -> Code Debugger after the user chooses a fix path.
13
- - Code Debugger -> Diagnosing Bugs when implementation discovers the bug is unclear, flaky, or lacks a red-capable signal.
14
- - Diagnosing Bugs -> original intent: return diagnosis to the explainer path, or apply the fix through Code Debugger.
12
+ - Explainer -> Implement after the user chooses a fix path.
13
+ - Implement -> Diagnosing Bugs when implementation discovers the bug is unclear, flaky, or lacks a reliable signal.
14
+ - Diagnosing Bugs -> original intent: return diagnosis to the explainer path, or apply the fix through Implement.
15
15
 
16
16
  Inside an active user-authorized implementation/TDD flow, keep a reviewer repair
17
17
  in that same TDD activation only when the defect is high-confidence,
18
18
  source-required, inside the approved behavior/Seam, and has bounded trigger,
19
19
  cause, and repair. Group related cases by protected invariant before one
20
- consolidated repair batch. New product intent, a new Seam, or a risky trade-off
21
- stops for user approval. A separately authorized fix with no active flow, or an
22
- ambiguous reproduction/cause/repair, still routes through `code-debugger`.
20
+ consolidated repair batch for the current reviewed revision. Continue repair,
21
+ affected proof, and targeted Review until approval. New product intent, a new
22
+ Seam, or a risky trade-off stops for user approval. A separately authorized fix with no active flow routes
23
+ through `implement`; ambiguous reproduction/cause/repair first routes through
24
+ `diagnosing-bugs`.
23
25
 
24
26
  Do not collapse diagnosis-only, feedback-loop construction, and implementation into one response unless the user explicitly asked for that full flow.
@@ -1,123 +1,173 @@
1
1
  # Coding Skill Routing
2
2
 
3
- This file is the normative global route and ownership policy. Keep repository
4
- commands, domain facts, credentials, deployment assumptions, and product
5
- behavior in repository `AGENTS.md`, `CONTEXT.md`, ADRs, or local skills.
6
-
7
- ## Ownership
8
-
9
- - Personal skills live only in `../../skills`.
10
- - `agents/openai.yaml` owns invocation metadata; `agents/*.toml` owns named-role
11
- model and effort.
12
- - A shared rule has one owner. Callers link to it instead of copying its prose.
13
- - Skill-specific detail belongs in that skill's `references/` and loads only
14
- when its branch is active.
15
- - Root owns the user dialogue, decisions, critical path, integration, and final
16
- handoff. A reviewer child owns independent review.
17
-
18
- ## Default Implementation Route
19
-
20
- `medium` is the default for behavior-changing implementation. Use:
21
-
22
- - `simple` only for a tiny local change with one obvious proof;
23
- - `medium` for a clear coherent outcome with settled authority, one ownership
24
- path, and credible affected validation—even across several files, modules,
25
- API, persistence, or shared state;
26
- - `high` only when a sensitive mechanism has both a material failure
27
- consequence and an uncertainty amplifier such as unclear ownership,
28
- cross-trust effects, non-local recovery, an unproven external contract, or
29
- proof that cannot isolate the dangerous state.
30
-
31
- Prefer direct root implementation for `simple` and ordinary `medium` work. Use
32
- `$implementation-spec-maker` only for a real execution decision or coordination
33
- gap. Use `$tickets-orchestrator` only for an approved ticket graph, real delivery
34
- dependencies or disjoint parallel slices, or an explicit orchestration request.
35
- Do not manufacture PRDs, tickets, specs, agents, or review checkpoints from
36
- file count or generic risk labels.
37
-
38
- ## Core Routes
39
-
40
- | Situation | Route |
3
+ This file is the normative global coding route. Repository commands, product
4
+ facts, runtime constraints, and domain language remain in repository policy.
5
+
6
+ ## Shared Kernel
7
+
8
+ - **Authority** — perform only the requested outcome and actions authorized by
9
+ the user, Parent PRD, executable ticket, and repository policy. Planning or
10
+ publication never authorizes implementation. Normal direct and single-ticket
11
+ Implement authority includes one scoped local commit after proof and
12
+ applicable Review unless user or repository policy explicitly forbids or
13
+ reserves Git. Push and PR require separate authority.
14
+ - **Preservation** — preserve unrelated and concurrent work plus user-owned
15
+ runtimes. Dirty overlapping or unisolatable scope blocks the affected write
16
+ and commit.
17
+ - **Proof** — claim only the observable outcome proved through the real caller
18
+ seam. Missing required proof or independent-review approval blocks completion
19
+ and the affected commit.
20
+
21
+ These principles are checks, not workflow state or durable artifacts.
22
+
23
+ ## Main Route
24
+
25
+ The user-facing coding flow is Plan, Implement, Review.
26
+
27
+ - [`$plan`](../../skills/plan/SKILL.md) owns product decisions and multi-ticket
28
+ planning composition. It is the sole planning-composition entrypoint.
29
+ - [`$implement`](../../skills/implement/SKILL.md) is the single execution owner
30
+ for clear features, fixes, obvious local edits, and executable tickets.
31
+ - [`$code-review`](../../skills/code-review/SKILL.md) is the direct Review
32
+ entrypoint.
33
+
34
+ Direct non-ticket work stays in the current root context. Do not create a spec,
35
+ ticket, or worker merely because a change spans files or touches an important
36
+ contract. A real product or ownership decision gap returns to Plan. An approved
37
+ multi-ticket dependency graph routes to `$tickets-orchestrator`; a single
38
+ executable ticket never does.
39
+
40
+ Grilling uses dependency-aware frontier rounds. `$grilling` is the model-invoked,
41
+ write-free interview primitive; `$domain-modeling` is the model-invoked sole
42
+ domain-document writer. `$grill-me`, `$grill-with-docs`, and `$wait-what` are
43
+ explicit-only. Explicitly typed wrappers win: `$grill-me` remains write-free
44
+ even when its prompt mentions terminology or ADRs, while `$grill-with-docs`
45
+ keeps the whole `$grilling` interview write-free and invokes
46
+ `$domain-modeling` only after an empty frontier and explicit confirmation.
47
+ Implicit `$grilling` is write-free and cannot mutate domain docs.
48
+
49
+ `$writing-for-agents` is a model-invoked discipline for agent-consumed
50
+ documents and complements `$skill-creator`; it is not a router. Use
51
+ `$resolving-merge-conflicts` only for actual Git conflicts or an explicit
52
+ conflict-resolution request. It composes `$domain-modeling` as sole content
53
+ writer for domain-document conflicts, while each Git operation remains gated by
54
+ existing authority. `$documentation-gardener` may audit domain docs but routes
55
+ every authorized domain-document mutation through `$domain-modeling`.
56
+
57
+ Diagnosis-only work uses `$bug-root-cause-explainer`. Hard, flaky, unclear, or
58
+ performance bugs use `$diagnosing-bugs` to establish a reliable signal, then
59
+ return to Implement when a fix is authorized. External multi-source uncertainty
60
+ uses `$research`. A requested throwaway logic or UI experiment uses `$prototype`
61
+ to answer one bounded design question, then returns production delivery to
62
+ Implement. Grilling, TDD, to-spec, to-tickets, diagnosis, research, prototype,
63
+ and the graph-only Tickets Orchestrator are internal or side primitives.
64
+ Specialized runtime and platform skills remain side tools and do not compete
65
+ with the main flow.
66
+
67
+ ## Phase boundaries
68
+
69
+ Make a context choice only when one phase has actually completed and another is
70
+ about to begin; never interrupt a live phase merely because the conversation is
71
+ long. Apply these conditions in order:
72
+
73
+ 1. **Continue** when the next phase uses the current conversation as a primary
74
+ source and the same active owner and authority still apply.
75
+ 2. **Use an existing fresh-context route** only when the active workflow already
76
+ requires one, such as the single-ticket implementer or independent Review.
77
+ 3. **Use an authorized stable-role child** only for bounded AFK work that the
78
+ active skill already permits; root retains dialogue and integration.
79
+ 4. **Compact through the platform** only when relevant same-owner work must
80
+ continue in the current conversation and observed context pressure makes a
81
+ lossless continuation unreliable.
82
+
83
+ These are branch conditions, not a token threshold, a new handoff skill, or a
84
+ competing router. Plan, Implement, Review, prototype, `$to-spec`, and
85
+ `$to-tickets` keep their existing ownership and stop boundaries.
86
+
87
+ ## Implement Context
88
+
89
+ - A direct request is implemented by root in the current context.
90
+ - One executable ticket launches exactly one fresh `implementer` child. Root
91
+ supplies the complete ticket, Parent PRD, applicable repository policy, and
92
+ bounded write scope; verifies the child identity and completed wait; and
93
+ remains the only Git owner.
94
+ - Multiple executable tickets are graph work and remain owned by
95
+ `$tickets-orchestrator`.
96
+
97
+ Children do not talk to the user, spawn grandchildren, or perform Git actions.
98
+ Root integrates only isolated worker output and never overwrites unrelated work.
99
+
100
+ ## TDD, Proof, And Review
101
+
102
+ Use [`$tdd`](../../skills/tdd/SKILL.md) where possible: an observable behavior
103
+ change has a natural public seam and a meaningful failing signal can precede
104
+ the implementation. Otherwise use direct proof. Do not manufacture tests for
105
+ docs, copy, formatting, mechanical config, deletion, or an outcome that cannot
106
+ fail meaningfully before the edit.
107
+
108
+ A change is substantial when its behavior or contract goes beyond an obvious
109
+ local edit. This includes a public API, persistence, auth or payment,
110
+ concurrency or shared state, and cross-module interaction. Substantial settled
111
+ work launches one fresh `spec_reviewer` and one fresh `standards_reviewer` in
112
+ parallel. Spec checks the result against the request, issue, or Parent PRD;
113
+ Standards checks correctness, repository rules, cleanup, and legacy or duplicate
114
+ ownership.
115
+
116
+ The wait must complete and the coordinator must approve the settled diff from
117
+ the completed independent evidence.
118
+ Docs, copy, formatting, mechanical config, and an obvious local correction may
119
+ finish with direct proof and no reviewer. Reviewer failure or timeout blocks
120
+ approval; root does not replace independent review with self-review.
121
+
122
+ A Review blocker requires a concrete defect or proof gap causally linked to an
123
+ explicit obligation, existing invariant, or mandatory repository rule. The
124
+ reviewer verdict is evidence, not authority: coordinator approval exists when
125
+ the completed independent review has no verified blocker or required-proof
126
+ gap. A Fowler smell, general improvement, preference, or uncertain concern is a
127
+ non-blocking observation without that link. Review remains read-only. Initial
128
+ Review runs both lenses; targeted Review runs only the affected lens, or both
129
+ when the repair is mixed or cannot be isolated.
130
+
131
+ For direct or single-ticket substantial work, Implement consolidates every
132
+ verified blocker into one repair batch for that revision. The original
133
+ Implement authority covers every verified repair that does not widen the
134
+ authorized outcome without another user
135
+ confirmation. After repair it reruns affected proof and launches the affected
136
+ fresh reviewer or reviewer pair over only the repair delta, repaired blockers,
137
+ direct impact cone, and affected proof. Untouched previously approved scope
138
+ retains approval. This loop continues until approval; reviewer count is not a
139
+ stop condition. A complete two-lens Review repeats only when repair impact
140
+ cannot be isolated from previously approved scope.
141
+
142
+ Graph children receive no per-ticket delivery Review. After every checkpoint,
143
+ Tickets Orchestrator owns the initial cumulative review over the complete Parent
144
+ range, then follows the same unlimited repair loop with targeted Review of each
145
+ repair delta and directly affected Parent obligations.
146
+
147
+ Proof and applicable review must approve before staging or commit. Direct and
148
+ single-ticket Implement creates one scoped local commit by default when the
149
+ worktree is isolatable. Explicit user or repository policy that forbids or
150
+ reserves Git overrides that default. Push and PR always require separate
151
+ authority.
152
+
153
+ ## Stable Roles
154
+
155
+ Skills request only these stable roles; concrete model and reasoning effort are
156
+ central configuration details:
157
+
158
+ | Need | Role |
41
159
  | --- | --- |
42
- | Tiny, clear, low-risk edit | `$small-task-implementer` after its Fit Gate |
43
- | Clear feature or fix | Apply the TDD Fit Gate; when it fits, use Root + one `$tdd` activation, otherwise affected validation |
44
- | Missing execution detail | `$implementation-spec-maker` -> artifact review -> `$spec-implementer` |
45
- | Approved implementation spec | `$spec-implementer` |
46
- | Approved dependency graph or explicit orchestration | `$tickets-orchestrator` |
47
- | Product discovery or ticket decomposition | `$to-spec`, `$spec-to-tickets`, or `$wayfinder` as applicable; stop before delivery |
48
- | Explain-only bug | `$bug-root-cause-explainer`; no edits |
49
- | Confirmed bounded bug fix | Apply the TDD Fit Gate, then `$tdd` + `$code-debugger` when it fits; otherwise `$code-debugger` + affected validation |
50
- | Hard, flaky, unclear, or performance bug | `$diagnosing-bugs` before the explain/fix route |
51
- | Review request | `$code-review` in the profile-selected reviewer child |
52
- | External multi-source uncertainty | `$research`; narrow documentation lookup stays inline |
53
- | Commit request | `$commit`; push/PR still require separate authority |
54
-
55
- Generated planning artifacts and labels never authorize implementation. One
56
- deterministic approved ticket may run directly; a graph follows the authorized
57
- delivery workflow.
58
-
59
- ## TDD And Review
60
-
61
- Apply `$tdd` only when all three Fit Gate conditions hold:
62
-
63
- 1. The change alters observable behavior.
64
- 2. A natural public seam can prove that behavior.
65
- 3. A new test will fail before the change for the intended behavioral reason.
66
-
67
- Otherwise use existing regression tests plus affected validation. Do not invoke
68
- TDD merely because implementation files change, and do not manufacture RED
69
- tests for behavior-preserving cleanup, dead-code deletion, documentation, copy,
70
- formatting, generated assets, package maintenance, simple config, builds, or
71
- read-only work. For mixed tasks, activate `$tdd` only for the behavioral slice.
72
- An absence or architecture guard added after cleanup is validation, not a TDD
73
- cycle.
74
-
75
- Review applicability lives in [`review-gates.md`](review-gates.md). Shared
76
- Full/Closure mechanics live in [`review-protocol.md`](review-protocol.md).
77
- Artifact review is owned by
78
- [`implementation-spec-review/references/review-loop.md`](../../skills/implementation-spec-review/references/review-loop.md);
79
- approved-spec implementation review is owned by
80
- [`spec-implementer/references/review-loop.md`](../../skills/spec-implementer/references/review-loop.md).
81
-
82
- Root never substitutes self-review for a required independent reviewer. Use one
83
- `reviewer_fast` for `simple`, one `reviewer_standard` for `medium`, and two
84
- disjoint `reviewer_deep` tracks for `high`. A reviewer Adapter executes inline
85
- only after it is already inside that assigned child.
86
-
87
- ## Delegation
88
-
89
- Run work inline by default. Delegate only when the user, an invoked skill, or
90
- repository policy authorizes it and the task benefits from independent review,
91
- isolated deep analysis, or disjoint implementation ownership.
92
-
93
- | Need | Named role |
94
- | --- | --- |
95
- | Mechanical inventory | `explorer_quick` |
96
- | Bounded cross-module trace | `explorer_fast` |
97
- | Ambiguous architecture, contract, or cause | `analyst_deep` |
98
- | Primary-source external research | `researcher_standard` |
99
- | Independent review | `reviewer_fast`, `reviewer_standard`, or `reviewer_deep` by profile |
100
- | Approved isolated implementation slice | `implementer_standard`; `implementer_deep` only for material uncertainty |
101
-
102
- Keep the root critical path local. Use at most two explorers for disjoint
103
- questions and at most two parallel implementers with disjoint write scopes.
104
- Children do not conduct user dialogue or spawn grandchildren.
105
-
106
- ## Validation And Runtime Safety
107
-
108
- Use targeted behavior proof plus the smallest affected integration check for
109
- simple and medium work. Run a full repository suite only when repository policy
110
- requires it, a broad shared contract cannot be isolated, or the task is `high`.
111
-
112
- For Flutter UI, follow [`tool-usage.md`](tool-usage.md): platform QA owns UI
113
- work and `$flutter-attach-session` is only the attach-safe runtime layer. Treat
114
- live app, IDE, VM Service, and `flutter run` sessions as user-owned.
115
-
116
- Read local evidence before external search: applicable `AGENTS.md`, `CONTEXT.md`,
117
- ADRs, manifests, lockfiles, tests, scripts, and code owners. Mark missing facts
118
- unconfirmed rather than inventing them.
119
-
120
- Contract-risk implementation uses
121
- [`contract-test-ledger.md`](contract-test-ledger.md) only for material
122
- invariants. Long framework lenses, examples, and recipes remain skill-local and
123
- load on demand.
160
+ | Root dialogue, integration, and Git ownership | `root` |
161
+ | One isolated executable ticket | `implementer` |
162
+ | Requirement fidelity and scope drift | `spec_reviewer` |
163
+ | Correctness and repository standards | `standards_reviewer` |
164
+ | Bounded repository exploration | `explorer` |
165
+ | Independent bounded design alternative | `explorer` |
166
+ | Primary-source external research | `researcher` |
167
+
168
+ ## Runtime Safety
169
+
170
+ For Flutter UI, follow [`tool-usage.md`](tool-usage.md). Treat live app, IDE, VM
171
+ Service, and `flutter run` sessions as user-owned. Use the documented
172
+ non-destructive attach path only after ownership discovery; never replace or
173
+ terminate a user-owned session without explicit authority.
@@ -4,13 +4,13 @@
4
4
 
5
5
  Local Flutter UI work always uses a platform QA skill plus the runtime ownership gate:
6
6
 
7
- - Android emulator: use `$flutter-android-debug` as the lifecycle orchestrator and
8
- `test-android-apps:android-emulator-qa` for navigation, interaction, UI trees,
9
- screenshots, and logcat.
7
+ - Android emulator: always use `$flutter-android-debug` as the lifecycle orchestrator
8
+ and `test-android-apps:android-emulator-qa` for navigation, interaction, UI trees,
9
+ screenshots, and logcat, including inspection-only and screenshot-only requests.
10
10
  - iOS Simulator: use `$flutter-ios-debug` for environment/login gates, navigation,
11
11
  interaction, screenshots, and visual comparison.
12
- - Use `$flutter-attach-session` only to discover the runtime owner and perform
13
- reload/restart when that session is safe for this agent to control.
12
+ - Use `$flutter-attach-session` to discover the runtime owner and perform
13
+ reload/restart against the selected live session.
14
14
 
15
15
  Before any install, launch, attach, reload, restart, terminate, or force-stop:
16
16
 
@@ -18,17 +18,28 @@ Before any install, launch, attach, reload, restart, terminate, or force-stop:
18
18
  owner, and expected backend/environment.
19
19
  2. Treat any live PID, VM Service, IDE debug adapter, `flutter run`, or visible app as
20
20
  user-owned state.
21
- 3. If an IDE/debug adapter or machine run owns DevFS, do not create a second attach
22
- controller. Use the owning IDE/terminal reload action or ask the user to trigger it.
23
- A standalone interactive terminal run is the only exception, and only when
24
- discovery marks it attach-safe and the helper receives its confirmed PID.
21
+ 3. If an IDE/debug adapter or machine run owns DevFS, never attach a second Flutter
22
+ compiler/controller. Preserve the session, continue only with inspection or in-app
23
+ navigation, and ask the user to trigger `r`/`R` through the owning controller when
24
+ reload/restart is necessary. Unknown owners remain blocked.
25
25
  4. Use `r` for widget/layout/style/rendering changes and `R` for startup state,
26
26
  dependency injection, providers, globals, routes, or initialization changes.
27
- 5. For a safely attachable standalone runtime, pass the PID confirmed by
28
- `discover --json` as `--expected-pid`; never execute the raw discovered attach command.
29
- 6. Verify that the same PID and expected environment remain after each runtime or
27
+ 5. Always pass the PID confirmed by `discover --json` as `--expected-pid`; never
28
+ execute the raw discovered attach command.
29
+ 6. Confirm that the generated attach command preserves the selected run's effective
30
+ target and every `--dart-define` / `--dart-define-from-file`. Missing compile
31
+ identity must block runtime control rather than fall back to defaults.
32
+ Treat raw define values as sensitive; inspect only the helper's redacted display.
33
+ 7. Verify that the same PID and expected environment remain after each runtime or
30
34
  navigation action.
31
35
 
36
+ The bundled `flutter_session_ctl.py` helper is the only allowed reload/restart
37
+ transport for safely attachable standalone sessions. Never activate, open, focus, or
38
+ automate an IDE for this purpose. Do not use IDE menus, Command Palette, keyboard
39
+ shortcuts, `osascript`, System Events, or other GUI automation. If the helper fails or
40
+ times out, preserve the app process and report its exact failure; do not fall back to
41
+ the IDE.
42
+
32
43
  Requests to inspect, debug, navigate, capture screenshots, or verify UI do not authorize
33
44
  build/install, uninstall, app-data clearing, force-stop, process termination, replacement
34
45
  launch, or a new `flutter run`. Use those only when no live target exists and the user