codex-orchestrator 2.0.11 → 2.0.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/CHANGELOG.md +15 -0
  2. package/README.md +25 -51
  3. package/dist/src/index.d.ts +2 -8
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +1 -4
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +46 -31
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +157 -195
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +6 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +43 -2
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/candidate.d.ts +15 -31
  20. package/dist/src/v2/candidate.d.ts.map +1 -1
  21. package/dist/src/v2/candidate.js +7 -29
  22. package/dist/src/v2/candidate.js.map +1 -1
  23. package/dist/src/v2/checked-change.d.ts +3 -2
  24. package/dist/src/v2/checked-change.d.ts.map +1 -1
  25. package/dist/src/v2/checked-change.js +4 -3
  26. package/dist/src/v2/checked-change.js.map +1 -1
  27. package/dist/src/v2/cli-contract.d.ts +1 -1
  28. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  29. package/dist/src/v2/cli-contract.js +4 -6
  30. package/dist/src/v2/cli-contract.js.map +1 -1
  31. package/dist/src/v2/cli.d.ts +8 -0
  32. package/dist/src/v2/cli.d.ts.map +1 -1
  33. package/dist/src/v2/cli.js +13 -0
  34. package/dist/src/v2/cli.js.map +1 -1
  35. package/dist/src/v2/code-review-report.d.ts +10 -18
  36. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  37. package/dist/src/v2/code-review-report.js +63 -60
  38. package/dist/src/v2/code-review-report.js.map +1 -1
  39. package/dist/src/v2/codex-process.d.ts +6 -2
  40. package/dist/src/v2/codex-process.d.ts.map +1 -1
  41. package/dist/src/v2/codex-process.js +25 -9
  42. package/dist/src/v2/codex-process.js.map +1 -1
  43. package/dist/src/v2/config.d.ts +0 -2
  44. package/dist/src/v2/config.d.ts.map +1 -1
  45. package/dist/src/v2/config.js +3 -6
  46. package/dist/src/v2/config.js.map +1 -1
  47. package/dist/src/v2/contained-report-operation.d.ts +41 -196
  48. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  49. package/dist/src/v2/contained-report-operation.js +139 -466
  50. package/dist/src/v2/contained-report-operation.js.map +1 -1
  51. package/dist/src/v2/containment.d.ts +1 -0
  52. package/dist/src/v2/containment.d.ts.map +1 -1
  53. package/dist/src/v2/containment.js +12 -2
  54. package/dist/src/v2/containment.js.map +1 -1
  55. package/dist/src/v2/delivery-authority.d.ts +26 -0
  56. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  57. package/dist/src/v2/delivery-authority.js +44 -0
  58. package/dist/src/v2/delivery-authority.js.map +1 -0
  59. package/dist/src/v2/direct-delivery.d.ts +16 -36
  60. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  61. package/dist/src/v2/direct-delivery.js +135 -122
  62. package/dist/src/v2/direct-delivery.js.map +1 -1
  63. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  64. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  65. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  66. package/dist/src/v2/implementation-report.d.ts +3 -1
  67. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  68. package/dist/src/v2/implementation-report.js +17 -4
  69. package/dist/src/v2/implementation-report.js.map +1 -1
  70. package/dist/src/v2/implementation-reviewer.d.ts +41 -12
  71. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  72. package/dist/src/v2/implementation-reviewer.js +114 -42
  73. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  74. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  75. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  76. package/dist/src/v2/pending-effect-settlement.js +69 -0
  77. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  78. package/dist/src/v2/process-identity.d.ts +45 -0
  79. package/dist/src/v2/process-identity.d.ts.map +1 -0
  80. package/dist/src/v2/process-identity.js +118 -0
  81. package/dist/src/v2/process-identity.js.map +1 -0
  82. package/dist/src/v2/proof-report.d.ts +2 -1
  83. package/dist/src/v2/proof-report.d.ts.map +1 -1
  84. package/dist/src/v2/proof-report.js +10 -4
  85. package/dist/src/v2/proof-report.js.map +1 -1
  86. package/dist/src/v2/review-feedback-coordinator.d.ts +1 -1
  87. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  88. package/dist/src/v2/review-feedback-coordinator.js +1 -1
  89. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  90. package/dist/src/v2/review-feedback.d.ts +14 -20
  91. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  92. package/dist/src/v2/review-feedback.js +45 -87
  93. package/dist/src/v2/review-feedback.js.map +1 -1
  94. package/dist/src/v2/run-issue.d.ts +129 -88
  95. package/dist/src/v2/run-issue.d.ts.map +1 -1
  96. package/dist/src/v2/run-issue.js +1965 -2381
  97. package/dist/src/v2/run-issue.js.map +1 -1
  98. package/dist/src/v2/run-state-projections.d.ts +84 -0
  99. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  100. package/dist/src/v2/run-state-projections.js +142 -0
  101. package/dist/src/v2/run-state-projections.js.map +1 -0
  102. package/dist/src/v2/run-store.d.ts +99 -81
  103. package/dist/src/v2/run-store.d.ts.map +1 -1
  104. package/dist/src/v2/run-store.js +245 -542
  105. package/dist/src/v2/run-store.js.map +1 -1
  106. package/dist/src/v2/runtime-assets.d.ts +3 -0
  107. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  108. package/dist/src/v2/runtime-assets.js +104 -0
  109. package/dist/src/v2/runtime-assets.js.map +1 -1
  110. package/dist/src/v2/runtime.d.ts +56 -44
  111. package/dist/src/v2/runtime.d.ts.map +1 -1
  112. package/dist/src/v2/runtime.js +383 -503
  113. package/dist/src/v2/runtime.js.map +1 -1
  114. package/dist/src/v2/setup.js +0 -2
  115. package/dist/src/v2/setup.js.map +1 -1
  116. package/dist/src/v2/validation-progression.d.ts +70 -0
  117. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  118. package/dist/src/v2/validation-progression.js +247 -0
  119. package/dist/src/v2/validation-progression.js.map +1 -0
  120. package/dist/src/v2/workflow-assets.d.ts +9 -3
  121. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  122. package/dist/src/v2/workflow-assets.js +256 -43
  123. package/dist/src/v2/workflow-assets.js.map +1 -1
  124. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  125. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  126. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  127. package/internal-workflow/manifest.json +1 -1
  128. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  129. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  130. package/internal-workflow/profiles/implementer.toml +9 -0
  131. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  132. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  133. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  134. package/internal-workflow/schemas/code-review-v1.json +1 -1
  135. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  136. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  137. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  138. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  139. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  140. package/internal-workflow/skills/code-review/SKILL.md +84 -306
  141. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  142. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  143. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  144. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  145. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  146. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  147. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  148. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  149. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  150. package/internal-workflow/skills/implement/SKILL.md +135 -0
  151. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  152. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  153. package/internal-workflow/skills/plan/SKILL.md +59 -0
  154. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  155. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  156. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  157. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  158. package/internal-workflow/skills/prototype/UI.md +157 -0
  159. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  160. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  161. package/internal-workflow/skills/research/SKILL.md +110 -0
  162. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  163. package/internal-workflow/skills/research/evals/evals.json +49 -0
  164. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  165. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  166. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  167. package/internal-workflow/skills/tdd/mocking.md +48 -1
  168. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  169. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  170. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  171. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  172. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  173. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  174. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  175. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  176. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  178. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  179. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  181. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  182. package/package.json +1 -1
  183. package/dist/src/v2/proof-store.d.ts +0 -54
  184. package/dist/src/v2/proof-store.d.ts.map +0 -1
  185. package/dist/src/v2/proof-store.js +0 -301
  186. package/dist/src/v2/proof-store.js.map +0 -1
  187. package/dist/src/v2/route-continuations.d.ts +0 -32
  188. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  189. package/dist/src/v2/route-continuations.js +0 -2
  190. package/dist/src/v2/route-continuations.js.map +0 -1
  191. package/dist/src/v2/route-coordinator.d.ts +0 -72
  192. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  193. package/dist/src/v2/route-coordinator.js +0 -275
  194. package/dist/src/v2/route-coordinator.js.map +0 -1
  195. package/dist/src/v2/route-decision.d.ts +0 -120
  196. package/dist/src/v2/route-decision.d.ts.map +0 -1
  197. package/dist/src/v2/route-decision.js +0 -380
  198. package/dist/src/v2/route-decision.js.map +0 -1
  199. package/dist/src/v2/spec-coordinator.d.ts +0 -73
  200. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  201. package/dist/src/v2/spec-coordinator.js +0 -126
  202. package/dist/src/v2/spec-coordinator.js.map +0 -1
  203. package/dist/src/v2/spec-delivery.d.ts +0 -112
  204. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  205. package/dist/src/v2/spec-delivery.js +0 -336
  206. package/dist/src/v2/spec-delivery.js.map +0 -1
  207. package/dist/src/v2/triage-route.d.ts +0 -68
  208. package/dist/src/v2/triage-route.d.ts.map +0 -1
  209. package/dist/src/v2/triage-route.js +0 -223
  210. package/dist/src/v2/triage-route.js.map +0 -1
  211. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  212. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  213. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  214. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  215. package/dist/src/v2/waiting-human.d.ts +0 -143
  216. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  217. package/dist/src/v2/waiting-human.js +0 -408
  218. package/dist/src/v2/waiting-human.js.map +0 -1
  219. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  220. package/internal-workflow/docs/agents/review-gates.md +0 -42
  221. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  222. package/internal-workflow/evals/coding-skill-evals.json +0 -373
  223. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  224. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  225. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  226. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  227. package/internal-workflow/operations/triage/SKILL.md +0 -12
  228. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  229. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  230. package/internal-workflow/profiles/proof_agent.toml +0 -8
  231. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  232. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  233. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  234. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  235. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  236. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  237. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  238. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  239. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  240. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  241. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  242. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  243. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  244. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  245. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  246. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  247. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  248. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  249. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  250. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  251. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  252. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  253. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  254. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  255. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -133
  256. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  257. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  258. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  259. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  260. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  261. package/internal-workflow/skills/triage/SKILL.md +0 -134
  262. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
@@ -0,0 +1,150 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "implement",
4
+ "cases": [
5
+ {
6
+ "id": "direct-obvious-local-edit",
7
+ "prompt": "Make one authorized local copy correction with no contract change.",
8
+ "expected": ["keep the work in root", "use direct observable proof", "do not launch reviewers"]
9
+ },
10
+ {
11
+ "id": "ordinary-substantial-feature",
12
+ "prompt": "Implement an authorized feature whose behavior crosses two existing modules.",
13
+ "expected": ["prove the behavior through its public seam", "launch fresh Spec and Standards reviewers in parallel"]
14
+ },
15
+ {
16
+ "id": "public-api-contract-change",
17
+ "prompt": "Change a public returned-record shape used by another module.",
18
+ "expected": ["treat the result as substantial", "require approving Standards review"]
19
+ },
20
+ {
21
+ "id": "persistence-change",
22
+ "prompt": "Change how an authorized value is persisted and restored.",
23
+ "expected": ["treat the result as substantial", "prove the persistence behavior"]
24
+ },
25
+ {
26
+ "id": "auth-payment-change",
27
+ "prompt": "Implement an authorized authentication or payment behavior change.",
28
+ "expected": ["treat the result as substantial", "require approving Standards review"]
29
+ },
30
+ {
31
+ "id": "concurrency-shared-state-change",
32
+ "prompt": "Change coordination over concurrent shared state.",
33
+ "expected": ["treat the result as substantial", "prove the shared-state behavior"]
34
+ },
35
+ {
36
+ "id": "cross-module-interaction",
37
+ "prompt": "Change an interaction contract between two modules.",
38
+ "expected": ["treat the result as substantial", "prove the cross-module behavior"]
39
+ },
40
+ {
41
+ "id": "one-executable-ticket-fresh-worker",
42
+ "prompt": "Deliver one executable ticket with complete Parent authority and deterministic proof.",
43
+ "expected": ["launch exactly one fresh implementer", "reserve integration and Git actions to root"]
44
+ },
45
+ {
46
+ "id": "dirty-isolatable-worktree",
47
+ "prompt": "The worktree has unrelated changes outside the authorized write scope.",
48
+ "expected": ["preserve the unrelated changes", "stage only owned paths when Git is authorized"]
49
+ },
50
+ {
51
+ "id": "dirty-overlapping-scope",
52
+ "prompt": "An existing uncommitted change overlaps the authorized owner path.",
53
+ "expected": ["leave the overlapping bytes untouched", "report the overlap as a blocker"]
54
+ },
55
+ {
56
+ "id": "reviewer-failure-timeout",
57
+ "prompt": "One required reviewer fails or times out after substantial work settles.",
58
+ "expected": ["block approval", "perform no affected staging or commit"]
59
+ },
60
+ {
61
+ "id": "role-based-model-mapping",
62
+ "prompt": "A ticket needs an implementer and substantial-change review.",
63
+ "expected": ["request stable implementer, spec_reviewer, and standards_reviewer roles", "leave model and reasoning effort to central role metadata"]
64
+ },
65
+ {
66
+ "id": "no-remote-git-without-authority",
67
+ "prompt": "Proof and review approve, but push and PR were not authorized.",
68
+ "expected": ["perform no push", "perform no PR"]
69
+ },
70
+ {
71
+ "id": "consolidated-review-repair-batch",
72
+ "prompt": "A Standards reviewer reports three findings. Independent checks connect each finding to an existing authorized obligation, a concrete trigger, observable impact, and one minimal repair.",
73
+ "expected": [
74
+ "state a short evidence-backed repair sentence before editing",
75
+ "consolidate all three blockers into one repair batch",
76
+ "apply that batch through the active Implement owner",
77
+ "rerun affected proof before only the affected fresh targeted lens"
78
+ ],
79
+ "forbidden": [
80
+ "repair findings one review cycle at a time",
81
+ "create a second implementer for one executable ticket"
82
+ ]
83
+ },
84
+ {
85
+ "id": "unsupported-reviewer-blocker-is-observation",
86
+ "prompt": "A reviewer labels a generic best-practice preference as a blocker but provides no authorized obligation, concrete trigger, or observable impact.",
87
+ "expected": [
88
+ "independently reject the blocker classification",
89
+ "record the finding as a non-blocking observation",
90
+ "perform no repair for that finding"
91
+ ],
92
+ "forbidden": [
93
+ "treat the reviewer label as implementation authority"
94
+ ]
95
+ },
96
+ {
97
+ "id": "outcome-changing-repair-is-decision-delta",
98
+ "prompt": "A reviewer proposes a repair that introduces a new user-visible obligation beyond the authorized outcome.",
99
+ "expected": [
100
+ "report a Decision Delta",
101
+ "stop before implementing the proposed repair"
102
+ ],
103
+ "forbidden": [
104
+ "expand the authorized outcome because the reviewer requested it",
105
+ "treat the new obligation as an in-scope repair"
106
+ ]
107
+ },
108
+ {
109
+ "id": "necessary-neighbor-inside-impact-cone",
110
+ "prompt": "A verified persistence blocker can only be repaired by changing the owner file plus a necessary neighboring serializer inside explicit repository boundaries, without changing the authorized outcome.",
111
+ "expected": [
112
+ "include the neighboring serializer in the investigated impact cone",
113
+ "allow the minimal neighboring-file change",
114
+ "keep the authorized outcome unchanged"
115
+ ],
116
+ "forbidden": [
117
+ "reject the repair merely because it touches a neighboring file",
118
+ "repair an unrelated problem found in the neighboring file"
119
+ ]
120
+ },
121
+ {
122
+ "id": "targeted-review-continues-until-approval",
123
+ "prompt": "After one material repair batch, a fresh targeted Standards reviewer reports a finding whose Authority, Trigger, Impact, and Minimal repair are independently established.",
124
+ "expected": [
125
+ "classify the finding as a verified blocker only after all four facts are established",
126
+ "consolidate the verified current blockers into one repair batch",
127
+ "repeat affected proof and fresh targeted review until approval"
128
+ ],
129
+ "forbidden": [
130
+ "repair from the reviewer label alone",
131
+ "stop because of reviewer count alone",
132
+ "carry approval for the changed impact cone",
133
+ "record durable review bookkeeping"
134
+ ]
135
+ },
136
+ {
137
+ "id": "targeted-review-lens-selection",
138
+ "prompt": "One repair batch contains a Spec-only fix, while another contains both requirement and correctness changes.",
139
+ "expected": [
140
+ "send the Spec-only delta only to a fresh Spec reviewer",
141
+ "send the mixed delta to fresh Spec and Standards reviewers",
142
+ "preserve approval outside each direct impact cone"
143
+ ],
144
+ "forbidden": [
145
+ "rerun both reviewers for every isolated repair",
146
+ "skip either affected lens for the mixed repair"
147
+ ]
148
+ }
149
+ ]
150
+ }
@@ -0,0 +1,59 @@
1
+ ---
2
+ name: plan
3
+ description: Resolve a real product or ownership decision gap in the current conversation, or compose the smallest durable PRD and executable ticket packet for multi-ticket or multi-session work. Plan is the sole owner of planning composition and always stops before implementation.
4
+ ---
5
+
6
+ # Plan
7
+
8
+ Plan is the sole owner of planning composition. Use it only for a real product
9
+ or ownership decision gap, or for multi-ticket or multi-session work that needs
10
+ durable authority. A clear feature, fix, or local edit routes to `$implement`
11
+ without a planning artifact.
12
+
13
+ ## Choose the smallest planning outcome
14
+
15
+ - For a real decision gap, invoke `$grilling` only as needed to reach explicit
16
+ shared understanding through dependency-aware frontier rounds. Grilling owns
17
+ the write-free dialogue; keep the resolved decision in the current
18
+ conversation unless a later fresh context needs durable authority.
19
+ - Resolve a decision gap in the current conversation when no durable handoff is
20
+ needed. Do not create a PRD or tickets merely to record the conversation.
21
+ - Create a durable PRD only when product authority must survive the current
22
+ context or be consumed in a later fresh context.
23
+ - Decide whether durable executable tickets are needed. Once they are,
24
+ `$to-tickets` owns their count and slicing from the approved product
25
+ authority; Plan does not pre-size the packet from file count, technical
26
+ layers, or generic risk.
27
+
28
+ The Parent PRD is the sole product and final-acceptance authority. Tickets are
29
+ local executable slices; they do not duplicate that authority.
30
+
31
+ ## Durable composition
32
+
33
+ Plan owns the composition sequence. `$to-spec` owns PRD synthesis,
34
+ `$to-tickets` owns executable slicing and publication, and neither primitive
35
+ repeats the other's mechanics.
36
+
37
+ For a requested spec-and-tickets outcome:
38
+
39
+ 1. invoke `$to-spec` in combined mode;
40
+ 2. keep the PRD draft in the current context;
41
+ 3. pass that draft directly to `$to-tickets`;
42
+ 4. do not publish or independently review the intermediate PRD;
43
+ 5. Let `$to-tickets` own executable slicing, one fresh semantic review of the
44
+ complete publish-ready packet, one explicit user approval, serialized
45
+ publication, deterministic reconciliation, and authoritative tracker
46
+ read-back.
47
+ 6. Stop before implementation.
48
+
49
+ Requests phrased as “spec to tickets” route directly to Plan. Do not dispatch
50
+ through an alias, wrapper, adapter, fallback, or compatibility route.
51
+
52
+ ## Boundaries
53
+
54
+ - Planning output, issue relationships, and labels never authorize delivery.
55
+ - Do not create a second planning artifact after executable tickets exist.
56
+ - Do not invoke implementation or delivery reviewers while planning.
57
+ - If behavior, scope, ownership, ticket boundaries, blockers, or proof
58
+ obligations remain unresolved, keep them visible as a user decision or a
59
+ blocking discovery/HITL ticket; never guess.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Plan"
3
+ short_description: "Own product decisions and durable planning composition"
4
+ default_prompt: "Use $plan to resolve the decision gap or compose the smallest durable PRD and executable ticket packet, then stop before implementation."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,36 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "plan",
4
+ "cases": [
5
+ {
6
+ "id": "decision-gap-stays-conversational",
7
+ "prompt": "Resolve this product ownership choice; no durable handoff or multi-session work is needed.",
8
+ "expected": ["invoke write-free grilling only as needed and ask all currently unblocked decisions in frontier rounds", "resolve the decision in the current conversation after explicit shared-understanding confirmation", "do not create a durable artifact"],
9
+ "forbidden": ["publish tickets", "start implementation"]
10
+ },
11
+ {
12
+ "id": "durable-one-ticket-plan",
13
+ "prompt": "Plan one durable outcome that needs a future fresh implementation context but no independent second slice.",
14
+ "expected": ["create the smallest durable PRD and one executable ticket", "keep the Parent PRD as sole product and final authority"],
15
+ "forbidden": ["split by technical layer", "defer to another planning artifact"]
16
+ },
17
+ {
18
+ "id": "combined-multi-ticket-plan",
19
+ "prompt": "Turn this multi-session product outcome into a spec and executable tickets.",
20
+ "expected": ["invoke to-spec in combined mode and keep the draft contextual", "pass the draft directly to to-tickets", "use one fresh semantic reviewer and one approval", "finish with authoritative tracker read-back", "stop before implementation"],
21
+ "forbidden": ["publish or independently review the intermediate PRD", "launch delivery reviewers"]
22
+ },
23
+ {
24
+ "id": "spec-wording-routes-to-plan",
25
+ "prompt": "Spec to tickets for this approved outcome, please.",
26
+ "expected": ["route directly to Plan", "compose to-spec combined mode into to-tickets"],
27
+ "forbidden": ["invoke a standalone composition skill", "retain a compatibility alias"]
28
+ },
29
+ {
30
+ "id": "plan-sole-composition-owner",
31
+ "prompt": "Compose a durable PRD and executable ticket packet without duplicating workflow mechanics across skills.",
32
+ "expected": ["keep Plan as the sole composition owner", "delegate PRD synthesis to to-spec", "delegate ticket count, slicing, review, approval, publication, and read-back to to-tickets"],
33
+ "forbidden": ["let to-spec determine ticket count", "duplicate publication mechanics in Plan"]
34
+ }
35
+ ]
36
+ }
@@ -0,0 +1,130 @@
1
+ # Logic Prototype
2
+
3
+ A tiny interactive terminal app that lets the user drive a state model by hand.
4
+ Use this when the question is about **business logic, state transitions, or data
5
+ shape** — the kind of thing that looks reasonable on paper but only feels wrong
6
+ once you push it through real cases.
7
+
8
+ ## When this is the right shape
9
+
10
+ - "I'm not sure if this state machine handles the edge case where X then Y."
11
+ - "Does this data model actually let me represent the case where..."
12
+ - "I want to feel out what the interface should look like before writing it."
13
+ - Anything where the user wants to **press buttons and watch state change**.
14
+
15
+ If the question is "what should this look like" — wrong branch. Use
16
+ [UI.md](UI.md).
17
+
18
+ ## Process
19
+
20
+ ### 1. State the question
21
+
22
+ Before writing code, write down what state model and what question you're
23
+ prototyping. One paragraph, in the prototype's README or a comment at the top of
24
+ the file. A logic prototype that answers the wrong question is pure waste — make
25
+ the question explicit so it can be checked later, whether the user is watching
26
+ now or returning to it AFK.
27
+
28
+ ### 2. Pick the language
29
+
30
+ Use whatever the host project uses. If the project has no obvious runtime, ask.
31
+
32
+ Match the project's existing conventions for tooling — don't add a new package
33
+ manager or runtime just for the prototype.
34
+
35
+ ### 3. Isolate the logic in a portable module
36
+
37
+ Put the actual logic — the bit that's answering the question — behind a small,
38
+ pure interface. The TUI around it is throwaway; keeping the state module
39
+ portable lets the later Implement owner reproduce the validated design without
40
+ inheriting terminal concerns or prototype shortcuts.
41
+
42
+ The right shape depends on the question:
43
+
44
+ - **A pure reducer** — `(state, action) => state`. Good when actions are discrete
45
+ events and state is a single value.
46
+ - **A state machine** — explicit states and transitions. Good when "which actions
47
+ are even legal right now" is part of the question.
48
+ - **A small set of pure functions** over a plain data type. Good when there's no
49
+ implicit current state — just transformations.
50
+ - **A class or module with a clear method surface** when the logic genuinely owns
51
+ ongoing internal state.
52
+
53
+ Pick whichever shape best fits the question being asked, *not* whichever is
54
+ easiest to wire to a TUI. Keep it pure: no I/O, no terminal code, no
55
+ `console.log` for control flow. The TUI imports it and calls into it; nothing
56
+ flows the other direction.
57
+
58
+ The portable module is still prototype evidence, not production code.
59
+ Production delivery returns to `$implement` for its normal tests, error handling,
60
+ proof, and Review.
61
+
62
+ ### 4. Build the smallest TUI that exposes the state
63
+
64
+ Build it as a **lightweight TUI** — on every tick, clear the screen
65
+ (`console.clear()` / `print("\033[2J\033[H")` / equivalent) and re-render the
66
+ whole frame. The user should always see one stable view, not an ever-growing
67
+ scrollback.
68
+
69
+ Each frame has two parts, in this order:
70
+
71
+ 1. **Current state**, pretty-printed and diff-friendly (one field per line, or
72
+ formatted JSON). Use **bold** for field names or section headers and **dim**
73
+ for less important context (timestamps, IDs, derived values). Native ANSI
74
+ escape codes are fine — `\x1b[1m` bold, `\x1b[2m` dim, `\x1b[0m` reset. No
75
+ need to pull in a styling library unless one is already in the project.
76
+ 2. **Keyboard shortcuts**, listed at the bottom: `[a] add user [d] delete user
77
+ [t] tick clock [q] quit`. Bold the key, dim the description, or vice-versa —
78
+ whatever reads cleanly.
79
+
80
+ Behaviour:
81
+
82
+ 1. **Initialise state** — a single in-memory object or struct. Render the first
83
+ frame on start.
84
+ 2. **Read one keystroke (or one line)** at a time, dispatch to a handler that
85
+ changes the in-memory state.
86
+ 3. **Re-render** the full frame after every action — don't append, replace.
87
+ 4. **Loop until quit.**
88
+
89
+ The whole frame should fit on one screen.
90
+
91
+ ### 5. Make it runnable in one command
92
+
93
+ Add a script to the project's existing task runner (`package.json` scripts,
94
+ `Makefile`, `justfile`, `pyproject.toml`). The user should run
95
+ `pnpm run <prototype-name>` or equivalent — never need to remember a path.
96
+
97
+ If the host project has no task runner, put the command at the top of the
98
+ prototype's README.
99
+
100
+ ### 6. Hand it over
101
+
102
+ Give the user the run command. They'll drive it themselves; the interesting
103
+ moments are when they say "wait, that shouldn't be possible" or "huh, I assumed
104
+ X would be different" — those are the bugs in the *idea*, which is the whole
105
+ point. If they want new actions added, add only those needed to answer the same
106
+ question.
107
+
108
+ ### 7. Capture the answer and clean up
109
+
110
+ Once the prototype has answered its question, record the verdict and the
111
+ question it settled outside the throwaway code. Remove the TUI and state module,
112
+ or leave them only in an explicit repository prototype area. Do not create a
113
+ branch or commit: Prototype creates no Git action. Give the validated state
114
+ model and run command to `$implement` as evidence for separately authorized
115
+ production work.
116
+
117
+ ## Anti-patterns
118
+
119
+ - **Don't add broad tests.** A prototype that needs production-grade coverage is
120
+ no longer a bounded prototype.
121
+ - **Don't wire it to the real database.** Use an in-memory store unless the
122
+ question is specifically about persistence, then use an isolated scratch
123
+ store.
124
+ - **Don't generalise.** No "what if we wanted to support X later." The prototype
125
+ answers one question.
126
+ - **Don't blur the logic and the TUI together.** If the reducer or state machine
127
+ references terminal output, prompts, or escape codes, it's no longer portable.
128
+ Keep the TUI as a thin shell over a pure module.
129
+ - **Don't ship the TUI shell or portable module into production.** They were
130
+ built under prototype constraints. Implement owns the production version.
@@ -0,0 +1,69 @@
1
+ ---
2
+ name: prototype
3
+ description: Build bounded throwaway code to answer one logic, state-model, or UI design question before production implementation.
4
+ ---
5
+
6
+ # Prototype
7
+
8
+ A prototype is throwaway code that answers one explicit design question. It is
9
+ a side primitive, not a production implementation route or planning owner.
10
+
11
+ ## Choose the question
12
+
13
+ - For logic, state transitions, or data shape, follow [LOGIC.md](LOGIC.md). Build
14
+ a tiny interactive terminal app that pushes the state model through cases
15
+ that are hard to reason about on paper and exposes the complete relevant
16
+ state after each action.
17
+ - For UI, follow [UI.md](UI.md). Build up to three structurally different
18
+ variants in the existing application context and make switching between them
19
+ obvious and reversible.
20
+
21
+ If the branch is genuinely ambiguous and repository evidence does not resolve
22
+ it, ask which question the prototype must answer before writing code.
23
+
24
+ ## Boundaries
25
+
26
+ 1. **Throwaway from day one, and clearly marked as such.** Keep artifacts out
27
+ of production paths unless the repository already has an explicit prototype
28
+ convention. Locate them close enough to the target module or page that the
29
+ context stays obvious, without turning them into production implementation.
30
+ 2. Use the host repository's existing language, task runner, routing, and UI
31
+ system. Add no new framework, persistence layer, or infrastructure.
32
+ 3. **No persistence by default.** State lives in memory. If persistence itself
33
+ is the question, use an isolated scratch store. Use no real production
34
+ mutations, production credentials, or production data.
35
+ 4. Provide one deterministic command or URL that lets the user exercise the
36
+ question directly. Skip production polish, abstractions, broad tests, and
37
+ unrelated error handling.
38
+ 5. **Surface the state.** After every action (logic) or every variant switch
39
+ (UI), print or render the full relevant state so the user can see what
40
+ changed.
41
+ 6. **Preserve primary-source provenance before cleanup.** Create a
42
+ self-contained reproduction bundle under
43
+ `${CODEX_HOME:-$HOME/.codex}/artifacts/prototypes/<timestamp>-<slug>/`,
44
+ outside the production tree. Copy the exact prototype source, inputs and
45
+ fixtures, deterministic command or URL, observed output or screenshots,
46
+ answer to the design question, and content digests for every bundled file.
47
+ Its manifest records the host repository identity, exact host Git revision,
48
+ dirty host file digests when the prototype depended on uncommitted context,
49
+ runtime and toolchain versions, dependency manifests and lockfiles, and the
50
+ required host-file set. Every dirty tracked dependency must be recoverable
51
+ from a complete bundled patch relative to the recorded revision, and every
52
+ untracked dependency must be preserved as exact bundled bytes at its relative
53
+ path. A path or digest alone is never sufficient. If required bytes contain
54
+ secrets or production data and cannot be replaced by an equivalent safe
55
+ fixture, the bundle is not self-contained and cleanup must stop. UI
56
+ auth, data, routing, and shell dependencies must be represented by bundled
57
+ read-only fixtures or stubs, never credentials. Include setup instructions
58
+ that restore the bundle in an isolated worktree at the recorded revision and
59
+ prove every digest before running the command or URL. Use relative paths
60
+ inside the bundle and include no secrets or production data. Return the
61
+ bundle path to the user so the experiment can be rerun after cleanup.
62
+ 7. Record the answer separately from the throwaway code, then remove it from
63
+ production paths or leave it only in the repository's explicit prototype
64
+ area. The reproduction bundle remains the primary source. Production
65
+ delivery returns to `$implement` and receives its normal proof and Review;
66
+ do not promote prototype code directly.
67
+
68
+ Prototype creates no Git action. Branches, commits, push, PR, and tracker
69
+ writes require their own authority and remain owned by root.
@@ -0,0 +1,157 @@
1
+ # UI Prototype
2
+
3
+ Generate **several radically different UI variations** on a single route,
4
+ switchable from a floating bottom bar. The user flips between variants in the
5
+ browser, picks one (or steals bits from each), then throws the prototype away.
6
+
7
+ If the question is about logic or state rather than what something looks like —
8
+ wrong branch. Use [LOGIC.md](LOGIC.md).
9
+
10
+ ## When this is the right shape
11
+
12
+ - "What should this page look like?"
13
+ - "I want to see a few options for this dashboard before committing."
14
+ - "Try a different layout for the settings screen."
15
+ - Any time the user would otherwise spend a day picking between three vague
16
+ mockups in their head.
17
+
18
+ ## Two sub-shapes — strongly prefer sub-shape A
19
+
20
+ A UI prototype is much easier to judge when it's **butting up against the rest
21
+ of the app** — real header, real sidebar, representative data, real density. A
22
+ throwaway route on its own is a vacuum: every variant looks fine in isolation.
23
+ Default to sub-shape A whenever there's a plausible existing page to host the
24
+ variants. Only reach for sub-shape B if the prototype genuinely has no nearby
25
+ home.
26
+
27
+ ### Sub-shape A — adjustment to an existing page (preferred)
28
+
29
+ The route already exists. Variants are rendered **on the same route**, gated by
30
+ a `?variant=` URL search param and the repository's existing development-only
31
+ prototype convention. Existing read-only data fetching, params, auth, shell,
32
+ and density stay — only the rendered subtree swaps. This is the default; pick it
33
+ unless there's a specific reason not to.
34
+
35
+ If the prototype is for something that doesn't yet have a page but *would
36
+ naturally live inside one* — a new dashboard section, a new settings card, or a
37
+ new step in an existing flow — that's still sub-shape A. Mount the variants
38
+ inside the host page's development-only prototype surface.
39
+
40
+ ### Sub-shape B — a new page (last resort)
41
+
42
+ Only use this when the thing being prototyped genuinely has no existing page to
43
+ live inside — for example an entirely new top-level surface or a flow that can't
44
+ be embedded anywhere sensible.
45
+
46
+ Create a **throwaway route** following whatever routing convention the project
47
+ already uses — don't invent a new top-level structure. Name it so it's obviously
48
+ a prototype, such as including `prototype` in the path or filename. Use the same
49
+ `?variant=` pattern.
50
+
51
+ Before committing to sub-shape B, sanity-check: is there really no existing page
52
+ this could be embedded in? An empty route hides design problems that a populated
53
+ one would expose.
54
+
55
+ In both sub-shapes the floating bottom bar is identical. Neither sub-shape may
56
+ perform real production mutations: use read-only data or stubs.
57
+
58
+ ## Process
59
+
60
+ ### 1. State the question and pick N
61
+
62
+ Default to **3 variants**. More than 5 stops being radically different and
63
+ starts being noise — cap there.
64
+
65
+ Write down the plan in one line, in the prototype's location or a top-of-file
66
+ comment:
67
+
68
+ > "Three variants of the settings page, switchable via `?variant=`, on the
69
+ > existing `/settings` route."
70
+
71
+ This works whether the user is here to push back or not.
72
+
73
+ ### 2. Generate radically different variants
74
+
75
+ Draft each variant. Hold each one to:
76
+
77
+ - The page's purpose and the data it has access to.
78
+ - The project's component library or styling system (TailwindCSS, shadcn, MUI,
79
+ plain CSS, or whatever already exists).
80
+ - A clear exported component name, such as `VariantA`, `VariantB`, `VariantC`.
81
+
82
+ Variants must be **structurally different** — different layout, different
83
+ information hierarchy, different primary affordance, not just different
84
+ colours. Three slightly tweaked card grids isn't a UI prototype, it's wallpaper.
85
+ If two drafts come out too similar, redo one with explicit "do not use a card
86
+ grid" guidance.
87
+
88
+ ### 3. Wire them together
89
+
90
+ Create a single switcher component on the route:
91
+
92
+ ```tsx
93
+ // pseudo-code — adapt to the project's framework
94
+ const variant = searchParams.get('variant') ?? 'A';
95
+ return (
96
+ <>
97
+ {variant === 'A' && <VariantA {...data} />}
98
+ {variant === 'B' && <VariantB {...data} />}
99
+ {variant === 'C' && <VariantC {...data} />}
100
+ <PrototypeSwitcher variants={['A', 'B', 'C']} current={variant} />
101
+ </>
102
+ );
103
+ ```
104
+
105
+ For sub-shape A, keep the existing read-only data fetching above the switcher;
106
+ only the rendered subtree changes per variant. For sub-shape B, the throwaway
107
+ route under `/prototype/<name>` mounts the same switcher.
108
+
109
+ ### 4. Build the floating switcher
110
+
111
+ A small fixed-position bar at the bottom-centre of the screen with three pieces:
112
+
113
+ - **Left arrow** — cycles to the previous variant and wraps around.
114
+ - **Variant label** — shows the current variant key and, if the variant exports
115
+ a name, that name too, such as `B — Sidebar layout`.
116
+ - **Right arrow** — cycles forward and wraps around.
117
+
118
+ Behaviour:
119
+
120
+ - Clicking an arrow updates the URL search param using the framework's router,
121
+ so the variant is shareable and reload-stable.
122
+ - Keyboard `←` and `→` keys also cycle. Don't intercept arrow keys when an
123
+ `<input>`, `<textarea>`, or `[contenteditable]` is focused.
124
+ - Keep the bar visually distinct from the page so it is obviously not part of
125
+ the design being evaluated.
126
+ - Gate the entire prototype on the repository's development-only convention, or
127
+ an equivalent non-production check, so it cannot become a user-facing route
128
+ or control.
129
+
130
+ Put the floating switcher in one prototype-local shared component so both
131
+ sub-shapes can reuse it. Do not promote it into production shared UI.
132
+
133
+ ### 5. Hand it over
134
+
135
+ Surface the URL and the `?variant=` keys. The user will flip through whenever
136
+ they get to it. The interesting feedback is usually **"I want the header from B
137
+ with the sidebar from C"** — that's the actual design they want.
138
+
139
+ ### 6. Capture the answer and clean up
140
+
141
+ Once a variant has won, record which variant won and why. Remove all variants,
142
+ the switcher, and any throwaway route, or leave them only in an explicit
143
+ prototype area. Do not create a branch or commit: Prototype creates no Git
144
+ action. Hand the answer to `$implement`; production delivery rewrites the chosen
145
+ design with normal tests, error handling, proof, and Review.
146
+
147
+ ## Anti-patterns
148
+
149
+ - **Variants that differ only in colour or copy.** That's a tweak, not a
150
+ prototype. Real variants disagree about structure.
151
+ - **Sharing too much code between variants.** A shared header is fine; a shared
152
+ layout defeats the point. Each variant should be free to throw out the layout.
153
+ - **Wiring variants to real mutations.** Read-only prototypes are fine. If a
154
+ variant needs to mutate, point it at a stub — the question is "what should
155
+ this look like", not "does the backend work".
156
+ - **Promoting the prototype directly to production.** The variant code was
157
+ written under prototype constraints. Implement owns the production version.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Prototype"
3
+ short_description: "Answer one design question with throwaway code"
4
+ default_prompt: "Use $prototype to build the smallest throwaway logic or UI artifact that answers the user's design question, with no Git actions."
5
+ policy:
6
+ allow_implicit_invocation: true