codex-orchestrator 2.0.11 → 2.0.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/CHANGELOG.md +15 -0
  2. package/README.md +25 -51
  3. package/dist/src/index.d.ts +2 -8
  4. package/dist/src/index.d.ts.map +1 -1
  5. package/dist/src/index.js +1 -4
  6. package/dist/src/index.js.map +1 -1
  7. package/dist/src/v2/acceptance-proof.d.ts +46 -31
  8. package/dist/src/v2/acceptance-proof.d.ts.map +1 -1
  9. package/dist/src/v2/acceptance-proof.js +157 -195
  10. package/dist/src/v2/acceptance-proof.js.map +1 -1
  11. package/dist/src/v2/active-attempt.d.ts +94 -0
  12. package/dist/src/v2/active-attempt.d.ts.map +1 -0
  13. package/dist/src/v2/active-attempt.js +200 -0
  14. package/dist/src/v2/active-attempt.js.map +1 -0
  15. package/dist/src/v2/adapters/command.d.ts +6 -0
  16. package/dist/src/v2/adapters/command.d.ts.map +1 -1
  17. package/dist/src/v2/adapters/command.js +43 -2
  18. package/dist/src/v2/adapters/command.js.map +1 -1
  19. package/dist/src/v2/candidate.d.ts +15 -31
  20. package/dist/src/v2/candidate.d.ts.map +1 -1
  21. package/dist/src/v2/candidate.js +7 -29
  22. package/dist/src/v2/candidate.js.map +1 -1
  23. package/dist/src/v2/checked-change.d.ts +3 -2
  24. package/dist/src/v2/checked-change.d.ts.map +1 -1
  25. package/dist/src/v2/checked-change.js +4 -3
  26. package/dist/src/v2/checked-change.js.map +1 -1
  27. package/dist/src/v2/cli-contract.d.ts +1 -1
  28. package/dist/src/v2/cli-contract.d.ts.map +1 -1
  29. package/dist/src/v2/cli-contract.js +4 -6
  30. package/dist/src/v2/cli-contract.js.map +1 -1
  31. package/dist/src/v2/cli.d.ts +8 -0
  32. package/dist/src/v2/cli.d.ts.map +1 -1
  33. package/dist/src/v2/cli.js +13 -0
  34. package/dist/src/v2/cli.js.map +1 -1
  35. package/dist/src/v2/code-review-report.d.ts +10 -18
  36. package/dist/src/v2/code-review-report.d.ts.map +1 -1
  37. package/dist/src/v2/code-review-report.js +63 -60
  38. package/dist/src/v2/code-review-report.js.map +1 -1
  39. package/dist/src/v2/codex-process.d.ts +6 -2
  40. package/dist/src/v2/codex-process.d.ts.map +1 -1
  41. package/dist/src/v2/codex-process.js +25 -9
  42. package/dist/src/v2/codex-process.js.map +1 -1
  43. package/dist/src/v2/config.d.ts +0 -2
  44. package/dist/src/v2/config.d.ts.map +1 -1
  45. package/dist/src/v2/config.js +3 -6
  46. package/dist/src/v2/config.js.map +1 -1
  47. package/dist/src/v2/contained-report-operation.d.ts +41 -196
  48. package/dist/src/v2/contained-report-operation.d.ts.map +1 -1
  49. package/dist/src/v2/contained-report-operation.js +139 -466
  50. package/dist/src/v2/contained-report-operation.js.map +1 -1
  51. package/dist/src/v2/containment.d.ts +1 -0
  52. package/dist/src/v2/containment.d.ts.map +1 -1
  53. package/dist/src/v2/containment.js +12 -2
  54. package/dist/src/v2/containment.js.map +1 -1
  55. package/dist/src/v2/delivery-authority.d.ts +26 -0
  56. package/dist/src/v2/delivery-authority.d.ts.map +1 -0
  57. package/dist/src/v2/delivery-authority.js +44 -0
  58. package/dist/src/v2/delivery-authority.js.map +1 -0
  59. package/dist/src/v2/direct-delivery.d.ts +16 -36
  60. package/dist/src/v2/direct-delivery.d.ts.map +1 -1
  61. package/dist/src/v2/direct-delivery.js +135 -122
  62. package/dist/src/v2/direct-delivery.js.map +1 -1
  63. package/dist/src/v2/immutable-workflow-publisher.d.ts.map +1 -1
  64. package/dist/src/v2/immutable-workflow-publisher.js +3 -1
  65. package/dist/src/v2/immutable-workflow-publisher.js.map +1 -1
  66. package/dist/src/v2/implementation-report.d.ts +3 -1
  67. package/dist/src/v2/implementation-report.d.ts.map +1 -1
  68. package/dist/src/v2/implementation-report.js +17 -4
  69. package/dist/src/v2/implementation-report.js.map +1 -1
  70. package/dist/src/v2/implementation-reviewer.d.ts +41 -12
  71. package/dist/src/v2/implementation-reviewer.d.ts.map +1 -1
  72. package/dist/src/v2/implementation-reviewer.js +114 -42
  73. package/dist/src/v2/implementation-reviewer.js.map +1 -1
  74. package/dist/src/v2/pending-effect-settlement.d.ts +44 -0
  75. package/dist/src/v2/pending-effect-settlement.d.ts.map +1 -0
  76. package/dist/src/v2/pending-effect-settlement.js +69 -0
  77. package/dist/src/v2/pending-effect-settlement.js.map +1 -0
  78. package/dist/src/v2/process-identity.d.ts +45 -0
  79. package/dist/src/v2/process-identity.d.ts.map +1 -0
  80. package/dist/src/v2/process-identity.js +118 -0
  81. package/dist/src/v2/process-identity.js.map +1 -0
  82. package/dist/src/v2/proof-report.d.ts +2 -1
  83. package/dist/src/v2/proof-report.d.ts.map +1 -1
  84. package/dist/src/v2/proof-report.js +10 -4
  85. package/dist/src/v2/proof-report.js.map +1 -1
  86. package/dist/src/v2/review-feedback-coordinator.d.ts +1 -1
  87. package/dist/src/v2/review-feedback-coordinator.d.ts.map +1 -1
  88. package/dist/src/v2/review-feedback-coordinator.js +1 -1
  89. package/dist/src/v2/review-feedback-coordinator.js.map +1 -1
  90. package/dist/src/v2/review-feedback.d.ts +14 -20
  91. package/dist/src/v2/review-feedback.d.ts.map +1 -1
  92. package/dist/src/v2/review-feedback.js +45 -87
  93. package/dist/src/v2/review-feedback.js.map +1 -1
  94. package/dist/src/v2/run-issue.d.ts +129 -88
  95. package/dist/src/v2/run-issue.d.ts.map +1 -1
  96. package/dist/src/v2/run-issue.js +1965 -2381
  97. package/dist/src/v2/run-issue.js.map +1 -1
  98. package/dist/src/v2/run-state-projections.d.ts +84 -0
  99. package/dist/src/v2/run-state-projections.d.ts.map +1 -0
  100. package/dist/src/v2/run-state-projections.js +142 -0
  101. package/dist/src/v2/run-state-projections.js.map +1 -0
  102. package/dist/src/v2/run-store.d.ts +99 -81
  103. package/dist/src/v2/run-store.d.ts.map +1 -1
  104. package/dist/src/v2/run-store.js +245 -542
  105. package/dist/src/v2/run-store.js.map +1 -1
  106. package/dist/src/v2/runtime-assets.d.ts +3 -0
  107. package/dist/src/v2/runtime-assets.d.ts.map +1 -1
  108. package/dist/src/v2/runtime-assets.js +104 -0
  109. package/dist/src/v2/runtime-assets.js.map +1 -1
  110. package/dist/src/v2/runtime.d.ts +56 -44
  111. package/dist/src/v2/runtime.d.ts.map +1 -1
  112. package/dist/src/v2/runtime.js +383 -503
  113. package/dist/src/v2/runtime.js.map +1 -1
  114. package/dist/src/v2/setup.js +0 -2
  115. package/dist/src/v2/setup.js.map +1 -1
  116. package/dist/src/v2/validation-progression.d.ts +70 -0
  117. package/dist/src/v2/validation-progression.d.ts.map +1 -0
  118. package/dist/src/v2/validation-progression.js +247 -0
  119. package/dist/src/v2/validation-progression.js.map +1 -0
  120. package/dist/src/v2/workflow-assets.d.ts +9 -3
  121. package/dist/src/v2/workflow-assets.d.ts.map +1 -1
  122. package/dist/src/v2/workflow-assets.js +256 -43
  123. package/dist/src/v2/workflow-assets.js.map +1 -1
  124. package/internal-workflow/docs/agents/bug-workflow-routing.md +9 -7
  125. package/internal-workflow/docs/agents/coding-skill-routing.md +170 -120
  126. package/internal-workflow/docs/agents/tool-usage.md +23 -12
  127. package/internal-workflow/manifest.json +1 -1
  128. package/internal-workflow/operations/code-review/SKILL.md +34 -15
  129. package/internal-workflow/operations/implementation/SKILL.md +21 -16
  130. package/internal-workflow/profiles/implementer.toml +9 -0
  131. package/internal-workflow/profiles/review_coordinator.toml +9 -0
  132. package/internal-workflow/profiles/spec_reviewer.toml +9 -0
  133. package/internal-workflow/profiles/standards_reviewer.toml +9 -0
  134. package/internal-workflow/schemas/code-review-v1.json +1 -1
  135. package/internal-workflow/schemas/implementation-report-v1.json +1 -1
  136. package/internal-workflow/schemas/proof-report-v1.json +1 -1
  137. package/internal-workflow/skills/bug-root-cause-explainer/SKILL.md +114 -0
  138. package/internal-workflow/skills/bug-root-cause-explainer/agents/openai.yaml +7 -0
  139. package/internal-workflow/skills/bug-root-cause-explainer/evals/evals.json +18 -0
  140. package/internal-workflow/skills/code-review/SKILL.md +84 -306
  141. package/internal-workflow/skills/code-review/agents/openai.yaml +5 -3
  142. package/internal-workflow/skills/code-review/evals/evals.json +83 -0
  143. package/internal-workflow/skills/code-review/references/standards-smells.md +41 -0
  144. package/internal-workflow/skills/diagnosing-bugs/SKILL.md +69 -32
  145. package/internal-workflow/skills/diagnosing-bugs/agents/openai.yaml +2 -2
  146. package/internal-workflow/skills/diagnosing-bugs/evals/evals.json +63 -0
  147. package/internal-workflow/skills/grilling/SKILL.md +51 -0
  148. package/internal-workflow/skills/grilling/agents/openai.yaml +6 -0
  149. package/internal-workflow/skills/grilling/evals/evals.json +47 -0
  150. package/internal-workflow/skills/implement/SKILL.md +135 -0
  151. package/internal-workflow/skills/implement/agents/openai.yaml +6 -0
  152. package/internal-workflow/skills/implement/evals/evals.json +150 -0
  153. package/internal-workflow/skills/plan/SKILL.md +59 -0
  154. package/internal-workflow/skills/plan/agents/openai.yaml +6 -0
  155. package/internal-workflow/skills/plan/evals/evals.json +36 -0
  156. package/internal-workflow/skills/prototype/LOGIC.md +130 -0
  157. package/internal-workflow/skills/prototype/SKILL.md +69 -0
  158. package/internal-workflow/skills/prototype/UI.md +157 -0
  159. package/internal-workflow/skills/prototype/agents/openai.yaml +6 -0
  160. package/internal-workflow/skills/prototype/evals/evals.json +67 -0
  161. package/internal-workflow/skills/research/SKILL.md +110 -0
  162. package/internal-workflow/skills/research/agents/openai.yaml +6 -0
  163. package/internal-workflow/skills/research/evals/evals.json +49 -0
  164. package/internal-workflow/skills/tdd/SKILL.md +72 -67
  165. package/internal-workflow/skills/tdd/agents/openai.yaml +2 -2
  166. package/internal-workflow/skills/tdd/evals/evals.json +12 -0
  167. package/internal-workflow/skills/tdd/mocking.md +48 -1
  168. package/internal-workflow/skills/tdd/refactoring.md +3 -3
  169. package/internal-workflow/skills/tickets-orchestrator/SKILL.md +199 -0
  170. package/internal-workflow/skills/tickets-orchestrator/agents/openai.yaml +6 -0
  171. package/internal-workflow/skills/tickets-orchestrator/evals/evals.json +126 -0
  172. package/internal-workflow/skills/tickets-orchestrator/references/delegate-integrate.md +83 -0
  173. package/internal-workflow/skills/tickets-orchestrator/references/finish-delivery.md +69 -0
  174. package/internal-workflow/skills/tickets-orchestrator/references/stop-completion.md +63 -0
  175. package/internal-workflow/skills/to-spec/SKILL.md +133 -0
  176. package/internal-workflow/skills/to-spec/agents/openai.yaml +6 -0
  177. package/internal-workflow/skills/to-spec/evals/evals.json +24 -0
  178. package/internal-workflow/skills/to-tickets/SKILL.md +189 -0
  179. package/internal-workflow/skills/to-tickets/agents/openai.yaml +6 -0
  180. package/internal-workflow/skills/to-tickets/evals/evals.json +79 -0
  181. package/internal-workflow/skills/to-tickets/references/publishing-details.md +117 -0
  182. package/package.json +1 -1
  183. package/dist/src/v2/proof-store.d.ts +0 -54
  184. package/dist/src/v2/proof-store.d.ts.map +0 -1
  185. package/dist/src/v2/proof-store.js +0 -301
  186. package/dist/src/v2/proof-store.js.map +0 -1
  187. package/dist/src/v2/route-continuations.d.ts +0 -32
  188. package/dist/src/v2/route-continuations.d.ts.map +0 -1
  189. package/dist/src/v2/route-continuations.js +0 -2
  190. package/dist/src/v2/route-continuations.js.map +0 -1
  191. package/dist/src/v2/route-coordinator.d.ts +0 -72
  192. package/dist/src/v2/route-coordinator.d.ts.map +0 -1
  193. package/dist/src/v2/route-coordinator.js +0 -275
  194. package/dist/src/v2/route-coordinator.js.map +0 -1
  195. package/dist/src/v2/route-decision.d.ts +0 -120
  196. package/dist/src/v2/route-decision.d.ts.map +0 -1
  197. package/dist/src/v2/route-decision.js +0 -380
  198. package/dist/src/v2/route-decision.js.map +0 -1
  199. package/dist/src/v2/spec-coordinator.d.ts +0 -73
  200. package/dist/src/v2/spec-coordinator.d.ts.map +0 -1
  201. package/dist/src/v2/spec-coordinator.js +0 -126
  202. package/dist/src/v2/spec-coordinator.js.map +0 -1
  203. package/dist/src/v2/spec-delivery.d.ts +0 -112
  204. package/dist/src/v2/spec-delivery.d.ts.map +0 -1
  205. package/dist/src/v2/spec-delivery.js +0 -336
  206. package/dist/src/v2/spec-delivery.js.map +0 -1
  207. package/dist/src/v2/triage-route.d.ts +0 -68
  208. package/dist/src/v2/triage-route.d.ts.map +0 -1
  209. package/dist/src/v2/triage-route.js +0 -223
  210. package/dist/src/v2/triage-route.js.map +0 -1
  211. package/dist/src/v2/waiting-human-coordinator.d.ts +0 -49
  212. package/dist/src/v2/waiting-human-coordinator.d.ts.map +0 -1
  213. package/dist/src/v2/waiting-human-coordinator.js +0 -509
  214. package/dist/src/v2/waiting-human-coordinator.js.map +0 -1
  215. package/dist/src/v2/waiting-human.d.ts +0 -143
  216. package/dist/src/v2/waiting-human.d.ts.map +0 -1
  217. package/dist/src/v2/waiting-human.js +0 -408
  218. package/dist/src/v2/waiting-human.js.map +0 -1
  219. package/internal-workflow/docs/agents/contract-test-ledger.md +0 -71
  220. package/internal-workflow/docs/agents/review-gates.md +0 -42
  221. package/internal-workflow/docs/agents/review-protocol.md +0 -98
  222. package/internal-workflow/evals/coding-skill-evals.json +0 -373
  223. package/internal-workflow/operations/ambiguity-review/SKILL.md +0 -5
  224. package/internal-workflow/operations/qualification-repair/SKILL.md +0 -17
  225. package/internal-workflow/operations/spec-author/SKILL.md +0 -12
  226. package/internal-workflow/operations/spec-review/SKILL.md +0 -12
  227. package/internal-workflow/operations/triage/SKILL.md +0 -12
  228. package/internal-workflow/profiles/analyst_deep.toml +0 -9
  229. package/internal-workflow/profiles/implementer_standard.toml +0 -9
  230. package/internal-workflow/profiles/proof_agent.toml +0 -8
  231. package/internal-workflow/profiles/reviewer_deep.toml +0 -9
  232. package/internal-workflow/profiles/reviewer_standard.toml +0 -9
  233. package/internal-workflow/schemas/ambiguity-review-v1.json +0 -1
  234. package/internal-workflow/schemas/spec-author-v1.json +0 -1
  235. package/internal-workflow/schemas/spec-review-v1.json +0 -30
  236. package/internal-workflow/schemas/triage-route-v1.json +0 -1
  237. package/internal-workflow/skills/agent-auto/SKILL.md +0 -19
  238. package/internal-workflow/skills/agent-auto/agents/openai.yaml +0 -6
  239. package/internal-workflow/skills/code-debugger/SKILL.md +0 -122
  240. package/internal-workflow/skills/code-debugger/agents/openai.yaml +0 -7
  241. package/internal-workflow/skills/code-review/references/bug-classes.md +0 -56
  242. package/internal-workflow/skills/code-review/references/cleanup-lens.md +0 -52
  243. package/internal-workflow/skills/code-review/references/framework-lenses.md +0 -34
  244. package/internal-workflow/skills/code-review/references/targeted-recipes.md +0 -49
  245. package/internal-workflow/skills/implementation-spec-maker/SKILL.md +0 -107
  246. package/internal-workflow/skills/implementation-spec-maker/agents/openai.yaml +0 -6
  247. package/internal-workflow/skills/implementation-spec-maker/references/source-modes.md +0 -32
  248. package/internal-workflow/skills/implementation-spec-maker/references/spec-template.md +0 -146
  249. package/internal-workflow/skills/implementation-spec-review/SKILL.md +0 -131
  250. package/internal-workflow/skills/implementation-spec-review/agents/openai.yaml +0 -6
  251. package/internal-workflow/skills/implementation-spec-review/evals/evals.json +0 -78
  252. package/internal-workflow/skills/implementation-spec-review/references/review-loop.md +0 -121
  253. package/internal-workflow/skills/small-task-implementer/SKILL.md +0 -112
  254. package/internal-workflow/skills/small-task-implementer/agents/openai.yaml +0 -6
  255. package/internal-workflow/skills/spec-implementer/SKILL.md +0 -133
  256. package/internal-workflow/skills/spec-implementer/agents/openai.yaml +0 -6
  257. package/internal-workflow/skills/spec-implementer/evals/evals.json +0 -30
  258. package/internal-workflow/skills/spec-implementer/references/review-loop.md +0 -100
  259. package/internal-workflow/skills/triage/AGENT-BRIEF.md +0 -192
  260. package/internal-workflow/skills/triage/OUT-OF-SCOPE.md +0 -101
  261. package/internal-workflow/skills/triage/SKILL.md +0 -134
  262. package/internal-workflow/skills/triage/agents/openai.yaml +0 -6
@@ -0,0 +1,67 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "prototype",
4
+ "cases": [
5
+ {
6
+ "id": "logic-tui-portable-state",
7
+ "prompt": "Prototype whether a small state machine feels right before production implementation.",
8
+ "expected": [
9
+ "state the single logic question",
10
+ "isolate the state model behind a small pure portable interface",
11
+ "build a lightweight TUI that re-renders the complete state after every action",
12
+ "provide one command and hand the validated answer to Implement"
13
+ ],
14
+ "forbidden": [
15
+ "mix terminal I/O into the state module",
16
+ "connect to production persistence",
17
+ "promote the prototype directly to production"
18
+ ]
19
+ },
20
+ {
21
+ "id": "ui-existing-page-variants",
22
+ "prompt": "Explore three layouts for a settings area that already has a real host page.",
23
+ "expected": [
24
+ "use the existing page first under a development-only prototype convention",
25
+ "build three structurally different variants selected by ?variant=",
26
+ "provide a floating switcher with URL and keyboard cycling",
27
+ "keep existing read-only page context visible"
28
+ ],
29
+ "forbidden": [
30
+ "create an isolated empty route without checking the existing page",
31
+ "vary only colours or copy",
32
+ "wire controls to real production mutations"
33
+ ]
34
+ },
35
+ {
36
+ "id": "no-production-or-git-mutation",
37
+ "prompt": "The prototype answered its question and one variant won.",
38
+ "expected": [
39
+ "record the answer separately from throwaway code",
40
+ "remove or isolate every prototype artifact",
41
+ "return production delivery to Implement"
42
+ ],
43
+ "forbidden": [
44
+ "create a branch or commit",
45
+ "ship the prototype code",
46
+ "write to a tracker"
47
+ ]
48
+ },
49
+ {
50
+ "id": "reproducible-primary-source-bundle",
51
+ "prompt": "The throwaway prototype answered its question and will now be cleaned from the product tree.",
52
+ "expected": [
53
+ "preserve a self-contained reproduction bundle outside the production tree before cleanup",
54
+ "include exact source, inputs, deterministic command or URL, observed output, answer, content digests, host revision, toolchain, lockfiles, and required host files",
55
+ "bundle a complete patch for every dirty tracked dependency and exact bytes for every untracked dependency; paths or digests alone are insufficient",
56
+ "restore UI dependencies with read-only fixtures in an isolated worktree at the recorded revision",
57
+ "return the bundle path while production delivery remains with Implement"
58
+ ],
59
+ "forbidden": [
60
+ "create a prototype branch or commit",
61
+ "copy secrets or production data",
62
+ "clean up while a required dirty or untracked dependency is not recoverable from the bundle",
63
+ "promote prototype source into production"
64
+ ]
65
+ }
66
+ ]
67
+ }
@@ -0,0 +1,110 @@
1
+ ---
2
+ name: research
3
+ description: Research material external API, SDK, specification, service, or source-code questions using primary sources and save one cited repository artifact. Use for requested durable/delegated research or multi-source contract uncertainty; not for narrow lookups, repo-only work, bug reproduction, or specialized docs tasks.
4
+ ---
5
+
6
+ # Research
7
+
8
+ Resolve one external question into reusable evidence for downstream coding work.
9
+ The invoked skill authorizes one `researcher` child; root owns source
10
+ verification, artifact integration, user communication, and later decisions.
11
+ Research authorizes only the one cited evidence artifact described below. It
12
+ does not authorize implementation or production mutation, Git actions, tracker
13
+ writes, or the decision that consumes the evidence.
14
+
15
+ ## Route Proportionately
16
+
17
+ - Read local evidence first: manifests, lockfiles, installed source, tests,
18
+ configs, ADRs, and repository docs.
19
+ - Keep one narrow documentation lookup inline unless the user explicitly requests delegation or a durable artifact. When the lookup stays inline, use the owning specialized docs skill or tool and answer in chat without creating an artifact.
20
+ - Invoke this workflow when the user requests delegated reading or a saved research result, or when a material decision needs multi-source comparison, freshness checking, or external contract synthesis.
21
+ - Use repo exploration or bug-diagnosis skills when the owning evidence is local
22
+ code or runtime behavior. Research may supply one external contract input but
23
+ never owns bug reproduction or implementation.
24
+
25
+ ## Build The Research Capsule
26
+
27
+ Before delegation, record:
28
+
29
+ - the exact question and decision it must unblock;
30
+ - relevant verified local context;
31
+ - in-scope and excluded products, versions, environments, and claims;
32
+ - allowed primary-source types and required freshness;
33
+ - the repository output path.
34
+
35
+ Use the repository's existing research-note convention. If none exists, choose
36
+ `docs/research/YYYY-MM-DD/HHMM-<slug>.md`.
37
+
38
+ ## Delegate One Bounded Question
39
+
40
+ Launch one fresh `researcher` child with the Research Capsule and no
41
+ inherited conclusions. The child is read-only and must return:
42
+
43
+ 1. a short answer;
44
+ 2. a claim-to-source ledger for every material fact;
45
+ 3. source version or publication/update date when available;
46
+ 4. conflicts, uncertainty, and missing evidence;
47
+ 5. clearly labelled inferences for the repository decision.
48
+
49
+ While it reads, continue only independent local work. Do not make or implement
50
+ the blocked decision before the research returns. If the named role is
51
+ unavailable, perform the same bounded workflow inline and report the fallback;
52
+ do not substitute an unrelated code explorer or reviewer.
53
+
54
+ ## Source Standard
55
+
56
+ Prefer the source that owns the claim:
57
+
58
+ 1. official documentation or specifications;
59
+ 2. first-party source code, changelogs, release notes, or issue trackers;
60
+ 3. first-party APIs or published schemas.
61
+
62
+ Use secondary material only to discover primary sources or to expose a disputed
63
+ interpretation. Never promote it to authority when an owning source exists.
64
+ Cite the exact page or repository location that supports each material claim.
65
+ Separate sourced fact from inference, and state when current behavior cannot be
66
+ confirmed.
67
+
68
+ Use specialized source adapters when applicable: for example, `$openai-docs`
69
+ for OpenAI products, Context7 for precise package documentation, and site
70
+ parsers for extraction. Their output still must satisfy this source standard.
71
+
72
+ ## Verify And Save
73
+
74
+ Root must open and verify every source behind a claim that changes architecture,
75
+ scope, implementation, security, cost, or compatibility. Repair unsupported or
76
+ overstated claims, then save exactly one Markdown artifact:
77
+
78
+ ```markdown
79
+ # <Research question>
80
+
81
+ ## Decision To Unblock
82
+ <decision and relevant local context>
83
+
84
+ ## Short Answer
85
+ <concise answer>
86
+
87
+ ## Findings
88
+ | Claim | Primary Source | Version / Date | Confidence |
89
+ | --- | --- | --- | --- |
90
+ | ... | ... | ... | ... |
91
+
92
+ ## Repository Implications
93
+ <clearly labelled inferences and affected plans/specs/tickets>
94
+
95
+ ## Conflicts And Unknowns
96
+ <conflicting sources, stale evidence, and unresolved questions>
97
+ ```
98
+
99
+ Do not include credentials, private tokens, or copied secrets. Link or cite
100
+ sources instead of reproducing long copyrighted passages.
101
+
102
+ ## Downstream Contract
103
+
104
+ - Return the saved path and the decision it now supports.
105
+ - Let plans, PRDs, and tickets cite the artifact as their external evidence
106
+ instead of repeating the research.
107
+ - Re-read only claims invalidated by changed versions, dates, contracts, or
108
+ source conflicts.
109
+ - Treat the artifact as evidence, not implementation authority. Behavior-changing
110
+ work still follows the normal TDD, implementation, and review routes.
@@ -0,0 +1,6 @@
1
+ interface:
2
+ display_name: "Research"
3
+ short_description: "Research from high-trust primary sources"
4
+ default_prompt: "Use $research to investigate this external question and save a cited repository artifact."
5
+ policy:
6
+ allow_implicit_invocation: true
@@ -0,0 +1,49 @@
1
+ {
2
+ "schema_version": 1,
3
+ "skill": "research",
4
+ "cases": [
5
+ {
6
+ "id": "narrow-lookup-stays-inline",
7
+ "prompt": "Find one current option name in the official documentation for a package already pinned by the repository. No durable note was requested.",
8
+ "expected": [
9
+ "read the pinned local version first",
10
+ "use the owning specialized documentation source inline",
11
+ "answer in chat without delegation or a repository artifact"
12
+ ],
13
+ "forbidden": [
14
+ "launch a researcher for one narrow lookup",
15
+ "create a research capsule artifact",
16
+ "change production code"
17
+ ]
18
+ },
19
+ {
20
+ "id": "fresh-primary-source-capsule",
21
+ "prompt": "Compare a material external SDK contract across its current specification, first-party source, and release notes, and save reusable evidence.",
22
+ "expected": [
23
+ "record the decision, pinned local context, scope, exclusions, and required freshness",
24
+ "delegate one bounded read-only question to a researcher",
25
+ "cite primary sources with version or date, conflicts, unknowns, and labelled inferences",
26
+ "save exactly one Markdown research artifact"
27
+ ],
28
+ "forbidden": [
29
+ "use a secondary article as authority",
30
+ "make the blocked implementation decision before research returns",
31
+ "perform a production mutation"
32
+ ]
33
+ },
34
+ {
35
+ "id": "root-verifies-material-claims",
36
+ "prompt": "A researcher returns a cited claim that changes architecture and compatibility.",
37
+ "expected": [
38
+ "root opens the exact source behind the material claim",
39
+ "root repairs unsupported or overstated wording",
40
+ "the artifact separates sourced fact from repository inference"
41
+ ],
42
+ "forbidden": [
43
+ "accept the child summary as source verification",
44
+ "treat the artifact as implementation authority",
45
+ "perform Git or tracker writes"
46
+ ]
47
+ }
48
+ ]
49
+ }
@@ -1,73 +1,78 @@
1
1
  ---
2
2
  name: tdd
3
- description: Test-driven development for changes that alter observable behavior, have a natural public test seam, and can produce a meaningful failing test before implementation. Use after the global TDD Fit Gate passes, or when the user explicitly requests red-green-refactor, test-first development, or TDD.
3
+ description: Use test-driven development where possible for observable behavior changes with a natural public seam and a meaningful pre-change failure.
4
4
  ---
5
5
 
6
6
  # Test-Driven Development
7
7
 
8
- Use short vertical RED -> GREEN cycles. Make each test prove observable behavior through the same public seam real callers use.
9
-
10
- ## Fit
11
-
12
- Use this skill only when the change alters observable behavior, a natural public
13
- seam exists, and the pre-change test will fail for the intended behavioral
14
- reason. If an implicit activation fails this gate, stop the TDD route and use
15
- existing regression tests plus affected validation. For mixed tasks, apply TDD
16
- only to the behavioral slice.
17
-
18
- Behavior-preserving cleanup, dead-code deletion, documentation, copy,
19
- formatting, generated assets, package maintenance, simple config, builds, and
20
- read-only work do not need TDD. Absence and architecture guards added after a
21
- cleanup are validation, not RED proofs.
22
-
23
- ## Core Contract
24
-
25
- - Lock expected behavior from the request, specification, design, bug report, or existing product behavior before changing implementation.
26
- - Derive expected values from an independent source, never from the production algorithm.
27
- - Prove RED on the old behavior for the same observable reason the user reported or requested.
28
- - Add only enough implementation to make the current test pass; do not anticipate later tests.
29
- - Keep tests stable across behavior-preserving refactors.
30
- - After sufficient GREEN, stop by default. Refactor only to reduce concrete complexity introduced by the change.
31
-
32
- Read [tests.md](tests.md) when choosing or reviewing test shape. Read [mocking.md](mocking.md) before introducing test doubles.
33
-
34
- ## Before the First RED
35
-
36
- 1. Read local instructions, domain language, existing tests, and relevant ADRs.
37
- 2. List the prioritized observable behaviors, not implementation steps.
38
- 3. Select the public seam where callers observe each behavior.
39
- 4. Ask the user only when the seam changes the public contract, product intent is unclear, or behavior priorities materially conflict.
40
- 5. Use the shared [Contract Test Ledger](../../docs/agents/contract-test-ledger.md) only when its material-delta and missed-failure gate passes.
41
- 6. If no natural public seam exists, stop the TDD route. Consult [interface-design.md](interface-design.md) only when changing the interface is itself required by the task.
42
-
43
- For UI behavior, define proof at the rendered seam: visible content and order, interaction result, semantics, or screenshot when layout direction or scrolling matters.
44
-
45
- ## RED -> GREEN Cycle
46
-
47
- For each behavior:
48
-
49
- 1. **RED:** Write one test through the selected seam.
50
- 2. Confirm it fails on current behavior for the expected reason. A passing test or an internal-only failure is not valid RED.
51
- 3. **GREEN:** Add the minimal implementation required for that test.
52
- 4. Run the proof and update the ledger status to `red`, `green`, or `blocked` with the missing seam or evidence.
53
-
54
- Keep each cycle to one seam, one behavior, one test, and one minimal implementation. Do not rewrite the test to fit the code. For state, async, lifecycle, retry, cache, or auth defects, include the competing condition when feasible.
55
-
56
- Handle reviewer repairs inside the same activation only under [bug workflow routing](../../docs/agents/bug-workflow-routing.md); group related cases by protected invariant.
57
-
58
- ## After GREEN
59
-
60
- GREEN is a valid stopping point. If the current change created concrete local complexity, use [refactoring.md](refactoring.md) and rerun affected tests.
61
-
62
- ## Cycle Checklist
63
-
64
- ```text
65
- [ ] Behavior is proved through the caller's public seam
66
- [ ] Expected behavior is locked and the expected value is independent
67
- [ ] RED fails on old behavior for the correct observable reason
68
- [ ] Test was not fitted to implementation details
69
- [ ] GREEN uses only the code needed for the current behavior
70
- [ ] Final outcome and relevant competing condition are proved
71
- [ ] Contract Test Ledger is current when applicable
72
- [ ] Any refactor is local and reduces current-change complexity
73
- ```
8
+ Use TDD where possible. TDD is the RED -> GREEN loop. Work in short vertical
9
+ cycles through the same public seam real callers use. This skill is the
10
+ reference that makes that loop produce tests worth keeping: what a good test
11
+ is, where tests go, the anti-patterns, and the rules of the loop.
12
+
13
+ When exploring the codebase, read `CONTEXT.md` (if it exists) so test names and
14
+ interface vocabulary match the project's domain language, and respect ADRs in
15
+ the area you're touching.
16
+
17
+ TDD is an internal proof discipline. It does not own or start the
18
+ implementation workflow, production mutations, Review, or Git; the invoking
19
+ Implement owner retains those responsibilities.
20
+
21
+ ## What a good test is
22
+
23
+ Tests verify behavior through public interfaces, not implementation details.
24
+ Code can change entirely; tests shouldn't. A good test reads like a
25
+ specification — "user can checkout with valid cart" tells you exactly what
26
+ capability exists — and survives refactors because it doesn't care about
27
+ internal structure.
28
+
29
+ See [tests.md](tests.md) for examples and [mocking.md](mocking.md) for mocking
30
+ guidelines.
31
+
32
+ ## Seams — where tests go
33
+
34
+ A **seam** is the natural public boundary you test at: the interface where you
35
+ observe behavior without reaching inside. Tests live at seams, never against
36
+ internals. Choose that seam from repository evidence and the real caller path;
37
+ ask only when choosing it would change product behavior or ownership.
38
+
39
+ ## Contract
40
+
41
+ - Lock expected behavior from the authorized request, issue, Parent PRD, or
42
+ existing product contract.
43
+ - Choose the natural public seam from repository evidence. Ask only when the
44
+ seam itself changes product behavior or ownership.
45
+ - Write one behavior test and confirm it fails before implementation for the
46
+ intended observable reason.
47
+ - Add only enough production code to make that test pass, then repeat for the
48
+ next behavior.
49
+ - Derive expected values independently; do not reproduce the production
50
+ algorithm in the assertion.
51
+ - Prefer tests that survive internal refactors. Do not test private methods or
52
+ introduce production indirection solely for mocks.
53
+ - Refactor only after GREEN and only to remove concrete complexity introduced
54
+ by the change.
55
+
56
+ ## Anti-patterns
57
+
58
+ - **Implementation-coupled** — mocks internal collaborators, tests private methods, or verifies through a side channel (querying the database instead of using the interface). The tell: the test breaks when you refactor but behavior hasn't changed.
59
+ - **Tautological** — the assertion recomputes the expected value the way the code does (`expect(add(a, b)).toBe(a + b)`, a snapshot derived by hand the same way, a constant asserted equal to itself), so it passes by construction and can never disagree with the code. Expected values must come from an independent source of truth — a known-good literal, a worked example, the spec.
60
+ - **Horizontal slicing** — writing all tests first, then all implementation. Bulk tests verify _imagined_ behavior: you test the _shape_ of things rather than user-facing behavior, the tests go insensitive to real changes, and you commit to test structure before understanding the implementation. Work in **vertical slices** instead — one test -> one implementation -> repeat, each test a **tracer bullet** that responds to what the last cycle taught you.
61
+
62
+ ## Rules of the loop
63
+
64
+ - **RED before GREEN.** Confirm a meaningful pre-change failure for the
65
+ intended observable reason, then add only enough code to pass it. Do not
66
+ anticipate future tests or add speculative features.
67
+ - **One slice at a time.** One seam, one test, one minimal implementation per
68
+ cycle.
69
+ - **Refactor only after GREEN.** Use [refactoring.md](refactoring.md) and act
70
+ only on specific observed complexity. Refactoring is not a mandatory phase.
71
+
72
+ If no natural public seam or meaningful pre-change failure exists, use direct
73
+ observable proof instead. Do not manufacture RED for docs, copy, formatting,
74
+ mechanical config, generated files, deletion, builds, or read-only work.
75
+
76
+ Read [interface-design.md](interface-design.md) before changing a production
77
+ external seam for testability, and [mocking.md](mocking.md) before introducing
78
+ test doubles.
@@ -1,6 +1,6 @@
1
1
  interface:
2
2
  display_name: "Test-Driven Development"
3
- short_description: "Use TDD only when its behavioral fit gate passes"
4
- default_prompt: "Use $tdd after confirming an observable behavior change, a public test seam, and a meaningful pre-change failure."
3
+ short_description: "Use behavior-first RED to GREEN where possible"
4
+ default_prompt: "Use $tdd where possible to prove one observable behavior at a time through its natural public seam."
5
5
  policy:
6
6
  allow_implicit_invocation: true
@@ -8,6 +8,18 @@
8
8
  "expected": ["stop after green", "keep the current structure"],
9
9
  "forbidden": ["add helpers, classes, or value objects", "refactor unrelated code"]
10
10
  },
11
+ {
12
+ "id": "meaningful-red-at-public-seam",
13
+ "prompt": "A behavior change has a natural public caller seam, but the first proposed test fails only because its fixture path is wrong.",
14
+ "expected": [
15
+ "repair the test setup until RED fails for the intended observable behavior",
16
+ "use the same natural public seam for GREEN"
17
+ ],
18
+ "forbidden": [
19
+ "count an infrastructure or fixture error as meaningful RED",
20
+ "test a private helper instead"
21
+ ]
22
+ },
11
23
  {
12
24
  "id": "no-test-only-seam",
13
25
  "prompt": "A behavior test can use the existing public seam, but dependency injection would make mocking easier.",
@@ -17,4 +17,51 @@ Don't mock:
17
17
 
18
18
  Use the existing public or system-boundary seam first. Add dependency injection,
19
19
  an adapter, or an SDK wrapper only when production ownership or the requested
20
- contract requires it—not only to make a test easier to mock.
20
+ contract requires it, not solely to make a test easier to mock. A production
21
+ seam must earn its place with production variation or ownership, not a test-only
22
+ route.
23
+
24
+ At an existing system boundary, design interfaces that are easy to mock:
25
+
26
+ **1. Use dependency injection**
27
+
28
+ Pass external dependencies in rather than creating them internally:
29
+
30
+ ```typescript
31
+ // Easy to mock
32
+ function processPayment(order, paymentClient) {
33
+ return paymentClient.charge(order.total);
34
+ }
35
+
36
+ // Hard to mock
37
+ function processPayment(order) {
38
+ const client = new StripeClient(process.env.STRIPE_KEY);
39
+ return client.charge(order.total);
40
+ }
41
+ ```
42
+
43
+ **2. Prefer SDK-style interfaces over generic fetchers**
44
+
45
+ Create specific functions for each external operation instead of one generic
46
+ function with conditional logic:
47
+
48
+ ```typescript
49
+ // GOOD: Each function is independently mockable
50
+ const api = {
51
+ getUser: (id) => fetch(`/users/${id}`),
52
+ getOrders: (userId) => fetch(`/users/${userId}/orders`),
53
+ createOrder: (data) => fetch('/orders', { method: 'POST', body: data }),
54
+ };
55
+
56
+ // BAD: Mocking requires conditional logic inside the mock
57
+ const api = {
58
+ fetch: (endpoint, options) => fetch(endpoint, options),
59
+ };
60
+ ```
61
+
62
+ The SDK approach means:
63
+
64
+ - Each mock returns one specific shape
65
+ - No conditional logic in test setup
66
+ - Easier to see which endpoints a test exercises
67
+ - Type safety per endpoint
@@ -1,8 +1,8 @@
1
1
  # Refactoring After GREEN
2
2
 
3
- Stop when GREEN code is clear and local. Refactor only when the current change
4
- introduced concrete duplication, confusion, or misplaced ownership and the edit
5
- reduces total complexity.
3
+ Stop when GREEN code is clear and local. Refactor only after GREEN and only when
4
+ the current change exposes specific concrete observed complexity: duplication,
5
+ confusion, or misplaced ownership that the edit will reduce.
6
6
 
7
7
  Keep it local. Do not add helpers, classes, value objects, deeper modules, or
8
8
  unrelated cleanup from pattern preference alone. Rerun affected tests.