@iowarp/clio-coder 0.4.1 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (604) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/CONTRIBUTING.md +142 -52
  3. package/README.md +434 -473
  4. package/SECURITY.md +2 -1
  5. package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
  6. package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
  7. package/dist/assets/codewiki.json +1 -1
  8. package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
  9. package/dist/builtins-IA7V7FUC.js +22 -0
  10. package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
  11. package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
  12. package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
  13. package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
  14. package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
  15. package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
  16. package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
  17. package/dist/chunk-3EBYEESD.js +314 -0
  18. package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
  19. package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
  20. package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
  21. package/dist/chunk-4UVU7BJ5.js +39 -0
  22. package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
  23. package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
  24. package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
  25. package/dist/chunk-5MEZN6CB.js +1334 -0
  26. package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
  27. package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
  28. package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
  29. package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
  30. package/dist/chunk-7DRAWPTZ.js +360 -0
  31. package/dist/chunk-7E7I3WLS.js +3762 -0
  32. package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
  33. package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
  34. package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
  35. package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
  36. package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
  37. package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
  38. package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
  39. package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
  40. package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
  41. package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
  42. package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
  43. package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
  44. package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
  45. package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
  46. package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
  47. package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
  48. package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
  49. package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
  50. package/dist/chunk-GYV6VZOC.js +26 -0
  51. package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
  52. package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
  53. package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
  54. package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
  55. package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
  56. package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
  57. package/dist/chunk-IJNZMHLA.js +101 -0
  58. package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
  59. package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
  60. package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
  61. package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
  62. package/dist/chunk-JEQ3XTHC.js +42 -0
  63. package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
  64. package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
  65. package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
  66. package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
  67. package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
  68. package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
  69. package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
  70. package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
  71. package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
  72. package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
  73. package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
  74. package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
  75. package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
  76. package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
  77. package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
  78. package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
  79. package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
  80. package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
  81. package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
  82. package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
  83. package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
  84. package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
  85. package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
  86. package/dist/chunk-R6Q67RJH.js +134 -0
  87. package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
  88. package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
  89. package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
  90. package/dist/chunk-SKHCAU7K.js +385 -0
  91. package/dist/chunk-SZAA6XDG.js +30 -0
  92. package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
  93. package/dist/chunk-UOIZ7DA4.js +41 -0
  94. package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
  95. package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
  96. package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
  97. package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
  98. package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
  99. package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
  100. package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
  101. package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
  102. package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
  103. package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
  104. package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
  105. package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
  106. package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
  107. package/dist/chunk-XRZT5WY5.js +47 -0
  108. package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
  109. package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
  110. package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
  111. package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
  112. package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
  113. package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
  114. package/dist/cli/index.js +41 -39
  115. package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
  116. package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
  117. package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
  118. package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
  119. package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
  120. package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
  121. package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
  122. package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
  123. package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
  124. package/dist/context-map-COB37XXN.js +505 -0
  125. package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
  126. package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
  127. package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
  128. package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
  129. package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
  130. package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
  131. package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
  132. package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
  133. package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
  134. package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
  135. package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
  136. package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
  137. package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
  138. package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
  139. package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
  140. package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
  141. package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
  142. package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
  143. package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
  144. package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
  145. package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
  146. package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
  147. package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
  148. package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
  149. package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
  150. package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
  151. package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
  152. package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
  153. package/dist/reset-EAJFFJVB.js +344 -0
  154. package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
  155. package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
  156. package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
  157. package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
  158. package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
  159. package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
  160. package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
  161. package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
  162. package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
  163. package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
  164. package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
  165. package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
  166. package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
  167. package/dist/uninstall-5PEVOE5B.js +408 -0
  168. package/dist/upgrade-M4WXY6KN.js +303 -0
  169. package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
  170. package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
  171. package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
  172. package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
  173. package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
  174. package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
  175. package/dist/worker/entry.js +90 -74
  176. package/docs/README.md +176 -81
  177. package/docs/{acp.md → architecture/acp.md} +36 -20
  178. package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
  179. package/docs/{architecture.md → architecture/architecture.md} +43 -22
  180. package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
  181. package/docs/architecture/artifact-versions.md +90 -0
  182. package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
  183. package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
  184. package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
  185. package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
  186. package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
  187. package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
  188. package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
  189. package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
  190. package/docs/{observability.md → architecture/observability.md} +38 -14
  191. package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
  192. package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
  193. package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
  194. package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
  195. package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
  196. package/docs/architecture/time-conventions.md +125 -0
  197. package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
  198. package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
  199. package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
  200. package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
  201. package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
  202. package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
  203. package/docs/guide/configuration-reference.md +1163 -0
  204. package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
  205. package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
  206. package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
  207. package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
  208. package/docs/{glossary.md → guide/glossary.md} +14 -11
  209. package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
  210. package/docs/guide/panes-and-files.md +290 -0
  211. package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
  212. package/docs/{resource-library.md → guide/resource-library.md} +13 -4
  213. package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
  214. package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
  215. package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
  216. package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
  217. package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
  218. package/docs/process/development-pipeline.md +152 -0
  219. package/docs/process/documentation-coverage.md +100 -0
  220. package/docs/process/documentation-guide.md +187 -0
  221. package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
  222. package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
  223. package/docs/{evolution.md → process/evolution.md} +2 -2
  224. package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
  225. package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
  226. package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
  227. package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
  228. package/evals/README.md +2 -2
  229. package/evals/behavioral-model.yaml +3 -2
  230. package/package.json +10 -8
  231. package/skills/README.md +52 -41
  232. package/skills/coding/ast-grep/SKILL.md +102 -31
  233. package/skills/coding/ast-grep/evals.md +26 -0
  234. package/skills/coding/coding-standards/SKILL.md +41 -6
  235. package/skills/coding/coding-standards/evals.md +23 -0
  236. package/skills/coding/prototype/SKILL.md +88 -29
  237. package/skills/coding/prototype/evals.md +19 -0
  238. package/skills/coding/tdd/SKILL.md +81 -54
  239. package/skills/coding/tdd/evals.md +20 -0
  240. package/skills/context/context-handoff/SKILL.md +44 -3
  241. package/skills/context/context-handoff/evals.md +44 -0
  242. package/skills/context/context-prime/SKILL.md +46 -16
  243. package/skills/context/context-prime/evals.md +45 -0
  244. package/skills/git/branch-closeout/SKILL.md +132 -0
  245. package/skills/git/branch-closeout/evals.md +133 -0
  246. package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
  247. package/skills/git/file-ticket/SKILL.md +78 -64
  248. package/skills/git/file-ticket/assets/issue-template.md +22 -0
  249. package/skills/git/file-ticket/evals.md +31 -26
  250. package/skills/git/file-ticket/references/issue-discovery.md +49 -0
  251. package/skills/git/fix-issue/SKILL.md +88 -65
  252. package/skills/git/fix-issue/evals.md +35 -31
  253. package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
  254. package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
  255. package/skills/git/resolve-merge-conflicts/evals.md +52 -25
  256. package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
  257. package/skills/git/ship/SKILL.md +103 -67
  258. package/skills/git/ship/assets/pr-template.md +21 -0
  259. package/skills/git/ship/evals.md +44 -28
  260. package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
  261. package/skills/git/worktree-create/SKILL.md +80 -50
  262. package/skills/git/worktree-create/evals.md +40 -33
  263. package/skills/git/worktree-create/references/worktree-setup.md +62 -66
  264. package/skills/git/worktree-merge/SKILL.md +112 -65
  265. package/skills/git/worktree-merge/evals.md +42 -34
  266. package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
  267. package/skills/meta/clio-coder-dev/SKILL.md +9 -5
  268. package/skills/meta/clio-coder-dev/evals.md +3 -2
  269. package/skills/meta/clio-coder-test/SKILL.md +102 -95
  270. package/skills/meta/clio-coder-test/evals.md +9 -4
  271. package/skills/meta/clio-coder-test/references/harness.md +100 -124
  272. package/skills/meta/clio-coder-test/references/test-map.md +77 -50
  273. package/skills/meta/credentials/SKILL.md +2 -2
  274. package/skills/meta/find-skills/SKILL.md +2 -2
  275. package/skills/meta/herdr/SKILL.md +2 -2
  276. package/skills/meta/skill-craft/SKILL.md +22 -16
  277. package/skills/planning/archify/SKILL.md +196 -0
  278. package/skills/planning/archify/evals.md +65 -0
  279. package/skills/planning/architecture/SKILL.md +62 -13
  280. package/skills/planning/architecture/evals.md +65 -0
  281. package/skills/planning/backlog/SKILL.md +131 -15
  282. package/skills/planning/backlog/evals.md +142 -0
  283. package/skills/planning/prd/SKILL.md +47 -7
  284. package/skills/planning/prd/evals.md +54 -0
  285. package/skills/planning/product-intent/SKILL.md +58 -3
  286. package/skills/planning/product-intent/evals.md +70 -0
  287. package/skills/planning/tech-spec/SKILL.md +54 -3
  288. package/skills/planning/tech-spec/evals.md +73 -0
  289. package/skills/registry.yaml +70 -62
  290. package/skills/remote.yaml +13 -0
  291. package/skills/research/arxiv-literature/SKILL.md +77 -19
  292. package/skills/research/arxiv-literature/evals.md +50 -0
  293. package/skills/research/experiment-protocol/SKILL.md +21 -2
  294. package/skills/research/experiment-protocol/evals.md +23 -0
  295. package/skills/research/scientific-debugging/SKILL.md +24 -2
  296. package/skills/research/scientific-debugging/evals.md +18 -0
  297. package/skills/research/scientific-modernization/SKILL.md +27 -2
  298. package/skills/research/scientific-modernization/evals.md +27 -0
  299. package/skills/skill-marketplace.json +97 -62
  300. package/skills/workflow/cut-it/SKILL.md +66 -6
  301. package/skills/workflow/cut-it/evals.md +101 -0
  302. package/skills/workflow/design-council/SKILL.md +118 -28
  303. package/skills/workflow/design-council/evals.md +161 -0
  304. package/skills/workflow/grill-me/SKILL.md +87 -11
  305. package/skills/workflow/grill-me/evals.md +153 -0
  306. package/skills/workflow/workflow-distiller/SKILL.md +77 -18
  307. package/skills/workflow/workflow-distiller/evals.md +118 -0
  308. package/src/cli/args.ts +2 -2
  309. package/src/cli/bootstrap-generate.ts +1 -1
  310. package/src/cli/config-inspect.ts +65 -12
  311. package/src/cli/configure-interop.ts +105 -13
  312. package/src/cli/configure-oauth.ts +57 -0
  313. package/src/cli/configure-onboarding.ts +980 -0
  314. package/src/cli/configure-target.ts +594 -0
  315. package/src/cli/configure.ts +1082 -532
  316. package/src/cli/context-map.ts +114 -0
  317. package/src/cli/context.ts +4 -0
  318. package/src/cli/docs.ts +22 -14
  319. package/src/cli/doctor-naming.ts +5 -5
  320. package/src/cli/doctor-toolchain.ts +3 -3
  321. package/src/cli/eval.ts +1 -2
  322. package/src/cli/extensions.ts +2 -1
  323. package/src/cli/fleet.ts +1 -1
  324. package/src/cli/index.ts +3 -1
  325. package/src/cli/internal-dispatch.ts +3 -4
  326. package/src/cli/lifecycle-presenter.ts +436 -0
  327. package/src/cli/models.ts +10 -2
  328. package/src/cli/modes/print.ts +5 -1
  329. package/src/cli/panes.ts +19 -5
  330. package/src/cli/reset.ts +228 -106
  331. package/src/cli/run.ts +9 -4
  332. package/src/cli/select.ts +664 -0
  333. package/src/cli/share.ts +5 -1
  334. package/src/cli/skills-eval.ts +3 -3
  335. package/src/cli/skills.ts +9 -2
  336. package/src/cli/targets.ts +5 -6
  337. package/src/cli/trace.ts +55 -4
  338. package/src/cli/uninstall.ts +233 -165
  339. package/src/cli/upgrade.ts +204 -149
  340. package/src/cli/usage.ts +86 -27
  341. package/src/cli/validate-model.ts +3 -3
  342. package/src/cli/wiki-generate.ts +1 -1
  343. package/src/core/artifact-paths.ts +1 -1
  344. package/src/core/bash-exec.ts +131 -86
  345. package/src/core/bus-events.ts +51 -6
  346. package/src/core/config.ts +61 -1
  347. package/src/core/defaults.ts +7 -4
  348. package/src/core/dispatch-outcome.ts +16 -0
  349. package/src/core/external-diagnostic.ts +44 -0
  350. package/src/core/gateway-routing.ts +157 -0
  351. package/src/core/guardrails.ts +10 -49
  352. package/src/core/prompt-hint.ts +9 -0
  353. package/src/core/safe-exec.ts +17 -2
  354. package/src/core/skill-activation.ts +89 -2
  355. package/src/domains/agents/builtins/architect.md +2 -3
  356. package/src/domains/agents/builtins/coder.md +3 -2
  357. package/src/domains/agents/builtins/debugger.md +2 -2
  358. package/src/domains/agents/builtins/documenter.md +2 -2
  359. package/src/domains/agents/builtins/git-master.md +1 -1
  360. package/src/domains/agents/builtins/oracle.md +1 -1
  361. package/src/domains/agents/builtins/provenance.md +1 -1
  362. package/src/domains/agents/builtins/researcher.md +1 -1
  363. package/src/domains/agents/builtins/scout.md +1 -1
  364. package/src/domains/agents/builtins/tester.md +2 -2
  365. package/src/domains/agents/builtins/verifier.md +2 -2
  366. package/src/domains/agents/builtins/wiki-writer.md +1 -1
  367. package/src/domains/agents/builtins/world-knowledge.md +31 -0
  368. package/src/domains/agents/catalog.ts +13 -15
  369. package/src/domains/agents/contract.ts +2 -0
  370. package/src/domains/agents/extension.ts +23 -1
  371. package/src/domains/agents/result-contract.ts +70 -0
  372. package/src/domains/config/keybindings.ts +8 -0
  373. package/src/domains/context/extension.ts +0 -3
  374. package/src/domains/context/wiki/map-seed.ts +589 -0
  375. package/src/domains/context/wiki/plan.ts +2 -2
  376. package/src/domains/context/working-set/path-index.ts +1 -0
  377. package/src/domains/dispatch/admission.ts +29 -0
  378. package/src/domains/dispatch/agent-candidates.ts +10 -0
  379. package/src/domains/dispatch/budget-envelope.ts +86 -1
  380. package/src/domains/dispatch/capability-match.ts +11 -0
  381. package/src/domains/dispatch/capacity-lease.ts +17 -0
  382. package/src/domains/dispatch/contract.ts +11 -1
  383. package/src/domains/dispatch/extension.ts +237 -49
  384. package/src/domains/dispatch/host-verification.ts +435 -39
  385. package/src/domains/dispatch/intent-requirements.ts +10 -0
  386. package/src/domains/dispatch/intent.ts +18 -1
  387. package/src/domains/dispatch/path-scope.ts +235 -24
  388. package/src/domains/dispatch/run-event-journal.ts +4 -15
  389. package/src/domains/dispatch/state.ts +2 -3
  390. package/src/domains/dispatch/transport.ts +45 -21
  391. package/src/domains/dispatch/types.ts +58 -3
  392. package/src/domains/dispatch/worker-model-metadata.ts +38 -0
  393. package/src/domains/eval/artifacts/store.ts +5 -0
  394. package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
  395. package/src/domains/eval/metrics/token-stream.ts +201 -31
  396. package/src/domains/eval/metrics/tracked.ts +40 -4
  397. package/src/domains/eval/runners/clio-run.ts +5 -2
  398. package/src/domains/eval/schema/suite.ts +28 -0
  399. package/src/domains/eval/schema/verdict.ts +2 -2
  400. package/src/domains/eval/store.ts +8 -1
  401. package/src/domains/eval/suites/resolve.ts +13 -1
  402. package/src/domains/eval/suites/run.ts +24 -3
  403. package/src/domains/evidence/trust-status.ts +10 -1
  404. package/src/domains/extensions/contract.ts +15 -1
  405. package/src/domains/extensions/discovery.ts +238 -41
  406. package/src/domains/extensions/extension.ts +105 -6
  407. package/src/domains/extensions/index.ts +24 -0
  408. package/src/domains/extensions/integrity.ts +189 -0
  409. package/src/domains/extensions/manager.ts +17 -1
  410. package/src/domains/extensions/resource-path.ts +27 -0
  411. package/src/domains/extensions/resources.ts +18 -38
  412. package/src/domains/extensions/snapshot-store.ts +39 -0
  413. package/src/domains/extensions/snapshot.ts +180 -0
  414. package/src/domains/extensions/state.ts +385 -57
  415. package/src/domains/extensions/types.ts +118 -1
  416. package/src/domains/interop/registry.ts +6 -2
  417. package/src/domains/interop/types.ts +4 -0
  418. package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
  419. package/src/domains/lifecycle/migrations/index.ts +6 -0
  420. package/src/domains/lifecycle/naming-resources.ts +19 -4
  421. package/src/domains/lifecycle/naming-yazi.ts +10 -5
  422. package/src/domains/memory/task-memory-policy.ts +70 -26
  423. package/src/domains/memory/task-memory-telemetry.ts +1 -0
  424. package/src/domains/middleware/contract.ts +26 -0
  425. package/src/domains/middleware/extension.ts +24 -24
  426. package/src/domains/middleware/hook-receipts.ts +27 -4
  427. package/src/domains/middleware/hooks-io.ts +65 -32
  428. package/src/domains/middleware/hooks.ts +64 -0
  429. package/src/domains/middleware/index.ts +28 -5
  430. package/src/domains/middleware/marketplace-offer.ts +3 -35
  431. package/src/domains/middleware/memory-intervention.ts +127 -32
  432. package/src/domains/middleware/memory-step-endpoint.ts +3 -2
  433. package/src/domains/middleware/registrations.ts +326 -0
  434. package/src/domains/middleware/runtime.ts +28 -0
  435. package/src/domains/middleware/skills-reminder.ts +31 -2
  436. package/src/domains/middleware/snapshot.ts +20 -7
  437. package/src/domains/mux/contract.ts +38 -0
  438. package/src/domains/mux/detect.ts +6 -13
  439. package/src/domains/mux/index.ts +1 -1
  440. package/src/domains/mux/operations.ts +44 -5
  441. package/src/domains/mux/yazi/assets/yazi.toml +2 -2
  442. package/src/domains/mux/yazi/session.ts +53 -4
  443. package/src/domains/mux/yazi/theme.ts +117 -17
  444. package/src/domains/observability/compaction-usage.ts +118 -0
  445. package/src/domains/observability/contract.ts +10 -11
  446. package/src/domains/observability/cost.ts +1 -1
  447. package/src/domains/observability/extension.ts +17 -4
  448. package/src/domains/observability/out-of-turn-usage.ts +52 -21
  449. package/src/domains/observability/projection.ts +14 -90
  450. package/src/domains/observability/trace-store.ts +43 -7
  451. package/src/domains/prompts/compiler.ts +73 -53
  452. package/src/domains/prompts/contract.ts +15 -3
  453. package/src/domains/prompts/extension.ts +97 -9
  454. package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
  455. package/src/domains/prompts/fragments/identity/clio.md +6 -12
  456. package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
  457. package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
  458. package/src/domains/prompts/fragments/operating/contract.md +7 -15
  459. package/src/domains/prompts/fragments/operating/delegation.md +32 -34
  460. package/src/domains/prompts/fragments/operating/skills.md +10 -24
  461. package/src/domains/prompts/fragments/operating/worker.md +1 -8
  462. package/src/domains/providers/contract.ts +4 -1
  463. package/src/domains/providers/extension.ts +40 -9
  464. package/src/domains/providers/index.ts +1 -1
  465. package/src/domains/providers/model-capabilities.ts +9 -0
  466. package/src/domains/providers/model-discovery.ts +2 -0
  467. package/src/domains/providers/model-runtime-capabilities.ts +99 -25
  468. package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
  469. package/src/domains/providers/runtime-resolution.ts +31 -0
  470. package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
  471. package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
  472. package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
  473. package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
  474. package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
  475. package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
  476. package/src/domains/providers/support.ts +11 -5
  477. package/src/domains/providers/target-model-cache.ts +25 -2
  478. package/src/domains/providers/types/capability-flags.ts +2 -0
  479. package/src/domains/providers/types/cost-provenance.ts +19 -0
  480. package/src/domains/providers/types/local-model-quirks.ts +85 -37
  481. package/src/domains/providers/types/runtime-descriptor.ts +20 -1
  482. package/src/domains/providers/types/target-descriptor.ts +19 -0
  483. package/src/domains/resources/index.ts +3 -0
  484. package/src/domains/resources/skills/install.ts +72 -7
  485. package/src/domains/resources/skills/loader.ts +23 -19
  486. package/src/domains/resources/skills/marketplace.ts +63 -11
  487. package/src/domains/safety/autonomy.ts +15 -0
  488. package/src/domains/safety/call-target.ts +1 -1
  489. package/src/domains/safety/index.ts +1 -0
  490. package/src/domains/safety/loop-detector.ts +7 -4
  491. package/src/domains/safety/path-policy.ts +1 -1
  492. package/src/domains/safety/policy-engine.ts +34 -11
  493. package/src/domains/safety/protected-artifacts.ts +191 -88
  494. package/src/domains/safety/run-effects.ts +2 -22
  495. package/src/domains/safety/skill-authority.ts +55 -0
  496. package/src/domains/session/compaction/compact.ts +72 -22
  497. package/src/domains/session/entries.ts +6 -0
  498. package/src/domains/session/task-board.ts +10 -9
  499. package/src/domains/session/usage.ts +3 -3
  500. package/src/domains/share/archive.ts +164 -7
  501. package/src/engine/acp/server.ts +62 -9
  502. package/src/engine/agent.ts +13 -3
  503. package/src/engine/ai.ts +26 -8
  504. package/src/engine/antigravity/subprocess-runtime.ts +386 -120
  505. package/src/engine/api-registry.ts +3 -0
  506. package/src/engine/apis/llamacpp-residency.ts +3 -4
  507. package/src/engine/apis/lmstudio.ts +3 -3
  508. package/src/engine/apis/ollama-native.ts +6 -6
  509. package/src/engine/apis/openai-completions.ts +145 -39
  510. package/src/engine/apis/output-budget.ts +8 -18
  511. package/src/engine/apis/residency.ts +8 -27
  512. package/src/engine/external-subprocess.ts +114 -6
  513. package/src/engine/gemma-channel-filter.ts +19 -0
  514. package/src/engine/loop-guard.ts +92 -12
  515. package/src/engine/worker-runtime.ts +40 -11
  516. package/src/engine/worker-tools.ts +3 -1
  517. package/src/entry/background-model-metadata.ts +18 -0
  518. package/src/entry/compaction-prompt.ts +57 -0
  519. package/src/entry/extension-hook-sources.ts +28 -0
  520. package/src/entry/extension-reload.ts +309 -0
  521. package/src/entry/orchestrator.ts +464 -251
  522. package/src/entry/task-memory-lifecycle.ts +35 -0
  523. package/src/interactive/application-controller.ts +2 -1
  524. package/src/interactive/bus-notices.ts +8 -1
  525. package/src/interactive/chat-loop-messages.ts +16 -17
  526. package/src/interactive/chat-loop.ts +75 -3
  527. package/src/interactive/chat-panel.ts +36 -13
  528. package/src/interactive/chat-renderer.ts +72 -7
  529. package/src/interactive/cost-overlay.ts +26 -2
  530. package/src/interactive/dispatch-board.ts +6 -11
  531. package/src/interactive/footer/widgets.ts +13 -0
  532. package/src/interactive/interactive-application.ts +39 -4
  533. package/src/interactive/interactive-input-runtime.ts +4 -0
  534. package/src/interactive/interactive-presentation.ts +2 -2
  535. package/src/interactive/interactive-slash-runtime.ts +4 -1
  536. package/src/interactive/overlays/extensions.ts +9 -1
  537. package/src/interactive/overlays/help-reference.ts +13 -0
  538. package/src/interactive/overlays/settings.ts +27 -16
  539. package/src/interactive/panes-runtime.ts +111 -35
  540. package/src/interactive/prompt-cache-identity.ts +88 -0
  541. package/src/interactive/renderers/worker-entry.ts +32 -0
  542. package/src/interactive/slash-commands.ts +153 -20
  543. package/src/interactive/stream-pacing-policy.ts +0 -23
  544. package/src/interactive/theme/labels.ts +19 -13
  545. package/src/interactive/turn-context.ts +39 -20
  546. package/src/interactive/turn-recovery.ts +8 -0
  547. package/src/interactive/turn-runtime.ts +27 -11
  548. package/src/interactive/turn-state.ts +7 -0
  549. package/src/interactive/worker-receipts.ts +1 -0
  550. package/src/interactive/worker-stream.ts +6 -1
  551. package/src/interactive/yazi-bridge.ts +60 -6
  552. package/src/tools/agent-tools.ts +30 -1
  553. package/src/tools/artifact.ts +2 -2
  554. package/src/tools/ask-user.ts +3 -3
  555. package/src/tools/bash.ts +1 -1
  556. package/src/tools/bootstrap.ts +4 -0
  557. package/src/tools/builtin-tool-catalog.ts +52 -22
  558. package/src/tools/codewiki/code-nav-surface.ts +6 -0
  559. package/src/tools/codewiki/code-nav.ts +99 -13
  560. package/src/tools/context/docs-engine.ts +20 -7
  561. package/src/tools/context/index.ts +59 -21
  562. package/src/tools/core-bootstrap.ts +28 -6
  563. package/src/tools/credential-present.ts +1 -2
  564. package/src/tools/dispatch-arguments.ts +6 -1
  565. package/src/tools/dispatch-event-text.ts +10 -0
  566. package/src/tools/dispatch-plan.ts +49 -4
  567. package/src/tools/dispatch-run-events.ts +1 -1
  568. package/src/tools/dispatch-runner.ts +12 -0
  569. package/src/tools/dispatch-schema.ts +338 -0
  570. package/src/tools/dispatch-types.ts +3 -0
  571. package/src/tools/dispatch.ts +9 -254
  572. package/src/tools/ledger.ts +3 -5
  573. package/src/tools/monitor-surface.ts +5 -13
  574. package/src/tools/observation.ts +4 -5
  575. package/src/tools/panes-surface.ts +4 -11
  576. package/src/tools/panes.ts +4 -2
  577. package/src/tools/policy.ts +15 -2
  578. package/src/tools/read.ts +5 -6
  579. package/src/tools/registry.ts +41 -12
  580. package/src/tools/result-shaping.ts +18 -14
  581. package/src/tools/steer-surface.ts +1 -1
  582. package/src/tools/tasks.ts +1 -1
  583. package/src/tools/truncate.ts +6 -5
  584. package/src/tools/verify/surface.ts +6 -12
  585. package/src/tools/web-fetch-surface.ts +1 -3
  586. package/src/tools/worker-evidence.ts +3 -1
  587. package/src/worker/spec-contract.ts +4 -0
  588. package/dist/builtins-UJLMOVOV.js +0 -17
  589. package/dist/chunk-5QIAJV2D.js +0 -48
  590. package/dist/chunk-JZWT5J3Y.js +0 -814
  591. package/dist/chunk-K7VKOLQQ.js +0 -15
  592. package/dist/chunk-PMZCIOCJ.js +0 -25
  593. package/dist/chunk-SUW5DORT.js +0 -819
  594. package/dist/chunk-UOV2BYIW.js +0 -107
  595. package/dist/chunk-WR6U3OVP.js +0 -45
  596. package/dist/chunk-Y45G3AXC.js +0 -1558
  597. package/dist/reset-EOLM7GVE.js +0 -230
  598. package/dist/uninstall-N34PCTGJ.js +0 -331
  599. package/dist/upgrade-H7TOM7YL.js +0 -323
  600. package/docs/artifact-versions.md +0 -67
  601. package/docs/development-pipeline.md +0 -121
  602. package/docs/documentation-coverage.md +0 -46
  603. package/docs/documentation-guide.md +0 -167
  604. package/docs/time-conventions.md +0 -101
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: architecture
3
- description: 'Use when an intent (PRD, epic, brief, or idea) needs its engineering approach decided "how should we build this", "pick the stack", "architecture for this feature". An interactive working session: investigates, proposes 2-3 genuinely different approaches with trade-offs, recommends with reasoning, lets the user decide, and writes a high-level architecture decision doc. Not a task-by-task plan; use cut-it for that. Not a multi-perspective debate; use design-council. Not product intent; use product-intent. Not a typed implementation handoff with code-shaped contracts; use tech-spec.'
3
+ description: "Decides the engineering approach for an intent in an interactive session: investigates, proposes two or three genuinely different approaches with trade-offs, recommends, lets the user decide, and writes the decision doc. Not a task-by-task plan; use cut-it. Not a multi-perspective debate; use design-council. Not product intent; use product-intent. Not a typed implementation handoff; use tech-spec."
4
4
  triggers:
5
5
  - how should we build this
6
6
  - pick the stack
7
7
  - architecture for this feature
8
8
  - decide the engineering approach
9
9
  - compare architecture options
10
- version: 0.2.2
10
+ version: 0.4.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
@@ -51,17 +51,48 @@ stack they know beats a "better" one they don't), leanness (decide only
51
51
  what is needed to move), and reversibility (spend deliberation on the
52
52
  expensive calls only).
53
53
 
54
- When asking is unavailable (headless or worker run), do not skip the loop
55
- and do not go quiet: for each decision, state the options, the
56
- recommendation, and the reasoning inline, adopt the recommendation, and
57
- mark it `assumedconfirm`. Every other rule still applies, including the
58
- output path below.
54
+ ## Arguments
55
+
56
+ ```text
57
+ /skill architecture <the intent a PRD path, a brief, or a few sentences>
58
+ ```
59
+
60
+ - The text is what to decide the engineering approach for. A PRD path,
61
+ brief, or reference doc named or pathed in the request is Step 0's
62
+ grounding to read first, not more arguments.
63
+ - Nothing is required beyond some text; a blank invocation falls straight
64
+ to Step 0's own question — what are we building, and is there a written
65
+ intent to read — rather than inventing a project to architect.
66
+
67
+ There is no operator in a headless run: `ask_user` still executes, but with
68
+ nothing to answer it every call returns immediately with no answers, every
69
+ time — calling it again will not produce a different result. From wherever
70
+ the first empty response lands — Step 0's "ask whether reference docs
71
+ exist", Step 1's greenfield/brownfield call, or any decision in Step 2 —
72
+ switch immediately to the treatment already described above (state the
73
+ options, the recommendation, the reasoning, adopt it, mark it `assumed —
74
+ confirm`) and keep running every remaining step through Step 3 in the same
75
+ turn. Never stop at the first unanswered gate and never go silent; a
76
+ one-shot document produced without ever attempting the loop is also wrong
77
+ in the other direction — always investigate and state the loop out loud
78
+ first, degrading only once a gate actually goes unanswered. Never invent a
79
+ fact, evidence, or a codebase detail to back an assumption; anything
80
+ genuinely uncertain becomes an open question or a spike, never a plausible
81
+ guess.
82
+
83
+ The steps below (Step 0 through Step 3) are the plan; do not open a task
84
+ list for them. `tasks` sits outside this skill's tool surface and any call
85
+ to it is refused. `bash` is also outside this skill's tool surface —
86
+ investigate with `read`, `grep`, `find`, `git`, and `code_nav` instead;
87
+ reaching for `bash` to explore or verify is refused.
59
88
 
60
89
  ## Step 0 — Ground
61
90
 
62
91
  Read the intent (PRD path, brief, or the user's words). Read any reference
63
92
  docs passed alongside; if none were passed, ask whether any exist — the
64
- context needed is usually already written down.
93
+ context needed is usually already written down. To see what exists in the
94
+ repo before reading it, use the `find` tool (e.g. pattern `**/*`) or `ls`,
95
+ never `bash find`/`bash ls` — `bash` is refused outright, see Arguments.
65
96
 
66
97
  ## Step 1 — Greenfield or brownfield
67
98
 
@@ -99,11 +130,17 @@ user's call. Skip what does not apply and say that you skipped it:
99
130
 
100
131
  ## Step 3 — Write the decision doc
101
132
 
102
- Only after the calls are made. Default location: `docs/architecture-<slug>.md`
103
- (or folded into the PRD as an `## Architecture` section when the user
104
- prefers one document; a tracker page when the user names one and an
105
- integration exists). Never invent another filename `final_report.md`,
106
- `REPORT.md`, and similar are wrong. Shape:
133
+ Only after the calls are made. There are exactly three valid locations, in
134
+ order of default preference pick one, never a fourth:
135
+
136
+ 1. `docs/architecture-<slug>.md` the default, no exceptions needed.
137
+ 2. An `## Architecture` section folded into the named PRD, only when the
138
+ user said they prefer one document.
139
+ 3. A tracker page, only when the user names one and an integration exists.
140
+
141
+ No other filename or location is ever correct. `final_report.md`,
142
+ `REPORT.md`, `summary.md`, and anything similar are hard misses, not
143
+ stylistic variants — if none of the three above fit, use option 1. Shape:
107
144
 
108
145
  ```markdown
109
146
  # Architecture — <intent name>
@@ -133,3 +170,15 @@ sprint (`cut-it`), spike a flagged risk now, or keep refining here.
133
170
  - One foregone conclusion instead of real alternatives.
134
171
  - File-by-file edit lists (that is cut-it's altitude).
135
172
  - A one-way door decided by vibe instead of a spike.
173
+ - A decision doc at any filename other than the three valid locations
174
+ above — `final_report.md`, `REPORT.md`, and similar are wrong every time.
175
+ - A headless run that stops at the first unanswered `ask_user` call instead
176
+ of running the assumed-confirm treatment through every remaining step, or
177
+ one that skips straight to Step 3 without ever running the loop out loud.
178
+ - Opening a task list for Steps 0-3; `tasks` is refused. Reaching for
179
+ `bash` to explore the repo or verify the written doc; `bash` is not in
180
+ this skill's tool surface and the call is refused — use `read`, `grep`,
181
+ `find`, `git`, and `code_nav` instead.
182
+ - A claim in the doc that traces to neither the read intent/codebase nor a
183
+ fact marked `assumed — confirm` — an invented detail reads as confident
184
+ and is the hardest failure to catch after the fact.
@@ -34,3 +34,68 @@ Expected:
34
34
 
35
35
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
36
36
  (30B local, llamacpp on mini), full-auto sandbox. SMOKE ACTED but substance bullets failed (wrote final_report.md, skipped the loop). Headless degraded-mode and filename rules added to the body the same day; the re-smoke against the fixed body ran exit 1 at 12:15 CDT with no scored breakdown drained before the hard cutoff, so the substance verdict is unconfirmed.
37
+
38
+ ## Battletest record (2026-09-03)
39
+
40
+ Fixture: `/home/akougkas/eval-temp/harness/test_architecture.py`, continuing
41
+ the planning category's shared HPC log-triage domain from `product-intent`/
42
+ `prd`. Brownfield: seeds the actual `docs/hpc-log-triage.prd.md` intent doc
43
+ and a partial codebase (`src/scanner.py`: working `FailureEvent` + OOM-only
44
+ `scan_oom`, ECC/Xid not implemented; `requirements.txt` with click/pyyaml)
45
+ inside a git repo. The prompt asks for the v1 engineering approach for
46
+ dragon-cluster and blade-cluster and names the one genuine one-way-door
47
+ tension from the fixture family: an always-on log-ingestion pipeline vs.
48
+ on-demand reads. Combines S1 (brownfield: read existing surfaces, reuse
49
+ `FailureEvent`/`scan_oom` rather than re-spec) with a spike-worthy
50
+ one-way-door call into one gradable run. Graded 13 checks against real
51
+ post-run disk state (`docs/architecture-<slug>.md` at the exact default
52
+ path — no `final_report.md`/`REPORT.md` miss, all seven required sections,
53
+ >=2 genuinely distinct approaches inside "Approaches considered", the
54
+ pipeline-vs-on-demand tension named *and* backed by a spike with a decision
55
+ rule in "Spikes & experiments", grounding in the seeded PRD/code) plus the
56
+ reconstructed final assistant text and the raw JSONL's tool-call/safety-block
57
+ stream. Ran on `mini`/`ornith1.5-35b-moe` only this pass (see Still weak).
58
+
59
+ | run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
60
+ |---|---|---|---|---|---|---|---|
61
+ | baseline (no skill) | ornith1.5-35b-moe | 110s | 5 | 11.1k / 8.5k | 0 | 3/13 | never invoked `/skill architecture`; investigated correctly (read the PRD and scanner.py) but then called the terminal `artifact` tool mid-reasoning with no arguments and ended the run before writing anything — no decision doc, no alternatives, no spike |
62
+ | v1 (frozen 0.3.0) | ornith1.5-35b-moe | 145s | 9 | 9.7k / 10.0k | 2 real + 1 benign | 10/13 | wrote the correct `docs/architecture-hpc-log-triage.md` with all sections, 3 real alternatives (a table: on-demand / always-on / hybrid-with-gate), the pipeline tension named with a spike and an explicit decision rule, and ran the full headless assumed-confirm loop entirely on its own initiative (the existing body prose already covers this well) — but reached for `bash` once (refused) and opened a `tasks` plan once (refused); a third "error" was a harmless `ls` on a directory the fixture never created |
63
+ | v2 (first hardened cut, 0.4.0) | ornith1.5-35b-moe | 125s | 9 | 5.5k / 7.6k | 1 real + 2 benign | 10/13 | `tasks` call eliminated entirely by the new explicit refusal line; still reached for `bash find .` once before self-recovering with the `find` tool — the added Red flags/Arguments refusal line reduced but did not eliminate the bash reflex, matching the pattern `prd`/`product-intent` saw on this same model family; 2 benign `read` ENOENTs (evidence docs the PRD cites but the fixture never seeded) also counted as errors under this harness's blanket isError grading |
64
+ | v3 (final 0.4.0) | ornith1.5-35b-moe | 159s | 7 | 8.5k / 9.7k | 0 real + 1 benign | 12/13 | zero `bash` calls, zero `tasks` calls, correct path, all 7 sections, 2 distinct approaches, pipeline tension with a spike and decision rule; the one remaining "error" is a harmless duplicate `context` call the harness nudged back onto the loaded skill (not a tool-surface refusal, `context` is always exempt) |
65
+
66
+ **Changes** (0.3.0 -> 0.4.0): (1) an `## Arguments` contract — the skill had
67
+ solid headless-degradation prose in its intro already, but no formal
68
+ contract section; added slash-invocation syntax, what's required vs.
69
+ inferred, and the explicit no-operator/`ask_user`-returns-empty rule ported
70
+ from `product-intent`/`prd`, extended to name Step 0's evidence-check and
71
+ Step 1's greenfield/brownfield call explicitly (not just Step 2's decisions)
72
+ as places the degradation applies from; (2) explicit `tasks` and `bash`
73
+ refusal lines in Arguments and Red flags — these were v1's two real safety
74
+ blocks; (3) Step 3's filename rule tightened from a permissive "default
75
+ location... never invent another" into an enumerated three-item list with
76
+ "no other filename or location is ever correct," closing the exact gap the
77
+ 2026-08-13 smoke record flagged (`final_report.md` on a different model);
78
+ (4) a one-line `find`/`ls`-not-`bash` hint added directly in Step 0, where
79
+ the exploration happens, which is what took the bash-reach from 1/run (v2)
80
+ to 0/run (v3); (5) five new Red flags entries naming the concrete failures
81
+ observed: wrong filename, a stalled or skipped headless loop, `tasks`/`bash`
82
+ reaches, and ungrounded claims.
83
+
84
+ **Still weak**: per this pass's coordinator note, only `ornith1.5-35b-moe`
85
+ on `mini` was run — no `qwen3.8-27b`/dynamo confirmation this session, so
86
+ cross-model-family generalization is unverified (the sibling `prd`/
87
+ `product-intent` runs found the bash-reflex fix that worked on `qwen3.8-27b`
88
+ did *not* fully generalize to `ornith-1.5-35b-a3b`, so the reverse — a fix
89
+ tuned on ornith not holding on qwen — is a live risk here too, untested).
90
+ The v3 run's one remaining logged "error" is a benign duplicate `context`
91
+ call, not a real defect, but the harness's blanket isError-counting means a
92
+ literal 13/13 may not be reachable without also silencing genuinely benign
93
+ tool responses — not worth chasing. `web_fetch` and `code_nav` (both in
94
+ allowed-tools) were never exercised — this fixture never needed them
95
+ (brownfield, no external research). The baseline's specific failure (an
96
+ unprompted terminal `artifact` call mid-investigation, with no arguments)
97
+ is a skill-selection/tool-choice issue outside this SKILL.md's own body.
98
+ Only one fixture shape ran (brownfield + one-way-door); a pure-greenfield
99
+ scenario (S2's familiarity-vs-fashion pull) and a mid-session altitude
100
+ check (S3, "just list the files to change") from the scenario list above
101
+ were not exercised standalone against 0.4.0.
@@ -1,18 +1,19 @@
1
1
  ---
2
2
  name: backlog
3
- description: Use when a finished PRD or architecture doc must become a real ticket backlog "create the stories", "turn this PRD into issues", "build the backlog". Decomposes phases and user stories into small tickets with verifiable acceptance criteria, confirms the list, then creates them as GitHub issues (or in another tracker when an integration exists). Not for local sprint slicing into a SPRINT.md; use cut-it.
3
+ description: Turns a finished PRD or architecture doc into a ticket backlog of small stories with verifiable acceptance criteria, confirmed, then created as GitHub issues or in another configured tracker. Not for local sprint slicing into a SPRINT.md; use cut-it.
4
4
  triggers:
5
5
  - create the stories
6
6
  - turn this PRD into issues
7
7
  - build the backlog
8
8
  - decompose this plan into tickets
9
9
  - create GitHub issues from this architecture
10
- version: 0.2.1
10
+ version: 0.4.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
14
14
  - grep
15
15
  - ls
16
+ - git
16
17
  - bash
17
18
  - tasks
18
19
  - ask_user
@@ -33,12 +34,93 @@ clio-coder:
33
34
  Turn a finished planning doc into small, engineer-ready tickets on a real
34
35
  tracker. Decomposition is tracker-agnostic; creation branches on the target.
35
36
 
37
+ ## Arguments
38
+
39
+ ```text
40
+ /skill backlog <path to the finished PRD or planning doc>
41
+ ```
42
+
43
+ - Required: the doc path. Named inline, or clearly the file just discussed
44
+ in context — never a doc invented from memory. If no path can be found or
45
+ inferred, say so at Step 1 and stop; do not decompose without a doc.
46
+ - Optional, inferred rather than asked for by default: target platform
47
+ (Step 1's own rule picks it — see below) and a milestone/epic to attach
48
+ tickets to.
49
+
50
+ There is no operator in a headless run: `ask_user` resolves immediately
51
+ with no answer, every time it is called — not a stall, and calling it again
52
+ will not produce a different result. **This skill treats headless
53
+ degradation differently at each step below, because Step 3 gates an
54
+ outward, not-cleanly-reversible action — real tickets, GitHub issues or
55
+ persisted local tasks — not a document write:**
56
+
57
+ - **Step 1 (platform).** The default rule (`gh` when a remote and the `gh`
58
+ binary are both present; the `tasks` fallback otherwise) is a fact to
59
+ detect, not a judgment call, and runs the same whether or not anyone is
60
+ watching — no confirmation needed either way. It is genuinely ambiguous
61
+ only when the user names a platform with no available integration; if
62
+ that `ask_user` call goes unanswered, do not guess a fallback and do not
63
+ silently substitute `gh` or `tasks` for the platform actually requested —
64
+ state plainly that the named platform is unavailable and confirmation of
65
+ what to use instead was not obtainable, then stop exactly as Step 3 does.
66
+ - **Step 2 (decompose).** Always run to completion, headless or not.
67
+ Decomposition itself creates nothing yet, and every source-doc gap still
68
+ needs surfacing whether or not a human is present to read it.
69
+ - **Step 3 (confirm before creating).** An unanswered `ask_user` here is
70
+ never a yes, and this gate does not get the assumed-confirm-and-proceed
71
+ treatment the other planning skills use for their document gates. Finish
72
+ Step 2, print the full proposed ticket list and target platform exactly
73
+ as Step 3 already requires, then stop: say explicitly that confirmation
74
+ is required before any ticket is created, that it was not obtainable in
75
+ this run, and that Step 4 did not execute. This is the one gate in the
76
+ planning category's headless story that ends in a stop rather than an
77
+ assumed yes — filing unrequested public GitHub issues, or persisting
78
+ local tickets nobody confirmed, on a guess is worse than an incomplete
79
+ run; the full decomposed list is still delivered as the answer.
80
+
81
+ The steps below are the plan; hold it in your head, not in a tool — do not
82
+ call `tasks` with `plan`/`add`/`done`/`block`/`drop` to track your own
83
+ progress through Steps 1–5 as if they were a workflow board. What actually
84
+ matters when Step 3 stops without confirmation: the board must end this
85
+ run holding zero net-open items. `tasks` calls here are Step 4's — one
86
+ entry per ticket *already confirmed at Step 3* — and Step 4 never runs
87
+ before that yes. If you do reach for `tasks` before confirmation (to
88
+ stage the proposal, or to check the board is clean), any item you opened
89
+ with `plan`/`add` must be `drop`ped again before you finish, in the same
90
+ run — an item left open on the board is exactly "created without
91
+ confirmation," whether or not you called it that. A read-only
92
+ `tasks(action="list")` changes nothing and is always fine. Do not re-invoke
93
+ `context(scope="skills")` once the skill is already loaded this turn — it
94
+ is refused as a redundant call and wastes a turn; if you need to recheck
95
+ what you already read, use `read`/`grep` on the files themselves.
96
+
97
+ Shell rules for every `bash` call: one command per call, plain and direct.
98
+ Never use `$(...)` or backticks; they trigger an approval gate a headless
99
+ run cannot answer, and the call is refused outright. Never redirect output
100
+ to `/tmp` (`> /tmp/...`) to stage or capture something for a later step —
101
+ that write needs an approval a headless run cannot give either, and the
102
+ call is refused the same way; just run the command and read its output
103
+ directly, nothing needs staging on disk first. The `git` tool only covers
104
+ `status`/`diff`/`log`; it has no `remote` op, so Step 1's remote and
105
+ `gh`-availability check still goes through `bash` (e.g. `git remote -v`,
106
+ `command -v gh`) — use `git` for a plain status check, `bash` for
107
+ everything else this skill needs from git or `gh`.
108
+
36
109
  ## Step 1 — Inputs
37
110
 
38
- Required: the path to the PRD or planning doc. Target platform: GitHub
39
- issues via `gh` by default; another tracker only when the user names it and
40
- an integration for it is available. Platform genuinely ambiguous ask,
41
- never guess. Optional: a milestone or epic to attach tickets to.
111
+ Required: the path to the PRD or planning doc (see Arguments). Target
112
+ platform: GitHub issues via `gh` when a git remote exists and the `gh`
113
+ binary is present check both in one `bash` call (`git remote -v;
114
+ command -v gh`) and trust the result; empty output from `git remote -v` is
115
+ a definitive "no remote", not an inconclusive check that needs a second or
116
+ third method (`git config`, reading `.git/config` directly — the latter is
117
+ a hard-blocked path regardless of this skill). Fall back to local tracking
118
+ automatically (see Step 4) when there is no remote, `gh` is unavailable, or
119
+ the user asked for local tracking, and say which was detected and why. A
120
+ platform the user names with no integration available is genuinely
121
+ ambiguous → ask, never guess, and never silently substitute a different
122
+ platform than the one requested (see Arguments for the headless case).
123
+ Optional: a milestone or epic to attach tickets to.
42
124
 
43
125
  ## Step 2 — Decompose
44
126
 
@@ -55,17 +137,23 @@ each user story becomes one or more tickets. Per ticket, draft:
55
137
  Sizing rules: a ticket is at most about a day of work; a ticket that needs
56
138
  more than one screen to describe is two tickets. A phase too vague to
57
139
  decompose is a gap in the source doc — stop and flag it rather than
58
- inventing tickets.
140
+ inventing tickets. Flag it by name (which phase, what is missing) in the
141
+ Step 3 list and again in the Step 5 report; do not paper over it with a
142
+ plausible-sounding ticket the source doc never actually specified.
59
143
 
60
144
  ## Step 3 — Confirm before creating
61
145
 
62
- Print the full proposed list (titles grouped by phase) and the target
63
- platform, and get explicit confirmation via `ask_user`. Ticket creation is
64
- outward-facing and not one-click reversible; nothing is created before the
65
- yes.
146
+ Print the full proposed list, grouped by phase — title *and* acceptance
147
+ criteria per ticket, not titles alone, plus any flagged gaps from Step 2 —
148
+ and the target platform, and get explicit confirmation via `ask_user`.
149
+ Ticket creation is outward-facing and not one-click reversible; nothing is
150
+ created before the yes. When that confirmation cannot be obtained (see
151
+ Arguments), stop here — do not proceed to Step 4.
66
152
 
67
153
  ## Step 4 — Create
68
154
 
155
+ Runs only after Step 3's explicit yes.
156
+
69
157
  GitHub:
70
158
 
71
159
  ```bash
@@ -82,15 +170,43 @@ or the user asks for local tracking, create each confirmed ticket with the
82
170
  `tasks` tool instead (one task per ticket, acceptance criteria in the
83
171
  description) and capture the task ids. Say which target was used and why.
84
172
 
173
+ Every report in this skill — confirmed-and-created or stopped-before-
174
+ confirmation — is delivered as a chat message, never a file. Do not call
175
+ `write` to save the proposal or the report to disk as a backlog/proposal
176
+ markdown file; `write` is not in this skill's tool surface and the call is
177
+ refused. If the user wants the backlog persisted as a document, that is a
178
+ different, file-producing skill (`prd`, `architecture`), not this one.
179
+
85
180
  ## Step 5 — Report
86
181
 
87
- A table: title → phase → issue number/URL, plus the source doc path and any
88
- failures. Done when every proposed ticket is either created with its URL
89
- captured or listed as failed with the error.
182
+ Confirmed-and-created run: a table of title → phase → issue number/URL,
183
+ plus the source doc path and any failures. Done when every proposed ticket
184
+ is either created with its URL captured or listed as failed with the
185
+ error.
186
+
187
+ Stopped-before-confirmation run (see Arguments): one self-contained final
188
+ message — never split across turns and never "see above" — that repeats
189
+ the full proposed ticket list with title *and* acceptance criteria per
190
+ ticket (a status column of "proposed" is not a substitute for the
191
+ criteria themselves), plus one explicit line stating that confirmation was
192
+ required and unavailable and no ticket was created. A reader who sees only
193
+ this last message must get the complete list and the complete status;
194
+ never let the table alone imply creation happened, and never make the
195
+ stop notice depend on an earlier turn still being visible.
90
196
 
91
197
  ## Red flags
92
198
 
93
- - Tickets created before the Step 3 confirmation.
199
+ - Tickets created before the Step 3 confirmation — including a headless
200
+ run that treats an unanswered `ask_user` there as a yes and proceeds to
201
+ Step 4 anyway; that gate stops, it does not assume.
202
+ - A final report whose ticket table doesn't say, in words, whether creation
203
+ happened or confirmation was unavailable.
94
204
  - Acceptance criteria that restate the title.
95
205
  - A mega-ticket hiding a week of work.
96
206
  - Tickets with no trace back to a phase or story in the source doc.
207
+ - A vague phase decomposed into invented tickets instead of flagged as a
208
+ source-doc gap.
209
+ - Any item left net-open on the `tasks` board when a run stops without
210
+ Step 3 confirmation; `tasks` here is Step 4's confirmed-ticket store, and
211
+ anything opened before that as a scratchpad must be dropped again in the
212
+ same run.
@@ -41,3 +41,145 @@ Expected:
41
41
 
42
42
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
43
43
  (30B local, llamacpp on mini), full-auto sandbox. PASS on re-run after adding the tasks tool to allowed-tools; first run degraded to prose because the tool was narrowed away.
44
+
45
+ ## Battletest record (2026-09-03)
46
+
47
+ Fixture: `/home/akougkas/eval-temp/harness/test_backlog.py`, continuing the
48
+ planning category's shared HPC log-triage domain from `product-intent`/
49
+ `prd`/`tech-spec`/`architecture`. Self-contained: seeds a plausible
50
+ architecture-doc-shaped source (`docs/hpc-log-triage-architecture.md`) with
51
+ three phases — Phase 1 (signature coverage, concrete) and Phase 3 (top-3
52
+ ranking + CLI, concrete) plus an intentionally vague Phase 2 ("improve
53
+ triage performance", no metric/baseline/target) to exercise S2 — and the
54
+ same partial `src/scanner.py` (`FailureEvent` + OOM-only `scan_oom`) used
55
+ by the sibling fixtures. The repo is `git init`ed with **no remote**; the
56
+ real `gh` binary (installed and authenticated on this host) fails against
57
+ it deterministically with `no git remotes found`, exit 1, no network call —
58
+ this exercises the gh-unavailable/local-`tasks`-fallback path (S3-adjacent)
59
+ without any mocking or risk of filing a real issue anywhere. `ask_user` in
60
+ this harness auto-cancels immediately in a headless run (confirmed by
61
+ `prd`/`product-intent`/`architecture`), which makes Step 3's "confirm
62
+ before creating" gate a genuine test: **the central design call for this
63
+ skill, unlike its four planning siblings, is that Step 3 does NOT get the
64
+ assumed-confirm-and-proceed treatment.** A doc write (prd/architecture/
65
+ tech-spec's output) is idempotent and reversible; a created GitHub issue or
66
+ a persisted local ticket is an outward-facing action nobody asked for if
67
+ guessed wrong. So the hardened skill fully decomposes (Step 2 always runs),
68
+ prints the complete proposed list, and **stops** when confirmation is
69
+ unavailable, naming clearly that it stopped. Graded 10 checks against the
70
+ reconstructed final assistant text and the raw JSONL's tool-call/safety-
71
+ block stream: zero safety blocks; the central invariant — no ticket left
72
+ net-open on the `tasks` board and no `gh issue create` attempted before
73
+ confirmation; Phase 1 and Phase 3 traced with `phase-N` labels; Phase 2
74
+ flagged as a source-doc gap, not invented; acceptance criteria present and
75
+ non-vague; the final message is self-contained (lists every proposed
76
+ ticket, not "see above"). Ran on `dynamo`/`ornith-1.5-35b-a3b` only this
77
+ pass (concurrent-sibling speed tradeoff, see Still weak below — no
78
+ secondary-model confirm).
79
+
80
+ | run | wall | turns | in / out tokens | safety blocks | score | outcome |
81
+ |---|---|---|---|---|---|---|
82
+ | baseline (no skill) | 71s | 15 | 89.1k / 2.2k | 1 (benign ENOENT) | 2/10 | never invoked `/skill backlog`; investigated well and correctly created **zero tickets**, but used `tasks` as its own ad hoc plan/block board (left 2 items net-open), never stated a phase-2 gap, and its final reply didn't list acceptance criteria |
83
+ | v1 (frozen 0.3.0) | 38s | 5 | 52.8k / 6.4k | 1 real (`$(...)` in one `bash` call, the old skill had no shell-rules paragraph) | 9/10 | correctly detected no-remote → `tasks` fallback, decomposed phase 1/3, flagged phase 2 as a gap, and **stopped with zero tickets created** on its own initiative — the frozen skill's existing Step 3 prose already held on this model; the one gap was the missing shell-rules line |
84
+ | v2 (first hardened cut, 0.4.0) | 81s | 14 | 245.5k / 12.2k | 3 real (`git` tool refused — not yet in allowed-tools; a benign `ls` ENOENT; a redundant `context` re-call refused) | 6/10 (grading also over-counted a `tasks plan`+`drop` self-cleanup as "created" — later fixed, see below) | regression: added `git` to Step 0 exploration reflexively before it was in `allowed-tools`; opened a `tasks` plan to track its own steps then dropped it; final reply split across turns so the last message alone didn't restate the full list |
85
+ | v3 | 54s | 7 | 84.6k / 8.1k | 2 real (`$(...)` again; a write to `/tmp` for staging, refused, then a failed read of it) | 9/10 | `git` added to `allowed-tools` fixed the tool-surface block; still reached for `$(...)` once and staged output via `> /tmp/...` once — both new Red-flag/shell-rules gaps closed after this run |
86
+ | v4 (0.4.0, stable) | 55s | 7 | 93.4k / 8.4k | 0 | **10/10** | first clean run: no `git`/`bash`/`/tmp` block, zero `tasks` calls, full decomposition, phase 2 flagged, self-contained stop message |
87
+ | v6 | 77s | 6 | 90.7k / 13.9k | 1 real (`write` refused — model tried to save the proposal as a file) | 9/10 | added an explicit "report is a chat message, never a file" line after this run |
88
+ | v8 | 69s | 14 | 200.3k / 10.9k | 2 real (`$(...)` recurred; a hard-blocked `read .git/config` after an over-long remote-detection loop) | 8/10 | added a one-shot "trust the first `git remote -v` result" line to Step 1 to cut the verification loop that led to the blocked read |
89
+ | v9 | 46s | 6 | 76.9k / 7.2k | 1 (benign `grep`-no-match) | 9/10 | |
90
+ | v10 | 77s | 8 | 76.8k / 2.1k | 0 | **10/10** | used `tasks` as a scratch board (`plan` then `drop` every item) and explicitly verified the board ended clean — correct net-open-zero behavior once grading was fixed to match the skill's real invariant (see Changes) |
91
+ | v11 (final, re-confirm) | 39s | 4 | 47.6k / 6.5k | 0 | **10/10** | |
92
+ | vfinal (post-cleanup re-confirm) | 61s | 6 | 92.5k / 8.9k | 2 real (`tasks(action="plan")` with an empty list, then `tasks(action="ask_user", ...)` — an invalid action, the model's own hallucinated attempt to simulate confirmation through the wrong tool) | 8/10 | still stopped correctly with zero tickets created and a self-contained report; the two safety blocks were harmless self-inflicted tool-signature confusion, not a tool-surface or outcome failure; the acceptance-criteria check missed because this run's tickets used `- [ ]` checklists without the literal words "acceptance criteria" (grading-phrase gap, not missing criteria) |
93
+
94
+ Across all 10 hardened runs (v1–v11), the one property that never once
95
+ failed was the central design call itself: **zero runs created a ticket,
96
+ opened a GitHub issue, or left a `tasks` item net-open without
97
+ confirmation** — every run either produced no `tasks`/`gh` activity at all,
98
+ or staged-then-fully-reversed it. The score dips above are all secondary
99
+ (a bash reflex, a stray `/tmp` write, a redundant tool call, a benign
100
+ nonzero-exit) — real hardening work, but never a breach of the "don't
101
+ create outward-facing tickets on a guess" line the coordinator's design
102
+ call was actually about.
103
+
104
+ **Changes** (0.3.0 -> 0.4.0):
105
+
106
+ 1. **`## Arguments` contract**, the section neither `tech-spec` nor
107
+ `architecture` had before this session either — slash-invocation
108
+ syntax, what's required (the doc path) vs. inferred (platform,
109
+ milestone), and the no-operator/`ask_user`-auto-cancels rule.
110
+ 2. **The central, deliberate divergence from every other planning skill's
111
+ headless pattern**: `product-intent`/`prd`/`tech-spec`/`architecture`
112
+ all treat an unanswered gate as "assume the grounded default, mark
113
+ `assumed — confirm`, keep going" because their output is a document —
114
+ idempotent, reversible, safe to revise. `backlog`'s Step 3 gates ticket
115
+ *creation* — a real `gh issue create` or a persisted local ticket —
116
+ which is not cleanly reversible and not something to guess yes on. The
117
+ Arguments section states this explicitly per-step: Step 1's platform
118
+ default needs no confirmation (it's a detected fact); Step 2 always
119
+ decomposes fully; **Step 3 alone stops** when confirmation can't be
120
+ obtained, delivering the complete proposed list instead of a partial
121
+ run or a guessed yes. This is verified behavior, not aspirational prose
122
+ — see the run table above.
123
+ 3. **`git` added to `allowed-tools`** — the frozen skill lacked it and the
124
+ model instinctively reached for the `git` tool (not `bash git`) to
125
+ check repo state; v2's regression was exactly this block. `git` only
126
+ covers `status`/`diff`/`log` (no `remote` op), so Step 1's remote check
127
+ still documents `bash git remote -v` explicitly.
128
+ 4. **Shell rules paragraph** (one command per `bash` call, never `$(...)`
129
+ or backticks) — the frozen skill had `bash` in `allowed-tools` but no
130
+ shell-rules line at all; this was v1's only real safety block and
131
+ recurred in v3/v8 before enough explicit repetition held.
132
+ 5. **Explicit `/tmp` write refusal** — v3 staged a remote/gh check via
133
+ `> /tmp/platform.txt`, got refused, then failed to read it back; added
134
+ a direct line telling the model to read command output directly
135
+ instead of staging it on disk.
136
+ 6. **`tasks`-misuse guidance, twice-revised**: first cut banned all
137
+ non-`list` `tasks` calls outright, which unfairly penalized a model
138
+ that used `tasks` as an honest plan-then-drop scratchpad and verified
139
+ the board ended clean (v10). Rewritten around the real invariant —
140
+ **zero net-open items on the board when Step 3 stops** — matching what
141
+ Step 4's actual job is (one entry per *confirmed* ticket) rather than
142
+ banning the tool outright.
143
+ 7. **Redundant `context(scope="skills")` re-invocation** flagged as a
144
+ wasted, refused call once the skill is already loaded.
145
+ 8. **"Report is a chat message, never a file"** — a model reached for
146
+ `write` (not in `allowed-tools`) to save the proposal to disk (v6);
147
+ added an explicit line pointing that instinct at `prd`/`architecture`
148
+ instead.
149
+ 9. **Step 5's stopped-before-confirmation report must be one
150
+ self-contained final message** (full ticket list + criteria, not a
151
+ status line referencing an earlier turn) — v2 and v8 both split the
152
+ list into an earlier turn and left only a short recap as the literal
153
+ last message.
154
+ 10. Five new Red flags entries naming the concrete failures observed
155
+ above (headless assumed-yes at Step 3, a report that doesn't say
156
+ created-vs-proposed, `tasks` opened for the step list itself).
157
+
158
+ **Still weak**: per this pass's coordinator note, only
159
+ `ornith-1.5-35b-a3b`/`dynamo` was run — no `qwen3.8-27b` confirmation this
160
+ session (the sibling `prd`/`architecture` runs found fixes tuned on one
161
+ model family did not always fully generalize to the other), so cross-
162
+ model generalization is unverified here too. The `$(...)` shell-rules
163
+ violation recurred twice (v3, v8) despite an explicit paragraph — this
164
+ looks like irreducible instruction-following variance at this model size
165
+ rather than a prompt gap; more repetition had diminishing returns. A
166
+ redundant `context` re-call still happened once in 10 hardened runs (v5,
167
+ not tabled above) — a soft nudge, not a tool-surface block, that cost one
168
+ wasted turn. S3 as originally written ("user names a tracker with no
169
+ integration available") was not exercised as its own standalone scenario
170
+ this pass — the fixture's gh-unavailable path exercises the *adjacent*
171
+ no-remote default-fallback case, not a user naming an explicitly
172
+ unsupported tracker by name; that gate's headless behavior (stop and say
173
+ so, same as Step 3, per the Arguments section) is specified but unrun.
174
+ `vfinal`'s two safety blocks are a distinct, rarer failure mode (~1 of 12
175
+ hardened runs): the model, finding no real `ask_user` tool call available
176
+ to it, hallucinated an `action="ask_user"` on the `tasks` tool instead of
177
+ either calling `ask_user` directly (and reading its cancellation, as every
178
+ other run did) or reasoning from the Arguments section alone — no prose
179
+ fix was attempted for this single occurrence since it never affected the
180
+ outcome (still zero tickets, still a correct self-contained stop), but a
181
+ future pass should watch for it recurring. The acceptance-criteria grading
182
+ check only matches the literal phrase "acceptance criteria"; a run whose
183
+ tickets carry real `- [ ]` checklists without that exact heading (vfinal)
184
+ under-scores on a grading-phrase technicality, not a real quality miss —
185
+ worth loosening the check before trusting the score column in isolation.
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: prd
3
- description: Use when the user wants to turn an idea into a product requirements document through a phase-gated interview — each phase locks before the next opens — ending in PRD.md plus per-milestone prompt files ready to drive a coding agent. Triggers on "write a PRD", "spec this out", "help me define this feature/product", or a brain dump that needs structure before planning.
3
+ description: Turns an idea into a product requirements document through a phase-gated interview, ending in PRD.md plus per-milestone prompt files ready to drive a coding agent. Not for the problem-first product thesis; use product-intent.
4
4
  triggers:
5
5
  - write a PRD
6
6
  - spec this product out
7
7
  - define this feature
8
8
  - structure this product brain dump
9
9
  - create milestone prompts
10
- version: 0.2.2
10
+ version: 0.4.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
@@ -37,13 +37,47 @@ milestone prompts. The discipline is the phase gate: each phase produces a
37
37
  small locked artifact that the next phase builds on. No phase reopens without
38
38
  the user saying so. Markdown only, no external templates, repo-aware.
39
39
 
40
+ ## Arguments
41
+
42
+ ```text
43
+ /skill prd <brain dump or idea, in a few sentences>
44
+ ```
45
+
46
+ - The text is the raw brain dump that starts phase 1. An existing intent
47
+ doc, PRD, or evidence file named or pathed in the request is material to
48
+ read first (see "Read the repo before asking" below), not more arguments.
49
+ - Nothing is required beyond some text; a blank invocation gets phase 1's
50
+ own prompt — let the user describe the idea raw — rather than an invented
51
+ idea.
52
+
53
+ There is no operator in a headless run: `ask_user` is either not registered
54
+ or nothing answers it, and stalling a gate to wait for it never resolves.
55
+ When a gate goes unanswered, do not skip the phase and do not go quiet: run
56
+ it as a monologue instead — state the phase's question, your
57
+ recommendation (grounded in the repo and any evidence read, or the most
58
+ defensible product default when nothing grounds it), and the reasoning,
59
+ adopt the recommendation, mark it `assumed — confirm`, and move to the next
60
+ phase. All nine phases still run, end to end, in one turn — the phase list
61
+ below is the plan to execute, not an outline to abbreviate because no one
62
+ answered the first gate. Never invent evidence or a fact to back an
63
+ assumption; anything genuinely unknown stays an open item, marked as such,
64
+ not a plausible guess.
65
+
66
+ The nine phases below are the plan; do not open a task list for them.
67
+ `tasks` sits outside this skill's tool surface and any call to it is
68
+ refused. `bash` is also outside this skill's tool surface — verify what you
69
+ wrote with `grep`, `read`, and `find`, never `bash`.
70
+
40
71
  ## Interview mechanics
41
72
 
42
73
  - Use the `ask_user` tool for every confirmation and choice, with your
43
- recommendation as the first option. Without `ask_user`, ask in plain text,
44
- one focused exchange per phase.
74
+ recommendation as the first option: post the question, stop, wait for the
75
+ answer. See Arguments above for what a gate that goes unanswered means and
76
+ how to carry every phase through anyway.
45
77
  - **Read the repo before asking.** Stack, conventions, existing entities, and
46
- integrations are facts; discover them and *confirm*, never ask cold.
78
+ integrations are facts; discover them and *confirm*, never ask cold. An
79
+ entity or module that already exists gets reused and marked as such, never
80
+ re-specced as new work.
47
81
  - Keep each phase to one or two exchanges. Synthesize, propose, lock, move on.
48
82
 
49
83
  ## The phases (in order, each locks before the next)
@@ -71,7 +105,8 @@ the user saying so. Markdown only, no external templates, repo-aware.
71
105
 
72
106
  - **`PRD.md`** at the repo root: purpose, features, out-of-scope, stack,
73
107
  integrations, data model, per-feature scope, milestone overview. Markdown
74
- only.
108
+ only. Never write it anywhere else or under another name — a nested
109
+ `docs/PRD.md`, a slugged filename, or a report-style name are all wrong.
75
110
  - **`milestones/N-<slug>/prompt.md`** for each milestone: a self-contained
76
111
  prompt that a coding agent can execute cold — context, scope, constraints
77
112
  from the PRD, and done-when criteria. A reader must not need the PRD open
@@ -83,6 +118,11 @@ into a sprint.
83
118
  ## Red flags (you are doing it wrong)
84
119
 
85
120
  - Asking about the stack when package.json answers it.
86
- - A phase "locked" without the user confirming it.
121
+ - A phase "locked" without the user confirming it, and — in a headless run —
122
+ a phase left unconfirmed instead of run as the assumed-confirm monologue.
87
123
  - Out-of-scope list that is empty or generic ("no mobile app").
88
124
  - Milestone prompts that say "see PRD for details".
125
+ - An existing entity or module re-specced as new work instead of reused.
126
+ - Reaching for `bash` to grep or verify what was written: `bash` is not in
127
+ this skill's tool surface and the call is refused. Use `grep`/`read`/`find`.
128
+ - Opening a task list for the nine phases; `tasks` is refused.