@iowarp/clio-coder 0.4.1 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (604) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/CONTRIBUTING.md +142 -52
  3. package/README.md +434 -473
  4. package/SECURITY.md +2 -1
  5. package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
  6. package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
  7. package/dist/assets/codewiki.json +1 -1
  8. package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
  9. package/dist/builtins-IA7V7FUC.js +22 -0
  10. package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
  11. package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
  12. package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
  13. package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
  14. package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
  15. package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
  16. package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
  17. package/dist/chunk-3EBYEESD.js +314 -0
  18. package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
  19. package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
  20. package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
  21. package/dist/chunk-4UVU7BJ5.js +39 -0
  22. package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
  23. package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
  24. package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
  25. package/dist/chunk-5MEZN6CB.js +1334 -0
  26. package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
  27. package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
  28. package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
  29. package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
  30. package/dist/chunk-7DRAWPTZ.js +360 -0
  31. package/dist/chunk-7E7I3WLS.js +3762 -0
  32. package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
  33. package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
  34. package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
  35. package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
  36. package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
  37. package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
  38. package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
  39. package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
  40. package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
  41. package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
  42. package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
  43. package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
  44. package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
  45. package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
  46. package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
  47. package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
  48. package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
  49. package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
  50. package/dist/chunk-GYV6VZOC.js +26 -0
  51. package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
  52. package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
  53. package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
  54. package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
  55. package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
  56. package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
  57. package/dist/chunk-IJNZMHLA.js +101 -0
  58. package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
  59. package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
  60. package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
  61. package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
  62. package/dist/chunk-JEQ3XTHC.js +42 -0
  63. package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
  64. package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
  65. package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
  66. package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
  67. package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
  68. package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
  69. package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
  70. package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
  71. package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
  72. package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
  73. package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
  74. package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
  75. package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
  76. package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
  77. package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
  78. package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
  79. package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
  80. package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
  81. package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
  82. package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
  83. package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
  84. package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
  85. package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
  86. package/dist/chunk-R6Q67RJH.js +134 -0
  87. package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
  88. package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
  89. package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
  90. package/dist/chunk-SKHCAU7K.js +385 -0
  91. package/dist/chunk-SZAA6XDG.js +30 -0
  92. package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
  93. package/dist/chunk-UOIZ7DA4.js +41 -0
  94. package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
  95. package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
  96. package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
  97. package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
  98. package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
  99. package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
  100. package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
  101. package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
  102. package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
  103. package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
  104. package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
  105. package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
  106. package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
  107. package/dist/chunk-XRZT5WY5.js +47 -0
  108. package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
  109. package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
  110. package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
  111. package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
  112. package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
  113. package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
  114. package/dist/cli/index.js +41 -39
  115. package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
  116. package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
  117. package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
  118. package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
  119. package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
  120. package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
  121. package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
  122. package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
  123. package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
  124. package/dist/context-map-COB37XXN.js +505 -0
  125. package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
  126. package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
  127. package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
  128. package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
  129. package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
  130. package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
  131. package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
  132. package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
  133. package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
  134. package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
  135. package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
  136. package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
  137. package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
  138. package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
  139. package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
  140. package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
  141. package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
  142. package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
  143. package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
  144. package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
  145. package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
  146. package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
  147. package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
  148. package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
  149. package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
  150. package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
  151. package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
  152. package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
  153. package/dist/reset-EAJFFJVB.js +344 -0
  154. package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
  155. package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
  156. package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
  157. package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
  158. package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
  159. package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
  160. package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
  161. package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
  162. package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
  163. package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
  164. package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
  165. package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
  166. package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
  167. package/dist/uninstall-5PEVOE5B.js +408 -0
  168. package/dist/upgrade-M4WXY6KN.js +303 -0
  169. package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
  170. package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
  171. package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
  172. package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
  173. package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
  174. package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
  175. package/dist/worker/entry.js +90 -74
  176. package/docs/README.md +176 -81
  177. package/docs/{acp.md → architecture/acp.md} +36 -20
  178. package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
  179. package/docs/{architecture.md → architecture/architecture.md} +43 -22
  180. package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
  181. package/docs/architecture/artifact-versions.md +90 -0
  182. package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
  183. package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
  184. package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
  185. package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
  186. package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
  187. package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
  188. package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
  189. package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
  190. package/docs/{observability.md → architecture/observability.md} +38 -14
  191. package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
  192. package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
  193. package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
  194. package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
  195. package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
  196. package/docs/architecture/time-conventions.md +125 -0
  197. package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
  198. package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
  199. package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
  200. package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
  201. package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
  202. package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
  203. package/docs/guide/configuration-reference.md +1163 -0
  204. package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
  205. package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
  206. package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
  207. package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
  208. package/docs/{glossary.md → guide/glossary.md} +14 -11
  209. package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
  210. package/docs/guide/panes-and-files.md +290 -0
  211. package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
  212. package/docs/{resource-library.md → guide/resource-library.md} +13 -4
  213. package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
  214. package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
  215. package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
  216. package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
  217. package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
  218. package/docs/process/development-pipeline.md +152 -0
  219. package/docs/process/documentation-coverage.md +100 -0
  220. package/docs/process/documentation-guide.md +187 -0
  221. package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
  222. package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
  223. package/docs/{evolution.md → process/evolution.md} +2 -2
  224. package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
  225. package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
  226. package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
  227. package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
  228. package/evals/README.md +2 -2
  229. package/evals/behavioral-model.yaml +3 -2
  230. package/package.json +10 -8
  231. package/skills/README.md +52 -41
  232. package/skills/coding/ast-grep/SKILL.md +102 -31
  233. package/skills/coding/ast-grep/evals.md +26 -0
  234. package/skills/coding/coding-standards/SKILL.md +41 -6
  235. package/skills/coding/coding-standards/evals.md +23 -0
  236. package/skills/coding/prototype/SKILL.md +88 -29
  237. package/skills/coding/prototype/evals.md +19 -0
  238. package/skills/coding/tdd/SKILL.md +81 -54
  239. package/skills/coding/tdd/evals.md +20 -0
  240. package/skills/context/context-handoff/SKILL.md +44 -3
  241. package/skills/context/context-handoff/evals.md +44 -0
  242. package/skills/context/context-prime/SKILL.md +46 -16
  243. package/skills/context/context-prime/evals.md +45 -0
  244. package/skills/git/branch-closeout/SKILL.md +132 -0
  245. package/skills/git/branch-closeout/evals.md +133 -0
  246. package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
  247. package/skills/git/file-ticket/SKILL.md +78 -64
  248. package/skills/git/file-ticket/assets/issue-template.md +22 -0
  249. package/skills/git/file-ticket/evals.md +31 -26
  250. package/skills/git/file-ticket/references/issue-discovery.md +49 -0
  251. package/skills/git/fix-issue/SKILL.md +88 -65
  252. package/skills/git/fix-issue/evals.md +35 -31
  253. package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
  254. package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
  255. package/skills/git/resolve-merge-conflicts/evals.md +52 -25
  256. package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
  257. package/skills/git/ship/SKILL.md +103 -67
  258. package/skills/git/ship/assets/pr-template.md +21 -0
  259. package/skills/git/ship/evals.md +44 -28
  260. package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
  261. package/skills/git/worktree-create/SKILL.md +80 -50
  262. package/skills/git/worktree-create/evals.md +40 -33
  263. package/skills/git/worktree-create/references/worktree-setup.md +62 -66
  264. package/skills/git/worktree-merge/SKILL.md +112 -65
  265. package/skills/git/worktree-merge/evals.md +42 -34
  266. package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
  267. package/skills/meta/clio-coder-dev/SKILL.md +9 -5
  268. package/skills/meta/clio-coder-dev/evals.md +3 -2
  269. package/skills/meta/clio-coder-test/SKILL.md +102 -95
  270. package/skills/meta/clio-coder-test/evals.md +9 -4
  271. package/skills/meta/clio-coder-test/references/harness.md +100 -124
  272. package/skills/meta/clio-coder-test/references/test-map.md +77 -50
  273. package/skills/meta/credentials/SKILL.md +2 -2
  274. package/skills/meta/find-skills/SKILL.md +2 -2
  275. package/skills/meta/herdr/SKILL.md +2 -2
  276. package/skills/meta/skill-craft/SKILL.md +22 -16
  277. package/skills/planning/archify/SKILL.md +196 -0
  278. package/skills/planning/archify/evals.md +65 -0
  279. package/skills/planning/architecture/SKILL.md +62 -13
  280. package/skills/planning/architecture/evals.md +65 -0
  281. package/skills/planning/backlog/SKILL.md +131 -15
  282. package/skills/planning/backlog/evals.md +142 -0
  283. package/skills/planning/prd/SKILL.md +47 -7
  284. package/skills/planning/prd/evals.md +54 -0
  285. package/skills/planning/product-intent/SKILL.md +58 -3
  286. package/skills/planning/product-intent/evals.md +70 -0
  287. package/skills/planning/tech-spec/SKILL.md +54 -3
  288. package/skills/planning/tech-spec/evals.md +73 -0
  289. package/skills/registry.yaml +70 -62
  290. package/skills/remote.yaml +13 -0
  291. package/skills/research/arxiv-literature/SKILL.md +77 -19
  292. package/skills/research/arxiv-literature/evals.md +50 -0
  293. package/skills/research/experiment-protocol/SKILL.md +21 -2
  294. package/skills/research/experiment-protocol/evals.md +23 -0
  295. package/skills/research/scientific-debugging/SKILL.md +24 -2
  296. package/skills/research/scientific-debugging/evals.md +18 -0
  297. package/skills/research/scientific-modernization/SKILL.md +27 -2
  298. package/skills/research/scientific-modernization/evals.md +27 -0
  299. package/skills/skill-marketplace.json +97 -62
  300. package/skills/workflow/cut-it/SKILL.md +66 -6
  301. package/skills/workflow/cut-it/evals.md +101 -0
  302. package/skills/workflow/design-council/SKILL.md +118 -28
  303. package/skills/workflow/design-council/evals.md +161 -0
  304. package/skills/workflow/grill-me/SKILL.md +87 -11
  305. package/skills/workflow/grill-me/evals.md +153 -0
  306. package/skills/workflow/workflow-distiller/SKILL.md +77 -18
  307. package/skills/workflow/workflow-distiller/evals.md +118 -0
  308. package/src/cli/args.ts +2 -2
  309. package/src/cli/bootstrap-generate.ts +1 -1
  310. package/src/cli/config-inspect.ts +65 -12
  311. package/src/cli/configure-interop.ts +105 -13
  312. package/src/cli/configure-oauth.ts +57 -0
  313. package/src/cli/configure-onboarding.ts +980 -0
  314. package/src/cli/configure-target.ts +594 -0
  315. package/src/cli/configure.ts +1082 -532
  316. package/src/cli/context-map.ts +114 -0
  317. package/src/cli/context.ts +4 -0
  318. package/src/cli/docs.ts +22 -14
  319. package/src/cli/doctor-naming.ts +5 -5
  320. package/src/cli/doctor-toolchain.ts +3 -3
  321. package/src/cli/eval.ts +1 -2
  322. package/src/cli/extensions.ts +2 -1
  323. package/src/cli/fleet.ts +1 -1
  324. package/src/cli/index.ts +3 -1
  325. package/src/cli/internal-dispatch.ts +3 -4
  326. package/src/cli/lifecycle-presenter.ts +436 -0
  327. package/src/cli/models.ts +10 -2
  328. package/src/cli/modes/print.ts +5 -1
  329. package/src/cli/panes.ts +19 -5
  330. package/src/cli/reset.ts +228 -106
  331. package/src/cli/run.ts +9 -4
  332. package/src/cli/select.ts +664 -0
  333. package/src/cli/share.ts +5 -1
  334. package/src/cli/skills-eval.ts +3 -3
  335. package/src/cli/skills.ts +9 -2
  336. package/src/cli/targets.ts +5 -6
  337. package/src/cli/trace.ts +55 -4
  338. package/src/cli/uninstall.ts +233 -165
  339. package/src/cli/upgrade.ts +204 -149
  340. package/src/cli/usage.ts +86 -27
  341. package/src/cli/validate-model.ts +3 -3
  342. package/src/cli/wiki-generate.ts +1 -1
  343. package/src/core/artifact-paths.ts +1 -1
  344. package/src/core/bash-exec.ts +131 -86
  345. package/src/core/bus-events.ts +51 -6
  346. package/src/core/config.ts +61 -1
  347. package/src/core/defaults.ts +7 -4
  348. package/src/core/dispatch-outcome.ts +16 -0
  349. package/src/core/external-diagnostic.ts +44 -0
  350. package/src/core/gateway-routing.ts +157 -0
  351. package/src/core/guardrails.ts +10 -49
  352. package/src/core/prompt-hint.ts +9 -0
  353. package/src/core/safe-exec.ts +17 -2
  354. package/src/core/skill-activation.ts +89 -2
  355. package/src/domains/agents/builtins/architect.md +2 -3
  356. package/src/domains/agents/builtins/coder.md +3 -2
  357. package/src/domains/agents/builtins/debugger.md +2 -2
  358. package/src/domains/agents/builtins/documenter.md +2 -2
  359. package/src/domains/agents/builtins/git-master.md +1 -1
  360. package/src/domains/agents/builtins/oracle.md +1 -1
  361. package/src/domains/agents/builtins/provenance.md +1 -1
  362. package/src/domains/agents/builtins/researcher.md +1 -1
  363. package/src/domains/agents/builtins/scout.md +1 -1
  364. package/src/domains/agents/builtins/tester.md +2 -2
  365. package/src/domains/agents/builtins/verifier.md +2 -2
  366. package/src/domains/agents/builtins/wiki-writer.md +1 -1
  367. package/src/domains/agents/builtins/world-knowledge.md +31 -0
  368. package/src/domains/agents/catalog.ts +13 -15
  369. package/src/domains/agents/contract.ts +2 -0
  370. package/src/domains/agents/extension.ts +23 -1
  371. package/src/domains/agents/result-contract.ts +70 -0
  372. package/src/domains/config/keybindings.ts +8 -0
  373. package/src/domains/context/extension.ts +0 -3
  374. package/src/domains/context/wiki/map-seed.ts +589 -0
  375. package/src/domains/context/wiki/plan.ts +2 -2
  376. package/src/domains/context/working-set/path-index.ts +1 -0
  377. package/src/domains/dispatch/admission.ts +29 -0
  378. package/src/domains/dispatch/agent-candidates.ts +10 -0
  379. package/src/domains/dispatch/budget-envelope.ts +86 -1
  380. package/src/domains/dispatch/capability-match.ts +11 -0
  381. package/src/domains/dispatch/capacity-lease.ts +17 -0
  382. package/src/domains/dispatch/contract.ts +11 -1
  383. package/src/domains/dispatch/extension.ts +237 -49
  384. package/src/domains/dispatch/host-verification.ts +435 -39
  385. package/src/domains/dispatch/intent-requirements.ts +10 -0
  386. package/src/domains/dispatch/intent.ts +18 -1
  387. package/src/domains/dispatch/path-scope.ts +235 -24
  388. package/src/domains/dispatch/run-event-journal.ts +4 -15
  389. package/src/domains/dispatch/state.ts +2 -3
  390. package/src/domains/dispatch/transport.ts +45 -21
  391. package/src/domains/dispatch/types.ts +58 -3
  392. package/src/domains/dispatch/worker-model-metadata.ts +38 -0
  393. package/src/domains/eval/artifacts/store.ts +5 -0
  394. package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
  395. package/src/domains/eval/metrics/token-stream.ts +201 -31
  396. package/src/domains/eval/metrics/tracked.ts +40 -4
  397. package/src/domains/eval/runners/clio-run.ts +5 -2
  398. package/src/domains/eval/schema/suite.ts +28 -0
  399. package/src/domains/eval/schema/verdict.ts +2 -2
  400. package/src/domains/eval/store.ts +8 -1
  401. package/src/domains/eval/suites/resolve.ts +13 -1
  402. package/src/domains/eval/suites/run.ts +24 -3
  403. package/src/domains/evidence/trust-status.ts +10 -1
  404. package/src/domains/extensions/contract.ts +15 -1
  405. package/src/domains/extensions/discovery.ts +238 -41
  406. package/src/domains/extensions/extension.ts +105 -6
  407. package/src/domains/extensions/index.ts +24 -0
  408. package/src/domains/extensions/integrity.ts +189 -0
  409. package/src/domains/extensions/manager.ts +17 -1
  410. package/src/domains/extensions/resource-path.ts +27 -0
  411. package/src/domains/extensions/resources.ts +18 -38
  412. package/src/domains/extensions/snapshot-store.ts +39 -0
  413. package/src/domains/extensions/snapshot.ts +180 -0
  414. package/src/domains/extensions/state.ts +385 -57
  415. package/src/domains/extensions/types.ts +118 -1
  416. package/src/domains/interop/registry.ts +6 -2
  417. package/src/domains/interop/types.ts +4 -0
  418. package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
  419. package/src/domains/lifecycle/migrations/index.ts +6 -0
  420. package/src/domains/lifecycle/naming-resources.ts +19 -4
  421. package/src/domains/lifecycle/naming-yazi.ts +10 -5
  422. package/src/domains/memory/task-memory-policy.ts +70 -26
  423. package/src/domains/memory/task-memory-telemetry.ts +1 -0
  424. package/src/domains/middleware/contract.ts +26 -0
  425. package/src/domains/middleware/extension.ts +24 -24
  426. package/src/domains/middleware/hook-receipts.ts +27 -4
  427. package/src/domains/middleware/hooks-io.ts +65 -32
  428. package/src/domains/middleware/hooks.ts +64 -0
  429. package/src/domains/middleware/index.ts +28 -5
  430. package/src/domains/middleware/marketplace-offer.ts +3 -35
  431. package/src/domains/middleware/memory-intervention.ts +127 -32
  432. package/src/domains/middleware/memory-step-endpoint.ts +3 -2
  433. package/src/domains/middleware/registrations.ts +326 -0
  434. package/src/domains/middleware/runtime.ts +28 -0
  435. package/src/domains/middleware/skills-reminder.ts +31 -2
  436. package/src/domains/middleware/snapshot.ts +20 -7
  437. package/src/domains/mux/contract.ts +38 -0
  438. package/src/domains/mux/detect.ts +6 -13
  439. package/src/domains/mux/index.ts +1 -1
  440. package/src/domains/mux/operations.ts +44 -5
  441. package/src/domains/mux/yazi/assets/yazi.toml +2 -2
  442. package/src/domains/mux/yazi/session.ts +53 -4
  443. package/src/domains/mux/yazi/theme.ts +117 -17
  444. package/src/domains/observability/compaction-usage.ts +118 -0
  445. package/src/domains/observability/contract.ts +10 -11
  446. package/src/domains/observability/cost.ts +1 -1
  447. package/src/domains/observability/extension.ts +17 -4
  448. package/src/domains/observability/out-of-turn-usage.ts +52 -21
  449. package/src/domains/observability/projection.ts +14 -90
  450. package/src/domains/observability/trace-store.ts +43 -7
  451. package/src/domains/prompts/compiler.ts +73 -53
  452. package/src/domains/prompts/contract.ts +15 -3
  453. package/src/domains/prompts/extension.ts +97 -9
  454. package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
  455. package/src/domains/prompts/fragments/identity/clio.md +6 -12
  456. package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
  457. package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
  458. package/src/domains/prompts/fragments/operating/contract.md +7 -15
  459. package/src/domains/prompts/fragments/operating/delegation.md +32 -34
  460. package/src/domains/prompts/fragments/operating/skills.md +10 -24
  461. package/src/domains/prompts/fragments/operating/worker.md +1 -8
  462. package/src/domains/providers/contract.ts +4 -1
  463. package/src/domains/providers/extension.ts +40 -9
  464. package/src/domains/providers/index.ts +1 -1
  465. package/src/domains/providers/model-capabilities.ts +9 -0
  466. package/src/domains/providers/model-discovery.ts +2 -0
  467. package/src/domains/providers/model-runtime-capabilities.ts +99 -25
  468. package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
  469. package/src/domains/providers/runtime-resolution.ts +31 -0
  470. package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
  471. package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
  472. package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
  473. package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
  474. package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
  475. package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
  476. package/src/domains/providers/support.ts +11 -5
  477. package/src/domains/providers/target-model-cache.ts +25 -2
  478. package/src/domains/providers/types/capability-flags.ts +2 -0
  479. package/src/domains/providers/types/cost-provenance.ts +19 -0
  480. package/src/domains/providers/types/local-model-quirks.ts +85 -37
  481. package/src/domains/providers/types/runtime-descriptor.ts +20 -1
  482. package/src/domains/providers/types/target-descriptor.ts +19 -0
  483. package/src/domains/resources/index.ts +3 -0
  484. package/src/domains/resources/skills/install.ts +72 -7
  485. package/src/domains/resources/skills/loader.ts +23 -19
  486. package/src/domains/resources/skills/marketplace.ts +63 -11
  487. package/src/domains/safety/autonomy.ts +15 -0
  488. package/src/domains/safety/call-target.ts +1 -1
  489. package/src/domains/safety/index.ts +1 -0
  490. package/src/domains/safety/loop-detector.ts +7 -4
  491. package/src/domains/safety/path-policy.ts +1 -1
  492. package/src/domains/safety/policy-engine.ts +34 -11
  493. package/src/domains/safety/protected-artifacts.ts +191 -88
  494. package/src/domains/safety/run-effects.ts +2 -22
  495. package/src/domains/safety/skill-authority.ts +55 -0
  496. package/src/domains/session/compaction/compact.ts +72 -22
  497. package/src/domains/session/entries.ts +6 -0
  498. package/src/domains/session/task-board.ts +10 -9
  499. package/src/domains/session/usage.ts +3 -3
  500. package/src/domains/share/archive.ts +164 -7
  501. package/src/engine/acp/server.ts +62 -9
  502. package/src/engine/agent.ts +13 -3
  503. package/src/engine/ai.ts +26 -8
  504. package/src/engine/antigravity/subprocess-runtime.ts +386 -120
  505. package/src/engine/api-registry.ts +3 -0
  506. package/src/engine/apis/llamacpp-residency.ts +3 -4
  507. package/src/engine/apis/lmstudio.ts +3 -3
  508. package/src/engine/apis/ollama-native.ts +6 -6
  509. package/src/engine/apis/openai-completions.ts +145 -39
  510. package/src/engine/apis/output-budget.ts +8 -18
  511. package/src/engine/apis/residency.ts +8 -27
  512. package/src/engine/external-subprocess.ts +114 -6
  513. package/src/engine/gemma-channel-filter.ts +19 -0
  514. package/src/engine/loop-guard.ts +92 -12
  515. package/src/engine/worker-runtime.ts +40 -11
  516. package/src/engine/worker-tools.ts +3 -1
  517. package/src/entry/background-model-metadata.ts +18 -0
  518. package/src/entry/compaction-prompt.ts +57 -0
  519. package/src/entry/extension-hook-sources.ts +28 -0
  520. package/src/entry/extension-reload.ts +309 -0
  521. package/src/entry/orchestrator.ts +464 -251
  522. package/src/entry/task-memory-lifecycle.ts +35 -0
  523. package/src/interactive/application-controller.ts +2 -1
  524. package/src/interactive/bus-notices.ts +8 -1
  525. package/src/interactive/chat-loop-messages.ts +16 -17
  526. package/src/interactive/chat-loop.ts +75 -3
  527. package/src/interactive/chat-panel.ts +36 -13
  528. package/src/interactive/chat-renderer.ts +72 -7
  529. package/src/interactive/cost-overlay.ts +26 -2
  530. package/src/interactive/dispatch-board.ts +6 -11
  531. package/src/interactive/footer/widgets.ts +13 -0
  532. package/src/interactive/interactive-application.ts +39 -4
  533. package/src/interactive/interactive-input-runtime.ts +4 -0
  534. package/src/interactive/interactive-presentation.ts +2 -2
  535. package/src/interactive/interactive-slash-runtime.ts +4 -1
  536. package/src/interactive/overlays/extensions.ts +9 -1
  537. package/src/interactive/overlays/help-reference.ts +13 -0
  538. package/src/interactive/overlays/settings.ts +27 -16
  539. package/src/interactive/panes-runtime.ts +111 -35
  540. package/src/interactive/prompt-cache-identity.ts +88 -0
  541. package/src/interactive/renderers/worker-entry.ts +32 -0
  542. package/src/interactive/slash-commands.ts +153 -20
  543. package/src/interactive/stream-pacing-policy.ts +0 -23
  544. package/src/interactive/theme/labels.ts +19 -13
  545. package/src/interactive/turn-context.ts +39 -20
  546. package/src/interactive/turn-recovery.ts +8 -0
  547. package/src/interactive/turn-runtime.ts +27 -11
  548. package/src/interactive/turn-state.ts +7 -0
  549. package/src/interactive/worker-receipts.ts +1 -0
  550. package/src/interactive/worker-stream.ts +6 -1
  551. package/src/interactive/yazi-bridge.ts +60 -6
  552. package/src/tools/agent-tools.ts +30 -1
  553. package/src/tools/artifact.ts +2 -2
  554. package/src/tools/ask-user.ts +3 -3
  555. package/src/tools/bash.ts +1 -1
  556. package/src/tools/bootstrap.ts +4 -0
  557. package/src/tools/builtin-tool-catalog.ts +52 -22
  558. package/src/tools/codewiki/code-nav-surface.ts +6 -0
  559. package/src/tools/codewiki/code-nav.ts +99 -13
  560. package/src/tools/context/docs-engine.ts +20 -7
  561. package/src/tools/context/index.ts +59 -21
  562. package/src/tools/core-bootstrap.ts +28 -6
  563. package/src/tools/credential-present.ts +1 -2
  564. package/src/tools/dispatch-arguments.ts +6 -1
  565. package/src/tools/dispatch-event-text.ts +10 -0
  566. package/src/tools/dispatch-plan.ts +49 -4
  567. package/src/tools/dispatch-run-events.ts +1 -1
  568. package/src/tools/dispatch-runner.ts +12 -0
  569. package/src/tools/dispatch-schema.ts +338 -0
  570. package/src/tools/dispatch-types.ts +3 -0
  571. package/src/tools/dispatch.ts +9 -254
  572. package/src/tools/ledger.ts +3 -5
  573. package/src/tools/monitor-surface.ts +5 -13
  574. package/src/tools/observation.ts +4 -5
  575. package/src/tools/panes-surface.ts +4 -11
  576. package/src/tools/panes.ts +4 -2
  577. package/src/tools/policy.ts +15 -2
  578. package/src/tools/read.ts +5 -6
  579. package/src/tools/registry.ts +41 -12
  580. package/src/tools/result-shaping.ts +18 -14
  581. package/src/tools/steer-surface.ts +1 -1
  582. package/src/tools/tasks.ts +1 -1
  583. package/src/tools/truncate.ts +6 -5
  584. package/src/tools/verify/surface.ts +6 -12
  585. package/src/tools/web-fetch-surface.ts +1 -3
  586. package/src/tools/worker-evidence.ts +3 -1
  587. package/src/worker/spec-contract.ts +4 -0
  588. package/dist/builtins-UJLMOVOV.js +0 -17
  589. package/dist/chunk-5QIAJV2D.js +0 -48
  590. package/dist/chunk-JZWT5J3Y.js +0 -814
  591. package/dist/chunk-K7VKOLQQ.js +0 -15
  592. package/dist/chunk-PMZCIOCJ.js +0 -25
  593. package/dist/chunk-SUW5DORT.js +0 -819
  594. package/dist/chunk-UOV2BYIW.js +0 -107
  595. package/dist/chunk-WR6U3OVP.js +0 -45
  596. package/dist/chunk-Y45G3AXC.js +0 -1558
  597. package/dist/reset-EOLM7GVE.js +0 -230
  598. package/dist/uninstall-N34PCTGJ.js +0 -331
  599. package/dist/upgrade-H7TOM7YL.js +0 -323
  600. package/docs/artifact-versions.md +0 -67
  601. package/docs/development-pipeline.md +0 -121
  602. package/docs/documentation-coverage.md +0 -46
  603. package/docs/documentation-guide.md +0 -167
  604. package/docs/time-conventions.md +0 -101
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: cut-it
3
- description: Use when a plan, PRD, or milestone must become an executable sprint dependency-ordered vertical slices sized for one focused agent run each, with done-when verification per slice. Never fabricates a plan; if none exists or it is too vague to slice, says so and recommends an interview first. Triggers on "cut it", "slice this plan", "make this executable", "turn this into a sprint".
3
+ description: Slices an existing plan, PRD, or milestone into an executable sprint of dependency-ordered vertical slices sized for one agent run each, with done-when verification per slice; never fabricates a plan. Not for deciding the approach; use architecture.
4
4
  triggers:
5
5
  - cut it
6
6
  - slice this plan
7
7
  - make this plan executable
8
8
  - turn this milestone into a sprint
9
9
  - write dependency-ordered vertical slices
10
- version: 0.2.3
10
+ version: 0.4.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
@@ -18,7 +18,6 @@ allowed-tools:
18
18
  - context
19
19
  - code_nav
20
20
  - write
21
- - artifact
22
21
  - ask_user
23
22
  clio-coder:
24
23
  registry-id: iowarp/clio-coder
@@ -39,6 +38,43 @@ can run one at a time, leaving the build green after every slice. The output
39
38
  is a `SPRINT.md` another agent can execute cold — no conversation context
40
39
  required.
41
40
 
41
+ ## Arguments
42
+
43
+ ```text
44
+ cut it [<path to plan>]
45
+ ```
46
+
47
+ There is no flag syntax; the trigger is conversational — "cut it", "slice
48
+ this plan", "turn this milestone into a sprint". A path the user names in
49
+ the same request (a specific `PLAN.md`, `PRD.md`, or `milestones/*/prompt.md`)
50
+ is the plan to slice; a path with no plan words near it, or a bare
51
+ destination like "write it to docs/SPRINT.md", is Step 3's output location,
52
+ not the input. When neither is named, Step 1 finds the plan and Step 3
53
+ writes to the repo-root default.
54
+
55
+ This run has no back-and-forth. `ask_user` still executes — it is registered
56
+ and the call succeeds — but nothing answers it in a headless run: every round
57
+ returns `{cancelled: true}` immediately, as an ordinary result, not an error.
58
+ The happy path here rarely needs a question at all — Step 1's "stop and say
59
+ so" for a missing or vague plan is already headless-safe, and Step 3's output
60
+ path defaults without asking. If several plans or milestones are plausible
61
+ candidates and the choice matters, do not stop on an open question: pick the
62
+ most recently modified or most specifically named one, state that choice and
63
+ the alternative you set aside, mark it `assumed — confirm`, and keep going in
64
+ the same turn. Do not end a turn on an unanswered question in `ask_user` or
65
+ in plain text.
66
+
67
+ The steps below are the plan; do not open a task list for them. This skill's
68
+ tool surface is exactly `read`, `grep`, `ls`, `find`, `git`, `context`,
69
+ `code_nav`, `write`, and `ask_user` (`context` and `ask_user` are always
70
+ available regardless). `tasks` and `bash` both sit outside it and any call to
71
+ either is refused — track progress by walking the steps below, not a task
72
+ board; check for a build/test/lint setup (a `package.json`, a Makefile, a
73
+ node/toolchain version) with `find`, `ls`, and `read`, not `bash node
74
+ --version` or `bash find`. The read-only `git` tool (`status`, `log`) is
75
+ useful in Step 1 when locating the plan benefits from recent history or
76
+ uncommitted changes.
77
+
42
78
  ## Step 1 — Locate the plan
43
79
 
44
80
  In priority order: a file the user names, a plan in the conversation,
@@ -60,10 +96,14 @@ slicing of a vague plan hides gaps; flagging them is the deliverable.
60
96
  - **Self-contained.** Real file paths, real commands, concrete steps. A reader
61
97
  with zero conversation context can execute it.
62
98
 
63
- ## Step 3 — Write the artifact
99
+ ## Step 3 — Write SPRINT.md
64
100
 
65
- Default output is `SPRINT.md` at the repo root (honor a caller-supplied path).
66
- Format:
101
+ Use the `write` tool. `SPRINT.md` is a plain file in the working tree, not a
102
+ generated report — do not call a tool literally named `artifact` for this;
103
+ that tool is not on this skill's surface and, in this harness, is a
104
+ terminal call that ends the run the instant it is invoked, before you can
105
+ report back. Default output path is `SPRINT.md` at the repo root; honor a
106
+ caller-supplied path from Arguments instead. Format:
67
107
 
68
108
  ```markdown
69
109
  # Sprint: <name>
@@ -86,9 +126,29 @@ Format:
86
126
  "Done when" is the contract, not decoration. If you cannot write a testable
87
127
  done-when for a slice, the slice is not ready to cut — go back to the plan.
88
128
 
129
+ ## Step 4 — Report back
130
+
131
+ End the turn with a short final reply, not silence after the write: the path
132
+ you wrote (`SPRINT.md` or the caller-supplied path), the number of slices,
133
+ and a one-line summary of the battle order. This is the only confirmation
134
+ the caller gets that the write actually happened.
135
+
89
136
  ## Red flags (you are doing it wrong)
90
137
 
91
138
  - A slice whose steps say "and related changes" or "etc."
92
139
  - Done-when criteria that restate the goal instead of naming a check.
93
140
  - A slice that only compiles when a later slice lands.
94
141
  - Slicing a plan you had to invent on the spot.
142
+ - Calling a tool literally named `artifact` because Step 3 talks about "the
143
+ artifact" — that word here means "the deliverable document," not the
144
+ `artifact` tool. That tool is off this skill's surface and, in this
145
+ harness, terminates the run on the spot, writes to
146
+ `.clio-coder/artifacts/` instead of `SPRINT.md`, and skips Step 4 entirely.
147
+ Use `write`.
148
+ - Ending the run right after the write with no final reply — Step 4 is not
149
+ optional.
150
+ - Opening a `tasks` list for the steps above; `tasks` is refused.
151
+ - Reaching for `bash` (`node --version`, `find`, `ls -la`, ...) to check the
152
+ repo's toolchain or structure; `bash` is refused, use `find`/`ls`/`read`.
153
+ - Stopping to wait for an `ask_user` reply that headless runs never send;
154
+ see Arguments.
@@ -40,3 +40,104 @@ Expected:
40
40
 
41
41
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
42
42
  (30B local, llamacpp on mini), full-auto sandbox. PASS. Sliced the seeded PLAN.md; judge 4/4.
43
+
44
+ ## Battletest record (2026-09-03)
45
+
46
+ Fixture: `/home/akougkas/eval-temp/harness/test_cutit.py`. S1 reuses this
47
+ evals.md's own fixture text verbatim (`src/todos.js` stub + a concrete
48
+ three-feature `PLAN.md`); S2 is an empty repo with only a `README.md`; S3 is
49
+ a `PLAN.md` that says only "improve performance and clean up the code."
50
+ S1 is the primary grading fixture, scored on 12 checks against the raw
51
+ JSONL's tool-call/safety-block stream, the actual `SPRINT.md` written to
52
+ disk, and the reconstructed final assistant text: zero safety blocks, zero
53
+ real `artifact` tool calls, zero `tasks` calls, `SPRINT.md` exists with a
54
+ `## Battle order` and 2+ numbered slices, every slice carries all six
55
+ required fields (Goal/Depends on/Files/Steps/Done when/Out of scope), no
56
+ horizontal-layering red-flag language, slices trace to the plan's concrete
57
+ features, done-when blocks are command-shaped and testable, and the final
58
+ reply names the path and slice count. S2/S3 are graded on 5 checks each
59
+ (zero safety blocks, zero `artifact` calls, no `SPRINT.md` fabricated,
60
+ correctly flags the gap, recommends a next step). Primary model
61
+ `qwen3.8-27b` on `dynamo` (LM Studio); cross-model confirm on
62
+ `ornith1.5-35b-moe` on `mini` (llama.cpp), run against S1 and S2 both.
63
+
64
+ **The bug found reading the source, confirmed empirically first**: the
65
+ frozen skill (0.3.0) had `artifact` in `allowed-tools` and titled Step 3
66
+ "Write the artifact." `src/tools/artifact.ts` sets `terminate: true` on
67
+ every successful call — the run ends the instant the tool executes, with
68
+ no further LLM turn to confirm what happened. v1's run called `artifact`
69
+ with `kind: "plan"` and, by luck, an explicit `path: "SPRINT.md"` (the
70
+ model inferred this from "honor a caller-supplied path" even though no
71
+ caller supplied one) — so the file landed in the right place this time, but
72
+ the run still ended mid-sentence ("Writing the sprint:") with no
73
+ confirmation reply, and a separate `tasks` call (also off-surface) drew a
74
+ real safety block. Score 8/12: missing `reply_mentions_sprint_path`, one
75
+ real safety block. Had the model not guessed an explicit path, the default
76
+ would have been `.clio-coder/artifacts/PLAN.md` (kind defaults to `plan`,
77
+ see `core/artifact-paths.ts`) — wrong file, wrong location, same silent
78
+ termination. This is exactly the failure mode the planning category's
79
+ tech-spec baseline hit ("called `artifact` for an early exit... instead of
80
+ a spec").
81
+
82
+ | run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
83
+ |---|---|---|---|---|---|---|---|
84
+ | baseline (no skill) | qwen3.8-27b | 76s | 7 | 73.8k / 7.1k | 5 (repeated `ls ".cl"` truncated-path retries, benign) | 5/12 | never invoked `/skill cut-it`; read the plan and module correctly, reasoned to genuinely good vertical slices with real done-when checks in its head, then hit a tool-call loop guard and delivered the entire sprint as **prose in the reply, never wrote `SPRINT.md`** — the exact gap this skill exists to close |
85
+ | v1 (frozen 0.3.0) | qwen3.8-27b | 115s | 6 | 68.6k / 11.3k | 1 real (`tasks` refused) | 8/12 | called the real `artifact` tool for Step 3 as titled; terminated the turn immediately after writing, mid-sentence, with no confirmation reply — the artifact-tool bug, confirmed |
86
+ | v2 (first hardened cut, 0.4.0) | qwen3.8-27b | 70s | 6 | 72.4k / 6.5k | 0 | 11/12 (12/12 after a grading-regex fix, see below) | used `write` correctly, used the `git` tool in Step 1, reported the path and slice count in the final reply; the one score miss was a test-harness regex that didn't handle a `**Done when** (fresh state...):` label followed by bulleted checks on the next lines — the actual done-when content was already command-shaped and testable, fixed in the harness, not the skill |
87
+ | v3 (bash-reflex found) | qwen3.8-27b | — | — | — | 2 (1 benign ENOENT, 1 real: `bash` refused) | 11/12 | reached for `bash` (`node --version`, `find`, `ls -la` chained with `&&`) to survey the toolchain even though `bash` was never in `allowed-tools` and nothing in the body named it explicitly — added the same explicit `bash`-refusal line the `tasks` refusal already had, plus a Red flags entry |
88
+ | v4 (final, stable) | qwen3.8-27b | 60s | 6 | 71.4k / 5.6k | 0 | **12/12** | clean run: `context` → `read`/`ls` → `git status` → `write`, self-contained final reply naming path, slice count, and battle order |
89
+ | final-s2 (no plan) | qwen3.8-27b | 26s | 4 | 43.2k / 2.0k | 0 | **5/5** | checked all three plan locations plus `git log`/`status`, correctly stopped with no `SPRINT.md` written, cited the skill's own red-flag language, recommended `grill-me` |
90
+ | final-s3 (vague plan) | qwen3.8-27b | 28s | 5 | 56.2k / 2.5k | 0 | **5/5** | read the one-line `PLAN.md`, explicitly invoked "the skill's own test" (a testable done-when), listed three concrete missing pieces (object/baseline/target for "performance", a definition of "clean", the absent codebase), stopped without writing `SPRINT.md` |
91
+ | final-mini (cross-model, S1) | ornith1.5-35b-moe (mini) | 42s | 6 | 10.8k / 3.2k | 0 | **12/12** | same clean shape on the second model family: `context` → `ls`/`read` → `write`, self-contained final reply |
92
+ | final-mini-s2 (cross-model, S2) | ornith1.5-35b-moe (mini) | 19s | 6 | 4.0k / 1.1k | 0 | **5/5** | correctly found nothing to slice, recommended `grill-me`, offered to slice immediately if pointed at a plan |
93
+
94
+ **Changes** (0.3.0 -> 0.4.0):
95
+
96
+ 1. **`artifact` removed from `allowed-tools`.** cut-it never needs the real
97
+ `artifact` tool — `SPRINT.md` is a plain file, always written with
98
+ `write`. This is the fix for the bug above.
99
+ 2. **Step 3 retitled** "Write SPRINT.md" (was "Write the artifact") and its
100
+ body now says explicitly "use the `write` tool" and names the
101
+ `artifact`-tool confusion directly, including what it actually does
102
+ wrong (terminal call, wrong default path under `.clio-coder/artifacts/`,
103
+ skips the report-back step).
104
+ 3. **New Step 4 — Report back**, an explicit final-reply requirement (path
105
+ written, slice count, one-line battle-order summary). Nothing in the
106
+ frozen skill told the model to confirm after writing; every hardened run
107
+ now does.
108
+ 4. **`## Arguments` contract**, the section the frozen skill never had:
109
+ conversational trigger syntax, how a named path splits between "the plan
110
+ to slice" and "where to write `SPRINT.md`", the no-operator/`ask_user`-
111
+ auto-cancels rule (adapted from `grill-me` 0.5.0's "this run has no
112
+ back-and-forth" framing — stated as a fact about the run, not gated on a
113
+ cancellation response), and — since cut-it's happy path rarely needs a
114
+ question at all — explicit guidance for the one place it plausibly might
115
+ (choosing among several candidate plans/milestones): pick the best one,
116
+ state the alternative, mark `assumed — confirm`, keep going.
117
+ 5. **`tasks` and `bash` explicitly named as refused**, in Arguments and Red
118
+ Flags, both found empirically: v1's `tasks` call (tracking its own
119
+ steps) and v3's `bash` call (`node --version`, `find`, `ls -la` chained
120
+ with `&&`, to survey the toolchain) were both real safety blocks despite
121
+ neither tool ever having been in `allowed-tools`.
122
+ 6. Three new Red Flags entries for the concrete failures observed: the
123
+ `artifact`-tool confusion, ending the run with no final reply, and the
124
+ `bash` reflex (`tasks` already had informal coverage, now explicit too).
125
+
126
+ **Still weak**: `code_nav` (in `allowed-tools`) was never exercised — this
127
+ fixture's grounding fit entirely in one small stub file, so `read`/`grep`
128
+ sufficed; a plan referencing a larger call graph might exercise it, none
129
+ was built here. The `ask_user`-unavailable path and the "several candidate
130
+ plans" disambiguation guidance in Arguments are reasoned prose, not
131
+ empirically run — no fixture here seeds multiple plausible plan files or a
132
+ scenario where the model actually reaches for `ask_user`; every hardened
133
+ run reasoned straight to the assumed-confirm default or never needed a
134
+ question at all. S3's "flag as vague" grading is phrase-matching against a
135
+ fixed term list (`vague`, `underspecified`, `insufficient`, ...) — a run
136
+ that flags the same gap in different words would under-score on a
137
+ technicality, though neither observed run did. The v2 grading-regex miss
138
+ (`done_when_testable`, fixed in the harness before v4) is a reminder that
139
+ this fixture's automated score can undercount a genuinely correct skill
140
+ output; the raw `SPRINT.md` files are worth spot-reading, not just the
141
+ score column. No timing was captured for v3 (an incremental re-run to
142
+ confirm the `bash` finding, superseded immediately by v4) — not a gap in
143
+ the skill's coverage, just an artifact of the iteration order.
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: design-council
3
- description: Use when a design decision has real tradeoffs and needs several expert perspectives that challenge each other before code is written, such as architecture choices, API shapes, storage formats, parallelization strategies, or dependency decisions. Quick mode runs a single round for a fast perspective check. Triggers on "council", "debate this", "multiple perspectives", "weigh the options", "what would experts say". Not for a one-question-at-a-time interrogation of a plan; use grill-me. Not for splitting implementation work across workers; use dispatch directly.
3
+ description: Convenes several expert perspectives that challenge each other on a design decision with real trade-offs before code is written; quick mode runs a single round. Not for a one-question-at-a-time interrogation of a plan; use grill-me. Not for splitting implementation across workers; use dispatch directly.
4
4
  triggers:
5
5
  - convene a design council
6
6
  - debate this design
7
7
  - get multiple expert perspectives
8
8
  - weigh the architecture options
9
9
  - what would experts say
10
- version: 0.3.2
10
+ version: 0.5.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - dispatch
@@ -17,12 +17,13 @@ allowed-tools:
17
17
  - ls
18
18
  - context
19
19
  - code_nav
20
+ - ask_user
20
21
  clio-coder:
21
22
  registry-id: iowarp/clio-coder
22
23
  source-url: https://github.com/iowarp/clio-coder/tree/main/skills/workflow/design-council
23
24
  audit: pass
24
25
  provenance: designed
25
- eval-status: scenarios-recorded
26
+ eval-status: smoke-checked
26
27
  model-size: large
27
28
  agents:
28
29
  - scout
@@ -36,22 +37,95 @@ Run a bounded multi-perspective debate on a real design decision. The council
36
37
  surfaces the crux of a disagreement before code commits to one side. It is not
37
38
  a ritual: if experts would agree, do not convene it.
38
39
 
40
+ ## Arguments
41
+
42
+ ```text
43
+ convene a design council on <decision>
44
+ ```
45
+
46
+ There is no flag syntax; the trigger is conversational — "convene a design
47
+ council on X", "debate this design", "get multiple expert perspectives".
48
+ Whatever the user names is the decision. A referenced file, doc, or repo path
49
+ in the same request is Step 0/1's grounding to read first, not a separate
50
+ argument.
51
+
52
+ **Headless is the enforced default, not a suggestion.** A council is several
53
+ worker runs; nobody is waiting between rounds in a headless run, and every
54
+ round that actually runs costs real wall-clock time on top of the
55
+ orchestrator's own turns. `ask_user` still executes here even though it is
56
+ not in the list below — it is always available regardless — but nothing
57
+ answers it: the single question Step 1 asks comes back `{cancelled: true}`
58
+ immediately, as an ordinary result, not an error. Treat that cancellation (or
59
+ skip the call and reason from this paragraph directly — one is not more valid
60
+ than the other) as the fixed answer **quick mode: exactly three perspectives,
61
+ exactly one round (Positions) plus synthesis.** Do not compose four or five
62
+ perspectives headlessly and do not run a Responses or Convergence round
63
+ headlessly, no matter how contested the topic looks — four/five perspectives
64
+ and multi-round debate are for a live session with an operator who actually
65
+ asked for the deeper pass. This is the one rule the skill's own prior smoke
66
+ history says a model will not reliably self-infer from prose alone, so treat
67
+ the number 3 and the number 1 as hard, not as defaults to raise if the topic
68
+ seems to deserve more.
69
+
70
+ **Dispatch call shape.** Compose the round's perspectives, then make exactly
71
+ one `dispatch` call with all of them in `tasks` and `mode="parallel"`. If
72
+ that call comes back admission-denied for endpoint or target capacity (a
73
+ single local model instance commonly allows only one concurrent worker, so a
74
+ 3-task parallel wave can be denied outright rather than queued), retry the
75
+ identical `tasks` batch in one dispatch call with `mode="sequential"` instead
76
+ — the tool runs them one after another itself. Never split a round into
77
+ several separate one-task `dispatch` calls made one at a time waiting on each
78
+ result before deciding the next; that is the serial-perspectives pattern that
79
+ produced the round-trip cost the skill's own timeout history is about, and it
80
+ does not fix the capacity problem the parallel call already reported. Do not
81
+ call `dispatch(list:true)` to probe capacity first — it answers nothing about
82
+ concurrency and only spends a call.
83
+
84
+ Declare `intent: {read_roots: [...], relevant_paths: [...]}` with paths
85
+ relative to the repo root on every dispatch call instead of pasting an
86
+ absolute path into `task`/`briefing` prose (a config value, a mount point).
87
+ An absolute path token in briefing/task text with no declared `intent` is
88
+ rejected as `legacy_scope_path_absolute`. Keep a persona's argument in prose,
89
+ never literal shell syntax — a phrase like "you can `rm -rf` the directory"
90
+ inside a dispatch call's text can trip the same damage-control pattern that
91
+ blocks a real destructive shell command, even though nothing executes; say
92
+ "delete the directory" instead.
93
+
94
+ A dispatch call's own synchronous result already carries every worker's
95
+ output — do not follow it with a `bash`/`read` pass over the receipt file on
96
+ disk to re-read what you already have. The steps below are the plan; do not
97
+ open a `tasks` list for them. This skill's tool surface is exactly
98
+ `dispatch`, `read`, `grep`, `find`, `ls`, `context`, `code_nav`, and
99
+ `ask_user` (`context` and `ask_user` are always available regardless).
100
+ `tasks` and `bash` both sit outside it and any call to either is refused —
101
+ locate files with `find`/`ls`, not `bash find`/`bash ls`; inspect a receipt
102
+ with `read`, not `bash cat`. This skill never writes: `write` and `artifact`
103
+ are not on its surface, so the synthesis in Step 4 is chat output, never a
104
+ file.
105
+
39
106
  ## Step 0 — Check the question is contested
40
107
 
41
108
  Before composing anyone, ask: would credible experts actually disagree on the
42
109
  answer? If every perspective you can imagine picks the same option and differs
43
- only in caveats, stop here. Say the council is not needed, give the consensus
44
- answer with the caveats attached, and end.
110
+ only in caveats, stop here — do not dispatch anything. Say the council is not
111
+ needed, give the consensus answer with the caveats attached, and end. Do not
112
+ dispatch a round "just to confirm" a consensus call you already reached —
113
+ that is the ritual this step exists to skip, and it still costs the same
114
+ wall-clock time and worker slots as a real debate. Trust this self-check the
115
+ same way you trust the rest of your own reasoning; a council you convened to
116
+ double-check yourself is not more rigorous than the judgment behind it.
45
117
 
46
118
  ## Step 1 — Compose perspectives
47
119
 
48
120
  Derive perspectives from the topic itself, never from a generic role menu.
49
- Three is the default and the right number for almost every decision. Go to
50
- four or five only when the decision genuinely has that many independent
51
- stances, and never headless: each perspective is a worker run, and a model
52
- that dispatches the round serially instead of in parallel turns five
53
- perspectives into five sequential runs. If you are running without a user to
54
- wait on you, use three perspectives and one round.
121
+ On a genuinely contested topic, call `ask_user` once with `mode:
122
+ "single_question"` asking whether this should be a quick pass (three
123
+ perspectives, one round) or the full debate (up to five perspectives, up to
124
+ three rounds). See Arguments for what a headless run does with that call.
125
+ In a live session where the user answers, honor the requested depth. Compose
126
+ three perspectives by default; go to four or five only in that live full-
127
+ debate case, and only when the decision genuinely has that many independent
128
+ stances.
55
129
 
56
130
  Each perspective gets:
57
131
 
@@ -74,11 +148,13 @@ perspective from the read-only recipes in the live catalog:
74
148
  - `researcher`: stance leaning on external docs, standards, or papers.
75
149
  - `provenance`: stance arguing from runtime evidence and receipts.
76
150
 
77
- Run one round's perspectives in parallel: one `dispatch` call with the round's
78
- task prompts in `tasks` and `mode="parallel"`. Rounds are sequential. Each
79
- task prompt carries the persona block, the decision context, and the full
80
- transcript so far. Workers never edit files; the debate is analysis only.
81
- Dispatch receipts link every statement to a worker run.
151
+ See Arguments for the exact call shape (one batched `tasks` call, the
152
+ `mode="sequential"` capacity fallback, `intent` for scope, no literal shell
153
+ syntax). Rounds are sequential; a round's own perspectives are the one
154
+ dispatch call. Each task prompt carries the persona block, the decision
155
+ context, and the full transcript so far. Workers never edit files; the
156
+ debate is analysis only. Dispatch receipts link every statement to a worker
157
+ run.
82
158
 
83
159
  ## Step 3 — Run the rounds
84
160
 
@@ -90,14 +166,15 @@ Dispatch receipts link every statement to a worker run.
90
166
  still disagrees and why that crux is the crux, and its final
91
167
  recommendation.
92
168
 
93
- **Quick mode** (user asked for a light pass): round 1 plus synthesis. No
94
- responses round.
169
+ **Quick mode** (headless default, or a user asking for a light pass): round 1
170
+ plus synthesis. No responses round, no convergence round.
95
171
 
96
- **Early termination.** After round 1, judge disagreement on the decision
97
- question itself, not on side conditions. If every position picks the same
98
- option and differs only in caveats, toggles, or requests to measure later,
99
- that is consensus: skip rounds 2 and 3, report that the council was not
100
- needed, and return the consensus with caveats. Never manufacture friction.
172
+ **Early termination** (full-debate mode only). After round 1, judge
173
+ disagreement on the decision question itself, not on side conditions. If
174
+ every position picks the same option and differs only in caveats, toggles,
175
+ or requests to measure later, that is consensus: skip rounds 2 and 3, report
176
+ that the council was not needed, and return the consensus with caveats.
177
+ Never manufacture friction.
101
178
 
102
179
  ## Step 4 — Synthesize
103
180
 
@@ -124,10 +201,11 @@ synthesis line has a citation and the recommendation names its crux.
124
201
 
125
202
  ## Degraded mode
126
203
 
127
- If dispatch is unavailable or admission-denied, run the same rounds inline:
128
- write each perspective's contribution yourself, sequentially, same round
129
- structure and synthesis format. Label the output as degraded (single-model
130
- debate, no receipts).
204
+ If dispatch is unavailable or admission-denied for a reason other than the
205
+ capacity retry in Arguments (the tool itself is missing from the surface, or
206
+ every retry is denied), run the same rounds inline: write each perspective's
207
+ contribution yourself, sequentially, same round structure and synthesis
208
+ format. Label the output as degraded (single-model debate, no receipts).
131
209
 
132
210
  ## Boundaries
133
211
 
@@ -141,7 +219,19 @@ this skill when the debate needs the full round structure and receipts.
141
219
  ## Red Flags
142
220
 
143
221
  - Perspectives named "optimist" and "pessimist" (role menu, not topic).
144
- - More than five perspectives, or debate rounds beyond three.
222
+ - Composing four or five perspectives, or running a Responses/Convergence
223
+ round, in a headless run — the enforced default is exactly three and
224
+ exactly one, not a ceiling to raise because the topic looks deep.
225
+ - Splitting a round into several single-task `dispatch` calls issued one at a
226
+ time instead of one batched `tasks` call (retried as `mode="sequential"`
227
+ on a capacity denial).
228
+ - Calling `dispatch(list:true)` before dispatching the round.
229
+ - Pasting an absolute path into a dispatch `task`/`briefing` instead of a
230
+ declared `intent`, or quoting literal shell syntax inside a persona's
231
+ argument.
232
+ - Reaching for `bash`/`read` on a receipt file the dispatch call's own result
233
+ already contains.
234
+ - Opening a `tasks` list for the round structure; `tasks` is refused.
145
235
  - A synthesis that averages positions instead of naming the crux.
146
236
  - Manufactured disagreement on a settled question.
147
237
  - A worker asked to edit files as part of the debate.
@@ -95,3 +95,164 @@ default and tells a headless run to use three and one round, which is the
95
95
  blocker the timeout exposed. Not re-run, so `eval-status` stays
96
96
  `scenarios-recorded`; the next campaign has to confirm the shortened council
97
97
  fits the ceiling.
98
+
99
+ ## Battletest record (2026-09-03) — 0.4.0 -> 0.5.0
100
+
101
+ Fixture: `/home/akougkas/eval-temp/harness/test_designcouncil.py`, a
102
+ self-contained repo (`src/checkpoint.py` writing one raw `.npy` per rank per
103
+ step to a shared parallel filesystem, `src/config.py`/`config.yaml` using
104
+ `pyyaml`) adapted from this file's own worked examples. S1 = HDF5-vs-Zarr
105
+ (genuinely contested), S2 = hand-roll-YAML-vs-keep-pyyaml (consensus), S4 =
106
+ "poke holes in my plan" anti-trigger. `runner.py`'s `timeout=` was raised to
107
+ 1800s for this skill specifically — the 900s default is a `dispatch` fan-out
108
+ ceiling, not a model-quality one, and the mission called for confirming the
109
+ v0.3.0 "three perspectives, one round" headless fix on its own terms, not
110
+ routing around it. Primary: `qwen3.8-27b`/`dynamo`. Cross-model confirm:
111
+ `ornith1.5-35b-moe`/`mini`. Both runs shared the `dynamo` LM Studio endpoint
112
+ with concurrently active sibling battletest sessions (`grill-me`,
113
+ `workflow-distiller`) launched via `herdr` during this pass — real,
114
+ externally-caused contention, not a fixture artifact; see below.
115
+
116
+ | run | scenario | model | wall | turns | dispatch calls | safety blocks | score | outcome |
117
+ |---|---|---|---|---|---|---|---|---|
118
+ | baseline (no skill) | S1 | qwen3.8-27b | 121s | 5 | 0 | 0 | 3/11 | no skill invoked (ran under `--no-skills`); the model noticed the installed skill anyway and narrated "four positions, real cross-examination" as one inline monologue — a solid single-model analysis but no dispatch, no receipts, no named-recipe workers |
119
+ | v1 (frozen 0.4.0) | S1 | qwen3.8-27b | 1740s, **killed at the 1800s ceiling** (exit -9) | 22 | 12 | 7 | 4/11 | composed **four** perspectives and ran into **round 2**, both violating the stated headless "three perspectives, one round" default; opened a `tasks` plan (refused); two `dispatch` calls rejected for an absolute path token in `briefing` (`legacy_scope_path_absolute`); one denied for endpoint capacity; one **hard-blocked as `rm-recursive-or-force`** because a perspective's own round-1 argument prose said "...you can `ls`, checksum, `rm -rf`, and publish..." — the admission layer's damage-control scan matched the quoted shell syntax inside the debate text itself, not an executed command; reached for `bash` to inspect a receipt file (refused). This is the same failure class the 2026-08-13 smoke record flagged, confirmed still present, worse: the v0.3.0 fix was never actually followed |
120
+ | v2 (first hardened cut) | S1 | qwen3.8-27b | 416s | 11 | 3 | 4 | 8/11 | one batched `tasks`-array `dispatch` call, all 3 capacity/timeout-denied under real endpoint contention; correctly fell back to Degraded mode, cited the exact denial reasons, labeled the output degraded, produced a complete synthesis with agreements/crux/dissent — no `tasks` misuse, no shell-block trigger, no one-at-a-time call splitting |
121
+ | v2 (post Step 0 strengthen) | S1 | qwen3.8-27b | 304s | 7 | 3 | 4 | 8/11 | same shape: parallel wave denied, sequential retry denied/timed out, correct Degraded fallback, well-formed synthesis citing the transcript |
122
+ | v2 | S2 | qwen3.8-27b | 52.5s / 60.6s (before/after Step 0 edit) | 5 / 6 | **0** | 0 | **5/5** both | Step 0 stopped before any dispatch, stated the council was not needed, gave the consensus (keep `pyyaml`, `safe_load` only) with caveats, and named the repo's actually-contested question (the checkpoint format) as a pointer — no manufactured friction either run |
123
+ | v2 | S4 | qwen3.8-27b | 69.4s | — | 0 | 0 | 3/3 | did not convene a council; ran a grill-me-shaped one-question-at-a-time interrogation instead and named `grill-me` |
124
+ | v2 (cross-model) | S1 | ornith1.5-35b-moe/mini | 384s | 10 | 4 | 5 | 7/11 | 3 of 4 `dispatch` calls denied/timed out on the **same `dynamo` endpoint** (workers defaulted there even though the orchestrator itself ran on `mini`); one stray `monitor` call refused (outside the skill's surface); correctly diagnosed the capacity pattern in its own words ("times out at ~57s... despite reporting 0/4 slots in use... exhausted the capacity fallback") and ran Degraded mode with a clearly labeled warning banner |
125
+ | v2 (cross-model) | S2 | ornith1.5-35b-moe/mini | 239.5s / 188.1s (before/after Step 0 edit) | 6 | 3 / 2 | 3 / 2 | 2/5 both | did **not** skip Step 0's dispatch the way qwen did — ran one capacity-retried round anyway, but stopped there (no round 2/3), reached the same correct consensus ("keep pyyaml", supply-chain caveat preserved as a contingent dissent, not manufactured), and stated the council-not-needed conclusion explicitly; the Step 0 wording strengthen did not change this model's behavior |
126
+
127
+ **S3 (dispatch unavailable) — no clean forced trigger found, confirmed by
128
+ reading the CLI, not assumed**: `clio-coder run --help`'s `--tool-profile`
129
+ narrows a *dispatched sub-agent's own* tool set and requires `--agent`
130
+ (`clio-coder run: fleet dispatch flags require --agent <recipe-id>`); there
131
+ is no flag that strips `dispatch` from a top-level `--skill` orchestrator
132
+ run's own surface in this harness. Editing the skill's own `allowed-tools`
133
+ to omit `dispatch` would test a different skill, not this one. This is a
134
+ real, documented harness gap, not faked around. In its place, S3's expected
135
+ behavior was exercised **organically, repeatedly, for real**: every S1 run
136
+ on both models hit genuine `dispatch` admission denial or timeout from
137
+ endpoint capacity, and Degraded mode fired correctly every single time —
138
+ labeled degraded, same round structure, synthesis format intact, no
139
+ receipts claimed that did not exist.
140
+
141
+ **The dispatch-viability and recipe-existence questions, resolved
142
+ empirically:**
143
+
144
+ - `scout`, `researcher`, and `provenance` all exist as builtin shadow-agent
145
+ recipes (`src/domains/agents/builtins/*.md`, `capabilityClass: read-only`)
146
+ and are dispatchable — every hardened run that got a `dispatch` call
147
+ admitted used one of them by name and got real worker output back.
148
+ - Each recipe's own system prompt declares a rigid JSON-only result
149
+ contract (`{"findings":[...]}`, `{"source":...}`, `{"confirmedFacts":...}`)
150
+ that has nothing to do with a debate position. In practice this was
151
+ **not** a blocker: V1's successfully-admitted round 1 came back as full
152
+ argumentative prose (named claims, attacks, citations, mind-change
153
+ conditions) from `researcher`/`scout` workers, not the declared narrow
154
+ JSON shape — the persona in the task prompt won out over the recipe's
155
+ own stated output contract for the caller-facing transcript. Worth
156
+ knowing, not worth re-architecting around.
157
+ - The actual, dominant blocker is **`dispatch` admission capacity on a
158
+ single-instance local LM Studio target**. `~/.local/state/clio-coder/
159
+ endpoint-slots.json` records the `dynamo` endpoint (`100.104.197.69:1234`)
160
+ at `"slots": 1`. A 3-task parallel wave is denied outright
161
+ (`capacity exceeded (3/1 slots)`), and the `mode="sequential"` retry this
162
+ pass added to the skill is followed correctly by both models but still
163
+ gets denied or times out (`admission timed out after ~57s... 0/4 worker
164
+ slots in use` — a scheduling state that never resolves within the wait
165
+ window), evidently because the orchestrator's own foreground session
166
+ already holds the endpoint's only slot. This reproduced independently on
167
+ `mini` too: workers dispatched from an orchestrator running on `mini`
168
+ still routed to `dynamo`'s endpoint by default, so the same 1-slot
169
+ contention applied there as well — confirming the brief's suspicion that
170
+ worker fan-out may silently default to an unexpected node/target. Some
171
+ of this pass's contention was real concurrent load, not just self-
172
+ contention: `ps` showed sibling `grill-me`/`workflow-distiller`
173
+ battletest sessions actively running via `herdr` against the same
174
+ endpoint during these runs.
175
+ - The full-auto/`authorityBasis` auto-grant fact supplied going in
176
+ (`deps.getAutonomy?.() === "full-auto" ? "full-auto-policy" :
177
+ "operator-plan-approval"`) turned out to be **moot for this skill**:
178
+ that gate only applies to `agent:"auto"` dispatch requests
179
+ (`src/tools/dispatch-arguments.ts`, `agentSelection` is only populated
180
+ when `requestedAgent === "auto"`). Design-council always pins an explicit
181
+ recipe id (`scout`/`researcher`/`provenance`), so `agentSelection` stays
182
+ `undefined` and the operator-approval/full-auto-policy distinction never
183
+ engages — admission for this skill's calls is governed purely by the
184
+ capacity/reservation machinery above, independent of `--autonomy`.
185
+ - `legacy_scope_path_absolute`: an absolute path token (e.g. a value read
186
+ from `config.yaml`) pasted into `task`/`briefing` prose without a
187
+ declared `intent` is rejected. Fixed by telling the skill to declare
188
+ `intent.read_roots`/`relevant_paths` (relative) on every call instead.
189
+
190
+ **Changes (0.4.0 -> 0.5.0):**
191
+
192
+ 1. **`## Arguments` contract**, ported from `grill-me`/`cut-it`'s shape.
193
+ States headless is the *enforced* default (exactly three perspectives,
194
+ exactly one round), not a self-assessed suggestion — the v0.3.0 fix's
195
+ prose alone did not hold on either model (V1 composed four perspectives
196
+ and ran round 2 headlessly).
197
+ 2. **Dispatch call shape spelled out**: one batched `tasks`-array call per
198
+ round; on capacity denial, retry the *same* batch with
199
+ `mode="sequential"` in one call rather than splitting into several
200
+ single-task calls issued one at a time (V1's actual failure — 12
201
+ separate `dispatch` calls, a `list:true` probe, and manual receipt
202
+ reads via `bash`, all of which V2 stopped doing).
203
+ 3. **`intent.read_roots`/`relevant_paths` guidance** to avoid
204
+ `legacy_scope_path_absolute` rejections from paths quoted in prose.
205
+ 4. **No literal shell syntax in a persona's argument text** — added after
206
+ V1's `rm-recursive-or-force` hard block fired on debate prose, not a
207
+ real command.
208
+ 5. **`ask_user` added to `allowed-tools`** (already always-exempt, now
209
+ documented) with one call at Step 1 on a contested topic, asking
210
+ quick-vs-full-debate depth; headlessly it cancels, which *is* the
211
+ three-perspective/one-round confirmation, mirroring the auto-cancel
212
+ pattern the planning category established rather than relying on
213
+ unaided self-assessment.
214
+ 6. **Explicit `tasks`/`bash`/receipt-re-read refusal lines**, matching the
215
+ sibling skills' pattern (`tasks` opened a plan in V1; `bash` was reached
216
+ for twice across the runs above to re-inspect a receipt the dispatch
217
+ call's own result already contained).
218
+ 7. **Step 0 strengthened** against dispatching "just to confirm" a
219
+ consensus call already reached — tested on `ornith1.5-35b-moe`/`mini`
220
+ (still dispatches once) and `qwen3.8-27b`/`dynamo` (unaffected, already
221
+ correct); see Still weak.
222
+ 8. Five new Red flags entries naming the concrete failures observed above.
223
+
224
+ **Still weak:**
225
+
226
+ - **Step 0's short-circuit is model-family-dependent.** `qwen3.8-27b`
227
+ trusts its own contested-or-not judgment and skips `dispatch` entirely
228
+ for S2 (5/5, 0 dispatch calls, both before and after the Step 0 edit).
229
+ `ornith1.5-35b-moe` does not: it dispatches one (capacity-retried) round
230
+ "to confirm" even on the same consensus topic, both before and after the
231
+ strengthened wording — 2/5 unchanged. It still stops after that one
232
+ round, still reaches the correct consensus, still preserves the dissent
233
+ as a contingent caveat rather than manufacturing friction, so the
234
+ user-facing outcome is fine; the wasted dispatch cost is the actual gap,
235
+ and more prose did not move it, consistent with the planning category's
236
+ own finding that some model-family behaviors don't fully generalize no
237
+ matter how much repetition is added.
238
+ - **S1 never completed as a genuine 3-perspective/1-round council with
239
+ real receipts** on this pass — every attempt on both models hit real
240
+ endpoint capacity contention and degraded. The Degraded path is now
241
+ proven solid, but the "happy path" (dispatch succeeds, synthesis cites
242
+ real worker receipts) is unverified under this specific pass's
243
+ conditions; V1's transcript shows it *can* succeed (four real worker
244
+ runs completed with citable output before the round-2/shell-block
245
+ failure), so this reads as an availability problem this pass's timing
246
+ ran into, not a structural block — but it means the exact scoring
247
+ bullets that depend on `dispatch_receipts_ok` (perspective count,
248
+ receipts-ok) were not exercised clean this round.
249
+ `perspectives_dispatched`/`used_named_recipes` in the grading script
250
+ count attempted, not completed, dispatch tasks for this reason.
251
+ - **No purpose-built S3 fixture** — see above; a future pass with a
252
+ dedicated low-capacity or offline dispatch target would let this be
253
+ tested directly instead of relying on organic contention.
254
+ - `code_nav` and `grep` were barely exercised (fixture is small enough
255
+ that `read`/`ls`/`find` covered grounding in most runs).
256
+ - Only one fixture domain (HDF5-vs-Zarr / YAML-parser) ran this pass; the
257
+ richer four-perspective MPI-checkpoint example from "Observed live-smoke
258
+ results" above was not re-run against 0.5.0.