@iowarp/clio-coder 0.4.1 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (604) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/CONTRIBUTING.md +142 -52
  3. package/README.md +434 -473
  4. package/SECURITY.md +2 -1
  5. package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
  6. package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
  7. package/dist/assets/codewiki.json +1 -1
  8. package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
  9. package/dist/builtins-IA7V7FUC.js +22 -0
  10. package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
  11. package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
  12. package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
  13. package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
  14. package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
  15. package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
  16. package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
  17. package/dist/chunk-3EBYEESD.js +314 -0
  18. package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
  19. package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
  20. package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
  21. package/dist/chunk-4UVU7BJ5.js +39 -0
  22. package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
  23. package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
  24. package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
  25. package/dist/chunk-5MEZN6CB.js +1334 -0
  26. package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
  27. package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
  28. package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
  29. package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
  30. package/dist/chunk-7DRAWPTZ.js +360 -0
  31. package/dist/chunk-7E7I3WLS.js +3762 -0
  32. package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
  33. package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
  34. package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
  35. package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
  36. package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
  37. package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
  38. package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
  39. package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
  40. package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
  41. package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
  42. package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
  43. package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
  44. package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
  45. package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
  46. package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
  47. package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
  48. package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
  49. package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
  50. package/dist/chunk-GYV6VZOC.js +26 -0
  51. package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
  52. package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
  53. package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
  54. package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
  55. package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
  56. package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
  57. package/dist/chunk-IJNZMHLA.js +101 -0
  58. package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
  59. package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
  60. package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
  61. package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
  62. package/dist/chunk-JEQ3XTHC.js +42 -0
  63. package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
  64. package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
  65. package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
  66. package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
  67. package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
  68. package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
  69. package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
  70. package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
  71. package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
  72. package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
  73. package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
  74. package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
  75. package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
  76. package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
  77. package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
  78. package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
  79. package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
  80. package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
  81. package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
  82. package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
  83. package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
  84. package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
  85. package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
  86. package/dist/chunk-R6Q67RJH.js +134 -0
  87. package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
  88. package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
  89. package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
  90. package/dist/chunk-SKHCAU7K.js +385 -0
  91. package/dist/chunk-SZAA6XDG.js +30 -0
  92. package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
  93. package/dist/chunk-UOIZ7DA4.js +41 -0
  94. package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
  95. package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
  96. package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
  97. package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
  98. package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
  99. package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
  100. package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
  101. package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
  102. package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
  103. package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
  104. package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
  105. package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
  106. package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
  107. package/dist/chunk-XRZT5WY5.js +47 -0
  108. package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
  109. package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
  110. package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
  111. package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
  112. package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
  113. package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
  114. package/dist/cli/index.js +41 -39
  115. package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
  116. package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
  117. package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
  118. package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
  119. package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
  120. package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
  121. package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
  122. package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
  123. package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
  124. package/dist/context-map-COB37XXN.js +505 -0
  125. package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
  126. package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
  127. package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
  128. package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
  129. package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
  130. package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
  131. package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
  132. package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
  133. package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
  134. package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
  135. package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
  136. package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
  137. package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
  138. package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
  139. package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
  140. package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
  141. package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
  142. package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
  143. package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
  144. package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
  145. package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
  146. package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
  147. package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
  148. package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
  149. package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
  150. package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
  151. package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
  152. package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
  153. package/dist/reset-EAJFFJVB.js +344 -0
  154. package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
  155. package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
  156. package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
  157. package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
  158. package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
  159. package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
  160. package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
  161. package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
  162. package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
  163. package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
  164. package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
  165. package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
  166. package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
  167. package/dist/uninstall-5PEVOE5B.js +408 -0
  168. package/dist/upgrade-M4WXY6KN.js +303 -0
  169. package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
  170. package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
  171. package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
  172. package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
  173. package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
  174. package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
  175. package/dist/worker/entry.js +90 -74
  176. package/docs/README.md +176 -81
  177. package/docs/{acp.md → architecture/acp.md} +36 -20
  178. package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
  179. package/docs/{architecture.md → architecture/architecture.md} +43 -22
  180. package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
  181. package/docs/architecture/artifact-versions.md +90 -0
  182. package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
  183. package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
  184. package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
  185. package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
  186. package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
  187. package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
  188. package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
  189. package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
  190. package/docs/{observability.md → architecture/observability.md} +38 -14
  191. package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
  192. package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
  193. package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
  194. package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
  195. package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
  196. package/docs/architecture/time-conventions.md +125 -0
  197. package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
  198. package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
  199. package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
  200. package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
  201. package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
  202. package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
  203. package/docs/guide/configuration-reference.md +1163 -0
  204. package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
  205. package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
  206. package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
  207. package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
  208. package/docs/{glossary.md → guide/glossary.md} +14 -11
  209. package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
  210. package/docs/guide/panes-and-files.md +290 -0
  211. package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
  212. package/docs/{resource-library.md → guide/resource-library.md} +13 -4
  213. package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
  214. package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
  215. package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
  216. package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
  217. package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
  218. package/docs/process/development-pipeline.md +152 -0
  219. package/docs/process/documentation-coverage.md +100 -0
  220. package/docs/process/documentation-guide.md +187 -0
  221. package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
  222. package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
  223. package/docs/{evolution.md → process/evolution.md} +2 -2
  224. package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
  225. package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
  226. package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
  227. package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
  228. package/evals/README.md +2 -2
  229. package/evals/behavioral-model.yaml +3 -2
  230. package/package.json +10 -8
  231. package/skills/README.md +52 -41
  232. package/skills/coding/ast-grep/SKILL.md +102 -31
  233. package/skills/coding/ast-grep/evals.md +26 -0
  234. package/skills/coding/coding-standards/SKILL.md +41 -6
  235. package/skills/coding/coding-standards/evals.md +23 -0
  236. package/skills/coding/prototype/SKILL.md +88 -29
  237. package/skills/coding/prototype/evals.md +19 -0
  238. package/skills/coding/tdd/SKILL.md +81 -54
  239. package/skills/coding/tdd/evals.md +20 -0
  240. package/skills/context/context-handoff/SKILL.md +44 -3
  241. package/skills/context/context-handoff/evals.md +44 -0
  242. package/skills/context/context-prime/SKILL.md +46 -16
  243. package/skills/context/context-prime/evals.md +45 -0
  244. package/skills/git/branch-closeout/SKILL.md +132 -0
  245. package/skills/git/branch-closeout/evals.md +133 -0
  246. package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
  247. package/skills/git/file-ticket/SKILL.md +78 -64
  248. package/skills/git/file-ticket/assets/issue-template.md +22 -0
  249. package/skills/git/file-ticket/evals.md +31 -26
  250. package/skills/git/file-ticket/references/issue-discovery.md +49 -0
  251. package/skills/git/fix-issue/SKILL.md +88 -65
  252. package/skills/git/fix-issue/evals.md +35 -31
  253. package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
  254. package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
  255. package/skills/git/resolve-merge-conflicts/evals.md +52 -25
  256. package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
  257. package/skills/git/ship/SKILL.md +103 -67
  258. package/skills/git/ship/assets/pr-template.md +21 -0
  259. package/skills/git/ship/evals.md +44 -28
  260. package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
  261. package/skills/git/worktree-create/SKILL.md +80 -50
  262. package/skills/git/worktree-create/evals.md +40 -33
  263. package/skills/git/worktree-create/references/worktree-setup.md +62 -66
  264. package/skills/git/worktree-merge/SKILL.md +112 -65
  265. package/skills/git/worktree-merge/evals.md +42 -34
  266. package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
  267. package/skills/meta/clio-coder-dev/SKILL.md +9 -5
  268. package/skills/meta/clio-coder-dev/evals.md +3 -2
  269. package/skills/meta/clio-coder-test/SKILL.md +102 -95
  270. package/skills/meta/clio-coder-test/evals.md +9 -4
  271. package/skills/meta/clio-coder-test/references/harness.md +100 -124
  272. package/skills/meta/clio-coder-test/references/test-map.md +77 -50
  273. package/skills/meta/credentials/SKILL.md +2 -2
  274. package/skills/meta/find-skills/SKILL.md +2 -2
  275. package/skills/meta/herdr/SKILL.md +2 -2
  276. package/skills/meta/skill-craft/SKILL.md +22 -16
  277. package/skills/planning/archify/SKILL.md +196 -0
  278. package/skills/planning/archify/evals.md +65 -0
  279. package/skills/planning/architecture/SKILL.md +62 -13
  280. package/skills/planning/architecture/evals.md +65 -0
  281. package/skills/planning/backlog/SKILL.md +131 -15
  282. package/skills/planning/backlog/evals.md +142 -0
  283. package/skills/planning/prd/SKILL.md +47 -7
  284. package/skills/planning/prd/evals.md +54 -0
  285. package/skills/planning/product-intent/SKILL.md +58 -3
  286. package/skills/planning/product-intent/evals.md +70 -0
  287. package/skills/planning/tech-spec/SKILL.md +54 -3
  288. package/skills/planning/tech-spec/evals.md +73 -0
  289. package/skills/registry.yaml +70 -62
  290. package/skills/remote.yaml +13 -0
  291. package/skills/research/arxiv-literature/SKILL.md +77 -19
  292. package/skills/research/arxiv-literature/evals.md +50 -0
  293. package/skills/research/experiment-protocol/SKILL.md +21 -2
  294. package/skills/research/experiment-protocol/evals.md +23 -0
  295. package/skills/research/scientific-debugging/SKILL.md +24 -2
  296. package/skills/research/scientific-debugging/evals.md +18 -0
  297. package/skills/research/scientific-modernization/SKILL.md +27 -2
  298. package/skills/research/scientific-modernization/evals.md +27 -0
  299. package/skills/skill-marketplace.json +97 -62
  300. package/skills/workflow/cut-it/SKILL.md +66 -6
  301. package/skills/workflow/cut-it/evals.md +101 -0
  302. package/skills/workflow/design-council/SKILL.md +118 -28
  303. package/skills/workflow/design-council/evals.md +161 -0
  304. package/skills/workflow/grill-me/SKILL.md +87 -11
  305. package/skills/workflow/grill-me/evals.md +153 -0
  306. package/skills/workflow/workflow-distiller/SKILL.md +77 -18
  307. package/skills/workflow/workflow-distiller/evals.md +118 -0
  308. package/src/cli/args.ts +2 -2
  309. package/src/cli/bootstrap-generate.ts +1 -1
  310. package/src/cli/config-inspect.ts +65 -12
  311. package/src/cli/configure-interop.ts +105 -13
  312. package/src/cli/configure-oauth.ts +57 -0
  313. package/src/cli/configure-onboarding.ts +980 -0
  314. package/src/cli/configure-target.ts +594 -0
  315. package/src/cli/configure.ts +1082 -532
  316. package/src/cli/context-map.ts +114 -0
  317. package/src/cli/context.ts +4 -0
  318. package/src/cli/docs.ts +22 -14
  319. package/src/cli/doctor-naming.ts +5 -5
  320. package/src/cli/doctor-toolchain.ts +3 -3
  321. package/src/cli/eval.ts +1 -2
  322. package/src/cli/extensions.ts +2 -1
  323. package/src/cli/fleet.ts +1 -1
  324. package/src/cli/index.ts +3 -1
  325. package/src/cli/internal-dispatch.ts +3 -4
  326. package/src/cli/lifecycle-presenter.ts +436 -0
  327. package/src/cli/models.ts +10 -2
  328. package/src/cli/modes/print.ts +5 -1
  329. package/src/cli/panes.ts +19 -5
  330. package/src/cli/reset.ts +228 -106
  331. package/src/cli/run.ts +9 -4
  332. package/src/cli/select.ts +664 -0
  333. package/src/cli/share.ts +5 -1
  334. package/src/cli/skills-eval.ts +3 -3
  335. package/src/cli/skills.ts +9 -2
  336. package/src/cli/targets.ts +5 -6
  337. package/src/cli/trace.ts +55 -4
  338. package/src/cli/uninstall.ts +233 -165
  339. package/src/cli/upgrade.ts +204 -149
  340. package/src/cli/usage.ts +86 -27
  341. package/src/cli/validate-model.ts +3 -3
  342. package/src/cli/wiki-generate.ts +1 -1
  343. package/src/core/artifact-paths.ts +1 -1
  344. package/src/core/bash-exec.ts +131 -86
  345. package/src/core/bus-events.ts +51 -6
  346. package/src/core/config.ts +61 -1
  347. package/src/core/defaults.ts +7 -4
  348. package/src/core/dispatch-outcome.ts +16 -0
  349. package/src/core/external-diagnostic.ts +44 -0
  350. package/src/core/gateway-routing.ts +157 -0
  351. package/src/core/guardrails.ts +10 -49
  352. package/src/core/prompt-hint.ts +9 -0
  353. package/src/core/safe-exec.ts +17 -2
  354. package/src/core/skill-activation.ts +89 -2
  355. package/src/domains/agents/builtins/architect.md +2 -3
  356. package/src/domains/agents/builtins/coder.md +3 -2
  357. package/src/domains/agents/builtins/debugger.md +2 -2
  358. package/src/domains/agents/builtins/documenter.md +2 -2
  359. package/src/domains/agents/builtins/git-master.md +1 -1
  360. package/src/domains/agents/builtins/oracle.md +1 -1
  361. package/src/domains/agents/builtins/provenance.md +1 -1
  362. package/src/domains/agents/builtins/researcher.md +1 -1
  363. package/src/domains/agents/builtins/scout.md +1 -1
  364. package/src/domains/agents/builtins/tester.md +2 -2
  365. package/src/domains/agents/builtins/verifier.md +2 -2
  366. package/src/domains/agents/builtins/wiki-writer.md +1 -1
  367. package/src/domains/agents/builtins/world-knowledge.md +31 -0
  368. package/src/domains/agents/catalog.ts +13 -15
  369. package/src/domains/agents/contract.ts +2 -0
  370. package/src/domains/agents/extension.ts +23 -1
  371. package/src/domains/agents/result-contract.ts +70 -0
  372. package/src/domains/config/keybindings.ts +8 -0
  373. package/src/domains/context/extension.ts +0 -3
  374. package/src/domains/context/wiki/map-seed.ts +589 -0
  375. package/src/domains/context/wiki/plan.ts +2 -2
  376. package/src/domains/context/working-set/path-index.ts +1 -0
  377. package/src/domains/dispatch/admission.ts +29 -0
  378. package/src/domains/dispatch/agent-candidates.ts +10 -0
  379. package/src/domains/dispatch/budget-envelope.ts +86 -1
  380. package/src/domains/dispatch/capability-match.ts +11 -0
  381. package/src/domains/dispatch/capacity-lease.ts +17 -0
  382. package/src/domains/dispatch/contract.ts +11 -1
  383. package/src/domains/dispatch/extension.ts +237 -49
  384. package/src/domains/dispatch/host-verification.ts +435 -39
  385. package/src/domains/dispatch/intent-requirements.ts +10 -0
  386. package/src/domains/dispatch/intent.ts +18 -1
  387. package/src/domains/dispatch/path-scope.ts +235 -24
  388. package/src/domains/dispatch/run-event-journal.ts +4 -15
  389. package/src/domains/dispatch/state.ts +2 -3
  390. package/src/domains/dispatch/transport.ts +45 -21
  391. package/src/domains/dispatch/types.ts +58 -3
  392. package/src/domains/dispatch/worker-model-metadata.ts +38 -0
  393. package/src/domains/eval/artifacts/store.ts +5 -0
  394. package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
  395. package/src/domains/eval/metrics/token-stream.ts +201 -31
  396. package/src/domains/eval/metrics/tracked.ts +40 -4
  397. package/src/domains/eval/runners/clio-run.ts +5 -2
  398. package/src/domains/eval/schema/suite.ts +28 -0
  399. package/src/domains/eval/schema/verdict.ts +2 -2
  400. package/src/domains/eval/store.ts +8 -1
  401. package/src/domains/eval/suites/resolve.ts +13 -1
  402. package/src/domains/eval/suites/run.ts +24 -3
  403. package/src/domains/evidence/trust-status.ts +10 -1
  404. package/src/domains/extensions/contract.ts +15 -1
  405. package/src/domains/extensions/discovery.ts +238 -41
  406. package/src/domains/extensions/extension.ts +105 -6
  407. package/src/domains/extensions/index.ts +24 -0
  408. package/src/domains/extensions/integrity.ts +189 -0
  409. package/src/domains/extensions/manager.ts +17 -1
  410. package/src/domains/extensions/resource-path.ts +27 -0
  411. package/src/domains/extensions/resources.ts +18 -38
  412. package/src/domains/extensions/snapshot-store.ts +39 -0
  413. package/src/domains/extensions/snapshot.ts +180 -0
  414. package/src/domains/extensions/state.ts +385 -57
  415. package/src/domains/extensions/types.ts +118 -1
  416. package/src/domains/interop/registry.ts +6 -2
  417. package/src/domains/interop/types.ts +4 -0
  418. package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
  419. package/src/domains/lifecycle/migrations/index.ts +6 -0
  420. package/src/domains/lifecycle/naming-resources.ts +19 -4
  421. package/src/domains/lifecycle/naming-yazi.ts +10 -5
  422. package/src/domains/memory/task-memory-policy.ts +70 -26
  423. package/src/domains/memory/task-memory-telemetry.ts +1 -0
  424. package/src/domains/middleware/contract.ts +26 -0
  425. package/src/domains/middleware/extension.ts +24 -24
  426. package/src/domains/middleware/hook-receipts.ts +27 -4
  427. package/src/domains/middleware/hooks-io.ts +65 -32
  428. package/src/domains/middleware/hooks.ts +64 -0
  429. package/src/domains/middleware/index.ts +28 -5
  430. package/src/domains/middleware/marketplace-offer.ts +3 -35
  431. package/src/domains/middleware/memory-intervention.ts +127 -32
  432. package/src/domains/middleware/memory-step-endpoint.ts +3 -2
  433. package/src/domains/middleware/registrations.ts +326 -0
  434. package/src/domains/middleware/runtime.ts +28 -0
  435. package/src/domains/middleware/skills-reminder.ts +31 -2
  436. package/src/domains/middleware/snapshot.ts +20 -7
  437. package/src/domains/mux/contract.ts +38 -0
  438. package/src/domains/mux/detect.ts +6 -13
  439. package/src/domains/mux/index.ts +1 -1
  440. package/src/domains/mux/operations.ts +44 -5
  441. package/src/domains/mux/yazi/assets/yazi.toml +2 -2
  442. package/src/domains/mux/yazi/session.ts +53 -4
  443. package/src/domains/mux/yazi/theme.ts +117 -17
  444. package/src/domains/observability/compaction-usage.ts +118 -0
  445. package/src/domains/observability/contract.ts +10 -11
  446. package/src/domains/observability/cost.ts +1 -1
  447. package/src/domains/observability/extension.ts +17 -4
  448. package/src/domains/observability/out-of-turn-usage.ts +52 -21
  449. package/src/domains/observability/projection.ts +14 -90
  450. package/src/domains/observability/trace-store.ts +43 -7
  451. package/src/domains/prompts/compiler.ts +73 -53
  452. package/src/domains/prompts/contract.ts +15 -3
  453. package/src/domains/prompts/extension.ts +97 -9
  454. package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
  455. package/src/domains/prompts/fragments/identity/clio.md +6 -12
  456. package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
  457. package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
  458. package/src/domains/prompts/fragments/operating/contract.md +7 -15
  459. package/src/domains/prompts/fragments/operating/delegation.md +32 -34
  460. package/src/domains/prompts/fragments/operating/skills.md +10 -24
  461. package/src/domains/prompts/fragments/operating/worker.md +1 -8
  462. package/src/domains/providers/contract.ts +4 -1
  463. package/src/domains/providers/extension.ts +40 -9
  464. package/src/domains/providers/index.ts +1 -1
  465. package/src/domains/providers/model-capabilities.ts +9 -0
  466. package/src/domains/providers/model-discovery.ts +2 -0
  467. package/src/domains/providers/model-runtime-capabilities.ts +99 -25
  468. package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
  469. package/src/domains/providers/runtime-resolution.ts +31 -0
  470. package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
  471. package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
  472. package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
  473. package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
  474. package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
  475. package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
  476. package/src/domains/providers/support.ts +11 -5
  477. package/src/domains/providers/target-model-cache.ts +25 -2
  478. package/src/domains/providers/types/capability-flags.ts +2 -0
  479. package/src/domains/providers/types/cost-provenance.ts +19 -0
  480. package/src/domains/providers/types/local-model-quirks.ts +85 -37
  481. package/src/domains/providers/types/runtime-descriptor.ts +20 -1
  482. package/src/domains/providers/types/target-descriptor.ts +19 -0
  483. package/src/domains/resources/index.ts +3 -0
  484. package/src/domains/resources/skills/install.ts +72 -7
  485. package/src/domains/resources/skills/loader.ts +23 -19
  486. package/src/domains/resources/skills/marketplace.ts +63 -11
  487. package/src/domains/safety/autonomy.ts +15 -0
  488. package/src/domains/safety/call-target.ts +1 -1
  489. package/src/domains/safety/index.ts +1 -0
  490. package/src/domains/safety/loop-detector.ts +7 -4
  491. package/src/domains/safety/path-policy.ts +1 -1
  492. package/src/domains/safety/policy-engine.ts +34 -11
  493. package/src/domains/safety/protected-artifacts.ts +191 -88
  494. package/src/domains/safety/run-effects.ts +2 -22
  495. package/src/domains/safety/skill-authority.ts +55 -0
  496. package/src/domains/session/compaction/compact.ts +72 -22
  497. package/src/domains/session/entries.ts +6 -0
  498. package/src/domains/session/task-board.ts +10 -9
  499. package/src/domains/session/usage.ts +3 -3
  500. package/src/domains/share/archive.ts +164 -7
  501. package/src/engine/acp/server.ts +62 -9
  502. package/src/engine/agent.ts +13 -3
  503. package/src/engine/ai.ts +26 -8
  504. package/src/engine/antigravity/subprocess-runtime.ts +386 -120
  505. package/src/engine/api-registry.ts +3 -0
  506. package/src/engine/apis/llamacpp-residency.ts +3 -4
  507. package/src/engine/apis/lmstudio.ts +3 -3
  508. package/src/engine/apis/ollama-native.ts +6 -6
  509. package/src/engine/apis/openai-completions.ts +145 -39
  510. package/src/engine/apis/output-budget.ts +8 -18
  511. package/src/engine/apis/residency.ts +8 -27
  512. package/src/engine/external-subprocess.ts +114 -6
  513. package/src/engine/gemma-channel-filter.ts +19 -0
  514. package/src/engine/loop-guard.ts +92 -12
  515. package/src/engine/worker-runtime.ts +40 -11
  516. package/src/engine/worker-tools.ts +3 -1
  517. package/src/entry/background-model-metadata.ts +18 -0
  518. package/src/entry/compaction-prompt.ts +57 -0
  519. package/src/entry/extension-hook-sources.ts +28 -0
  520. package/src/entry/extension-reload.ts +309 -0
  521. package/src/entry/orchestrator.ts +464 -251
  522. package/src/entry/task-memory-lifecycle.ts +35 -0
  523. package/src/interactive/application-controller.ts +2 -1
  524. package/src/interactive/bus-notices.ts +8 -1
  525. package/src/interactive/chat-loop-messages.ts +16 -17
  526. package/src/interactive/chat-loop.ts +75 -3
  527. package/src/interactive/chat-panel.ts +36 -13
  528. package/src/interactive/chat-renderer.ts +72 -7
  529. package/src/interactive/cost-overlay.ts +26 -2
  530. package/src/interactive/dispatch-board.ts +6 -11
  531. package/src/interactive/footer/widgets.ts +13 -0
  532. package/src/interactive/interactive-application.ts +39 -4
  533. package/src/interactive/interactive-input-runtime.ts +4 -0
  534. package/src/interactive/interactive-presentation.ts +2 -2
  535. package/src/interactive/interactive-slash-runtime.ts +4 -1
  536. package/src/interactive/overlays/extensions.ts +9 -1
  537. package/src/interactive/overlays/help-reference.ts +13 -0
  538. package/src/interactive/overlays/settings.ts +27 -16
  539. package/src/interactive/panes-runtime.ts +111 -35
  540. package/src/interactive/prompt-cache-identity.ts +88 -0
  541. package/src/interactive/renderers/worker-entry.ts +32 -0
  542. package/src/interactive/slash-commands.ts +153 -20
  543. package/src/interactive/stream-pacing-policy.ts +0 -23
  544. package/src/interactive/theme/labels.ts +19 -13
  545. package/src/interactive/turn-context.ts +39 -20
  546. package/src/interactive/turn-recovery.ts +8 -0
  547. package/src/interactive/turn-runtime.ts +27 -11
  548. package/src/interactive/turn-state.ts +7 -0
  549. package/src/interactive/worker-receipts.ts +1 -0
  550. package/src/interactive/worker-stream.ts +6 -1
  551. package/src/interactive/yazi-bridge.ts +60 -6
  552. package/src/tools/agent-tools.ts +30 -1
  553. package/src/tools/artifact.ts +2 -2
  554. package/src/tools/ask-user.ts +3 -3
  555. package/src/tools/bash.ts +1 -1
  556. package/src/tools/bootstrap.ts +4 -0
  557. package/src/tools/builtin-tool-catalog.ts +52 -22
  558. package/src/tools/codewiki/code-nav-surface.ts +6 -0
  559. package/src/tools/codewiki/code-nav.ts +99 -13
  560. package/src/tools/context/docs-engine.ts +20 -7
  561. package/src/tools/context/index.ts +59 -21
  562. package/src/tools/core-bootstrap.ts +28 -6
  563. package/src/tools/credential-present.ts +1 -2
  564. package/src/tools/dispatch-arguments.ts +6 -1
  565. package/src/tools/dispatch-event-text.ts +10 -0
  566. package/src/tools/dispatch-plan.ts +49 -4
  567. package/src/tools/dispatch-run-events.ts +1 -1
  568. package/src/tools/dispatch-runner.ts +12 -0
  569. package/src/tools/dispatch-schema.ts +338 -0
  570. package/src/tools/dispatch-types.ts +3 -0
  571. package/src/tools/dispatch.ts +9 -254
  572. package/src/tools/ledger.ts +3 -5
  573. package/src/tools/monitor-surface.ts +5 -13
  574. package/src/tools/observation.ts +4 -5
  575. package/src/tools/panes-surface.ts +4 -11
  576. package/src/tools/panes.ts +4 -2
  577. package/src/tools/policy.ts +15 -2
  578. package/src/tools/read.ts +5 -6
  579. package/src/tools/registry.ts +41 -12
  580. package/src/tools/result-shaping.ts +18 -14
  581. package/src/tools/steer-surface.ts +1 -1
  582. package/src/tools/tasks.ts +1 -1
  583. package/src/tools/truncate.ts +6 -5
  584. package/src/tools/verify/surface.ts +6 -12
  585. package/src/tools/web-fetch-surface.ts +1 -3
  586. package/src/tools/worker-evidence.ts +3 -1
  587. package/src/worker/spec-contract.ts +4 -0
  588. package/dist/builtins-UJLMOVOV.js +0 -17
  589. package/dist/chunk-5QIAJV2D.js +0 -48
  590. package/dist/chunk-JZWT5J3Y.js +0 -814
  591. package/dist/chunk-K7VKOLQQ.js +0 -15
  592. package/dist/chunk-PMZCIOCJ.js +0 -25
  593. package/dist/chunk-SUW5DORT.js +0 -819
  594. package/dist/chunk-UOV2BYIW.js +0 -107
  595. package/dist/chunk-WR6U3OVP.js +0 -45
  596. package/dist/chunk-Y45G3AXC.js +0 -1558
  597. package/dist/reset-EOLM7GVE.js +0 -230
  598. package/dist/uninstall-N34PCTGJ.js +0 -331
  599. package/dist/upgrade-H7TOM7YL.js +0 -323
  600. package/docs/artifact-versions.md +0 -67
  601. package/docs/development-pipeline.md +0 -121
  602. package/docs/documentation-coverage.md +0 -46
  603. package/docs/documentation-guide.md +0 -167
  604. package/docs/time-conventions.md +0 -101
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: grill-me
3
- description: Use when the user wants a plan, design, or idea stress-tested through a phased one-question-at-a-time interview before any code is written, or when intent is too ambiguous to plan from. Scans available context first, reviews known facts, fills missing decisions, respects stop signals, and ends with a compact decision log. Triggers on "grill me", "interview me", "stress-test this plan", "poke holes in this".
3
+ description: Stress-tests a plan, design, or idea through a phased one-question-at-a-time interview before code is written, ending in a compact decision log. Not for a multi-perspective debate; use design-council.
4
4
  triggers:
5
5
  - grill me
6
6
  - interview me about this plan
7
7
  - stress-test this design
8
8
  - poke holes in this idea
9
9
  - clarify this plan one question at a time
10
- version: 0.3.3
10
+ version: 0.5.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
@@ -33,9 +33,54 @@ Run a rigorous, repo-aware interview that turns a vague plan into explicit
33
33
  decisions. The point is not to interrogate for sport; it is to surface hidden
34
34
  branches before anyone writes code.
35
35
 
36
+ ## Arguments
37
+
38
+ ```text
39
+ grill me on <plan, feature, or idea>
40
+ ```
41
+
42
+ There is no flag syntax; the trigger is conversational — "grill me on X",
43
+ "stress-test this design", "poke holes in this idea". Whatever the user
44
+ names is the subject. A referenced file, doc, or repo path in the same
45
+ request is Step 1's grounding to read first, not a separate argument.
46
+
47
+ **This run has no back-and-forth.** There is no second turn in which a user
48
+ reads your question and replies to it — whatever you ask, you must also
49
+ answer yourself, in this same turn, before it ends. Do not write a question
50
+ and stop to wait for a reply, in `ask_user` or in plain chat text; nothing
51
+ is coming. Ending the turn on an open question — even one question, even a
52
+ well-posed one — is this skill's single most common failure and worse than
53
+ skipping the interview format entirely.
54
+
55
+ There is no operator in a headless run. `ask_user` still executes — it is
56
+ registered and the call succeeds — but nothing answers it: every round
57
+ returns `{cancelled: true}` immediately, as an ordinary result, not an
58
+ error, every time, with no exceptions. Calling it again will not produce a
59
+ different result, so one call is enough to confirm it (not required —
60
+ reasoning from this paragraph alone is just as valid as calling it and
61
+ observing the cancellation). Whichever phase this lands in — even round
62
+ 1 — switch immediately to the assumed-confirm monologue for every phase
63
+ from here on, in the same turn: state the question you would have asked,
64
+ give your own best/recommended answer with the reasoning behind it, mark
65
+ it `assumed — confirm`, and move to the next phase. Do not re-call
66
+ `ask_user` hoping a later round behaves differently — that only burns the
67
+ `max_rounds` budget without ever converging. Keep working the phase map,
68
+ phase by phase, all the way through Step 5's decision log before ending
69
+ the turn — never end on "Answer 1/2/3" or any other place a reply is
70
+ expected.
71
+
72
+ The phase map below is the plan; do not open a task list for it. This
73
+ skill's tool surface is exactly `read`, `grep`, `ls`, `find`, `git`,
74
+ `context`, `code_nav`, and `ask_user` (`context` and `ask_user` are always
75
+ available regardless). `tasks` and `bash` both sit outside it and any call
76
+ to either is refused — inspect a file with `read`, not `bash cat`/`bash
77
+ head`/`bash wc`; locate files with `find` or `ls`, not `bash find`/`bash
78
+ ls`; check repo state with the `git` tool, not shell `git`.
79
+
36
80
  ## Operating Contract
37
81
 
38
- - Use `ask_user` for the interview whenever it is active.
82
+ - In a live session, use `ask_user` for the interview and actually wait for
83
+ the user's answer between rounds. In a headless run, see Arguments above.
39
84
  - For every interview round, call `ask_user` with `mode: "single_question"` and
40
85
  exactly one question.
41
86
  - On the first ask for a normal grill-me run, set `max_rounds` to a bounded
@@ -45,8 +90,6 @@ branches before anyone writes code.
45
90
  alternatives with short tradeoff descriptions.
46
91
  - The user answers in natural language. You translate answers into compact
47
92
  decision keys and rationale when you call `ask_user` with `action: "complete"`.
48
- - If `ask_user` is unavailable, ask in plain text, still one question at a
49
- time, and keep an internal decision log.
50
93
 
51
94
  ## Phase Map
52
95
 
@@ -69,7 +112,11 @@ or "fill" when the decision is genuinely missing.
69
112
 
70
113
  Read what the user already gave you. If the task references files, plans, code,
71
114
  tests, or project conventions, inspect them before asking. Prefer
72
- `context(scope="workspace")`, `grep`, `read`, and codewiki tools over guessing.
115
+ `context(scope="workspace")`, `grep`, `read`, `code_nav` (symbol and call-graph
116
+ lookups), and codewiki tools over guessing. Use the `git` tool (`status`,
117
+ `log`, not shell `git`) when the plan references repo state — recent
118
+ history, uncommitted changes, what "decided" actually means for this repo
119
+ right now.
73
120
 
74
121
  Privately build a phase map:
75
122
 
@@ -112,10 +159,26 @@ Good: "Which user should v1 optimize for first?"
112
159
  If an answer is vague, ask a follow-up on the same branch. Do not jump to a new
113
160
  branch while the current one is still unresolved.
114
161
 
162
+ **Headless: there is no reply coming, whether `ask_user` comes back
163
+ `cancelled` or you never call it at all.** Neither is a vague answer to
164
+ follow up on and neither is a signal to try again or to wait — both mean
165
+ there is no operator this run, from round 1 on. Do not call `ask_user`
166
+ again for this or any later phase, and do not phrase a question in plain
167
+ text as if a reply is pending. From here, run every remaining phase
168
+ (including this one) as the assumed-confirm monologue described in
169
+ Arguments, in this same turn, through to the Step 5 decision log. See
170
+ Arguments for the exact treatment.
171
+
115
172
  ### Step 4 - Respect Stop Signals
116
173
 
117
- Stop immediately when the user says "stop", "enough", "later", "done", "next
118
- time", or cancels the modal. Do not ask another question to confirm stopping.
174
+ This step applies to a live session with a real operator. Stop immediately
175
+ when the user says "stop", "enough", "later", "done", "next time", or cancels
176
+ the modal. Do not ask another question to confirm stopping.
177
+
178
+ A headless `cancelled` result is not a stop signal from a user — it is the
179
+ absence of an operator (see Step 3). Do not treat it as "the user cancelled
180
+ this session"; treat it as the cue to switch to the assumed-confirm
181
+ monologue and keep going to a complete decision log, not to stop early.
119
182
 
120
183
  If you have enough decisions to be useful, call:
121
184
 
@@ -140,8 +203,11 @@ state the partial decisions and the next unresolved root question.
140
203
 
141
204
  ### Step 5 - Complete
142
205
 
143
- Before final prose, call `ask_user` with `action: "complete"` and a compact
144
- `decisions` array. Then write the final decision log:
206
+ In a live session, close with `ask_user` `action: "complete"` and a compact
207
+ `decisions` array before final prose. In a headless run where `ask_user`
208
+ already came back cancelled, skip straight to the decision log below — do
209
+ not attempt another `ask_user` call just to close out; it will cancel too
210
+ and adds nothing. Write the final decision log:
145
211
 
146
212
  ```markdown
147
213
  ## Decision Log - <topic>
@@ -189,4 +255,14 @@ Use this ordering when several questions are possible:
189
255
  - Asking about facts discoverable from the repo.
190
256
  - Letting `ask_user` hit the round limit without completing the interview.
191
257
  - Ending with a summary paragraph instead of the decision log.
192
- - Treating cancellation as permission to keep asking.
258
+ - Re-calling `ask_user` for a later phase after an earlier round already came
259
+ back `cancelled` — the answer will not be different; that budget is wasted.
260
+ - Ending a turn on "Answer 1/2/3", "let me know which you prefer", or any
261
+ other wording that expects a reply in a headless run — there is no next
262
+ turn for a reply to land in. This is the single most common failure mode
263
+ of this skill and the one to watch hardest for: asking one question, then
264
+ stopping, instead of running the assumed-confirm monologue through every
265
+ remaining phase to the decision log in the same turn.
266
+ - Opening a `tasks` list for the phase map; `tasks` is refused.
267
+ - Treating a headless `cancelled` result as the user's stop signal (Step 4)
268
+ instead of the absence-of-operator cue it actually is.
@@ -76,3 +76,156 @@ Expected:
76
76
 
77
77
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
78
78
  (30B local, llamacpp on mini), full-auto sandbox. PASS (smoke). Skill loaded and interviewed; judge emitted nothing (truncation).
79
+
80
+ ## Battletest record (2026-09-03)
81
+
82
+ Fixture: `/home/akougkas/eval-temp/harness/test_grillme.py`, continuing the
83
+ planning category's shared HPC log-triage domain. Seeds the *decided* v1
84
+ architecture doc (`docs/hpc-log-triage-architecture.md`: on-demand reads
85
+ chosen over an always-on pipeline, alerting explicitly listed "Not
86
+ decided"), the partial `src/scanner.py` (`FailureEvent` + OOM-only
87
+ `scan_oom`), and a `pyproject.toml` declaring `pytest` as the test runner
88
+ — a repo fact the interview must find via `read`, not ask about (evals.md
89
+ S2). The prompt is S1's vague-but-grounded feature request: "grill me on
90
+ adding Slack alerting... nothing about notifications or Slack is decided
91
+ yet." `ask_user` in this harness auto-cancels immediately with no operator
92
+ (confirmed by the planning category and reconfirmed here), which makes
93
+ this skill's entire premise — a one-question-at-a-time live interview — the
94
+ thing under test. Graded 10 checks against the reconstructed final
95
+ assistant text and the raw JSONL's tool-call/safety-block stream: zero
96
+ safety blocks; zero `tasks` calls; no wasted `ask_user` retries (<=3
97
+ calls); repo facts (`scan_oom`, `FailureEvent`, on-demand, `click`,
98
+ `pytest`, `scan_ecc`/`scan_xid`, top-3) grounded in the final text; the
99
+ Step 5 decision-log shape present (`Decision Log`/`Deferred`/`Open
100
+ risks`/`Recommended next step`, not a summary paragraph); decisions
101
+ numbered; the `assumed — confirm` monologue used 3+ times; the Question
102
+ Priority order respected (user/problem framing before naming/polish); and
103
+ **the phase map actually completed** (4+ distinct phase numbers named,
104
+ not a stall after phase 0/1). S3 (user defers) and S5 (stop signal) are
105
+ not separately exercised — a headless run cannot produce a live "whatever
106
+ you think" or "stop" reply to react to; by construction, the
107
+ assumed-confirm monologue *is* the "whatever you think" case running for
108
+ every decision, so S3's expected behavior (record the recommendation,
109
+ say so explicitly) is exercised implicitly on every phase, every run.
110
+ S4 (long phased interview, near round-limit closeout) was not exercised
111
+ standalone. Primary model `qwen3.8-27b` on `dynamo` (LM Studio); cross-model
112
+ confirm on `ornith1.5-35b-moe` on `mini` (llama.cpp), required per this
113
+ session's brief because a prior pass in the same mission found fixes tuned
114
+ on one model family did not always transfer to another.
115
+
116
+ | run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
117
+ |---|---|---|---|---|---|---|---|
118
+ | baseline (no skill) | qwen3.8-27b | 42s | 3 | 32.5k / 3.7k | 0 | 5/10 | never invoked `/skill grill-me`; read both key files and grepped for "alerting", grounded correctly, but produced one live-style question and stopped — no phase map, no decision log (expected: this is what the skill exists to fix) |
119
+ | v1 (frozen 0.4.0) | qwen3.8-27b | 57s | 7 | 91.8k / 5.2k | 2 (1 real: `tasks` refused; 1 benign ENOENT on a fixture-dangling doc path) | 3/10 | opened a `tasks` plan for its own phase map (refused — `tasks` was never called out as off-surface in the frozen body), scanned thoroughly, then asked exactly **one** question in plain text and ended the turn — never called `ask_user` at all, never produced a decision log. This is the exact failure the session was built to catch: the frozen skill's fallback line ("if `ask_user` is unavailable, ask in plain text... keep an internal decision log") reads as permission to have an ordinary single-question live conversation, and the model took it literally |
120
+ | v2 (first hardened cut, 0.5.0) | qwen3.8-27b | 121s | 6 | 78.9k / 11.2k | 1 (benign ENOENT, same dangling doc path) | 9/10 | scanned via `git`/`read`/`ls`, reasoned "ask_user is not in my tool surface" (not quite accurate — it's always exempt — but harmless), and correctly ran the full assumed-confirm monologue through Phases 0-5 to a complete, correctly-shaped decision log, entirely without ever calling `ask_user`. Fixture's dangling PRD reference removed after this run to isolate real vs. benign blocks |
121
+ | v2 re-run (fixture fixed) | qwen3.8-27b | 134s | 7 | 101.0k / 12.1k | 1 real: `bash wc`/`bash head` on the sample log, refused | 9/10 | same strong monologue and decision log; the one new gap was a `bash` reflex the frozen and first-cut bodies never explicitly forbade (grill-me never had `bash` in `allowed-tools`, but nothing told the model not to reach for it) |
122
+ | v3 (bash-refusal added) | qwen3.8-27b | 97s | 5 | n/a | 0 | 6/10 | `bash` reflex gone, but a **regression**: stated "this harness has no `ask_user` tool" (still not quite accurate) and, without ever calling `ask_user`, asked one plain-text question and ended the turn on "Answer 1 / 2 / 3" — the exact frozen-skill failure recurring under the hardened body, proving the earlier fix's trigger ("switch on the first `cancelled` result") had a gap: a model that never calls `ask_user` at all never receives a `cancelled` result to trigger the switch |
123
+ | v4 (no-back-and-forth line added) | qwen3.8-27b | 124s | 6 | 85.1k / 11.0k | 0 | **10/10** | full 9-round monologue (Phases 0-5), correctly-shaped decision log, zero `tasks`/`bash` calls |
124
+ | v5 (stability re-run) | qwen3.8-27b | 95s | 5 | n/a | 0 | **10/10** | repeat of v4's result, confirming v3's regression was not the new steady state |
125
+ | final (cross-model confirm) | ornith1.5-35b-moe (mini) | 64s | 5 | 12.5k / 5.4k | 0 | **10/10** | full monologue and decision log on the second model family too; independently proposed a Slack-message trigger threshold not in the fixture and, in its closing line, distinguished what it would still "press on live" from what it correctly resolved on its own headless — the clearest sign the headless/live distinction actually landed as a real distinction, not just prose the model echoes back |
126
+
127
+ **Changes** (0.4.0 -> 0.5.0):
128
+
129
+ 1. **`## Arguments` contract**, the section the skill never had — slash-free
130
+ conversational syntax, and the ported no-operator/`ask_user`-auto-cancels
131
+ rule from the planning category, adapted for a skill whose entire
132
+ contract is "one question at a time" rather than a document write.
133
+ 2. **The critical fix, found empirically, not guessed up front**: the
134
+ frozen skill's fallback line ("if `ask_user` is unavailable, ask in
135
+ plain text... one question at a time... internal decision log") reads,
136
+ correctly, as "have a normal one-question conversation" — which is
137
+ exactly wrong when there is no second turn for a reply to land in. The
138
+ fix that actually held (v4/v5, and the cross-model run) is a **"this run
139
+ has no back-and-forth" statement that does not gate on receiving a
140
+ `cancelled` result** from `ask_user` — v3's first attempt gated the
141
+ switch-to-monologue on "the first round comes back cancelled," which
142
+ left a real gap: a model that reasons "`ask_user` isn't available" and
143
+ never calls it at all never receives that trigger, and fell straight
144
+ back into the frozen skill's exact failure. The held version states the
145
+ no-reply-coming rule as a fact about the run itself, independent of
146
+ whether `ask_user` was ever invoked.
147
+ 3. **Explicit `tasks` and `bash` refusal**, in Arguments and Red Flags —
148
+ `tasks` was v1's real safety block (opened a plan for its own phase
149
+ map); `bash` was a reflex on the v2 re-run (`bash wc`/`bash head` on a
150
+ file that should have been `read`). Both are outside `allowed-tools`
151
+ already; the gap was that nothing said so in the body.
152
+ 4. **Step 3 and Step 4 rewritten** to name the headless-cancellation
153
+ handling explicitly at the exact point in the workflow it applies (not
154
+ only in Arguments): Step 3 states that neither a `cancelled` result nor
155
+ never calling `ask_user` at all is a reason to wait; Step 4 draws the
156
+ line between a live stop signal (still respected) and a headless
157
+ `cancelled` result (not a stop signal, a cue to keep going).
158
+ 5. **`git` tool and `code_nav` called out explicitly in Step 1** for repo
159
+ state and symbol lookups respectively — `code_nav` remains unexercised
160
+ by every run this session (see Still weak); `git` was used in every
161
+ hardened run once named.
162
+ 6. Five Red Flags entries rewritten or added around the concrete failures
163
+ observed: `tasks`/`bash` reaches, re-calling `ask_user` after a
164
+ cancellation, treating `cancelled` as a user stop signal, and — the
165
+ headline one — ending a turn on "Answer 1/2/3" instead of running the
166
+ monologue to the decision log.
167
+
168
+ **Still weak**: `code_nav` (in `allowed-tools`) was never exercised by any
169
+ run this session — this fixture's grounding lived entirely in prose files
170
+ and one Python module small enough that `read`/`grep` sufficed; a fixture
171
+ with a larger call graph might exercise it, but none was built. S3 and S5
172
+ are reasoned about, not directly run (see the fixture note above) — a
173
+ harness that could inject a specific `ask_user` reply (rather than always
174
+ auto-cancelling) would let those two scenarios run for real instead of by
175
+ inference. S4's near-round-limit closeout behavior is unverified; every
176
+ hardened run here converged well under any plausible `max_rounds` value.
177
+ v3's regression is the one data point worth remembering past this session:
178
+ gating a headless-degradation rule on "the first tool result that comes
179
+ back a certain way" is fragile against a model that skips the tool call
180
+ entirely and reasons its way to the same wrong conclusion by a different
181
+ path — the fix needed to be a fact about the run, not a reaction to one
182
+ tool's return value. Two consecutive clean runs on the primary model and
183
+ one clean cross-model run is reasonable but not exhaustive evidence that
184
+ v3's failure mode is fully closed rather than just less frequent; only a
185
+ larger run count would raise that confidence further.
186
+
187
+ ## Live interactive confirmation (2026-09-03, Herdr pane)
188
+
189
+ The headless harness can only prove the assumed-confirm monologue path; it
190
+ cannot produce a genuine "whatever you think" or "stop" reply because
191
+ `ask_user` always auto-cancels with no operator. To exercise S3 and S5 for
192
+ real, this session ran the hardened 0.5.0 skill interactively: a Herdr pane
193
+ running `clio-coder` (target `mini`, model `ornith1.5-35b-moe`, switched
194
+ in-session via `/model`) against the same fixture repo the battletest used
195
+ (`/home/akougkas/eval-temp/grillme-final`, skill installed project-locally
196
+ at `.clio-coder/skills/grill-me/` so the live session resolved this
197
+ repo's edited SKILL.md rather than the separately npm-installed package
198
+ copy — `/skill grill-me` has no path-override flag the way `clio-coder run
199
+ --skill` does), with a human answering each `ask_user` round live via the
200
+ TUI's modal.
201
+
202
+ Result: Step 1 scanned the repo before asking anything (architecture doc,
203
+ scanner.py, sample log). Round 1 asked a single root-decision question
204
+ (trigger model) with the recommended option first and real tradeoffs on
205
+ the alternatives. After the human's real answer came back, the model's own
206
+ reasoning explicitly named the distinction the whole hardening pass turned
207
+ on: *"The modal returned round_answered — there is an operator here...
208
+ this is a live interview, not headless. I'll continue with one question
209
+ per round and wait for your answers."* — proof the headless/live branch is
210
+ a real fork in the model's behavior, not just prose it echoes. Five rounds
211
+ ran in priority order (trigger, success measure, non-goals, delivery/
212
+ secrets, payload shape); round 3 was answered "whatever you think is
213
+ best" and recorded the stated recommendation as the decision (S3,
214
+ confirmed live, not just implied by the monologue path); round 5 was
215
+ answered with an appended stop signal ("...; stop") and the model called
216
+ `ask_user` `action: "complete"` immediately, with no confirmation question
217
+ (S5, confirmed live). The final decision log matched the Step 5 shape
218
+ exactly (numbered decisions, Deferred, Open risks, Recommended next step)
219
+ and, unprompted, flagged a real cross-cutting risk the fixture didn't
220
+ spell out: the alerting feature's success measure depends on Phase 2's
221
+ ranking/CLI, which doesn't exist in `scanner.py` yet — a genuine "hole
222
+ poked," not filler. Zero safety blocks; zero `tasks` calls. One aside, not
223
+ a skill defect: the TUI's own usage nudge fired ("9+ read-only exploration
224
+ calls without a successful Scout dispatch") — Step 1's repo scan currently
225
+ reaches for `read`/`grep`/`ls` directly rather than delegating broad
226
+ reconnaissance to a Scout dispatch, worth a look in a future pass but out
227
+ of scope for this one since the scan itself was correct and grounded.
228
+
229
+ This closes the S3/S5-not-directly-run gap noted above for the specific
230
+ case of a genuine live operator; the headless assumed-confirm path
231
+ remains separately and repeatedly confirmed on its own terms.
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: workflow-distiller
3
- description: Use when a workflow that just happened should become reusable, when the user says "make this a skill", "package what we just did", "turn this into a workflow", or when the same multi-step process has been repeated across sessions. Reconstructs the workflow from the session record, interviews, checks overlap with installed skills, gates on approval, then writes the SKILL.md following skill-craft. Not for authoring a skill from scratch with no prior workflow; write the SKILL.md directly following skill-craft. Not for distilling into an agent recipe; propose that as a follow-up when the workflow is dispatch-shaped.
3
+ description: "Packages a workflow that just happened into a reusable SKILL.md: reconstructs it from the session record, checks overlap with installed skills, gates on approval, then writes it following skill-craft. Not for authoring a skill with no prior workflow; use skill-craft."
4
4
  triggers:
5
5
  - make this workflow a skill
6
6
  - package what we just did
7
7
  - turn this into a reusable workflow
8
8
  - distill this repeated process
9
9
  - create a skill from this session
10
- version: 0.2.2
10
+ version: 0.4.0
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - read
@@ -38,6 +38,37 @@ following skill-craft. skill-craft also governs how every skill this
38
38
  distiller produces is written: description, body, and pruning rules live
39
39
  there, not here.
40
40
 
41
+ ## Arguments
42
+
43
+ ```text
44
+ /skill workflow-distiller [<what to distill>]
45
+ ```
46
+
47
+ With arguments, the text names the workflow to distill; without, distill
48
+ whichever workflow just ran in this session. Everything else in the request
49
+ (the conversation so far) is the session record Phase 1 reconstructs from,
50
+ not more arguments.
51
+
52
+ A live operator can answer Phase 2's interview and Phase 4's design gate for
53
+ real - treat a real, non-cancelled `ask_user` reply as proof an operator is
54
+ present and continue asking one question per round. There is no operator in
55
+ a headless run: `ask_user` is not registered and every call resolves
56
+ `cancelled` immediately, whether or not it is ever called at all. When that
57
+ is the run's condition, do not wait for a reply that cannot come: answer
58
+ every remaining interview question and the Phase 4 gate yourself as an
59
+ assumed-confirm monologue (state the question, give your best-grounded
60
+ answer or design choice, mark it `assumed - confirm`) and proceed straight
61
+ through to Phase 5, the same way as a live "looks fine, write it." A skill
62
+ file is a plain, reversible artifact under version control, not an
63
+ irreversible external action - write it and say so in the final reply,
64
+ rather than stopping at the gate with nothing produced.
65
+
66
+ `tasks` sits outside this skill's tool surface (`read`, `grep`, `find`,
67
+ `ls`, `context`, `write`, `ask_user`) and any call is refused; the six
68
+ phases above are the plan, not a task list. This skill never has `bash` or
69
+ `git`, so it never commits or pushes the file it writes - say so in the
70
+ final reply and let the user commit it.
71
+
41
72
  ## Phase 1 - Reconstruct From Evidence
42
73
 
43
74
  Before asking anything, list the concrete steps that visibly executed in this
@@ -90,27 +121,47 @@ Validation scenario: <prompt, expected observable behavior>
90
121
  ```
91
122
 
92
123
  No skill file is written before the user approves. "Looks fine, but change X"
93
- means revise and re-present.
124
+ means revise and re-present. In a headless run with no operator (see
125
+ Arguments), treat the design as approved once it is internally consistent
126
+ with Phase 2's decisions, mark it `assumed - confirm` in your final reply,
127
+ and proceed to Phase 5 - do not stop the run with a presented-but-unwritten
128
+ design.
94
129
 
95
130
  ## Phase 5 - Create
96
131
 
97
- Write `SKILL.md` under `.clio-coder/skills/<approved-name>/`, following skill-craft
98
- for the frontmatter contract, a triggers-only third-person description, and
99
- the pruning pass, with `requires: [skill:<name>]` for every skill the overlap
100
- check referenced. Confirm it loads with `clio-coder skills validate`. Scope
101
- defaults to project; use the user skill store only when the user said the
102
- workflow crosses repositories. Placeholders replace every session-specific
103
- path, name, and value; distill the pattern, not the incident. Keep the
104
- generated skill under 120 lines; reference instead of inlining. If
105
- the session repeatedly dispatched the same worker pattern, also offer a recipe
106
- sketch for the agents surface, but do not write recipe files.
132
+ Write `SKILL.md` under `.clio-coder/skills/<approved-name>/` with the
133
+ frontmatter contract below - a triggers-only third-person description, and
134
+ the pruning pass - with `requires: [skill:<name>]` for every skill the
135
+ overlap check referenced. Do not try to load skill-craft's own file mid-run
136
+ to check this: only one skill can be active at a time, and a second
137
+ `context(scope="skills", name="skill-craft")` call is refused while this
138
+ skill is still pending. The frontmatter contract, current as of this
139
+ writing: `name`, `description` (third-person triggers, no "I"/"you"),
140
+ `triggers` (a non-empty list), `version` (start `0.1.0`), `license`,
141
+ `allowed-tools` (canonical lowercase Clio tool names only - `bash`/`git`
142
+ capitalized or spelled differently is rejected), and `requires` when Phase 3
143
+ found a reference. If in doubt about the exact shape, `read` an already-
144
+ installed skill's `SKILL.md` (this skill's own file is always available) and
145
+ mirror its frontmatter keys rather than guessing or trying to load
146
+ skill-craft. Scope defaults to project; use the user skill store only when
147
+ the user said the workflow crosses repositories. Placeholders replace every
148
+ session-specific path, name, and value; distill the pattern, not the
149
+ incident. Keep the generated skill under 120 lines; reference instead of
150
+ inlining. If the session repeatedly dispatched the same worker pattern, also
151
+ offer a recipe sketch for the agents surface, but do not write recipe files.
107
152
 
108
153
  ## Phase 6 - Validate
109
154
 
110
155
  Record one RED-GREEN scenario agreed with the user: the prompt, and the
111
- observable behavior that distinguishes with-skill from without. Run it once if
112
- cheap (a single small headless run); otherwise record it in the skill body's
113
- example section as the standing validation obligation.
156
+ observable behavior that distinguishes with-skill from without. `clio-coder
157
+ skills validate` needs `bash`, which is outside this skill's tool surface -
158
+ never attempt it; instead `read` the file back and confirm the frontmatter
159
+ contract from Phase 5 by eye (required keys present, `allowed-tools` entries
160
+ canonical lowercase, under the line budget), and say plainly that a real
161
+ `clio-coder skills validate` pass is still owed once bash is available.
162
+ Otherwise record the scenario in the skill body's example section as the
163
+ standing validation obligation. This skill has no `bash`/`git`, so it never
164
+ commits or pushes the file it writes; say so and let the user commit it.
114
165
 
115
166
  ## Worked Example
116
167
 
@@ -128,15 +179,23 @@ script, and verified row counts against the source, three sessions in a row.
128
179
  installed; no references.
129
180
  4. Gate: summary presented; user approves after tightening the description.
130
181
  5. Create: write `.clio-coder/skills/csv-ingest/SKILL.md`, placeholders for the
131
- export URL and column map; `clio-coder skills validate` passes.
182
+ export URL and column map; frontmatter contract confirmed by reading the
183
+ file back (no `bash`, so no `clio-coder skills validate` this turn).
132
184
  6. Validate: scenario "ingest this month's export" must show fetch,
133
185
  normalize, count-verify, spot-check in that order; recorded in the body.
134
186
 
135
187
  ## Red Flags
136
188
 
137
- - Writing any skill before the design gate is approved.
189
+ - Writing any skill before the design gate is approved by a live operator,
190
+ or before a headless run has marked it `assumed - confirm`.
191
+ - Stopping the run at Phase 4 with a presented-but-unwritten design when the
192
+ run is headless - that's the backlog pattern (guard an irreversible
193
+ action), and a written skill file is not that; it is reversible.
138
194
  - A reconstruction that lists steps nothing in the session shows.
139
195
  - Reimplementing an installed skill's job instead of referencing it.
140
196
  - Session-specific paths or values surviving into the generated skill.
141
197
  - Batching interview questions or ignoring a stop signal.
142
198
  - Distilling a one-off without asking about recurrence.
199
+ - Calling `bash` for `clio-coder skills validate`, or `context` with a
200
+ second skill's name to consult it mid-run: both are outside this skill's
201
+ tool surface and refused.
@@ -105,3 +105,121 @@ prompt; the phases still had to run in order.
105
105
 
106
106
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
107
107
  (30B local, llamacpp on mini), full-auto sandbox. PASS. Inline session-trace setup engaged; judge 5/6.
108
+
109
+ ## Battletest record (2026-09-03)
110
+
111
+ This skill's core loop (a several-turn interview, a live-or-headless design
112
+ gate, a write) is genuinely interactive, so it was tested primarily through
113
+ a live Herdr-driven `clio-coder` pane with a real human answering
114
+ `ask_user` rounds, not only through the headless harness the rest of this
115
+ category used — plus one headless confirm run to exercise the no-operator
116
+ path directly. Fixture: the shared HPC log-triage repo (`test_grillme.py`'s
117
+ `setup_fixture`, `docs/hpc-log-triage-architecture.md` + partial
118
+ `src/scanner.py`/`tests/test_scanner.py`), compounded with a real Part 1
119
+ task ("add `scan_ecc` mirroring `scan_oom`, test it, commit it") run to
120
+ completion *before* invoking `/skill workflow-distiller`, so Phase 1 had a
121
+ genuine session record to reconstruct from rather than a scripted one.
122
+
123
+ **v1 (frozen 0.3.0), live, `ornith1.5-35b-moe` on mini**: Phase 1
124
+ reconstruction was precise and correctly cited to real tool calls. The
125
+ 5-round interview (confirm+recurrence, varies/fixed, rigidity/failure,
126
+ scope cuts, name) ran cleanly, one question per round, recommended-first.
127
+ Phase 3's overlap check correctly found nothing to reference. Phase 4's
128
+ design gate presented the exact template shape and correctly waited for a
129
+ live "looks fine, write it" before writing anything. Two real bugs
130
+ surfaced in Phase 5/6: (1) "load skill-craft" is dead - `context(scope=
131
+ "skills", name="skill-craft")` while workflow-distiller is itself the
132
+ active pending skill is refused ("pending skill request(s)"); only one
133
+ skill can be active at a time. (2) Phase 6's "confirm it loads with
134
+ `clio-coder skills validate`" is dead - `bash` is outside this skill's
135
+ `allowed-tools`, so that command can never run. The model self-recovered
136
+ both times (read its own installed SKILL.md as a frontmatter mirror; did a
137
+ by-eye read-back instead of the blocked validate command) and reported the
138
+ gap honestly rather than fabricating a clean validate - a good sign for
139
+ model robustness, but the skill body should not depend on a model
140
+ inventing its own workaround for a dead instruction.
141
+
142
+ **Changes (0.3.0 -> 0.4.0)**: new `## Arguments` section stating the
143
+ live-vs-headless distinction explicitly (a real, non-cancelled `ask_user`
144
+ reply proves an operator is present and interview/gate proceed normally;
145
+ `ask_user` unregistered and auto-cancelling means assumed-confirm through
146
+ Phase 2 and Phase 4 and straight on to Phase 5 - a written skill file is a
147
+ reversible, version-controlled artifact, not the kind of irreversible
148
+ action `backlog`'s Step 3 guards, so it does not get that skill's
149
+ stop-and-report treatment); explicit `tasks` refusal and a "never commits"
150
+ note; Phase 4 gate prose updated with the headless branch; Phase 5 rewrote
151
+ the skill-craft consultation into a concrete, achievable frontmatter
152
+ contract (the required keys, and "`read` an already-installed skill's file
153
+ to mirror the shape, never try to load a second skill mid-run"); Phase 6
154
+ rewrote the validate step to a real, achievable by-eye check instead of
155
+ the dead `bash` command; the worked example's line claiming `clio-coder
156
+ skills validate` passes was corrected to match; two new Red Flags entries
157
+ for the two dead-instruction failure modes.
158
+
159
+ **v2 (hardened 0.4.0), live, `ornith1.5-35b-moe` on mini, fresh session**:
160
+ same compound Part 1 + Part 2 scenario end to end. Real finding along the
161
+ way, unrelated to this skill: the model cannot self-invoke `/skill` - it
162
+ is operator-gated UI, not an agent tool, confirmed twice (once via a
163
+ direct `context` call, once by embedding `/skill workflow-distiller` in a
164
+ larger message instead of sending it standalone) - both times the model
165
+ correctly recognized the constraint, said so, and asked the operator to
166
+ run it rather than retrying or faking activation. Once invoked properly:
167
+ Phase 1 reconstruction included the real mid-session detour (a genuine
168
+ case-sensitivity bug the model introduced and fixed via byte-level
169
+ debugging) rather than a cleaned-up story. The interview produced a
170
+ materially different design from v1's run on the same scenario (different
171
+ name chosen, "leave uncommitted" instead of "commit via bash" for the
172
+ generated skill) - real evidence the interview is actually deciding things,
173
+ not replaying a script. Phase 4 correctly said "this is a live run, so
174
+ I'll hold off writing until you approve" and, when a later `ask_user`
175
+ round had already closed, correctly fell back to a plain-text approval
176
+ request rather than stalling - explicitly restating, unprompted, every
177
+ constraint this session's hardening pass had just added (no commit, no
178
+ `clio-coder skills validate`, `tasks` out of surface). Phase 5 went
179
+ straight to `read`-ing its own installed SKILL.md to mirror the frontmatter
180
+ contract, with zero attempt to load skill-craft - bug #1 confirmed fixed.
181
+ The generated `.clio-coder/skills/add-signature-scanner/SKILL.md` (97
182
+ lines) has valid frontmatter, canonical-lowercase `allowed-tools`, real
183
+ placeholders, and a concrete validation scenario.
184
+
185
+ **A real, separate platform-level finding, not a skill-body bug**: during
186
+ this same v2 live session's Phase 6, a `bash wc -l` call executed
187
+ successfully (`exit 0`, green checkmark confirmed via `--format ansi`) even
188
+ though `bash` is not in workflow-distiller's `allowed-tools` - and the
189
+ identical class of call had been correctly refused earlier in this exact
190
+ mission's v1 session under the same skill ("bash is outside the tool
191
+ surface declared by the active skill(s)"). The model's own final report
192
+ then claimed "I can't run bash" in the same breath as having just run it.
193
+ This reads as skill tool-surface narrowing lapsing partway through a long,
194
+ many-turn interactive session (Phase 1 through 6 spans several real
195
+ conversation turns; the headless harness's single-turn monologue shape
196
+ never exercises this), not anything a SKILL.md can fix by itself. Worth a
197
+ maintainer look at the interactive admission path specifically, independent
198
+ of this skill or this mission's edits.
199
+
200
+ **Headless confirm, `qwen3.8-27b` on dynamo, fresh fixture**: single
201
+ `clio-coder run --skill ... --autonomy full-auto --json`. Real fixture
202
+ mismatch, deliberately not corrected: the prompt claimed a prior
203
+ `scan_ecc` commit that does not exist in this fresh fixture (no Part 1 ran
204
+ here). Phase 1 caught the discrepancy against real evidence (`grep` found
205
+ no `scan_ecc`, git log has one seed commit, the docstring still says "not
206
+ yet implemented"), tagged the claim `assumption - unverified`, and
207
+ grounded the distillation in the real architecture doc and `scan_oom`'s
208
+ actual code instead of fabricating verification of a commit that never
209
+ happened - the skill's stated identity ("runtime truth... reconstruction
210
+ wins") holding on a weaker/non-live model under direct pressure to just
211
+ agree. Ran the full assumed-confirm monologue through Phase 2 and Phase 4
212
+ (explicitly marked), wrote `.clio-coder/skills/signature-scanner/SKILL.md`
213
+ (92 lines, valid frontmatter, canonical-lowercase tools, real placeholders,
214
+ a concrete failure-behavior section, a validation scenario naming the still
215
+ -owed real `clio-coder skills validate` pass). Zero safety blocks.
216
+
217
+ **Still weak**: the skill-craft mid-run-load and dead-`bash`-validate bugs
218
+ are confirmed fixed by re-test, but only against this one fixture shape.
219
+ S2 (overlap with an installed skill) was not exercised this pass - this
220
+ fixture's only installed skill is workflow-distiller itself, so the overlap
221
+ check always correctly found nothing; a fixture with a second installed
222
+ skill covering one step is still owed. S3 (no recurrence, offer to stop)
223
+ was not exercised standalone. The tool-surface-lapse finding above is
224
+ real, reproduced, and unresolved - it is the single biggest risk this pass
225
+ surfaced, and it sits outside this skill (and outside `skills/`) entirely.
package/src/cli/args.ts CHANGED
@@ -166,12 +166,12 @@ export function parseRunCliArgs(argv: ReadonlyArray<string>): RunCliArgs {
166
166
  if (value !== null) parsed.agentId = value;
167
167
  continue;
168
168
  }
169
- if (arg === "--agent-profile" || arg === "--worker-profile" || arg === "--worker") {
169
+ if (arg === "--agent-profile") {
170
170
  const value = need(arg);
171
171
  if (value !== null) parsed.agentProfile = value;
172
172
  continue;
173
173
  }
174
- if (arg === "--agent-runtime" || arg === "--worker-runtime" || arg === "--runtime") {
174
+ if (arg === "--agent-runtime") {
175
175
  const value = need(arg);
176
176
  if (value !== null) parsed.agentRuntime = value;
177
177
  continue;
@@ -325,7 +325,7 @@ async function attemptBootstrapDispatch(
325
325
  // and returned zero bytes at 29s. The wiki documenter in this same directory
326
326
  // budgets minutes for the same shape of work. `internalDispatchTimeoutMs` is
327
327
  // the operator's knob for slow targets; nothing here knows better than it does.
328
- const deadline = armInternalDispatchDeadline(dispatch, handle.runId, "context bootstrap", process.env);
328
+ const deadline = armInternalDispatchDeadline(dispatch, handle.runId, "context bootstrap");
329
329
  let text = "";
330
330
  let receipt: RunReceipt | undefined;
331
331
  let parserAttempted = false;