@iowarp/clio-coder 0.4.2 → 0.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (523) hide show
  1. package/CHANGELOG.md +98 -0
  2. package/CONTRIBUTING.md +86 -19
  3. package/README.md +35 -6
  4. package/dist/{acp-TMDQZDIG.js → acp-WNAYYF4F.js} +12 -13
  5. package/dist/{agents-5N5NG3XG.js → agents-3OKXHLOI.js} +60 -57
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-Z5CCBXKQ.js → auth-VKNNMGPU.js} +21 -19
  8. package/dist/{builtins-K6TNDT24.js → builtins-WGALA46I.js} +9 -4
  9. package/dist/{chunk-XE3PCIXH.js → chunk-23L32XTI.js} +12 -9
  10. package/dist/{chunk-I64IFBLB.js → chunk-25QBEXRS.js} +18 -11
  11. package/dist/{chunk-CDNVLKUX.js → chunk-26QSH3EJ.js} +13 -7
  12. package/dist/{chunk-QQLGQY2A.js → chunk-2ASED4PZ.js} +22 -22
  13. package/dist/{chunk-MCEPRMZW.js → chunk-2CU2H6KE.js} +2 -2
  14. package/dist/chunk-2DSOYNFC.js +108 -0
  15. package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
  16. package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
  17. package/dist/{chunk-O3YUNJZ2.js → chunk-2ZSONWVL.js} +82 -25
  18. package/dist/{chunk-2NHR3NAY.js → chunk-36CT5VVL.js} +331 -42
  19. package/dist/{chunk-2X4RYJTJ.js → chunk-3GY4F45V.js} +3 -3
  20. package/dist/{chunk-ZW55JB7N.js → chunk-3ODX73FK.js} +4 -6
  21. package/dist/{chunk-PBP4B7XR.js → chunk-3UNOLWNZ.js} +3 -3
  22. package/dist/{chunk-4JDLP6ZS.js → chunk-3UUXNFEX.js} +14 -10
  23. package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
  24. package/dist/{chunk-ZW4HH5JJ.js → chunk-4M6Z5QVF.js} +6 -6
  25. package/dist/{chunk-K6BSR66V.js → chunk-4NSRCOYP.js} +4 -1
  26. package/dist/{chunk-M2DAX4F6.js → chunk-4WR7VSYB.js} +2 -2
  27. package/dist/{chunk-FSP7CMNU.js → chunk-54X7T7DK.js} +61 -6
  28. package/dist/{chunk-54ODD65L.js → chunk-5636DCO5.js} +4 -4
  29. package/dist/chunk-57XXR6DR.js +3763 -0
  30. package/dist/{chunk-3KIPBMUA.js → chunk-5ICU3EUH.js} +2 -2
  31. package/dist/{chunk-77QIVUZB.js → chunk-5MEZN6CB.js} +4 -4
  32. package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
  33. package/dist/{chunk-YJISEZKC.js → chunk-5TUB6SLS.js} +6 -6
  34. package/dist/{chunk-IMXMHHMQ.js → chunk-6OSVSQL5.js} +341 -57
  35. package/dist/{chunk-Q4XWMHX6.js → chunk-6PAZTBPA.js} +14 -2
  36. package/dist/{chunk-FVDGR2ZL.js → chunk-6Q3CYFD3.js} +112 -39
  37. package/dist/{chunk-IDNA72AH.js → chunk-6QOTUPRG.js} +155 -36
  38. package/dist/{chunk-X7IARSHT.js → chunk-6UINWWS6.js} +16 -10
  39. package/dist/{chunk-CYZW7JHJ.js → chunk-72YIHOZQ.js} +9 -9
  40. package/dist/{chunk-IKSLQ4XV.js → chunk-75W7L2E2.js} +752 -861
  41. package/dist/{chunk-CRFOIAX3.js → chunk-7UGL4MB5.js} +6 -6
  42. package/dist/{chunk-HIICAHCJ.js → chunk-AUPNRN7C.js} +2 -2
  43. package/dist/{chunk-7BHIY2MW.js → chunk-BJVFZO5U.js} +8 -50
  44. package/dist/{chunk-B74PXLU7.js → chunk-CUSRQKPU.js} +65 -3
  45. package/dist/chunk-DQOVN6KV.js +386 -0
  46. package/dist/{chunk-E7GT7O5N.js → chunk-DT3LWJOB.js} +7 -4
  47. package/dist/chunk-DXKJURES.js +671 -0
  48. package/dist/{chunk-JBCS7CRR.js → chunk-EL24TAU4.js} +10 -10
  49. package/dist/{chunk-TPEQIQIE.js → chunk-ELWDPP3Y.js} +8 -8
  50. package/dist/{chunk-NDINPTJ4.js → chunk-ELZVTCGV.js} +5 -4
  51. package/dist/chunk-EXLD33WO.js +381 -0
  52. package/dist/chunk-FEAXX7B6.js +101 -0
  53. package/dist/{chunk-RLYRBIYQ.js → chunk-FFUPXJC4.js} +90 -331
  54. package/dist/{chunk-5PFYMY2V.js → chunk-FTMGRKEF.js} +2 -2
  55. package/dist/{chunk-34BHNEE3.js → chunk-GHS5EBTQ.js} +58 -7
  56. package/dist/{chunk-DYHAXKHD.js → chunk-GWZNEVM2.js} +12 -8
  57. package/dist/chunk-GX5WYQO4.js +59 -0
  58. package/dist/chunk-GYV6VZOC.js +26 -0
  59. package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
  60. package/dist/chunk-I2DWJ4GM.js +390 -0
  61. package/dist/{chunk-TXOTCRLG.js → chunk-I5FWO7L5.js} +5 -5
  62. package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
  63. package/dist/{chunk-XQRY4DTA.js → chunk-IGLP3ODT.js} +10 -10
  64. package/dist/chunk-IRXAATOX.js +539 -0
  65. package/dist/chunk-IXIY2H4R.js +44 -0
  66. package/dist/{chunk-SSEYRH53.js → chunk-IZXGRF7P.js} +92 -147
  67. package/dist/{chunk-5TSRNF4G.js → chunk-JCI2ROMZ.js} +164 -6
  68. package/dist/{chunk-JWJGP5DQ.js → chunk-JEIYHLOR.js} +7 -7
  69. package/dist/{chunk-F2I26BDK.js → chunk-JQLNNIKT.js} +4 -4
  70. package/dist/{chunk-BYMNWQ7O.js → chunk-JSD46VO2.js} +315 -63
  71. package/dist/{chunk-AK5XEFVZ.js → chunk-JT2RFCC5.js} +64 -14
  72. package/dist/{chunk-MCMZMDAC.js → chunk-K6T2ZAMZ.js} +168 -6
  73. package/dist/{chunk-PGF63K6I.js → chunk-KFV5L5SK.js} +73 -4
  74. package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
  75. package/dist/{chunk-PJX3WQUQ.js → chunk-LLXSDWXS.js} +3 -3
  76. package/dist/{chunk-DZAW46HP.js → chunk-LTIKRKFL.js} +3 -3
  77. package/dist/{chunk-DZEK6CJN.js → chunk-N56KALIC.js} +21 -21
  78. package/dist/{chunk-B7HM5Z7T.js → chunk-NAI6ZFCY.js} +9 -5
  79. package/dist/{chunk-I66ZTYNP.js → chunk-NRO2BJRH.js} +2656 -2213
  80. package/dist/{chunk-ZGNYYXQ6.js → chunk-NXIMQY5W.js} +3 -3
  81. package/dist/{chunk-IKOZFYBN.js → chunk-NXYCB2VD.js} +149 -106
  82. package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
  83. package/dist/{chunk-462T4EGZ.js → chunk-O5CVSAG5.js} +2 -2
  84. package/dist/chunk-ODGTEFFI.js +50 -0
  85. package/dist/{chunk-3F7VUY77.js → chunk-OEJSLEPW.js} +2 -2
  86. package/dist/{chunk-KKOJXO6R.js → chunk-OMQNJVKW.js} +4 -2
  87. package/dist/{chunk-5KW52TEP.js → chunk-Q4WO54TA.js} +132 -77
  88. package/dist/{chunk-W6NIE6OW.js → chunk-QUFRYSWI.js} +13 -7
  89. package/dist/{chunk-42FMPA75.js → chunk-QZWQA4DE.js} +2 -2
  90. package/dist/chunk-R6Q67RJH.js +134 -0
  91. package/dist/{chunk-W4YEMFBX.js → chunk-RAY4OVGZ.js} +3 -3
  92. package/dist/{chunk-ZNT2M6TG.js → chunk-RQCKCSRL.js} +17 -17
  93. package/dist/{chunk-LJID3DYZ.js → chunk-RXTN6AKH.js} +3 -3
  94. package/dist/{chunk-P75RZCJW.js → chunk-RZDWV63N.js} +3 -3
  95. package/dist/{chunk-UH632ZYL.js → chunk-S6PYF2XF.js} +2 -2
  96. package/dist/{chunk-HJWWJ6IL.js → chunk-TOIVGRUX.js} +17 -5
  97. package/dist/{chunk-HLAFFSEK.js → chunk-TQAHXW6Y.js} +2 -2
  98. package/dist/{chunk-JIEGK6UF.js → chunk-U6TMQNSI.js} +48 -4
  99. package/dist/{chunk-2HFQNRV3.js → chunk-UEPWCCTY.js} +12 -12
  100. package/dist/chunk-UOIZ7DA4.js +41 -0
  101. package/dist/{chunk-UH347SHR.js → chunk-USR47QNF.js} +11 -11
  102. package/dist/{chunk-AZ4WMN4W.js → chunk-V6HJFQZE.js} +2 -2
  103. package/dist/chunk-V76WTFTW.js +318 -0
  104. package/dist/{chunk-NMJXSHBJ.js → chunk-W54I7H25.js} +2 -2
  105. package/dist/{chunk-KPXDY6QF.js → chunk-XRZT5WY5.js} +2 -2
  106. package/dist/{chunk-UBRFI4HS.js → chunk-XULDXHTN.js} +142 -50
  107. package/dist/chunk-XXYSBZIQ.js +283 -0
  108. package/dist/{chunk-HKMD33FO.js → chunk-Y55JBDO5.js} +405 -122
  109. package/dist/{chunk-XOXV5GKE.js → chunk-YD5GIKET.js} +17 -8
  110. package/dist/{chunk-XGDPUNND.js → chunk-YECAMM3D.js} +2 -2
  111. package/dist/{chunk-BO7Y52RY.js → chunk-YNFKXPEC.js} +7 -7
  112. package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
  113. package/dist/{chunk-M2WXEHER.js → chunk-ZA4VCIGV.js} +2 -2
  114. package/dist/cli/index.js +42 -40
  115. package/dist/{clio-7VB377CC.js → clio-QLICPCF5.js} +7 -7
  116. package/dist/{code-nav-YVLCYA7V.js → code-nav-IJR2DBPR.js} +9 -9
  117. package/dist/{components-UBWCQSRW.js → components-2TGAI2RC.js} +5 -6
  118. package/dist/{config-4HVOS65E.js → config-IUA6OYNS.js} +88 -81
  119. package/dist/{configure-PIWO7B24.js → configure-VEPX4NMX.js} +26 -25
  120. package/dist/{context-KQYIWPWT.js → context-2DKHWH2T.js} +60 -45
  121. package/dist/{context-IYEHL3WQ.js → context-4MPR7WKB.js} +78 -69
  122. package/dist/{context-N6ZE3LGJ.js → context-BOYF5EJM.js} +15 -11
  123. package/dist/{context-clear-G4OGZJDS.js → context-clear-S4ZJCQUX.js} +73 -65
  124. package/dist/context-map-COB37XXN.js +505 -0
  125. package/dist/{context-working-set-BWLF6LJP.js → context-working-set-3I3FYX6Y.js} +18 -17
  126. package/dist/detail-A7JAVSIG.js +98 -0
  127. package/dist/{dispatch-runner-2QQAITS3.js → dispatch-runner-RJ5I2F2O.js} +99 -75
  128. package/dist/{docs-PD3EXDKU.js → docs-SPOV3BAN.js} +3 -5
  129. package/dist/{doctor-LHBD36VU.js → doctor-DKICC2SN.js} +71 -48
  130. package/dist/{eval-C45FYRJ6.js → eval-OQOQUDHK.js} +308 -146
  131. package/dist/{eval-inventory-6DEJPLBF.js → eval-inventory-Y6QRFOH5.js} +4 -4
  132. package/dist/{evidence-6SHONYAF.js → evidence-4DQ25GUQ.js} +79 -175
  133. package/dist/evidence-4F5USFKH.js +208 -0
  134. package/dist/{evolve-KRKMV72X.js → evolve-GSS52E5J.js} +71 -67
  135. package/dist/{extensions-KPZ2UHBB.js → extensions-G7MFLYHT.js} +8 -9
  136. package/dist/{fleet-IVTCKDHT.js → fleet-Q37YHHAQ.js} +126 -118
  137. package/dist/{fleet-commands-EDWL3IT7.js → fleet-commands-G7E4N7SM.js} +16 -13
  138. package/dist/{fleet-decisions-YP3YEFGK.js → fleet-decisions-O7M6QBA2.js} +9 -8
  139. package/dist/{fleet-graph-ZFWKHY2M.js → fleet-graph-TOUBW6OW.js} +21 -20
  140. package/dist/{fleet-inspect-FVUNCBML.js → fleet-inspect-SW33JJNI.js} +65 -60
  141. package/dist/{fleet-preflight-UN5XED4R.js → fleet-preflight-CV2655TW.js} +4 -5
  142. package/dist/{fleet-validate-XOWC4HSX.js → fleet-validate-KESZX2YH.js} +25 -24
  143. package/dist/{fleet-verify-UN3SODEL.js → fleet-verify-UQPMTVE3.js} +66 -61
  144. package/dist/{fleet-view-TWHJKCN6.js → fleet-view-MN2VG4MR.js} +65 -60
  145. package/dist/{init-T2QORQ3Y.js → init-PXEXQSBF.js} +90 -82
  146. package/dist/{interop-IN5I2A66.js → interop-ZG5T62U3.js} +12 -13
  147. package/dist/inventory-C26CFDRR.js +101 -0
  148. package/dist/{library-LSCATDLZ.js → library-B2W4N74O.js} +29 -29
  149. package/dist/{memory-HYOKAGGJ.js → memory-YCANYS5A.js} +73 -69
  150. package/dist/{models-2GPMFYCM.js → models-GERTU3YI.js} +51 -47
  151. package/dist/{monitor-E4ASVUJH.js → monitor-CPNIUULB.js} +74 -67
  152. package/dist/{orchestrator-DDMPR3PY.js → orchestrator-J4BSH4WQ.js} +1288 -1606
  153. package/dist/{panes-E3RUXOW5.js → panes-BOHAEGYC.js} +4 -4
  154. package/dist/{panes-IXKLOKA2.js → panes-NXSLDQZ2.js} +10 -11
  155. package/dist/{paths-L7LGY6RN.js → paths-VSUWNC22.js} +6 -7
  156. package/dist/reset-TNWTB5LU.js +343 -0
  157. package/dist/{resources-OTRSN34L.js → resources-4PXNMD5G.js} +29 -22
  158. package/dist/{run-5DEYH5QK.js → run-D6XJ34CN.js} +132 -132
  159. package/dist/{share-IHWTLO3M.js → share-2NWMJJEE.js} +27 -27
  160. package/dist/{skills-IYMXMKW4.js → skills-KR7WON5G.js} +40 -33
  161. package/dist/{skills-eval-DROHSJAR.js → skills-eval-O2ZNOLDS.js} +81 -77
  162. package/dist/{skills-inventory-D7X4L4ZX.js → skills-inventory-ZZOUBK7O.js} +23 -22
  163. package/dist/{slash-commands-QBM7UZ3B.js → slash-commands-ZXPJD64J.js} +47 -37
  164. package/dist/{steer-Z5DO23FJ.js → steer-XA25PSCS.js} +4 -4
  165. package/dist/{support-U7QOWY26.js → support-7EMVWYG2.js} +6 -6
  166. package/dist/{targets-P2FUC4IL.js → targets-OMH2XCSN.js} +50 -49
  167. package/dist/tasks-IPAGMEIX.js +36 -0
  168. package/dist/{terminal-lease-YREJ3JX2.js → terminal-lease-C2J3JYRE.js} +4 -4
  169. package/dist/{tools-5B7RO6MV.js → tools-EFFEAIDP.js} +8 -9
  170. package/dist/{trace-YMGMUM6A.js → trace-FXMXUZUF.js} +7 -7
  171. package/dist/uninstall-HALS6BLF.js +407 -0
  172. package/dist/upgrade-MS72RJEP.js +306 -0
  173. package/dist/{usage-ME5MPXGX.js → usage-NHG6MCJM.js} +162 -108
  174. package/dist/{verifiers-BVZ7IWOO.js → verifiers-7AUNVXDY.js} +155 -22
  175. package/dist/{verify-5K7ZKQFC.js → verify-FWYGPKMR.js} +14 -12
  176. package/dist/{web-fetch-MPARV2K7.js → web-fetch-V4FKSDAV.js} +4 -4
  177. package/dist/{wiki-generate-F5W5QTYY.js → wiki-generate-743CIGJW.js} +99 -90
  178. package/dist/{with-panes-BYOJCLAM.js → with-panes-BDQEWBRT.js} +10 -10
  179. package/dist/worker/entry.js +72 -68
  180. package/docs/README.md +3 -2
  181. package/docs/architecture/acp.md +17 -0
  182. package/docs/architecture/artifact-placement.md +1 -0
  183. package/docs/architecture/artifact-versions.md +2 -2
  184. package/docs/architecture/context-engine.md +4 -0
  185. package/docs/architecture/dispatch-typed-intent.md +1 -1
  186. package/docs/architecture/evidence-and-memory.md +1 -1
  187. package/docs/architecture/middleware-and-components.md +1 -1
  188. package/docs/architecture/model-catalog.md +21 -10
  189. package/docs/architecture/observability.md +19 -2
  190. package/docs/architecture/prompt-envelope-and-tools.md +17 -5
  191. package/docs/architecture/provider-adapter-cookbook.md +63 -0
  192. package/docs/architecture/safety-model.md +25 -22
  193. package/docs/architecture/tui-design.md +1 -1
  194. package/docs/guide/built-in-agents.md +25 -11
  195. package/docs/guide/commands-and-modes.md +18 -3
  196. package/docs/guide/configuration-and-targets.md +100 -10
  197. package/docs/guide/configuration-reference.md +17 -7
  198. package/docs/guide/environment-variables.md +4 -2
  199. package/docs/guide/installation-and-lifecycle.md +37 -4
  200. package/docs/guide/proactive-memory.md +66 -55
  201. package/docs/guide/skills-marketplace.md +18 -0
  202. package/docs/guide/tool-usage.md +78 -3
  203. package/docs/history/config-knobs-audit.md +2 -2
  204. package/docs/process/development-pipeline.md +40 -2
  205. package/docs/process/eval-runner.md +67 -3
  206. package/docs/process/git-commit-provenance.md +15 -0
  207. package/docs/process/release-cut-checklist.md +207 -0
  208. package/docs/process/scientific-validation.md +18 -17
  209. package/evals/behavioral-machinery-support.ts +1 -0
  210. package/evals/behavioral-machinery.yaml +1 -1
  211. package/evals/behavioral-model.yaml +3 -2
  212. package/package.json +2 -2
  213. package/skills/README.md +7 -5
  214. package/skills/coding/ast-grep/SKILL.md +101 -30
  215. package/skills/coding/ast-grep/evals.md +26 -0
  216. package/skills/coding/coding-standards/SKILL.md +47 -2
  217. package/skills/coding/coding-standards/evals.md +23 -0
  218. package/skills/coding/prototype/SKILL.md +87 -28
  219. package/skills/coding/prototype/evals.md +19 -0
  220. package/skills/coding/tdd/SKILL.md +80 -53
  221. package/skills/coding/tdd/evals.md +20 -0
  222. package/skills/context/context-handoff/SKILL.md +43 -2
  223. package/skills/context/context-handoff/evals.md +44 -0
  224. package/skills/context/context-prime/SKILL.md +45 -15
  225. package/skills/context/context-prime/evals.md +45 -0
  226. package/skills/git/branch-closeout/SKILL.md +132 -0
  227. package/skills/git/branch-closeout/evals.md +133 -0
  228. package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
  229. package/skills/git/file-ticket/SKILL.md +77 -63
  230. package/skills/git/file-ticket/assets/issue-template.md +22 -0
  231. package/skills/git/file-ticket/evals.md +31 -26
  232. package/skills/git/file-ticket/references/issue-discovery.md +49 -0
  233. package/skills/git/fix-issue/SKILL.md +87 -64
  234. package/skills/git/fix-issue/evals.md +35 -31
  235. package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
  236. package/skills/git/resolve-merge-conflicts/SKILL.md +100 -51
  237. package/skills/git/resolve-merge-conflicts/evals.md +52 -25
  238. package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
  239. package/skills/git/ship/SKILL.md +103 -67
  240. package/skills/git/ship/assets/pr-template.md +21 -0
  241. package/skills/git/ship/evals.md +44 -28
  242. package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
  243. package/skills/git/worktree-create/SKILL.md +80 -50
  244. package/skills/git/worktree-create/evals.md +40 -33
  245. package/skills/git/worktree-create/references/worktree-setup.md +62 -66
  246. package/skills/git/worktree-merge/SKILL.md +112 -65
  247. package/skills/git/worktree-merge/evals.md +42 -34
  248. package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
  249. package/skills/planning/archify/SKILL.md +196 -0
  250. package/skills/planning/archify/evals.md +65 -0
  251. package/skills/planning/architecture/SKILL.md +61 -12
  252. package/skills/planning/architecture/evals.md +65 -0
  253. package/skills/planning/backlog/SKILL.md +130 -14
  254. package/skills/planning/backlog/evals.md +142 -0
  255. package/skills/planning/prd/SKILL.md +47 -6
  256. package/skills/planning/prd/evals.md +54 -0
  257. package/skills/planning/product-intent/SKILL.md +57 -2
  258. package/skills/planning/product-intent/evals.md +70 -0
  259. package/skills/planning/tech-spec/SKILL.md +53 -2
  260. package/skills/planning/tech-spec/evals.md +73 -0
  261. package/skills/registry.yaml +58 -50
  262. package/skills/remote.yaml +13 -0
  263. package/skills/research/arxiv-literature/SKILL.md +76 -18
  264. package/skills/research/arxiv-literature/evals.md +50 -0
  265. package/skills/research/experiment-protocol/SKILL.md +20 -1
  266. package/skills/research/experiment-protocol/evals.md +23 -0
  267. package/skills/research/scientific-debugging/SKILL.md +24 -1
  268. package/skills/research/scientific-debugging/evals.md +18 -0
  269. package/skills/research/scientific-modernization/SKILL.md +26 -1
  270. package/skills/research/scientific-modernization/evals.md +27 -0
  271. package/skills/skill-marketplace.json +63 -28
  272. package/skills/workflow/cut-it/SKILL.md +64 -5
  273. package/skills/workflow/cut-it/evals.md +101 -0
  274. package/skills/workflow/design-council/SKILL.md +112 -27
  275. package/skills/workflow/design-council/evals.md +161 -0
  276. package/skills/workflow/grill-me/SKILL.md +85 -10
  277. package/skills/workflow/grill-me/evals.md +153 -0
  278. package/skills/workflow/workflow-distiller/SKILL.md +76 -17
  279. package/skills/workflow/workflow-distiller/evals.md +118 -0
  280. package/src/cli/args.ts +0 -8
  281. package/src/cli/configure-interop.ts +105 -13
  282. package/src/cli/configure-oauth.ts +57 -0
  283. package/src/cli/configure-onboarding.ts +980 -0
  284. package/src/cli/configure-target.ts +594 -0
  285. package/src/cli/configure.ts +1084 -529
  286. package/src/cli/context-map.ts +114 -0
  287. package/src/cli/context.ts +4 -0
  288. package/src/cli/doctor-state-size.ts +1 -12
  289. package/src/cli/doctor-validation-contract.ts +28 -0
  290. package/src/cli/doctor.ts +5 -0
  291. package/src/cli/evidence-detail.ts +1 -75
  292. package/src/cli/evidence-inventory.ts +1 -167
  293. package/src/cli/index.ts +3 -0
  294. package/src/cli/lifecycle-presenter.ts +436 -0
  295. package/src/cli/models.ts +10 -2
  296. package/src/cli/modes/print.ts +5 -1
  297. package/src/cli/reset.ts +228 -106
  298. package/src/cli/run.ts +7 -4
  299. package/src/cli/select.ts +664 -0
  300. package/src/cli/skills.ts +9 -2
  301. package/src/cli/targets.ts +3 -0
  302. package/src/cli/tasks.ts +84 -0
  303. package/src/cli/uninstall.ts +233 -165
  304. package/src/cli/upgrade.ts +210 -150
  305. package/src/cli/usage.ts +92 -27
  306. package/src/cli/validate-model.ts +3 -3
  307. package/src/cli/verifiers.ts +147 -1
  308. package/src/cli/wiki-generate.ts +1 -0
  309. package/src/core/commit-attribution.ts +41 -1
  310. package/src/core/config.ts +56 -0
  311. package/src/core/external-diagnostic.ts +44 -0
  312. package/src/core/gateway-routing.ts +157 -0
  313. package/src/core/git-commit-attribution.ts +46 -3
  314. package/src/core/run-overrides.ts +0 -5
  315. package/src/core/safe-exec.ts +17 -2
  316. package/src/core/skill-activation.ts +92 -2
  317. package/src/core/tool-names.ts +5 -2
  318. package/src/domains/agents/builtins/architect.md +1 -1
  319. package/src/domains/agents/builtins/coder.md +1 -1
  320. package/src/domains/agents/builtins/documenter.md +1 -1
  321. package/src/domains/agents/builtins/git-master.md +1 -1
  322. package/src/domains/agents/builtins/provenance.md +7 -7
  323. package/src/domains/agents/builtins/tester.md +1 -1
  324. package/src/domains/agents/builtins/verifier.md +2 -2
  325. package/src/domains/agents/builtins/wiki-writer.md +4 -3
  326. package/src/domains/agents/builtins/world-knowledge.md +31 -0
  327. package/src/domains/agents/catalog.ts +1 -1
  328. package/src/domains/agents/result-contract.ts +70 -0
  329. package/src/domains/context/extension.ts +31 -7
  330. package/src/domains/context/refresh.ts +3 -0
  331. package/src/domains/context/wiki/frontmatter.ts +5 -2
  332. package/src/domains/context/wiki/generate.ts +6 -0
  333. package/src/domains/context/wiki/map-seed.ts +589 -0
  334. package/src/domains/context/wiki/plan.ts +2 -2
  335. package/src/domains/context/wiki/prompts.ts +43 -0
  336. package/src/domains/dispatch/active-route-planner.ts +4 -0
  337. package/src/domains/dispatch/admission.ts +29 -0
  338. package/src/domains/dispatch/agent-candidates.ts +10 -0
  339. package/src/domains/dispatch/budget-envelope.ts +86 -1
  340. package/src/domains/dispatch/capability-match.ts +1 -0
  341. package/src/domains/dispatch/capacity-lease.ts +17 -0
  342. package/src/domains/dispatch/code-step.ts +11 -4
  343. package/src/domains/dispatch/contract.ts +42 -8
  344. package/src/domains/dispatch/execution-scheduler.ts +2 -0
  345. package/src/domains/dispatch/extension.ts +184 -56
  346. package/src/domains/dispatch/fleet-commit-attribution.ts +5 -0
  347. package/src/domains/dispatch/fleet-run.ts +1 -0
  348. package/src/domains/dispatch/host-verification.ts +114 -13
  349. package/src/domains/dispatch/intent.ts +28 -18
  350. package/src/domains/dispatch/orphan-recovery.ts +2 -0
  351. package/src/domains/dispatch/receipt-integrity.ts +4 -0
  352. package/src/domains/dispatch/reservation-store.ts +5 -3
  353. package/src/domains/dispatch/state.ts +15 -2
  354. package/src/domains/dispatch/types.ts +20 -2
  355. package/src/domains/dispatch/worker-model-metadata.ts +38 -0
  356. package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
  357. package/src/domains/eval/metrics/token-stream.ts +201 -31
  358. package/src/domains/eval/metrics/tracked.ts +40 -4
  359. package/src/domains/eval/runners/clio-run.ts +17 -11
  360. package/src/domains/eval/runners/context-index.ts +2 -7
  361. package/src/domains/eval/runners/context-init.ts +3 -6
  362. package/src/domains/eval/runners/external-command.ts +29 -11
  363. package/src/domains/eval/schema/suite.ts +28 -0
  364. package/src/domains/eval/schema/verdict.ts +2 -2
  365. package/src/domains/eval/suites/resolve.ts +13 -1
  366. package/src/domains/eval/suites/run.ts +24 -3
  367. package/src/domains/evidence/build.ts +102 -15
  368. package/src/domains/evidence/detail.ts +69 -0
  369. package/src/domains/evidence/eval.ts +13 -1
  370. package/src/domains/evidence/finish-contract-map.ts +5 -1
  371. package/src/domains/evidence/inventory.ts +167 -0
  372. package/src/domains/evidence/store.ts +16 -0
  373. package/src/domains/evidence/types.ts +12 -0
  374. package/src/domains/extensions/resources.ts +7 -0
  375. package/src/domains/interop/registry.ts +6 -2
  376. package/src/domains/interop/types.ts +4 -0
  377. package/src/domains/lifecycle/migrations/index.ts +4 -0
  378. package/src/domains/memory/task-memory-policy.ts +70 -26
  379. package/src/domains/memory/task-memory-telemetry.ts +1 -0
  380. package/src/domains/middleware/index.ts +0 -1
  381. package/src/domains/middleware/marketplace-offer.ts +22 -35
  382. package/src/domains/middleware/memory-intervention.ts +127 -32
  383. package/src/domains/middleware/memory-step-endpoint.ts +3 -2
  384. package/src/domains/middleware/runtime.ts +7 -3
  385. package/src/domains/middleware/skills-reminder.ts +31 -2
  386. package/src/domains/mux/detect.ts +3 -6
  387. package/src/domains/observability/accountability.ts +15 -1
  388. package/src/domains/observability/compaction-usage.ts +118 -0
  389. package/src/domains/observability/contract.ts +52 -7
  390. package/src/domains/observability/cost.ts +1 -1
  391. package/src/domains/observability/evidence-index.ts +10 -0
  392. package/src/domains/observability/extension.ts +15 -6
  393. package/src/domains/observability/out-of-turn-usage.ts +52 -21
  394. package/src/domains/observability/projection.ts +394 -45
  395. package/src/{interactive → domains/observability}/worker-progress.ts +3 -3
  396. package/src/domains/prompts/fragments/operating/contract.md +2 -0
  397. package/src/domains/prompts/fragments/wiki/page.md +8 -0
  398. package/src/domains/providers/contract.ts +4 -1
  399. package/src/domains/providers/extension.ts +40 -9
  400. package/src/domains/providers/model-capabilities.ts +9 -0
  401. package/src/domains/providers/model-discovery.ts +3 -4
  402. package/src/domains/providers/model-runtime-capabilities.ts +15 -5
  403. package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +48 -26
  404. package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
  405. package/src/domains/providers/runtimes/claude/claude-code.ts +9 -0
  406. package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
  407. package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
  408. package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
  409. package/src/domains/providers/support.ts +11 -5
  410. package/src/domains/providers/target-model-cache.ts +25 -2
  411. package/src/domains/providers/types/capability-flags.ts +2 -0
  412. package/src/domains/providers/types/runtime-descriptor.ts +20 -1
  413. package/src/domains/providers/types/target-descriptor.ts +19 -0
  414. package/src/domains/resources/index.ts +3 -0
  415. package/src/domains/resources/skills/install.ts +72 -7
  416. package/src/domains/resources/skills/loader.ts +7 -0
  417. package/src/domains/resources/skills/marketplace.ts +63 -11
  418. package/src/domains/safety/action-classifier.ts +7 -0
  419. package/src/domains/safety/autonomy.ts +15 -0
  420. package/src/domains/safety/default-path-policy.ts +2 -0
  421. package/src/domains/safety/finish-contract-registration.ts +29 -14
  422. package/src/domains/safety/finish-contract.ts +252 -40
  423. package/src/domains/safety/index.ts +21 -1
  424. package/src/domains/safety/path-policy.ts +1 -1
  425. package/src/domains/safety/policy-engine.ts +60 -17
  426. package/src/domains/safety/protected-artifacts.ts +191 -88
  427. package/src/domains/safety/rigor.ts +53 -39
  428. package/src/domains/safety/run-effects.ts +2 -22
  429. package/src/domains/safety/skill-authority.ts +55 -0
  430. package/src/domains/safety/validation-contract.ts +388 -0
  431. package/src/domains/session/archive-readers.ts +10 -1
  432. package/src/domains/session/compaction/compact.ts +72 -22
  433. package/src/domains/session/decision-board.ts +101 -2
  434. package/src/domains/session/entries.ts +50 -7
  435. package/src/domains/session/extension.ts +4 -4
  436. package/src/domains/session/handoff.ts +2 -1
  437. package/src/domains/session/manager.ts +2 -3
  438. package/src/domains/session/task-board.ts +14 -1
  439. package/src/domains/session/tree/fork.ts +1 -2
  440. package/src/domains/session/tree/navigator.ts +1 -1
  441. package/src/domains/session/usage.ts +3 -3
  442. package/src/domains/user-tasks/acceptance.ts +56 -0
  443. package/src/domains/user-tasks/active-acceptance.ts +40 -0
  444. package/src/domains/user-tasks/store.ts +34 -3
  445. package/src/engine/acp/adapter.ts +24 -6
  446. package/src/engine/acp/server.ts +21 -4
  447. package/src/engine/acp/transport.ts +53 -8
  448. package/src/engine/acp/types.ts +4 -0
  449. package/src/engine/agent.ts +13 -3
  450. package/src/engine/ai.ts +26 -8
  451. package/src/engine/antigravity/subprocess-runtime.ts +386 -120
  452. package/src/engine/api-registry.ts +3 -0
  453. package/src/engine/apis/ollama-native.ts +15 -0
  454. package/src/engine/apis/openai-completions.ts +117 -14
  455. package/src/engine/claude/subprocess-runtime.ts +107 -60
  456. package/src/engine/external-subprocess.ts +122 -6
  457. package/src/entry/background-model-metadata.ts +18 -0
  458. package/src/entry/compaction-prompt.ts +57 -0
  459. package/src/entry/orchestrator.ts +416 -218
  460. package/src/entry/task-memory-lifecycle.ts +35 -0
  461. package/src/interactive/chat-loop-messages.ts +13 -4
  462. package/src/interactive/chat-loop.ts +65 -2
  463. package/src/interactive/chat-renderer.ts +1 -0
  464. package/src/interactive/cost-overlay.ts +26 -2
  465. package/src/interactive/dispatch-board.ts +46 -717
  466. package/src/interactive/fleet-run-preview.ts +2 -1
  467. package/src/interactive/interactive-application.ts +3 -2
  468. package/src/interactive/interactive-presentation.ts +55 -12
  469. package/src/interactive/interactive-slash-runtime.ts +4 -2
  470. package/src/interactive/oracle.ts +5 -2
  471. package/src/interactive/overlays/fleet-run-approval.ts +3 -2
  472. package/src/interactive/overlays/message-picker.ts +2 -2
  473. package/src/interactive/overlays/settings.ts +2 -2
  474. package/src/interactive/overlays/tree-selector.ts +2 -2
  475. package/src/interactive/renderers/branch-summary.ts +1 -1
  476. package/src/interactive/renderers/worker-entry.ts +32 -0
  477. package/src/interactive/slash-autocomplete.ts +4 -6
  478. package/src/interactive/slash-commands.ts +49 -45
  479. package/src/interactive/slash-spec.ts +28 -0
  480. package/src/interactive/theme/labels.ts +19 -13
  481. package/src/interactive/turn-context.ts +9 -5
  482. package/src/interactive/turn-recovery.ts +8 -0
  483. package/src/interactive/turn-runtime.ts +27 -11
  484. package/src/interactive/turn-state.ts +7 -0
  485. package/src/interactive/view/artifacts.ts +2 -0
  486. package/src/interactive/worker-receipts.ts +1 -0
  487. package/src/interactive/worker-stream.ts +13 -4
  488. package/src/tools/bootstrap.ts +4 -0
  489. package/src/tools/builtin-tool-catalog.ts +31 -0
  490. package/src/tools/compete-worktrees.ts +7 -1
  491. package/src/tools/context/index.ts +30 -9
  492. package/src/tools/core-bootstrap.ts +16 -0
  493. package/src/tools/decide.ts +136 -0
  494. package/src/tools/dispatch-admission.ts +21 -0
  495. package/src/tools/dispatch-arguments.ts +1 -0
  496. package/src/tools/dispatch-event-text.ts +10 -0
  497. package/src/tools/dispatch-plan.ts +6 -2
  498. package/src/tools/dispatch-runner.ts +27 -1
  499. package/src/tools/dispatch-types.ts +6 -0
  500. package/src/tools/evidence.ts +96 -0
  501. package/src/tools/limitation.ts +76 -0
  502. package/src/tools/policy.ts +9 -0
  503. package/src/tools/presentation.ts +3 -0
  504. package/src/tools/registry.ts +11 -5
  505. package/src/tools/result-shaping.ts +17 -5
  506. package/src/tools/task-worktree.ts +13 -3
  507. package/src/tools/tasks.ts +10 -1
  508. package/src/tools/verify/authoring.ts +170 -83
  509. package/src/tools/verify/catalog.ts +122 -5
  510. package/src/tools/verify/index.ts +2 -1
  511. package/src/tools/verify/numeric.ts +298 -0
  512. package/src/tools/verify/perf.ts +143 -0
  513. package/src/tools/verify/scripts.ts +229 -2
  514. package/src/tools/worker-evidence.ts +3 -1
  515. package/src/worker/spec-contract.ts +4 -0
  516. package/dist/chunk-2Z2IKEXI.js +0 -1554
  517. package/dist/chunk-RVG5JXAL.js +0 -41
  518. package/dist/chunk-T56WDKA5.js +0 -183
  519. package/dist/chunk-VN3SHNBN.js +0 -313
  520. package/dist/chunk-VPKWYKEY.js +0 -169
  521. package/dist/reset-OAQP3W4O.js +0 -230
  522. package/dist/uninstall-N34PCTGJ.js +0 -331
  523. package/dist/upgrade-PXK3S2YM.js +0 -325
@@ -6,54 +6,58 @@ skills:
6
6
  # ── coding ──
7
7
  - name: ast-grep
8
8
  path: coding/ast-grep
9
- version: 0.2.0
10
- sha256: d21a4b330d1488c43348eaee7fbeec24bb8a1d7c4e536059db02fd9377c62dfa
9
+ version: 0.3.0
10
+ sha256: 0a72f4c906303f550be78f80a927da7e24e5c2076f058ccafdec80dfbed275af
11
11
  - name: coding-standards
12
12
  path: coding/coding-standards
13
- version: 0.2.0
14
- sha256: da21ad373252575934ad484a2926e8c7827880c9d91b4e0c656fe92d71335063
13
+ version: 0.4.0
14
+ sha256: d573cdc5b430ae94ebfb5ef69c4a0ac9ac074aae815959ab4e6515012846ce3b
15
15
  - name: prototype
16
16
  path: coding/prototype
17
- version: 0.3.0
18
- sha256: 151bd153752874d5970dd06bc15df14c0037631283d148195a8fa75c5a59a45b
17
+ version: 0.4.0
18
+ sha256: 0f82df386c2ee565b0e968216ece746b1a11b4acd79676812074c1a406b698e1
19
19
  - name: tdd
20
20
  path: coding/tdd
21
- version: 0.3.0
22
- sha256: f3a03655864d4981791218b085d09f8fe836d0795012cf4c8c07b430730abb2a
21
+ version: 0.4.0
22
+ sha256: 63b88f29424091a92a8d8c0cc94474ae7e78af76491aea17e6f54af66f32db2d
23
23
  # ── context ──
24
24
  - name: context-handoff
25
25
  path: context/context-handoff
26
- version: 0.4.0
27
- sha256: ba0568719d0b58bb3e1631eee72094ec0165bf2fb7f69e1e67d223ff50b364d8
26
+ version: 0.5.1
27
+ sha256: e05a8dc9d09b4decbf79afade1f530d7c81d5412112723faab104a83893c34f8
28
28
  - name: context-prime
29
29
  path: context/context-prime
30
- version: 0.3.0
31
- sha256: d8413255688f1697d40231df830ce3c8947ac2a551f3e6fbb0d10158c4fa8d0e
30
+ version: 0.4.1
31
+ sha256: 3e6783d3fd95bd5635fe274ea0b9f73f3c4606ce64b832524cc242b114d8ef12
32
32
  # ── git ──
33
+ - name: branch-closeout
34
+ path: git/branch-closeout
35
+ version: 0.1.0
36
+ sha256: 3227f31b0693bb6428abb20a2d4f449aeba1d8e45a79cf0c2c7c333c30c10664
33
37
  - name: file-ticket
34
38
  path: git/file-ticket
35
- version: 0.2.0
36
- sha256: 07b5d996d459c6c8c99ea44f44421d418f4738075cccd0e4b759a6f8a7daf720
39
+ version: 0.3.0
40
+ sha256: 65029c2d9d545728750c8a713f92035bd6bdbfa9874de442df72bd059846cb7a
37
41
  - name: fix-issue
38
42
  path: git/fix-issue
39
- version: 0.2.0
40
- sha256: ec5e6a00056041494f9b100da8f0775455124aebc7000ed0edb7c9b8493f74c8
43
+ version: 0.3.0
44
+ sha256: 626e3aa8ca95bc603f2e0cdd2e7502aa96be85af6bfa92b635c39ef99d3845a5
41
45
  - name: resolve-merge-conflicts
42
46
  path: git/resolve-merge-conflicts
43
- version: 0.3.0
44
- sha256: a1df7dee85ef788f257f8b8fbf4c2564cac245c3010d3f60d3b62d721d185f5e
47
+ version: 0.4.0
48
+ sha256: 901f482aa2231552d64a721e0d1c58dc8f211347543b5a91fd2ab3c7e413c11c
45
49
  - name: ship
46
50
  path: git/ship
47
- version: 0.2.0
48
- sha256: 8a22f74e56a6e98c85a38051550d1e1a70c9ca5bb5cf271f4e6448f0f791e0ed
51
+ version: 0.5.0
52
+ sha256: 4d626718fdb8b67cf3f22d09ac5a228e5c8103befa1295a672587a6605f038b1
49
53
  - name: worktree-create
50
54
  path: git/worktree-create
51
- version: 0.3.0
52
- sha256: d40bb0bb9152c4f2e391de1285b6cbd2e67a45492e2cf865b1f82910e30a6bcc
55
+ version: 0.6.0
56
+ sha256: a4ae790d916a170977814b4334e14aef4f96fb304e3289f7beb5e31e8fb83881
53
57
  - name: worktree-merge
54
58
  path: git/worktree-merge
55
- version: 0.3.0
56
- sha256: bca803a2484b0eac90939377c81f9e66928242f36ac2d86cef9381b1ff997f3e
59
+ version: 0.6.0
60
+ sha256: 0fbe2297622ed955b2e4fb14306d75a51bd36db7eebf59e8e92fac1da5b6e64e
57
61
  # ── meta ──
58
62
  - name: clio-coder-dev
59
63
  path: meta/clio-coder-dev
@@ -80,57 +84,61 @@ skills:
80
84
  version: 0.3.0
81
85
  sha256: ba81e09412f26647a9384b07359201efb08949865c50934778a4ab3173d43096
82
86
  # ── planning ──
87
+ - name: archify
88
+ path: planning/archify
89
+ version: 0.1.0
90
+ sha256: d9e891eb5f3c27678de14165a0eb35d1eb2289be480e55009f9a57bcfe55be2d
83
91
  - name: architecture
84
92
  path: planning/architecture
85
- version: 0.3.0
86
- sha256: 0da15631b3a60946e90b63047047e0ee1d84b6705e4f9f5405036b87fc849454
93
+ version: 0.4.1
94
+ sha256: 0d62ded74021d148f3091e044c32d205edfa3d1298706e68c84ad0f032d0feba
87
95
  - name: backlog
88
96
  path: planning/backlog
89
- version: 0.3.0
90
- sha256: af92244d2ba8723b656ce00d531df88850c7886915653751e7ba59957b5454be
97
+ version: 0.4.1
98
+ sha256: 8fc700ad8f1b2f44fb6ea013c54ce06314cf9c4fdee63bd65c36686f5c25d092
91
99
  - name: prd
92
100
  path: planning/prd
93
- version: 0.3.0
94
- sha256: 505460dcac23b762a87fd0f7f3aaa99806e3b0fbedeaa54f9c48608c49c16356
101
+ version: 0.4.1
102
+ sha256: 3841f2f9635229c8208382e40123de1b5a496ec562cb851758d2ac013921c069
95
103
  - name: product-intent
96
104
  path: planning/product-intent
97
- version: 0.3.0
98
- sha256: d34300e6e2422fc1071e352f3e0f1b13a4e65dbcfd4eaf6dd9ef2b61f18d72ad
105
+ version: 0.4.1
106
+ sha256: 95d629fe8688003a9816a0774674f0c9fd5a922a82835415c3c1806799436832
99
107
  - name: tech-spec
100
108
  path: planning/tech-spec
101
- version: 0.2.0
102
- sha256: 28b1b0a0a3e9b48b29f679ab2377ee66d719c9a670eed622a38ac69be8b72b2e
109
+ version: 0.3.1
110
+ sha256: e8e6a1326aeb8b010647fc9fa9dc5ff827e13550f54524555f5e6a83341eb27f
103
111
  # ── research ──
104
112
  - name: arxiv-literature
105
113
  path: research/arxiv-literature
106
- version: 0.4.0
107
- sha256: b0b1bb80dc8d15e52ab180ab4f5645859310b45d7ebdf87649e0e5a7f7bca987
114
+ version: 0.5.1
115
+ sha256: 4668827aa0f13527de5fc34914911ef942d2103ed3d139008fbd1c0ce2c4bd3d
108
116
  - name: experiment-protocol
109
117
  path: research/experiment-protocol
110
- version: 0.2.0
111
- sha256: 126e0f6dc3534a141978f2ccd1c6ea659df447beb1015bc92e41856d318584a2
118
+ version: 0.3.1
119
+ sha256: da151931fb1961b19fcb6d4ae1957cef908f8b35692e301f109125488ad65302
112
120
  - name: scientific-debugging
113
121
  path: research/scientific-debugging
114
- version: 0.2.0
115
- sha256: f2b2b6442740ef23270fad94f75ba95ce522da1877a468ba112cbaff68e5fa2a
122
+ version: 0.3.1
123
+ sha256: ba94f8ad2655396a3791c7599631513f16651ffa61c8148044b996a25c0bc1cc
116
124
  - name: scientific-modernization
117
125
  path: research/scientific-modernization
118
- version: 0.3.0
119
- sha256: 66a99f38b38232735bb70d7924a85d0ab5addabfe3935cc82c0aaa3303d68773
126
+ version: 0.4.1
127
+ sha256: eb31099d764cba913c19726f6ba85d7081c5e01c81a3b2cd45eb136a4e47b858
120
128
  # ── workflow ──
121
129
  - name: cut-it
122
130
  path: workflow/cut-it
123
- version: 0.3.0
124
- sha256: 6429a4725acf874379c1df7736dd95db07acd97ad5bd6f6226649370cd03d177
131
+ version: 0.4.1
132
+ sha256: d0ce7dc6ea633d87e9e68c6b59ad402c24644b9eac580774410b235e0ca1c93f
125
133
  - name: design-council
126
134
  path: workflow/design-council
127
- version: 0.4.0
128
- sha256: 1ecd53397ce97ec78294cc9e8c7291b20e2e7d740bcde8ea999739d359c91055
135
+ version: 0.6.0
136
+ sha256: 42fbf2e54524253b2955b22576ae0aca998dc04a902b8a59440edcfd051e0f5a
129
137
  - name: grill-me
130
138
  path: workflow/grill-me
131
- version: 0.4.0
132
- sha256: 9cc07b89a4a8a3ecd4fb73d2cd9f7893d272d26f592ea3eb9883ff44069a2cea
139
+ version: 0.5.1
140
+ sha256: 3844770b507f9b2904aaa5c944c2b26f9ecf114383758dfff3bd7f39b2c717a6
133
141
  - name: workflow-distiller
134
142
  path: workflow/workflow-distiller
135
- version: 0.3.0
136
- sha256: 6aa6bb3088f0e4925946aaba3f217dfc5a6e4b8e31f2f051abeda599da91cd48
143
+ version: 0.4.1
144
+ sha256: 067b097299b0b630729a4a6ddde629dec5c798d5b4634f90f7feab380b8575e1
@@ -0,0 +1,13 @@
1
+ # Skills whose content lives in another repository at a pinned ref.
2
+ # Clio never vendors these. `npm run skills:pin` publishes each entry into
3
+ # skill-marketplace.json with the upstream tree as its sourceUrl, the catalog
4
+ # overlay whose files land on top of that tree at install, and the upstream
5
+ # top-level members the install drops. The overlay SKILL.md is pinned in
6
+ # registry.yaml like every other catalog skill.
7
+ version: 1
8
+ skills:
9
+ - name: archify
10
+ category: planning
11
+ sourceUrl: https://github.com/tt-a1i/archify/tree/v2.16.0/archify
12
+ overlay: skills/planning/archify
13
+ exclude: [test, package-lock.json]
@@ -7,7 +7,7 @@ triggers:
7
7
  - compare these papers
8
8
  - find recent research papers
9
9
  - build a literature survey
10
- version: 0.4.0
10
+ version: 0.5.1
11
11
  license: Apache-2.0
12
12
  allowed-tools:
13
13
  - web_fetch
@@ -16,7 +16,6 @@ allowed-tools:
16
16
  - grep
17
17
  - find
18
18
  - ls
19
- - artifact
20
19
  clio-coder:
21
20
  registry-id: iowarp/clio-coder
22
21
  source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/arxiv-literature
@@ -34,6 +33,20 @@ Find, summarize, or compare academic papers without flooding the main context
34
33
  window. Raw search results and paper text stay in a worker or get compressed
35
34
  immediately; only paper cards reach the user.
36
35
 
36
+ ## Arguments
37
+
38
+ ```text
39
+ /skill arxiv-literature <request>
40
+ ```
41
+
42
+ The request is free text: a paper URL/ID, a topic, or two or more IDs to
43
+ compare. There is no operator in a headless run: `ask_user` is not registered, so
44
+ any call is refused as an unregistered tool rather than answered. If the request is ambiguous (no
45
+ clear topic, an ID that doesn't resolve), state your best reading and
46
+ proceed; never stall waiting for clarification. `tasks` sits outside this
47
+ skill's tool surface and is refused; the steps below are the whole plan, do
48
+ not open a task list for them.
49
+
37
50
  ## Step 1 — Classify the request
38
51
 
39
52
  Pick exactly one:
@@ -45,29 +58,49 @@ Pick exactly one:
45
58
 
46
59
  ## Step 2 — Pick the vehicle
47
60
 
48
- - **Search, compare, survey**: dispatch the `researcher` shadow agent. Task
49
- prompt:
61
+ Default every request — single paper, search, compare, and survey alike to
62
+ `web_fetch` directly against arXiv:
50
63
 
51
- ```text
52
- Research arXiv literature for: <user goal>.
53
- Return only compact source-linked paper cards, comparison/synthesis,
54
- caveats, and read/skim/skip recommendations.
55
- ```
56
-
57
- - **Single paper, or dispatch unavailable**: use `web_fetch` directly.
58
- - Paper URL/ID: fetch the arXiv page. Clio normalizes it into structured
59
- metadata plus AlphaXiv enrichment when available.
60
- - Search query: fetch the arXiv Atom API; Clio compacts the XML into paper
61
- cards:
64
+ - Paper URL/ID: fetch the arXiv page. Clio normalizes it into structured
65
+ metadata plus AlphaXiv enrichment when available.
66
+ - Topic, compare, or survey: fetch the arXiv Atom API; Clio compacts the XML
67
+ into paper cards:
62
68
 
63
69
  ```text
64
70
  https://export.arxiv.org/api/query?search_query=all:QUERY&sortBy=submittedDate&sortOrder=descending&start=0&max_results=10
65
71
  ```
66
72
 
73
+ For a compare request, run one query per paper ID (`id_list=ID` instead of
74
+ `search_query`) or one broader query covering all of them — whichever stays
75
+ inside the fetch cap in Step 3.
76
+
67
77
  Useful categories for `search_query`: `cs.AI` (AI), `cs.LG` (ML), `cs.CL`
68
78
  (NLP/LLMs), `cs.CR` (security), `cs.SE` (software engineering), `cs.MA`
69
79
  (multi-agent), `cs.IR` (retrieval/RAG), `cs.CV` (vision), `cs.RO` (robotics).
70
80
 
81
+ **Only dispatch the `researcher` shadow agent when the user explicitly asks
82
+ for a deep or broad survey** ("survey the field", "don't just skim arXiv, go
83
+ wide") — never as the default for an ordinary search or compare. Left
84
+ unbounded, a dispatched worker has no arXiv-only restriction and no fetch
85
+ budget of its own, and will wander into Semantic Scholar, DBLP, OpenAlex, and
86
+ general web search, taking several minutes to return nothing useful. When you
87
+ do dispatch, state the same bound this skill uses directly, in the task
88
+ prompt itself:
89
+
90
+ ```text
91
+ Research arXiv literature for: <user goal>.
92
+ Search arXiv only (export.arxiv.org Atom API or arxiv.org paper pages) — do
93
+ not query Semantic Scholar, DBLP, OpenAlex, or general web search. One
94
+ round, at most two fetch attempts total (successes and failures both count).
95
+ On a timeout or HTTP error, do not retry with a different host, scheme, or
96
+ protocol — retry the identical URL at most once, then stop and report the
97
+ failure. Build cards from whatever you have; do not keep escalating.
98
+ Return only compact source-linked paper cards, comparison/synthesis,
99
+ caveats, and read/skim/skip recommendations.
100
+ ```
101
+
102
+ If dispatch is unavailable, fall back to the direct `web_fetch` path above.
103
+
71
104
  ## Step 3 — Return paper cards only
72
105
 
73
106
  Never paste raw Atom XML or full paper text into the response. Output format:
@@ -96,9 +129,14 @@ Never paste raw Atom XML or full paper text into the response. Output format:
96
129
  Done when every returned paper has a card with a working link and the
97
130
  recommendation section is filled in. Stop after one search round unless the
98
131
  user asks to go deeper; do not keep fetching to "be thorough". One round
99
- means at most two Atom API fetches: the initial query plus one refinement.
100
- Rewording the same query a third time is thrash build cards from what the
101
- first two returned.
132
+ means at most two Atom API fetch attempts total: the initial query plus one
133
+ refinement, or one retry of a failed fetch. Failed attempts (timeout, 429,
134
+ connection error) count against this cap the same as successful ones — a
135
+ timeout is not a free retry. Rewording the same query a third time, or
136
+ retrying through a different host/scheme (`http` vs `https`,
137
+ `export.arxiv.org` vs `arxiv.org/search`, an unofficial JSON mirror) after a
138
+ failure, is thrash: build cards from whatever the attempts returned, or
139
+ report the network failure plainly and stop.
102
140
 
103
141
  ## Gotchas
104
142
 
@@ -108,3 +146,23 @@ first two returned.
108
146
  - AlphaXiv is AI-generated enrichment: useful for scanning, never citable as
109
147
  authoritative.
110
148
  - Fetch or enrich only the top candidates, not every result.
149
+ - `export.arxiv.org` rate-limits (HTTP 429) under repeated hits; that is a
150
+ reason to stop at the fetch cap, not a reason to retry against a mirror.
151
+ - This skill's tool surface has no file-writing tool and no `artifact`. The
152
+ paper cards are the chat reply, never a document; do not go looking for a
153
+ way to save one.
154
+
155
+ ## Red flags
156
+
157
+ - Dispatching `researcher` for an ordinary search or compare "just in case"
158
+ it does a better job than a direct fetch — it is slower and unbounded by
159
+ default; reserve it for an explicit deep-survey ask.
160
+ - A dispatch task prompt without the arXiv-only, fetch-capped instruction —
161
+ that omission is what let a prior run wander into Semantic Scholar, DBLP,
162
+ and OpenAlex for minutes with nothing to show for it.
163
+ - Treating a timeout or 429 as free of the fetch cap and retrying against a
164
+ different host, scheme, or unofficial mirror instead of stopping.
165
+ - Raw Atom XML, a full abstract dump, or more than the top few candidates
166
+ reaching the final reply.
167
+ - Reaching for `artifact` or any write tool to "save" the result — it is not
168
+ in this skill's tool surface; the reply is the deliverable.
@@ -56,3 +56,53 @@ Expected:
56
56
 
57
57
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
58
58
  (30B local, llamacpp on mini), full-auto sandbox. PASS (driver FAIL overturned on transcript review): skill loaded, skill_surface blocked two curl attempts, retrieval ran through web_fetch on the Atom API per the designed dispatch-unavailable fallback. Judge misread the OR in bullet 1. Query thrash (8 fetches) led to the two-fetch cap now in the body.
59
+
60
+ ## Battletest record (2026-09-03)
61
+
62
+ S1 (topic search, speculative decoding) and S2 (single known paper,
63
+ `arxiv.org/abs/1706.03762`), `ornith1.5-35b-moe` on mini (llamacpp), `clio-coder
64
+ run --autonomy full-auto --json`, headless, real network access (no fixture).
65
+
66
+ | run | model | wall | turns | in / out tokens | safety blocks | outcome |
67
+ |---|---|---|---|---|---|---|
68
+ | baseline S1 (no skill) | ornith1.5-35b-moe | 286s (killed at timeout) | 9 | n/a (killed before `agent_end`) | 5 | never finished: fired raw `curl` via `bash` (permission-gated, refused) before falling back to `web_fetch`; hit real timeouts against `export.arxiv.org`; killed by the harness's 300s cap mid-turn with no cards produced |
69
+ | baseline S2 (no skill) | ornith1.5-35b-moe | 24s | 2 | 3.6k / 1.0k | 0 | fetched the real paper directly and wrote a good prose summary, but never labeled problem/method/evidence/limitation as distinct fields — the un-skilled gap S2 expects |
70
+ | v1 S1 (frozen v0.4.0, dispatch runaway) | ornith1.5-35b-moe | 286s (killed at timeout) | 6 | n/a (killed before `agent_end`) | 2 | opened a `tasks` plan (refused, outside surface), dispatched `researcher` twice (first dispatch flagged an absolute-path token in the briefing), then started re-fetching directly itself; killed by the harness's 300s cap before producing cards — the dispatch-runaway/no-cap bug reproduced live |
71
+ | v2 S1 (hardened v0.5.0) | ornith1.5-35b-moe | 109s | 4 | 2.9k / 3.9k | 2 (real `export.arxiv.org` 429 + timeout) | stayed on `web_fetch` only (no dispatch — topic search didn't ask for a "deep survey"), made exactly two fetch attempts against the identical URL per the tightened cap, hit a genuine rate limit then a timeout, **stopped at the cap**, explicitly refused to fabricate paper cards from invented IDs, and returned an honestly-labeled "established knowledge, not a fresh fetch" orientation instead — terminated cleanly on its own, no runaway |
72
+ | v2 S2 (hardened v0.5.0) | ornith1.5-35b-moe | 28s | 3 | 6.2k / 1.5k | 0 | direct `web_fetch` on the real paper page, full problem/method/evidence/limitation/relevance card, explicit read/skim/skip recommendation, AlphaXiv linked and labeled as enrichment, no `artifact` call — 5/5 on the S2 rubric |
73
+
74
+ Changes: removed `artifact` from `allowed-tools` — nothing in the procedure
75
+ ever called it, and it is a terminal tool that would end the run the moment
76
+ the model reached for it to "save" a result. Reworked Step 2 so `web_fetch`
77
+ directly against arXiv is the default vehicle for every request class
78
+ (single paper, search, compare, survey), and `dispatch` is reserved for an
79
+ explicitly requested deep/broad survey — the live probe that motivated this
80
+ pass showed a dispatched `researcher` wandering into Semantic Scholar, DBLP,
81
+ and OpenAlex with no arXiv-only restriction and no fetch budget, never
82
+ returning. When dispatch is used, its task-prompt template now states the
83
+ same arXiv-only, fetch-capped discipline the direct path uses, verbatim.
84
+ Tightened the Step 3 fetch cap to count failed attempts (timeout, 429,
85
+ connection error) against the same two-attempt budget as successes, and
86
+ banned escalating to a different host/scheme/mirror on failure — this closed
87
+ a real gap the v2 S1 run against a genuinely rate-limited `export.arxiv.org`
88
+ would otherwise have exploited (the same tightened wording is what let it
89
+ stop cleanly at 109s instead of retrying indefinitely). Added `## Arguments`
90
+ (free-text request, no operator headlessly, `tasks` refused) and a `##
91
+ Red flags` section naming the dispatch-runaway, uncapped-retry, and
92
+ artifact-reach failure modes actually observed.
93
+
94
+ Still weak: S3 (three-paper comparison) was not run against a real fixture
95
+ this pass — budget went to confirming the dispatch-runaway fix and the
96
+ fetch-cap fix on S1/S2, both of which reproduced live. The compare path's
97
+ "one query per ID or one broader query" guidance in Step 2 is new and
98
+ untested end-to-end. Both baseline and v1 runs for S1 were killed by the
99
+ harness's outer timeout rather than allowed to run to their own natural
100
+ (bad) conclusion — a longer timeout might show the old skill eventually
101
+ recovering, or might show it running further off scope; the fix (bounding
102
+ the fetch cap and defaulting off dispatch) is validated by the v2 behavior,
103
+ not by watching v1 fail for longer. `export.arxiv.org` rate-limited several
104
+ runs in this session from repeated hits in short succession; the 429s in the
105
+ v2 S1 run are a real external condition this pass ran into, not a fixture
106
+ simulation, but a quieter network day could produce a fully-populated
107
+ card set on the same prompt instead of the honest-failure path exercised
108
+ here — both are now handled, but only the failure path got a live rep.
@@ -8,7 +8,7 @@ triggers:
8
8
  - define numerical tolerances
9
9
  - reproduce these results
10
10
  - compare solver accuracy
11
- version: 0.2.0
11
+ version: 0.3.1
12
12
  license: Apache-2.0
13
13
  allowed-tools:
14
14
  - read
@@ -39,6 +39,25 @@ mode this protocol exists to prevent.
39
39
  Anti-trigger: if the question is "why is this output wrong", that is a
40
40
  diagnosis, not an experiment; use scientific-debugging.
41
41
 
42
+ ## Arguments
43
+
44
+ ```text
45
+ /skill experiment-protocol <what to benchmark, compare, or sweep>
46
+ ```
47
+
48
+ There is no operator in a headless run: `ask_user` is not registered, so
49
+ any call is refused as an unregistered tool rather than answered. If a threshold, tolerance, or environment detail is unstated,
50
+ pick the most defensible default, record it as an explicit assumption in the
51
+ pre-registration, and proceed; never stall Phase 0 waiting for confirmation.
52
+
53
+ The phases below are the plan; do not open a task list for them — `tasks`
54
+ sits outside this skill's tool surface and any call to it is refused.
55
+
56
+ Shell rules for every `bash` call: one command per call, plain and direct.
57
+ Never use `$(...)` or backticks; they trigger an approval gate that ends a
58
+ headless run. Capture checksums and environment facts with direct calls
59
+ (`sha256sum data/mesh.h5`), never command substitution.
60
+
42
61
  ## Phase 0 - Pre-register
43
62
 
44
63
  Before any measurement, write the protocol into the repository validation
@@ -89,3 +89,26 @@ Prompt: "Make this kernel faster."
89
89
 
90
90
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
91
91
  (30B local, llamacpp on mini), full-auto sandbox. PASS. Pre-registration written before touching the seeded kernel; judge 5/5.
92
+
93
+ ## Battletest record (2026-09-03)
94
+
95
+ Real `clio-coder run --json` against dynamo (LM Studio, `qwen3.8-27b`), the S1
96
+ kernel.py fixture above, git repo under
97
+ `/home/akougkas/eval-temp/expprotocol-fixture/`. Same gap as every other
98
+ research skill going in: no `## Arguments`, no `tasks`-refusal, no shell-rules
99
+ paragraph, no no-operator statement. Added all four.
100
+
101
+ | run | model | outcome |
102
+ |---|---|---|
103
+ | v0.3.0 (hardened) | qwen3.8-27b | environment capture (`python3 --version`, `uname -a`, CPU model), `numpy` version check, `sha256sum` on the input array and the frozen baseline copy, `.clio-coder/validation.yaml` written with thresholds/tolerance semantics before any benchmark, a real 20-rep baseline vs. a vectorized candidate, a bit-exact accuracy check between them, then a correctness spot-check on a second slice-based candidate before timing it — 34 tool calls, zero safety blocks, zero `$(...)`, zero `tasks`, `write`/`edit` used correctly (in this skill's surface, unlike scientific-debugging) instead of a heredoc |
104
+
105
+ The run did not reach a final Phase 3 verdict/report inside the 280s box used
106
+ this pass (31 API calls, ~750k cumulative input tokens — the box closes on
107
+ context-reprocessing volume, not model slowness); everything observed up to
108
+ that point followed Phase 0-2 exactly as specified, including registering a
109
+ 100x stretch target, measuring ~13x, and correctly continuing to iterate
110
+ rather than declaring victory early.
111
+
112
+ **Still weak**: no observed run reaching a written Phase 3 verdict this pass
113
+ (same time-box cause as scientific-debugging's record above). No cross-model
114
+ confirmation this pass.
@@ -8,7 +8,7 @@ triggers:
8
8
  - diagnose NaNs
9
9
  - debug with falsifiable hypotheses
10
10
  - scientific root cause
11
- version: 0.2.0
11
+ version: 0.3.1
12
12
  license: Apache-2.0
13
13
  allowed-tools:
14
14
  - read
@@ -38,6 +38,29 @@ Anti-trigger: if the failure is a typo, a missing import, or an error message
38
38
  that names its own cause, fix it directly and skip this workflow. The loop
39
39
  below is for failures that survived the first obvious fix.
40
40
 
41
+ ## Arguments
42
+
43
+ ```text
44
+ /skill scientific-debugging <failure description>
45
+ ```
46
+
47
+ Everything after the skill name is the failure report: the observed wrong
48
+ behavior and whatever has already been tried. There is no operator in a
49
+ headless run: `ask_user` is not registered, so any call is refused as an
50
+ unregistered tool rather than answered. If the goal,
51
+ a fault-class split, or a ranking call is ambiguous, state your best reading
52
+ in Step 1 or Step 3 and proceed; never stall a step waiting for confirmation.
53
+
54
+ The Loop below is the plan; do not open a task list for it — `tasks` sits
55
+ outside this skill's tool surface and any call to it is refused.
56
+
57
+ Shell rules for every `bash` call: one command per call, plain and direct.
58
+ Never use `$(...)` or backticks; they trigger an approval gate that ends a
59
+ headless run. This skill has no `write`/`edit` tool — the structured-
60
+ investigation file in the Tiers section below is written with a `bash`
61
+ heredoc (`cat > file <<'EOF' ... EOF`), never through an edit tool that
62
+ isn't in this skill's surface.
63
+
41
64
  ## The Loop
42
65
 
43
66
  1. **Goal.** One sentence stating the observable "fixed" state. "The regression
@@ -136,3 +136,21 @@ accumulation loop; regression check fails by 1.474e-4).
136
136
 
137
137
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
138
138
  (30B local, llamacpp on mini), full-auto sandbox. PASS. Judge 6/6 on the seeded numerical fixture; cleanest research run.
139
+
140
+ ## Battletest record (2026-09-03)
141
+
142
+ Real `clio-coder run --json` against dynamo (LM Studio, `qwen3.8-27b`), the S1
143
+ diffusion fixture above (order-sensitive summation regression, git repo under
144
+ `/home/akougkas/eval-temp/scidebug-fixture/`). No `## Arguments` section,
145
+ `tasks`-refusal, shell-rules paragraph, or no-operator statement existed in
146
+ the skill body going in — same gap every other hardened category this session
147
+ found. Added all four, matching skills/coding/prototype/SKILL.md's pattern,
148
+ and made explicit that the structured-investigation file goes through a
149
+ `bash` heredoc because this skill has no `write`/`edit` tool.
150
+
151
+ | run | model | outcome |
152
+ |---|---|---|
153
+ | baseline (no skill) | qwen3.8-27b | ran 9 API calls / ~180k tokens without converging inside a 200s box; genuinely still reasoning, not stalled |
154
+ | v0.3.0 (hardened) | qwen3.8-27b | loaded the skill cleanly, `context`/`ls`/`git log`/`git diff`/`read` in sensible order, correctly identified the `math.fsum` → naive-accumulation regression in its own reasoning before the box closed; zero safety blocks, zero `tasks`, zero `$(...)` |
155
+
156
+ **Still weak**: this session's harness runs are token-heavy (each tool round-trip reprocesses the full growing context) and neither the baseline nor the hardened run reached a written goal/hypotheses/verdict block inside the time box used this pass — the trajectory is correct and clean, but full-loop completion on this model under this box is unconfirmed, only strongly suggested. No cross-model confirmation this pass (time-boxed session). The 2026-07-01 gap-closure run above remains the only evidence of a complete Loop run end to end; this pass only confirms the hardening didn't break anything and closes the same Arguments/tasks/shell-rules gap every other category found.
@@ -8,7 +8,7 @@ triggers:
8
8
  - migrate the scientific build system
9
9
  - create a maintained fork
10
10
  - preserve scientific parity
11
- version: 0.3.0
11
+ version: 0.4.1
12
12
  license: Apache-2.0
13
13
  allowed-tools:
14
14
  - read
@@ -36,6 +36,31 @@ translation. Faster code, a clean build, and passing self-authored unit tests
36
36
  do not establish scientific equivalence. Work the stages below in order; each
37
37
  has an explicit exit condition.
38
38
 
39
+ ## Arguments
40
+
41
+ ```text
42
+ /skill scientific-modernization <what to modernize, port, rewrite, or replace, and why>
43
+ ```
44
+
45
+ There is no operator in a headless run: `ask_user` is not registered, so
46
+ any call is refused as an unregistered tool rather than answered. Stage 1's "the user has seen them" exit condition means,
47
+ headlessly: state the four bullets in your reply and proceed, never stall
48
+ waiting for acknowledgment. Every other stage gate below works the same way
49
+ — state the decision and its reasoning, then continue.
50
+
51
+ The six stages are the plan; do not open a task list for them — `tasks` sits
52
+ outside this skill's tool surface and any call to it is refused.
53
+
54
+ Shell rules for every `bash` call: one command per call, plain and direct.
55
+ Never use `$(...)` or backticks; they trigger an approval gate that ends a
56
+ headless run.
57
+
58
+ This is a long, multi-stage process on a small model or a time-boxed run. If
59
+ you are approaching your tool-call or time budget before reaching Stage 6,
60
+ stop at the current stage, state exactly which stage you reached and why you
61
+ stopped, and report the work as an incomplete prototype — never fabricate
62
+ completion of stages you did not actually reach.
63
+
39
64
  ## Stage 1 — Decide whether this work should exist
40
65
 
41
66
  Identify the upstream project: active maintainers, license, release cadence,
@@ -82,3 +82,30 @@ Expected:
82
82
 
83
83
  One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
84
84
  (30B local, llamacpp on mini), full-auto sandbox. NOT COMPLETED: loop guard at 75 tool calls. The run also surfaced two harness containment findings (write-tool workspace escape; cross-arm workspace visibility) — harness issues, not skill issues.
85
+
86
+ ## Battletest record (2026-09-03)
87
+
88
+ Same structural gap as the other three research skills: no `## Arguments`,
89
+ no `tasks`-refusal, no shell-rules paragraph, no no-operator statement. Added
90
+ all four, plus an explicit budget-awareness line (state which stage you
91
+ reached and report the work as an incomplete prototype rather than
92
+ fabricate completion) directly answering the 2026-08-13 record's loop-guard
93
+ finding above.
94
+
95
+ This mission's plan called for a small legacy-C fixture and a live A/B pass
96
+ on `mini`; that track was killed mid-run this session for taking materially
97
+ longer than the wall-clock budget available (this is a six-stage,
98
+ `model-size: large` skill — the prior smoke record already didn't complete
99
+ on a 30B model, and a fresh fixture never got built before the track was
100
+ stopped). **This pass is a static hardening pass only** — the four additions
101
+ above were applied and `npm run skills:pin && npm run skills:check && npm
102
+ run lint` verified green, but no fresh live run against this skill's
103
+ hardened body exists yet. Treat 0.4.0 as unverified beyond the structural
104
+ fix; it needs the small legacy-C fixture (a single-file numeric routine with
105
+ a captured reference output as the oracle) and a real baseline/hardened A/B
106
+ pass before its `eval-status` can move past `scenarios-recorded`.
107
+
108
+ **Still weak**: everything — no fixture built this pass, no run executed
109
+ against 0.4.0, no cross-model confirmation. This is the one skill in the
110
+ category that did not get real battletest evidence this session; say so
111
+ plainly rather than implying otherwise from the version bump.