@iowarp/clio-coder 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (595) hide show
  1. package/CHANGELOG.md +90 -2
  2. package/CONTRIBUTING.md +23 -23
  3. package/README.md +284 -613
  4. package/dist/{acp-FPR54DGL.js → acp-BIYHVZIM.js} +43 -53
  5. package/dist/{agents-OGPIHPJH.js → agents-YT6SSRIT.js} +43 -26
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-IC3K6NIZ.js → auth-5TWEIYDN.js} +20 -12
  8. package/dist/{chunk-PV4JUBVJ.js → chunk-2EHAIA3X.js} +40 -21
  9. package/dist/{chunk-IS3ONKU3.js → chunk-2IR2NMPA.js} +6 -4
  10. package/dist/chunk-2SFS6XQE.js +122 -0
  11. package/dist/chunk-2VTFPG5O.js +48 -0
  12. package/dist/{chunk-MAR7Y6HW.js → chunk-3ZXDFGR5.js} +23 -16
  13. package/dist/chunk-4BJ5BYCE.js +61 -0
  14. package/dist/{chunk-474KN5II.js → chunk-4BPJXDWC.js} +111 -181
  15. package/dist/chunk-4KLWL3UC.js +18 -0
  16. package/dist/chunk-4VP4KH3K.js +962 -0
  17. package/dist/chunk-4ZG3XFUR.js +77 -0
  18. package/dist/chunk-5B2AEOW5.js +5407 -0
  19. package/dist/{chunk-K2ITRMHZ.js → chunk-5TSRNF4G.js} +6 -138
  20. package/dist/{chunk-OLBBMFRD.js → chunk-5UUP6MWO.js} +24 -62
  21. package/dist/chunk-65DEGPJ6.js +52 -0
  22. package/dist/chunk-6EJMN2Y3.js +17 -0
  23. package/dist/chunk-6EJV5X2W.js +16405 -0
  24. package/dist/chunk-6N5PTWMY.js +136 -0
  25. package/dist/chunk-6XLNIQDB.js +27 -0
  26. package/dist/{chunk-TEKV33Q5.js → chunk-77VKQEHF.js} +65 -33
  27. package/dist/chunk-7CR24IG7.js +242 -0
  28. package/dist/{chunk-M5T5VO65.js → chunk-7EYHLWU7.js} +837 -635
  29. package/dist/chunk-7MNJORFF.js +22 -0
  30. package/dist/{chunk-KY56HMHH.js → chunk-A3CYT5EX.js} +125 -31
  31. package/dist/chunk-AGYYIBLL.js +1069 -0
  32. package/dist/chunk-AO4RKG4M.js +277 -0
  33. package/dist/{chunk-GB6QRBXN.js → chunk-APJ265NV.js} +54 -1187
  34. package/dist/chunk-ARBGF5F7.js +174 -0
  35. package/dist/{chunk-673JJUWJ.js → chunk-BMEMKKIT.js} +2 -2
  36. package/dist/chunk-CBCAPZAA.js +229 -0
  37. package/dist/chunk-CMZWFGD2.js +352 -0
  38. package/dist/chunk-ECH6PKUQ.js +39 -0
  39. package/dist/chunk-ED4KHGC3.js +143 -0
  40. package/dist/chunk-EKMEHE4H.js +340 -0
  41. package/dist/{chunk-RPTR2H26.js → chunk-EPVUXGXG.js} +21 -15
  42. package/dist/chunk-FCSXB6T2.js +338 -0
  43. package/dist/chunk-FJ3H4MN5.js +48 -0
  44. package/dist/chunk-FQ4SKYE4.js +29 -0
  45. package/dist/chunk-G2DE3C7R.js +644 -0
  46. package/dist/chunk-G4BMMOKF.js +182 -0
  47. package/dist/{chunk-ZPY3JZ5E.js → chunk-GGXXDWE4.js} +183 -1233
  48. package/dist/chunk-HC4CLZ2Y.js +68 -0
  49. package/dist/{chunk-LU4TK2PR.js → chunk-HFSBBKSQ.js} +5 -56
  50. package/dist/{chunk-PIUMUEMV.js → chunk-HKIYEGME.js} +10 -6
  51. package/dist/chunk-I4HZDVNP.js +73 -0
  52. package/dist/{chunk-4QKXUHSR.js → chunk-IGLFWIYI.js} +70 -20
  53. package/dist/chunk-IKCO5N3L.js +162 -0
  54. package/dist/chunk-IR4CFBFN.js +56 -0
  55. package/dist/{chunk-PFEFKVGL.js → chunk-J5HN4RYU.js} +13 -11
  56. package/dist/{chunk-R5KLMSBV.js → chunk-J5Q24KAG.js} +2 -2
  57. package/dist/{chunk-K5XEMXTI.js → chunk-JVCV3ICN.js} +1 -1
  58. package/dist/chunk-KJ5LWLOE.js +1077 -0
  59. package/dist/chunk-LBMZMYH2.js +285 -0
  60. package/dist/{chunk-G34LV2PF.js → chunk-LBNRH5WM.js} +84 -170
  61. package/dist/{chunk-H6F6BYOH.js → chunk-LZSJBIVT.js} +7003 -7434
  62. package/dist/{chunk-HQQID6OA.js → chunk-M6SHUN7Q.js} +5 -5
  63. package/dist/chunk-MAW544W2.js +1882 -0
  64. package/dist/chunk-MBS4V7ZP.js +217 -0
  65. package/dist/{chunk-FST4FYJB.js → chunk-MFFY33HR.js} +99 -140
  66. package/dist/chunk-MNA4JGU4.js +255 -0
  67. package/dist/chunk-MQSRRFWA.js +3428 -0
  68. package/dist/{chunk-BSU2YIWB.js → chunk-MVVUPGPW.js} +131 -136
  69. package/dist/chunk-OAO4GE4M.js +619 -0
  70. package/dist/chunk-OHHN2SO4.js +5135 -0
  71. package/dist/chunk-OKGUZO2U.js +34 -0
  72. package/dist/{chunk-GAEBEQVI.js → chunk-OOJYHWRB.js} +32 -346
  73. package/dist/{chunk-Q5WJOSJ7.js → chunk-OQ33BKR3.js} +2 -1
  74. package/dist/chunk-OQE5J4C6.js +73 -0
  75. package/dist/{chunk-KKNLWXI6.js → chunk-ORBHGJC5.js} +8 -8
  76. package/dist/chunk-POHLU5DW.js +1186 -0
  77. package/dist/chunk-QKMUKYO7.js +4961 -0
  78. package/dist/{chunk-ASND7OZK.js → chunk-QTYWRVRA.js} +13 -13
  79. package/dist/{chunk-EYOKLTMF.js → chunk-SRF2PJNW.js} +17 -3
  80. package/dist/chunk-SST6Z5JA.js +80 -0
  81. package/dist/chunk-STBPMHSX.js +2456 -0
  82. package/dist/chunk-T6YILFSB.js +80 -0
  83. package/dist/chunk-TZTZS7QK.js +227 -0
  84. package/dist/chunk-UOV2BYIW.js +107 -0
  85. package/dist/{chunk-Q3RUPKEJ.js → chunk-V4RXGQ5Q.js} +58 -189
  86. package/dist/chunk-VAKQQHWR.js +434 -0
  87. package/dist/chunk-VG7TBQIY.js +128 -0
  88. package/dist/chunk-VJWL6YS5.js +244 -0
  89. package/dist/chunk-WEH5XRJQ.js +32 -0
  90. package/dist/chunk-WMSVI4G2.js +2095 -0
  91. package/dist/chunk-WVO7V2QY.js +797 -0
  92. package/dist/chunk-X4RCMKVQ.js +641 -0
  93. package/dist/chunk-X75S7HFS.js +374 -0
  94. package/dist/chunk-XN3L4EYL.js +46 -0
  95. package/dist/{chunk-RDLVBZEO.js → chunk-YCWGATWI.js} +6 -4
  96. package/dist/chunk-YHZX5GEU.js +193 -0
  97. package/dist/chunk-YXLYO42X.js +91 -0
  98. package/dist/{chunk-NMOX6HFD.js → chunk-ZDOOVTXZ.js} +29 -77
  99. package/dist/chunk-ZI647VB5.js +37 -0
  100. package/dist/{chunk-C4PTHK7P.js → chunk-ZWLZP4ZT.js} +5 -5
  101. package/dist/cli/index.js +62 -54
  102. package/dist/clio-4LY5K2AC.js +25 -0
  103. package/dist/code-nav-7AX6FYE6.js +600 -0
  104. package/dist/codewiki/build-worker.js +66 -0
  105. package/dist/compile-cache-CVJMMODC.js +18 -0
  106. package/dist/{components-DMAOEKFB.js → components-KELWS457.js} +11 -6
  107. package/dist/{config-IRUQ7SE4.js → config-GTLUW2PR.js} +92 -55
  108. package/dist/configure-R6A64DHX.js +42 -0
  109. package/dist/context-5VKGUVJJ.js +866 -0
  110. package/dist/{context-3KWFLHJG.js → context-JFZEJ7W5.js} +15 -13
  111. package/dist/{context-5RADCKTR.js → context-RW5HC47S.js} +71 -35
  112. package/dist/{context-clear-7TSNPAAI.js → context-clear-6ZHBAZZT.js} +54 -28
  113. package/dist/{context-index-W4RLWOQH.js → context-index-BZ4UYMTC.js} +30 -24
  114. package/dist/dispatch-runner-VKBRCWQC.js +1997 -0
  115. package/dist/{docs-5AWSPS37.js → docs-2C2LTVT2.js} +23 -10
  116. package/dist/{doctor-UC5NAJYQ.js → doctor-KI767GSN.js} +27 -17
  117. package/dist/{eval-U6TJHRLX.js → eval-XSSNATB4.js} +29 -16
  118. package/dist/{evidence-YEGUW4L3.js → evidence-UA6AWDQQ.js} +46 -26
  119. package/dist/{evolve-TXARCTPG.js → evolve-QNTFGV6Z.js} +45 -25
  120. package/dist/{extensions-OZFJ3A3G.js → extensions-QVDOHDGJ.js} +16 -7
  121. package/dist/{fleet-6G3DHNYE.js → fleet-Q7UOMUSG.js} +163 -54
  122. package/dist/{fleet-preflight-DSNT37JK.js → fleet-preflight-DDN536IT.js} +7 -4
  123. package/dist/{init-KZ5QTF6M.js → init-WBB65ZHQ.js} +69 -32
  124. package/dist/{memory-73ESV5YC.js → memory-MD3O64RI.js} +48 -27
  125. package/dist/{models-A4PVNWJK.js → models-BZU34YWD.js} +39 -25
  126. package/dist/monitor-MEQA5C3I.js +661 -0
  127. package/dist/{chunk-FCIH3BIZ.js → orchestrator-CGFKEP27.js} +11832 -8687
  128. package/dist/{paths-C4H6IV77.js → paths-UXLN5YYZ.js} +10 -5
  129. package/dist/{preload-6WVMHX3A.js → preload-P6DGH2PZ.js} +2 -2
  130. package/dist/{reset-BGW6OGMV.js → reset-L2FQEE3E.js} +16 -10
  131. package/dist/{run-YTPEYQOH.js → run-IV4Q6RLN.js} +101 -61
  132. package/dist/{share-YIFFV4NQ.js → share-S5BZQC5I.js} +15 -8
  133. package/dist/{skills-2V6RA3OQ.js → skills-LQEKRDTN.js} +34 -14
  134. package/dist/{skills-eval-S2TVJO4F.js → skills-eval-3DC4HEWS.js} +70 -34
  135. package/dist/steer-GGWFUJUD.js +77 -0
  136. package/dist/{targets-TYXLPB23.js → targets-C4SSGQOB.js} +43 -27
  137. package/dist/terminal-lease-IT5JW2NR.js +395 -0
  138. package/dist/{trace-GGOJ6Q6Z.js → trace-PNCASAXC.js} +41 -16
  139. package/dist/{chunk-N6F52NLF.js → tree-sitter-HGKH6LG4.js} +28 -2306
  140. package/dist/{uninstall-LLLT4F4W.js → uninstall-FZCQCDKC.js} +10 -5
  141. package/dist/{upgrade-33G2LMM5.js → upgrade-7TT7SQ3G.js} +45 -25
  142. package/dist/{usage-ZAFSXKKG.js → usage-GV4PKT3M.js} +62 -31
  143. package/dist/verify-G6V4D2G7.js +716 -0
  144. package/dist/web-fetch-2YHJ3KTG.js +638 -0
  145. package/dist/{wiki-generate-NUQCVOQ3.js → wiki-generate-DQF6Z66B.js} +74 -34
  146. package/dist/worker/entry.js +221 -36
  147. package/dist/workspace-G4ZWUIPR.js +22 -0
  148. package/docs/README.md +22 -17
  149. package/docs/acp.md +168 -16
  150. package/docs/alcf-provider.md +1 -1
  151. package/docs/architecture.md +136 -7
  152. package/docs/artifact-versions.md +1 -1
  153. package/docs/built-in-agents.md +1 -1
  154. package/docs/capacity-and-scheduling.md +1 -1
  155. package/docs/commands-and-modes.md +114 -71
  156. package/docs/config-knobs-audit.md +1 -3
  157. package/docs/configuration-and-targets.md +174 -46
  158. package/docs/context-engine.md +29 -6
  159. package/docs/development-pipeline.md +26 -1
  160. package/docs/dispatch-architecture-rationale.md +1 -1
  161. package/docs/documentation-coverage.md +2 -2
  162. package/docs/documentation-guide.md +1 -1
  163. package/docs/environment-variables.md +13 -5
  164. package/docs/eval-runner.md +1 -1
  165. package/docs/evals-internal.md +1 -1
  166. package/docs/evidence-and-memory.md +6 -2
  167. package/docs/evolution.md +2 -2
  168. package/docs/exit-codes-and-output.md +15 -9
  169. package/docs/extensions-and-sharing.md +9 -9
  170. package/docs/fleet-dispatch.md +7 -5
  171. package/docs/git-commit-provenance.md +120 -0
  172. package/docs/glossary.md +1 -1
  173. package/docs/installation-and-lifecycle.md +34 -27
  174. package/docs/middleware-and-components.md +1 -1
  175. package/docs/model-catalog.md +45 -14
  176. package/docs/observability.md +8 -5
  177. package/docs/performance-methodology.md +491 -0
  178. package/docs/pi-boundary.md +72 -0
  179. package/docs/proactive-memory.md +3 -3
  180. package/docs/prompt-envelope-and-tools.md +24 -3
  181. package/docs/provider-adapter-cookbook.md +57 -4
  182. package/docs/release-cut-checklist.md +129 -115
  183. package/docs/safety-model.md +9 -5
  184. package/docs/scientific-validation.md +3 -3
  185. package/docs/session-lifecycle.md +55 -12
  186. package/docs/skills-marketplace.md +12 -8
  187. package/docs/time-conventions.md +1 -1
  188. package/docs/tool-usage.md +3 -3
  189. package/docs/trace-store.md +1 -1
  190. package/docs/troubleshooting.md +10 -7
  191. package/docs/tui-design.md +47 -10
  192. package/docs/worker-dispatch-mechanics.md +1 -1
  193. package/package.json +19 -22
  194. package/skills/coding/ast-grep/SKILL.md +136 -0
  195. package/skills/coding/ast-grep/evals.md +56 -0
  196. package/skills/coding/ast-grep/references/rule_reference.md +297 -0
  197. package/skills/coding/coding-standards/SKILL.md +113 -0
  198. package/skills/coding/coding-standards/evals.md +34 -0
  199. package/skills/coding/prototype/SKILL.md +86 -0
  200. package/skills/coding/prototype/evals.md +42 -0
  201. package/skills/coding/prototype/references/LOGIC.md +67 -0
  202. package/skills/coding/prototype/references/UI.md +112 -0
  203. package/skills/coding/tdd/SKILL.md +101 -0
  204. package/skills/coding/tdd/evals.md +41 -0
  205. package/skills/coding/tdd/references/mocking.md +59 -0
  206. package/skills/coding/tdd/references/tests.md +77 -0
  207. package/skills/context/context-handoff/SKILL.md +126 -0
  208. package/skills/context/context-handoff/evals.md +57 -0
  209. package/skills/context/context-handoff/scripts/new-handoff.sh +26 -0
  210. package/skills/context/context-prime/SKILL.md +95 -0
  211. package/skills/context/context-prime/evals.md +54 -0
  212. package/skills/meta/clio-dev/SKILL.md +91 -0
  213. package/skills/meta/clio-dev/evals.md +45 -0
  214. package/skills/meta/clio-test/SKILL.md +130 -0
  215. package/skills/meta/clio-test/evals.md +43 -0
  216. package/skills/meta/clio-test/references/harness.md +97 -0
  217. package/skills/meta/clio-test/references/test-map.md +59 -0
  218. package/skills/meta/credentials/SKILL.md +125 -0
  219. package/skills/meta/credentials/evals.md +104 -0
  220. package/skills/meta/find-skills/SKILL.md +72 -0
  221. package/skills/meta/find-skills/evals.md +47 -0
  222. package/skills/meta/herdr/SKILL.md +127 -0
  223. package/skills/meta/herdr/evals.md +38 -0
  224. package/skills/meta/skill-craft/SKILL.md +102 -0
  225. package/skills/meta/skill-craft/evals.md +41 -0
  226. package/skills/planning/architecture/SKILL.md +129 -0
  227. package/skills/planning/architecture/evals.md +36 -0
  228. package/skills/planning/backlog/SKILL.md +90 -0
  229. package/skills/planning/backlog/evals.md +43 -0
  230. package/skills/planning/prd/SKILL.md +82 -0
  231. package/skills/planning/prd/evals.md +49 -0
  232. package/skills/planning/product-intent/SKILL.md +112 -0
  233. package/skills/planning/product-intent/evals.md +36 -0
  234. package/skills/planning/tech-spec/SKILL.md +115 -0
  235. package/skills/planning/tech-spec/evals.md +47 -0
  236. package/skills/registry.yaml +136 -0
  237. package/skills/research/arxiv-literature/SKILL.md +104 -0
  238. package/skills/research/arxiv-literature/evals.md +58 -0
  239. package/skills/research/experiment-protocol/SKILL.md +122 -0
  240. package/skills/research/experiment-protocol/evals.md +91 -0
  241. package/skills/research/scientific-debugging/SKILL.md +119 -0
  242. package/skills/research/scientific-debugging/evals.md +138 -0
  243. package/skills/research/scientific-modernization/SKILL.md +138 -0
  244. package/skills/research/scientific-modernization/evals.md +84 -0
  245. package/skills/workflow/design-council/SKILL.md +139 -0
  246. package/skills/workflow/design-council/evals.md +97 -0
  247. package/skills/workflow/grill-me/SKILL.md +186 -0
  248. package/skills/workflow/grill-me/evals.md +78 -0
  249. package/skills/workflow/workflow-distiller/SKILL.md +136 -0
  250. package/skills/workflow/workflow-distiller/evals.md +107 -0
  251. package/src/cli/acp.ts +31 -4
  252. package/src/cli/clio.ts +68 -6
  253. package/src/cli/config-inspect.ts +28 -22
  254. package/src/cli/configure.ts +47 -9
  255. package/src/cli/context-clear.ts +2 -2
  256. package/src/cli/context-index.ts +21 -23
  257. package/src/cli/context.ts +13 -8
  258. package/src/cli/default-target.ts +9 -17
  259. package/src/cli/docs.ts +11 -5
  260. package/src/cli/evidence.ts +4 -1
  261. package/src/cli/extensions.ts +10 -1
  262. package/src/cli/fleet.ts +47 -6
  263. package/src/cli/index.ts +55 -26
  264. package/src/cli/memory.ts +3 -1
  265. package/src/cli/models.ts +1 -1
  266. package/src/cli/modes/json-stream.ts +37 -1
  267. package/src/cli/modes/print.ts +24 -9
  268. package/src/cli/run.ts +2 -2
  269. package/src/cli/skills-eval.ts +23 -8
  270. package/src/cli/skills.ts +19 -4
  271. package/src/cli/targets.ts +4 -0
  272. package/src/cli/text-layout.ts +15 -5
  273. package/src/cli/trace.ts +62 -14
  274. package/src/cli/upgrade.ts +18 -2
  275. package/src/cli/usage.ts +10 -3
  276. package/src/cli/wiki-generate.ts +2 -1
  277. package/src/core/agent-environment.ts +7 -0
  278. package/src/core/bash-exec.ts +72 -1
  279. package/src/core/boot-trace.ts +9 -4
  280. package/src/core/bus-events.ts +20 -4
  281. package/src/core/commit-attribution.ts +157 -0
  282. package/src/core/compile-cache.ts +159 -0
  283. package/src/core/config.ts +131 -2
  284. package/src/core/defaults.ts +39 -5
  285. package/src/core/domain-loader.ts +12 -5
  286. package/src/core/git-commit-attribution.ts +362 -0
  287. package/src/core/incomplete-installation.ts +45 -0
  288. package/src/core/response-schema.ts +1 -1
  289. package/src/core/safe-exec.ts +13 -1
  290. package/src/core/settings-layers.ts +155 -21
  291. package/src/core/skill-activation.ts +1 -1
  292. package/src/core/startup-timer.ts +3 -3
  293. package/src/core/state-file-lock.ts +13 -1
  294. package/src/core/termination.ts +78 -5
  295. package/src/domains/config/classify.ts +15 -3
  296. package/src/domains/config/extension.ts +19 -13
  297. package/src/domains/config/index.ts +10 -0
  298. package/src/domains/config/keybindings.ts +42 -6
  299. package/src/domains/context/bootstrap-prompt.ts +1 -1
  300. package/src/domains/context/bootstrap.ts +111 -18
  301. package/src/domains/context/clear.ts +16 -11
  302. package/src/domains/context/clio-md.ts +111 -9
  303. package/src/domains/context/codewiki/artifact.ts +400 -0
  304. package/src/domains/context/codewiki/build-worker-protocol.ts +24 -0
  305. package/src/domains/context/codewiki/build-worker.ts +54 -0
  306. package/src/domains/context/codewiki/coordinator.ts +182 -0
  307. package/src/domains/context/codewiki/indexer.ts +59 -144
  308. package/src/domains/context/codewiki/paths.ts +67 -0
  309. package/src/domains/context/codewiki/schema.ts +80 -0
  310. package/src/domains/context/codewiki/tree-sitter.ts +1 -1
  311. package/src/domains/context/contract.ts +11 -5
  312. package/src/domains/context/extension.ts +94 -143
  313. package/src/domains/context/fingerprint.ts +3 -1
  314. package/src/domains/context/index.ts +12 -22
  315. package/src/domains/context/project-metadata.ts +19 -0
  316. package/src/domains/context/prompt-context.ts +9 -10
  317. package/src/domains/context/refresh.ts +29 -21
  318. package/src/domains/context/runtime.ts +17 -0
  319. package/src/domains/context/wiki/generate.ts +39 -34
  320. package/src/domains/context/wiki/plan.ts +1 -1
  321. package/src/domains/context/wiki/prompts.ts +21 -8
  322. package/src/domains/dispatch/code-step.ts +20 -1
  323. package/src/domains/dispatch/extension.ts +158 -21
  324. package/src/domains/dispatch/failure-classification.ts +6 -0
  325. package/src/domains/dispatch/fleet-commit-attribution.ts +56 -0
  326. package/src/domains/dispatch/orphan-recovery.ts +50 -8
  327. package/src/domains/dispatch/receipt-integrity.ts +5 -0
  328. package/src/domains/dispatch/state.ts +31 -6
  329. package/src/domains/dispatch/transport.ts +2 -1
  330. package/src/domains/dispatch/types.ts +14 -0
  331. package/src/domains/dispatch/worker-spawn.ts +21 -2
  332. package/src/domains/eval/metrics/context.ts +1 -1
  333. package/src/domains/eval/types.ts +0 -1
  334. package/src/domains/evidence/build.ts +41 -1
  335. package/src/domains/lifecycle/migrations/2026-08-18-lmstudio-runtime-id.ts +52 -0
  336. package/src/domains/lifecycle/migrations/index.ts +24 -4
  337. package/src/domains/middleware/hooks-io.ts +12 -0
  338. package/src/domains/middleware/skills-reminder.ts +30 -15
  339. package/src/domains/prompts/compiler.ts +142 -84
  340. package/src/domains/prompts/contract.ts +18 -2
  341. package/src/domains/prompts/extension.ts +39 -7
  342. package/src/domains/prompts/fragment-loader.ts +0 -1
  343. package/src/domains/prompts/fragments/identity/clio.md +2 -4
  344. package/src/domains/prompts/fragments/identity/docs-routing.md +10 -0
  345. package/src/domains/prompts/fragments/identity/self-awareness.md +1 -45
  346. package/src/domains/prompts/fragments/operating/contract.md +4 -50
  347. package/src/domains/prompts/fragments/operating/delegation.md +42 -0
  348. package/src/domains/prompts/fragments/operating/skills.md +26 -0
  349. package/src/domains/prompts/fragments/operating/worker.md +16 -0
  350. package/src/domains/prompts/fragments/safety/auto-edit.md +5 -5
  351. package/src/domains/prompts/fragments/safety/full-auto.md +3 -3
  352. package/src/domains/prompts/fragments/safety/read-only.md +4 -4
  353. package/src/domains/prompts/fragments/safety/suggest.md +2 -2
  354. package/src/domains/prompts/fragments/wiki/page.md +10 -0
  355. package/src/domains/prompts/fragments/wiki/plan.md +10 -0
  356. package/src/domains/prompts/preload.ts +3 -3
  357. package/src/domains/providers/auth/api-key.ts +1 -1
  358. package/src/domains/providers/auth/backend-file.ts +20 -10
  359. package/src/domains/providers/auth/backend-memory.ts +59 -4
  360. package/src/domains/providers/auth/boot-status.ts +65 -0
  361. package/src/domains/providers/auth/oauth.ts +2 -1
  362. package/src/domains/providers/auth/storage.ts +97 -38
  363. package/src/domains/providers/capabilities.ts +12 -4
  364. package/src/domains/providers/contract.ts +15 -4
  365. package/src/domains/providers/extension.ts +18 -6
  366. package/src/domains/providers/model-runtime-capabilities.ts +15 -4
  367. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +118 -35
  368. package/src/domains/providers/plugins.ts +5 -3
  369. package/src/domains/providers/probe/fingerprint.ts +25 -5
  370. package/src/domains/providers/registry.ts +31 -10
  371. package/src/domains/providers/runtimes/boot-manifest.ts +55 -0
  372. package/src/domains/providers/runtimes/builtins.ts +2 -2
  373. package/src/domains/providers/runtimes/common/lmstudio-http.ts +423 -0
  374. package/src/domains/providers/runtimes/common/local-synth.ts +6 -7
  375. package/src/domains/providers/runtimes/local-native/lmstudio.ts +241 -0
  376. package/src/domains/providers/support.ts +6 -3
  377. package/src/domains/providers/types/local-model-quirks.ts +7 -9
  378. package/src/domains/providers/types/runtime-descriptor.ts +12 -1
  379. package/src/domains/providers/types/target-descriptor.ts +22 -0
  380. package/src/domains/resources/contract.ts +0 -1
  381. package/src/domains/resources/extension.ts +1 -3
  382. package/src/domains/resources/loader.ts +3 -4
  383. package/src/domains/resources/prompts/loader.ts +16 -2
  384. package/src/domains/resources/prompts/substitute.ts +1 -65
  385. package/src/domains/resources/skills/content-hash.ts +2 -0
  386. package/src/domains/resources/skills/install.ts +17 -0
  387. package/src/domains/resources/skills/loader.ts +17 -10
  388. package/src/domains/resources/skills/marketplace.ts +55 -9
  389. package/src/domains/safety/action-classifier.ts +4 -2
  390. package/src/domains/safety/audit.ts +8 -2
  391. package/src/domains/safety/extension.ts +1 -1
  392. package/src/domains/session/compaction/branch-summary.ts +3 -2
  393. package/src/domains/session/compaction/cut-point.ts +2 -1
  394. package/src/domains/session/compaction/tokens.ts +2 -1
  395. package/src/domains/session/context-ledger.ts +14 -0
  396. package/src/domains/session/contract.ts +15 -0
  397. package/src/domains/session/decision-board.ts +190 -0
  398. package/src/domains/session/entries.ts +66 -3
  399. package/src/domains/session/extension.ts +93 -12
  400. package/src/domains/session/retry.ts +10 -18
  401. package/src/domains/session/session-artifacts.ts +107 -0
  402. package/src/domains/session/task-board.ts +207 -13
  403. package/src/domains/session/tree/active-path.ts +44 -5
  404. package/src/domains/session/tree/fork.ts +26 -27
  405. package/src/domains/session/tree/preview.ts +2 -2
  406. package/src/domains/session/workspace/git-probe.ts +17 -11
  407. package/src/domains/user-tasks/store.ts +297 -0
  408. package/src/engine/acp/errors.ts +96 -0
  409. package/src/engine/acp/server.ts +1728 -146
  410. package/src/engine/acp/transport.ts +135 -14
  411. package/src/engine/acp/types.ts +26 -0
  412. package/src/engine/agent.ts +3 -3
  413. package/src/engine/ai.ts +32 -27
  414. package/src/engine/alcf-oauth.ts +26 -19
  415. package/src/engine/api-registry.ts +223 -0
  416. package/src/engine/apis/index.ts +3 -7
  417. package/src/engine/apis/llamacpp-residency.ts +49 -9
  418. package/src/engine/apis/lmstudio-residency.ts +5 -21
  419. package/src/engine/apis/lmstudio.ts +243 -0
  420. package/src/engine/apis/ollama-native.ts +24 -3
  421. package/src/engine/apis/openai-completions.ts +170 -91
  422. package/src/engine/apis/residency.ts +139 -3
  423. package/src/engine/apis/types.ts +16 -0
  424. package/src/engine/env-api-keys.ts +98 -0
  425. package/src/engine/gemma-channel-filter.ts +223 -0
  426. package/src/engine/instrumented-tui.ts +192 -0
  427. package/src/engine/messages.ts +14 -0
  428. package/src/engine/models.ts +42 -0
  429. package/src/engine/oauth.ts +16 -12
  430. package/src/engine/prompt-templates.ts +1 -0
  431. package/src/engine/provider-payload.ts +16 -59
  432. package/src/engine/strip-tokenizer-sentinels.ts +1 -1
  433. package/src/engine/truncate.ts +9 -0
  434. package/src/engine/tui.ts +17 -9
  435. package/src/engine/types.ts +3 -6
  436. package/src/engine/worker-runtime-capabilities.ts +5 -0
  437. package/src/engine/worker-runtime.ts +1 -1
  438. package/src/engine/worker-tools.ts +9 -4
  439. package/src/entry/boot-options.ts +50 -0
  440. package/src/entry/orchestrator.ts +288 -150
  441. package/src/interactive/application-controller.ts +89 -2
  442. package/src/interactive/chat-loop.ts +266 -41
  443. package/src/interactive/chat-panel.ts +173 -47
  444. package/src/interactive/chat-renderer.ts +262 -72
  445. package/src/interactive/clio-editor.ts +3 -8
  446. package/src/interactive/command-fallbacks.ts +2 -2
  447. package/src/interactive/context-overlay.ts +27 -1
  448. package/src/interactive/editor-submit.ts +228 -24
  449. package/src/interactive/export-html/ansi-to-html.ts +161 -0
  450. package/src/interactive/export-html/index.ts +51 -0
  451. package/src/interactive/export-html/template.ts +45 -0
  452. package/src/interactive/export-html/tool-renderer.ts +54 -0
  453. package/src/interactive/footer/dashboard.ts +4 -0
  454. package/src/interactive/footer/notifications.ts +1 -1
  455. package/src/interactive/footer/widgets.ts +20 -2
  456. package/src/interactive/footer-panel.ts +2 -2
  457. package/src/interactive/format-time.ts +14 -2
  458. package/src/interactive/interactive-application.ts +201 -17
  459. package/src/interactive/interactive-event-projection.ts +6 -1
  460. package/src/interactive/interactive-input-runtime.ts +50 -4
  461. package/src/interactive/interactive-presentation.ts +151 -19
  462. package/src/interactive/interactive-shell.ts +268 -14
  463. package/src/interactive/interactive-slash-runtime.ts +176 -114
  464. package/src/interactive/interactive-tickers.ts +38 -7
  465. package/src/interactive/keybinding-manager.ts +1 -1
  466. package/src/interactive/layout.ts +40 -3
  467. package/src/interactive/overlay-frame.ts +1 -1
  468. package/src/interactive/overlay-general-openers.ts +58 -1
  469. package/src/interactive/overlay-key-routing.ts +3 -0
  470. package/src/interactive/overlay-lifecycle.ts +13 -0
  471. package/src/interactive/overlay-permission-lifecycle.ts +2 -1
  472. package/src/interactive/overlay-session-lifecycle.ts +69 -12
  473. package/src/interactive/overlays/decisions.ts +300 -0
  474. package/src/interactive/overlays/help-reference.ts +15 -10
  475. package/src/interactive/overlays/model-selector.ts +34 -16
  476. package/src/interactive/overlays/session-selector.ts +18 -0
  477. package/src/interactive/overlays/settings.ts +105 -17
  478. package/src/interactive/overlays/skills-hub.ts +4 -4
  479. package/src/interactive/overlays/tree-selector.ts +41 -6
  480. package/src/interactive/render-trace.ts +499 -90
  481. package/src/interactive/renderers/compaction-summary.ts +2 -2
  482. package/src/interactive/renderers/diff.ts +115 -104
  483. package/src/interactive/renderers/mermaid.ts +53 -0
  484. package/src/interactive/renderers/tool-execution.ts +386 -133
  485. package/src/interactive/renderers/worker-entry.ts +20 -4
  486. package/src/interactive/session-switch-settlement.ts +10 -0
  487. package/src/interactive/slash-autocomplete.ts +6 -114
  488. package/src/interactive/slash-commands.ts +135 -47
  489. package/src/interactive/slash-spec.ts +9 -38
  490. package/src/interactive/status/controller.ts +5 -1
  491. package/src/interactive/stdout-backpressure.ts +99 -0
  492. package/src/interactive/stream-pacer.ts +530 -0
  493. package/src/interactive/stream-pacing-policy.ts +66 -0
  494. package/src/interactive/tasks-overlay.ts +368 -14
  495. package/src/interactive/terminal-lease.ts +485 -0
  496. package/src/interactive/theme/tokens.ts +1 -1
  497. package/src/interactive/turn-context.ts +4 -3
  498. package/src/interactive/turn-persistence.ts +30 -13
  499. package/src/interactive/turn-queues.ts +12 -0
  500. package/src/interactive/turn-recovery.ts +25 -8
  501. package/src/interactive/turn-runtime.ts +79 -12
  502. package/src/interactive/turn-state.ts +10 -0
  503. package/src/interactive/view/artifacts.ts +114 -4
  504. package/src/interactive/view/view-overlay.ts +3 -0
  505. package/src/interactive/welcome-dashboard.ts +17 -16
  506. package/src/interactive/worker-receipts.ts +52 -3
  507. package/src/interactive/worker-stream.ts +5 -1
  508. package/src/tools/agent-tools.ts +23 -3
  509. package/src/tools/artifact.ts +2 -2
  510. package/src/tools/ask-user.ts +23 -13
  511. package/src/tools/bash.ts +30 -2
  512. package/src/tools/bootstrap.ts +34 -431
  513. package/src/tools/builtin-tool-catalog.ts +265 -0
  514. package/src/tools/codewiki/code-nav-surface.ts +29 -0
  515. package/src/tools/codewiki/code-nav.ts +8 -22
  516. package/src/tools/codewiki/shared.ts +41 -38
  517. package/src/tools/context/docs-engine.ts +14 -3
  518. package/src/tools/context/index.ts +107 -28
  519. package/src/tools/context/surface.ts +19 -0
  520. package/src/tools/core-bootstrap.ts +168 -0
  521. package/src/tools/credential-present.ts +5 -5
  522. package/src/tools/dispatch-admission.ts +533 -0
  523. package/src/tools/dispatch-background.ts +54 -0
  524. package/src/tools/dispatch-event-text.ts +6 -0
  525. package/src/tools/dispatch-plan.ts +9 -4
  526. package/src/tools/dispatch-run-events.ts +238 -0
  527. package/src/tools/dispatch-runner.ts +2370 -0
  528. package/src/tools/dispatch-scout-admission.ts +295 -0
  529. package/src/tools/dispatch-types.ts +77 -0
  530. package/src/tools/dispatch.ts +67 -3161
  531. package/src/tools/find.ts +4 -2
  532. package/src/tools/grep.ts +2 -2
  533. package/src/tools/lazy-tool.ts +60 -0
  534. package/src/tools/ledger.ts +3 -3
  535. package/src/tools/monitor-surface.ts +36 -0
  536. package/src/tools/monitor.ts +2 -32
  537. package/src/tools/observers.ts +2 -2
  538. package/src/tools/registry.ts +39 -27
  539. package/src/tools/safe-exec.ts +2 -2
  540. package/src/tools/steer-surface.ts +17 -0
  541. package/src/tools/steer.ts +2 -13
  542. package/src/tools/tasks.ts +108 -11
  543. package/src/tools/truncate.ts +25 -184
  544. package/src/tools/verify/frontend.ts +3 -1
  545. package/src/tools/verify/index.ts +3 -38
  546. package/src/tools/verify/surface.ts +46 -0
  547. package/src/tools/web-fetch-surface.ts +23 -0
  548. package/src/tools/web-fetch.ts +2 -20
  549. package/src/tools/write.ts +7 -2
  550. package/src/worker/entry.ts +39 -2
  551. package/src/worker/spec-contract.ts +26 -5
  552. package/dist/chunk-7SS2CTV2.js +0 -61361
  553. package/dist/chunk-DKGKUHFA.js +0 -924
  554. package/dist/chunk-GEP36Y4X.js +0 -12796
  555. package/dist/chunk-XYWBQRDM.js +0 -137
  556. package/dist/clio-BZVGEUFJ.js +0 -58
  557. package/dist/configure-S7S6F6CL.js +0 -32
  558. package/docs/html/agents_blueprint.html +0 -936
  559. package/docs/html/alcf_blueprint.html +0 -324
  560. package/docs/html/architecture_blueprint.html +0 -850
  561. package/docs/html/commands_blueprint.html +0 -939
  562. package/docs/html/config_knobs_audit_blueprint.html +0 -178
  563. package/docs/html/configuration_blueprint.html +0 -1080
  564. package/docs/html/context_blueprint.html +0 -603
  565. package/docs/html/documentation_blueprint.html +0 -832
  566. package/docs/html/environment_blueprint.html +0 -404
  567. package/docs/html/eval_blueprint.html +0 -743
  568. package/docs/html/evals_internal_blueprint.html +0 -190
  569. package/docs/html/evolution_blueprint.html +0 -674
  570. package/docs/html/extensions_blueprint.html +0 -2065
  571. package/docs/html/fleet_dispatch_blueprint.html +0 -286
  572. package/docs/html/index.html +0 -919
  573. package/docs/html/lifecycle_blueprint.html +0 -723
  574. package/docs/html/memory_blueprint.html +0 -699
  575. package/docs/html/middleware_blueprint.html +0 -664
  576. package/docs/html/models_blueprint.html +0 -2366
  577. package/docs/html/observability_blueprint.html +0 -683
  578. package/docs/html/provider_adapter_blueprint.html +0 -245
  579. package/docs/html/safety_blueprint.html +0 -1386
  580. package/docs/html/shared.css +0 -571
  581. package/docs/html/shared.js +0 -143
  582. package/docs/html/skills_blueprint.html +0 -671
  583. package/docs/html/soak_blueprint.html +0 -182
  584. package/docs/html/tool_usage_blueprint.html +0 -350
  585. package/docs/html/tools_blueprint.html +0 -2249
  586. package/docs/html/trace_blueprint.html +0 -235
  587. package/docs/html/tui_design_blueprint.html +0 -374
  588. package/docs/html/validation_blueprint.html +0 -961
  589. package/docs/html/worker_dispatch_blueprint.html +0 -231
  590. package/src/core/release.ts +0 -2
  591. package/src/domains/providers/runtimes/common/lmstudio-logger.ts +0 -32
  592. package/src/domains/providers/runtimes/local-native/lmstudio-native.ts +0 -491
  593. package/src/engine/apis/lmstudio-native.ts +0 -1438
  594. package/src/engine/apis/thinking-replay.ts +0 -11
  595. package/src/tools/string-enum.ts +0 -15
@@ -0,0 +1,122 @@
1
+ ---
2
+ name: experiment-protocol
3
+ description: Use when running a performance study, numerical comparison, parameter sweep, kernel or solver benchmark, or any change justified by "faster" or "more accurate", and success criteria should be locked before results exist. Pre-registers thresholds, tolerances, environment pins, and verdict conditions into the repository validation contract before any measurement. Triggers on "benchmark", "compare implementations", "optimize", "tolerance", "reproduce results", "parameter sweep". Not for diagnosing a stalled bug; use scientific-debugging.
4
+ version: 0.1.2
5
+ license: Apache-2.0
6
+ allowed-tools:
7
+ - read
8
+ - write
9
+ - grep
10
+ - ls
11
+ - find
12
+ - git
13
+ - context
14
+ - code_nav
15
+ - bash
16
+ clio:
17
+ registry-id: iowarp/clio-coder
18
+ source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/experiment-protocol
19
+ audit: pass
20
+ provenance: designed
21
+ eval-status: smoke-checked
22
+ model-size: any
23
+ ---
24
+
25
+ # Experiment Protocol
26
+
27
+ Lock the success criteria before any result exists. Pre-registration that
28
+ happens after the first measurement is worthless: it can only ratify what was
29
+ already seen. Moving the threshold after seeing results is the exact failure
30
+ mode this protocol exists to prevent.
31
+
32
+ Anti-trigger: if the question is "why is this output wrong", that is a
33
+ diagnosis, not an experiment; use scientific-debugging.
34
+
35
+ ## Phase 0 - Pre-register
36
+
37
+ Before any measurement, write the protocol into the repository validation
38
+ contract: `.clio-coder/validation.yaml` (preferred) or `VALIDATION.md` at the repo
39
+ root. Creating this file raises Clio's repo-derived rigor to high immediately;
40
+ from the next turn onward, completion claims in this session must carry
41
+ validation evidence or state a limitation. That escalation is the point:
42
+ the contract arms the finish gate with the criteria you are about to commit to.
43
+
44
+ The pre-registration must contain:
45
+
46
+ - **Outcome**: one sentence, e.g. "the fused kernel reaches at least 1.8x the
47
+ baseline throughput on the pinned input at equal accuracy".
48
+ - **Thresholds**: minimum (below this is REFUTED), target, stretch.
49
+ - **Tolerance semantics per metric**: absolute for near-zero quantities,
50
+ relative elsewhere; a mixed scheme must say which applies where. "1e-6" with
51
+ no stated semantics is not a tolerance.
52
+ - **Environment pin**: compiler and flags, modules or package versions, node
53
+ class, scheduler context (partition, exclusivity).
54
+ - **Input identity**: paths plus checksums (`sha256sum`).
55
+ - **Verdict conditions**: what observation makes the result CONFIRMED, REFUTED,
56
+ or INCONCLUSIVE. Committed now, immutable after the first measurement.
57
+
58
+ ## Phase 1 - Baseline
59
+
60
+ Capture current behavior under the pinned environment before changing
61
+ anything. Store raw outputs and timings as artifacts; never edit them. A
62
+ speedup claim without a baseline captured under the same pin is a guess.
63
+
64
+ ## Phase 2 - Experiment
65
+
66
+ - One independent variable per run. A run that changes the algorithm and the
67
+ compiler flags answers no question.
68
+ - Size repetitions to the noise: shared nodes and networked filesystems need
69
+ more repetitions and a reported variance, not a single lucky run.
70
+ - Record scheduler identity (job id, node list) alongside every measurement.
71
+
72
+ ## Phase 3 - Analysis
73
+
74
+ Compare against the pre-registered thresholds only. If the protocol was
75
+ deviated from, log the deviation next to the result; do not silently absorb
76
+ it. Findings outside the registered outcome are marked exploratory and get
77
+ their own pre-registration if pursued.
78
+
79
+ ## Phase 4 - Iterate
80
+
81
+ Keep a dead-ends ledger in the contract file or beside it: one line per
82
+ rejected approach with the reason it was rejected. Read it before proposing
83
+ the next approach; re-proposing a ledger entry wastes a run. Stop when one of
84
+ the pre-registered stop conditions holds: target met, budget exhausted, or all
85
+ candidate strategies rejected.
86
+
87
+ ## Worked Example
88
+
89
+ Request: "make the halo exchange faster."
90
+
91
+ ```yaml
92
+ # .clio-coder/validation.yaml
93
+ experiment: halo-exchange-overlap
94
+ outcome: overlap communication with interior compute; >= 1.5x step throughput
95
+ thresholds: { minimum: 1.2x, target: 1.5x, stretch: 2.0x }
96
+ metrics:
97
+ step_time: { semantics: relative, tolerance: 5% run-to-run variance }
98
+ solution_l2: { semantics: absolute, tolerance: 1e-12 vs baseline }
99
+ environment: gcc 13.2 -O3, openmpi 4.1.6, 4x cpu-bind=cores, exclusive nodes
100
+ inputs: { mesh: data/mesh-256.h5, sha256: "<checksum>" }
101
+ verdicts:
102
+ confirmed: step_time speedup >= 1.2x AND solution_l2 within tolerance
103
+ refuted: speedup < 1.2x with variance < 5%, or accuracy loss
104
+ inconclusive: run-to-run variance > 5% (resize repetitions first)
105
+ dead_ends: []
106
+ ```
107
+
108
+ Baseline: 20 repetitions on exclusive nodes, mean and variance recorded with
109
+ job ids. Experiment: nonblocking exchange only, flags untouched. Result: 1.3x,
110
+ accuracy within 1e-12: CONFIRMED at minimum, target not met; ledger gains
111
+ "persistent requests: no gain over nonblocking here, latency-bound", and the
112
+ next variable (message aggregation) gets its own run.
113
+
114
+ ## Red Flags
115
+
116
+ - Any measurement taken before the contract file exists.
117
+ - A threshold, tolerance, or verdict condition edited after results appeared.
118
+ - Wall-clock numbers with no environment pin attached.
119
+ - A single-run victory claim on a shared machine.
120
+ - A REFUTED result reported as "promising"; refuted plus a ledger entry is a
121
+ successful experiment, say so plainly.
122
+ - Deleting or rewriting baseline artifacts.
@@ -0,0 +1,91 @@
1
+ # Evals - experiment-protocol
2
+
3
+ Baseline scenarios (run a subagent WITHOUT the skill to capture the gap, then
4
+ WITH the skill to confirm it closes). Rubric is pass/fail per bullet.
5
+
6
+ ## S1 - "make this kernel faster"
7
+
8
+ Setup: make the smoothing kernel in kernel.py faster.
9
+
10
+ Fixture:
11
+ ```bash
12
+ printf 'import time\n\ndef smooth(values, window):\n out = []\n for i in range(len(values)):\n lo = max(0, i - window)\n hi = min(len(values), i + window + 1)\n out.append(sum(values[lo:hi]) / (hi - lo))\n return out\n\nif __name__ == "__main__":\n data = [float(i %% 97) for i in range(200000)]\n t0 = time.perf_counter()\n smooth(data, 25)\n print("seconds:", round(time.perf_counter() - t0, 3))\n' > kernel.py
13
+ ```
14
+
15
+ Expected:
16
+
17
+ - Writes a pre-registration into `.clio-coder/validation.yaml` or `VALIDATION.md`
18
+ before running any benchmark or editing any code.
19
+ - The pre-registration contains thresholds (minimum/target/stretch) and
20
+ tolerance semantics stated per metric (absolute vs relative).
21
+ - Pins the environment (compiler, flags, versions, node/scheduler context)
22
+ and identifies inputs by path plus checksum.
23
+ - Captures a baseline under the same pin before changing anything.
24
+ - Changes one independent variable per experiment run.
25
+
26
+ ## S2 - results miss the target
27
+
28
+ Setup: the pre-registered target was 1.5x; the measured result is 1.1x, below
29
+ the registered minimum of 1.2x.
30
+
31
+ Expected:
32
+
33
+ - Reports REFUTED against the original pre-registered threshold.
34
+ - Does not restate the goal, lower the threshold, or reframe 1.1x as success.
35
+ - Adds a dead-ends ledger entry naming the rejected approach and the reason.
36
+ - Proposes the next approach only after reading the ledger.
37
+
38
+ ## S3 - noisy measurements on a shared machine
39
+
40
+ Setup: benchmark runs on a shared node; run-to-run variance exceeds the gap
41
+ being measured.
42
+
43
+ Expected:
44
+
45
+ - Sizes repetitions to the observed noise instead of reporting a single run.
46
+ - Reports variance alongside the mean, with scheduler identity recorded.
47
+ - Declares INCONCLUSIVE if variance swamps the effect, rather than picking
48
+ the best run.
49
+
50
+ ## S4 - anti-trigger: wrong output
51
+
52
+ Setup: user asks "why is this solver producing wrong values?"
53
+
54
+ Expected:
55
+
56
+ - Refers to scientific-debugging instead of starting a benchmark protocol.
57
+ - Does not write a validation contract for a diagnosis task.
58
+
59
+ ## Baseline failure modes to watch for (RED)
60
+
61
+ - Benchmarks first, defines success afterward from whatever the numbers show.
62
+ - "Faster" claimed from one run, no baseline, no environment pin.
63
+ - Threshold quietly adjusted after seeing results.
64
+ - Tolerance given as a bare number with no absolute/relative semantics.
65
+ - Rejected approaches vanish; the next session re-proposes them.
66
+ - Raw baseline artifacts edited or overwritten.
67
+
68
+ ## Observed gap closure
69
+
70
+ S1 run 2026-07-01, headless `clio-coder run` against a scratch git fixture (a pure
71
+ Python O(n^2) nearest-neighbor kernel with a single-shot bench script).
72
+ Prompt: "Make this kernel faster."
73
+
74
+ - RED (no skill): the agent edited the kernel immediately, reported a 7.3x
75
+ speedup from one timing run each way, wrote no contract, stated no
76
+ thresholds or tolerance semantics, and pinned nothing. Correctness was
77
+ checked ad hoc after the fact.
78
+ - GREEN (skill via `--skill` and `/skill <name>` invocation): the tool-call ledger
79
+ shows sha256 and environment capture, then `.clio-coder/validation.yaml` written
80
+ with min/target/stretch thresholds, per-metric tolerance semantics, and
81
+ verdict conditions, then the repetition-sized warm-up baseline, then
82
+ experiments, with the kernel edited only after a variant met the contract.
83
+ A float32 variant that beat the target but missed the accuracy tolerance
84
+ was rejected and logged in the dead-ends ledger. Writing the contract
85
+ raised repo rigor to high mid-session and the finish gate demanded
86
+ validation evidence before the turn settled. All five S1 bullets pass.
87
+
88
+ ## Smoke record (2026-08-13)
89
+
90
+ One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
91
+ (30B local, llamacpp on mini), full-auto sandbox. PASS. Pre-registration written before touching the seeded kernel; judge 5/5.
@@ -0,0 +1,119 @@
1
+ ---
2
+ name: scientific-debugging
3
+ description: Use when debugging has stalled after the first obvious fix, when a failure spans multiple systems, when someone is about to try random changes, or when a scientific or HPC code produces wrong numbers, NaNs, nondeterministic results, or an unexplained performance regression. Forces falsifiable hypotheses across distinct fault classes with evidence-cited verdicts before any fix. Triggers on "why is this failing", "wrong results", "flaky", "nondeterministic", "diagnose", "root cause". Not for designing benchmarks or pre-registered experiments; use experiment-protocol.
4
+ version: 0.1.2
5
+ license: Apache-2.0
6
+ allowed-tools:
7
+ - read
8
+ - grep
9
+ - ls
10
+ - find
11
+ - git
12
+ - context
13
+ - code_nav
14
+ - bash
15
+ clio:
16
+ registry-id: iowarp/clio-coder
17
+ source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/scientific-debugging
18
+ audit: pass
19
+ provenance: designed
20
+ eval-status: smoke-checked
21
+ model-size: any
22
+ ---
23
+
24
+ # Scientific Debugging
25
+
26
+ Debug by falsification, not by trying fixes. A fix attempted before a confirmed
27
+ diagnosis is an experiment run without a hypothesis; when it "works" you have
28
+ learned nothing, and when it does not you have contaminated the evidence.
29
+
30
+ Anti-trigger: if the failure is a typo, a missing import, or an error message
31
+ that names its own cause, fix it directly and skip this workflow. The loop
32
+ below is for failures that survived the first obvious fix.
33
+
34
+ ## The Loop
35
+
36
+ 1. **Goal.** One sentence stating the observable "fixed" state. "The regression
37
+ test matches the reference output within the documented tolerance on two
38
+ consecutive runs" is a goal; "make it work" is not.
39
+ 2. **Hypothesize.** Write at least three hypotheses. Each must name its fault
40
+ class and carry a falsification test: "this is WRONG if <observation>".
41
+ A hypothesis you cannot state a falsification test for is a hunch; refine it
42
+ until it is testable.
43
+ 3. **Rank.** Order by test cost times prior likelihood. Run the cheapest
44
+ decisive test first, not the most interesting one.
45
+ 4. **Test.** One variable per test. Preserve the raw failing output somewhere
46
+ untouched before you change anything.
47
+ 5. **Verdict.** Record CONFIRMED, REFUTED, or INCONCLUSIVE per hypothesis, each
48
+ citing the command and output that decided it. A verdict without a citable
49
+ observation is a guess.
50
+ 6. **Iterate.** Refuted everything? Generate new hypotheses from what the tests
51
+ revealed. Confirmed one? Only now edit code.
52
+
53
+ ## Fault Classes
54
+
55
+ Hypotheses must span at least two distinct classes. Anchoring on a single class
56
+ is the failure mode this rule exists to break: the debugger who is sure it is
57
+ "a race" stops seeing the stale module load in front of them.
58
+
59
+ | Class | Typical suspects |
60
+ |---|---|
61
+ | numerics | accumulation order, mixed precision, tolerance misuse, fastmath |
62
+ | data | format or layout drift, HDF5/NetCDF/Zarr metadata, units, corruption |
63
+ | concurrency | races, MPI collective mismatch, nondeterministic reduction order |
64
+ | environment | modules, compiler flags, library versions, scheduler context |
65
+ | resources | memory pressure, filesystem quirks, quota, node differences |
66
+ | regression | a recent change; bisect the history instead of staring at code |
67
+
68
+ ## Tiers
69
+
70
+ **Quick diagnosis** (default): the loop above, state held in conversation,
71
+ time-boxed at fifteen minutes of investigation. If the box expires without a
72
+ CONFIRMED verdict, escalate. Say that you are escalating; do not silently keep
73
+ poking.
74
+
75
+ **Structured investigation**: write an investigation file (for example
76
+ `INVESTIGATION.md` or `.clio-coder/investigation-<slug>.md` via bash heredoc since
77
+ this skill does not edit code) containing the goal, baseline measurements of
78
+ the failing behavior, and one experiment per hypothesis with its verdict
79
+ condition committed *before* the experiment runs. Update verdicts as evidence
80
+ arrives. The file is the state; the conversation is commentary.
81
+
82
+ ## Evidence Rule
83
+
84
+ The fix commit should cite the confirming observation, e.g. "confirmed by:
85
+ `OMP_NUM_THREADS=1` reproduces bitwise-identical results, run log above".
86
+ High-rigor repos will demand validation evidence at completion anyway; produce
87
+ it proactively rather than being re-prompted for it.
88
+
89
+ ## Worked Example
90
+
91
+ Report: "after the refactor, results differ from reference by 1e-4."
92
+
93
+ - Goal: `pytest tests/test_advection.py` passes against the pinned reference
94
+ within its stated rtol on a clean checkout plus the refactor commit.
95
+ - H1 (regression): the refactor changed the loop order and with it the
96
+ floating-point accumulation order. WRONG if the pre-refactor commit shows the
97
+ same 1e-4 drift. Test: `git stash && pytest ...` (cost: 1 min).
98
+ - H2 (numerics): the comparison uses an absolute tolerance where values near
99
+ zero need a relative one. WRONG if the drift is uniform across magnitudes.
100
+ Test: print elementwise error vs magnitude (cost: 5 min).
101
+ - H3 (environment): a different BLAS or compiler flag set is active in the new
102
+ environment. WRONG if `pip freeze`/module list matches the reference
103
+ environment pin. Test: diff environments (cost: 2 min).
104
+ - Order: H1, H3, H2. H1 verdict: REFUTED, pre-refactor commit is clean, output
105
+ cited. H3: REFUTED, environments identical. H2: CONFIRMED, error is constant
106
+ 1e-4 at all magnitudes, so near-zero elements fail the absolute check.
107
+ - Only now edit: fix the tolerance semantics, cite H2's observation in the
108
+ commit message.
109
+
110
+ ## Red Flags
111
+
112
+ - Editing code before any hypothesis has a CONFIRMED verdict.
113
+ - All hypotheses drawn from one fault class.
114
+ - A test that changes two variables at once.
115
+ - "It seems better now" presented as a verdict.
116
+ - Retrying a flaky test until it passes instead of making it deterministic.
117
+ - The fifteen-minute box expiring without an explicit escalation.
118
+ - Feeling certain: when a hypothesis feels obviously true, state its
119
+ falsification test anyway before touching the code.
@@ -0,0 +1,138 @@
1
+ # Evals - scientific-debugging
2
+
3
+ Baseline scenarios (run a subagent WITHOUT the skill to capture the gap, then
4
+ WITH the skill to confirm it closes). Rubric is pass/fail per bullet.
5
+
6
+ ## S1 - stalled numerical bug
7
+
8
+ Setup: a small numerical project with a reference-comparison test. The
9
+ workspace already contains the fixture; run `python -m unittest -q` for the
10
+ reference check. Prompt: "after a refactor our results differ from the
11
+ reference by about 1e-4 and the obvious fix did not help; diagnose it."
12
+
13
+ Fixture:
14
+
15
+ ```bash
16
+ mkdir -p tests
17
+ cat > diffusion.py <<'PY'
18
+ import math
19
+
20
+
21
+ def weighted_mean(values, weights):
22
+ weighted = [value * weight for value, weight in zip(values, weights)]
23
+ return math.fsum(weighted) / math.fsum(weights)
24
+ PY
25
+ cat > tests/test_diffusion.py <<'PY'
26
+ import math
27
+ import unittest
28
+
29
+ from diffusion import weighted_mean
30
+
31
+
32
+ class DiffusionReferenceTest(unittest.TestCase):
33
+ def test_weighted_mean_matches_reference(self):
34
+ values = [1.0e16, 6.0e-4, -1.0e16, 1.0e-4, 2.0e-4, -3.0e-4]
35
+ weights = [1.0, 1.0, 1.0, 1.0, 1.0, 1.0]
36
+ expected = math.fsum(value * weight for value, weight in zip(values, weights)) / math.fsum(weights)
37
+ observed = weighted_mean(values, weights)
38
+ self.assertLess(abs(observed - expected), 1.0e-12)
39
+
40
+
41
+ if __name__ == "__main__":
42
+ unittest.main()
43
+ PY
44
+ git init -q
45
+ git config user.name "Clio Eval"
46
+ git config user.email "clio-eval@example.invalid"
47
+ git add diffusion.py tests/test_diffusion.py
48
+ git commit -q -m "add stable weighted mean reference"
49
+ cat > diffusion.py <<'PY'
50
+ def weighted_mean(values, weights):
51
+ total = 0.0
52
+ for value, weight in zip(values, weights):
53
+ total += value * weight
54
+ return total / sum(weights)
55
+ PY
56
+ ```
57
+
58
+ Expected:
59
+
60
+ - States a one-sentence goal naming the observable fixed state before
61
+ investigating.
62
+ - Writes at least three hypotheses, each with an explicit "this is WRONG if"
63
+ falsification test.
64
+ - Hypotheses span at least two distinct fault classes, including numerics and
65
+ regression.
66
+ - Orders tests cheapest-first and runs one variable per test.
67
+ - Records a CONFIRMED/REFUTED/INCONCLUSIVE verdict per hypothesis, each citing
68
+ a command and its output.
69
+ - Edits no code before a hypothesis is CONFIRMED.
70
+
71
+ ## S2 - flaky parallel test
72
+
73
+ Setup: a test suite where one MPI/threaded test fails intermittently. Prompt:
74
+ "this test is flaky, sometimes it passes; figure out why."
75
+
76
+ Expected:
77
+
78
+ - Includes a concurrency-class hypothesis (race, collective mismatch, or
79
+ reduction order).
80
+ - Attempts a deterministic reproduction (pin threads, fix seeds, force
81
+ ordering) rather than rerunning until green.
82
+ - Preserves the raw failing output before changing anything.
83
+ - Does not present a lucky pass as a verdict.
84
+
85
+ ## S3 - escalation to structured tier
86
+
87
+ Setup: quick-tier investigation is not converging; all initial hypotheses come
88
+ back REFUTED and fifteen minutes of investigation have elapsed.
89
+
90
+ Expected:
91
+
92
+ - Explicitly announces escalation to the structured tier instead of silently
93
+ continuing.
94
+ - Writes an investigation file containing the goal, baseline measurements of
95
+ the failing behavior, and one experiment per hypothesis.
96
+ - Each experiment's verdict condition is committed before the experiment runs.
97
+ - New hypotheses are generated from what the refuted tests revealed.
98
+
99
+ ## S4 - anti-trigger: trivial failure
100
+
101
+ Setup: the failure is an obvious typo or missing import whose error message
102
+ names its own cause. Prompt: "why is this failing?"
103
+
104
+ Expected:
105
+
106
+ - Fixes it directly or says a quick fix is appropriate.
107
+ - Does not run the hypothesis ceremony for a self-explanatory failure.
108
+
109
+ ## Baseline failure modes to watch for (RED)
110
+
111
+ - Tries a fix immediately with no stated hypothesis.
112
+ - Single hypothesis, no falsification test, anchored on one fault class.
113
+ - Bundles the fix with the diagnosis in one edit.
114
+ - Verdicts asserted from intuition with no cited observation.
115
+ - Flaky test "resolved" by rerunning until it passes.
116
+ - Investigation drifts past the time box with no escalation and no file.
117
+
118
+ ## Observed gap closure
119
+
120
+ S1 run 2026-07-01, headless `clio-coder run` against a scratch git fixture (an
121
+ order-sensitive summation whose refactor replaced `math.fsum` with a plain
122
+ accumulation loop; regression check fails by 1.474e-4).
123
+
124
+ - RED (no skill): the agent found the correct root cause but with no stated
125
+ goal, no enumerated hypotheses, no falsification tests, and no verdicts; it
126
+ read the diff, asserted the cause, and benchmarked alternative summations ad
127
+ hoc. On a harder bug that first guess would have been unfalsified anchoring.
128
+ - GREEN (skill via `--skill` and `/skill <name>` invocation): one-sentence goal,
129
+ three hypotheses (numerics, data, environment; the refactor regression
130
+ folded into H1) each with a WRONG-if test, explicit cheapest-first ranking,
131
+ CONFIRMED/REFUTED verdicts citing command output, untested H3 marked N/A,
132
+ fix applied only after the CONFIRMED verdict, and a commit message citing
133
+ the confirming observation. All six S1 bullets pass.
134
+
135
+ ## Smoke record (2026-08-13)
136
+
137
+ One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
138
+ (30B local, llamacpp on mini), full-auto sandbox. PASS. Judge 6/6 on the seeded numerical fixture; cleanest research run.
@@ -0,0 +1,138 @@
1
+ ---
2
+ name: scientific-modernization
3
+ description: Use when modernizing, porting, rewriting, packaging, accelerating, or replacing established scientific software, especially across languages, build systems, CPU/GPU backends, or maintained forks. Establishes an external scientific oracle, preserves compatibility, delivers in independently validated stages, and settles upstream ownership and long-term stewardship before calling the work complete. Triggers on "modernize this scientific code", "rewrite in Rust", "port to GPU", "replace this research tool", "migrate the build", "maintained fork", and "scientific parity". Not for an isolated benchmark; use experiment-protocol. Not for diagnosing wrong results; use scientific-debugging.
4
+ version: 0.2.0
5
+ license: Apache-2.0
6
+ allowed-tools:
7
+ - read
8
+ - write
9
+ - grep
10
+ - ls
11
+ - find
12
+ - git
13
+ - context
14
+ - code_nav
15
+ - bash
16
+ clio:
17
+ registry-id: iowarp/clio-coder
18
+ source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/scientific-modernization
19
+ audit: pass
20
+ provenance: designed
21
+ eval-status: scenarios-recorded
22
+ model-size: large
23
+ ---
24
+
25
+ # Scientific Modernization
26
+
27
+ Modernization preserves scientific behavior and stewardship; it is not source
28
+ translation. Faster code, a clean build, and passing self-authored unit tests
29
+ do not establish scientific equivalence. Work the stages below in order; each
30
+ has an explicit exit condition.
31
+
32
+ ## Stage 1 — Decide whether this work should exist
33
+
34
+ Identify the upstream project: active maintainers, license, release cadence,
35
+ supported users, contribution path. Prefer improving the original project when
36
+ coordination is viable.
37
+
38
+ Record the decision in writing before any implementation:
39
+
40
+ - path chosen: upstream contribution, maintained successor, or explicitly
41
+ scoped fork;
42
+ - the accountable maintainer or organization by name;
43
+ - the compatibility surface users already rely on;
44
+ - release, deprecation, and migration intent.
45
+
46
+ Do not start a rewrite whose only durable plan is "the community can maintain
47
+ it later." Exit: the four bullets above are written down and the user has seen
48
+ them.
49
+
50
+ ## Stage 2 — Establish an independent scientific oracle
51
+
52
+ Write the acceptance contract in `.clio-coder/validation.yaml` (preferred) or
53
+ `VALIDATION.md` before changing behavior. Pick at least one oracle that does
54
+ not depend on the new implementation agreeing with itself:
55
+
56
+ - exact outputs from a trusted reference implementation;
57
+ - parity against the established tool over a representative corpus;
58
+ - known statistical behavior with pre-registered bounds;
59
+ - simulated data whose correct answer is fixed in advance;
60
+ - conserved quantities, analytical solutions, or domain invariants.
61
+
62
+ For every output, define: units, shapes, ordering, missing-value behavior,
63
+ determinism, absolute and relative tolerances, allowed platform variation.
64
+ Include adversarial and historically troublesome inputs.
65
+
66
+ If no credible oracle exists, stop and report that limitation. Implementation
67
+ velocity cannot repair an undefined truth condition. Keep performance
68
+ acceptance separate: pre-register speed claims through `experiment-protocol`.
69
+ Exit: the contract file exists and names its oracle(s).
70
+
71
+ ## Stage 3 — Freeze the compatibility envelope
72
+
73
+ Inventory observable behavior before migrating anything:
74
+
75
+ - CLI and API contracts, file formats, schemas, defaults, error behavior;
76
+ - packaging, fresh-install, upgrade, and uninstall paths;
77
+ - supported platforms, compilers, runtimes, accelerators, schedulers;
78
+ - resource scaling, reproducibility controls, provenance;
79
+ - undocumented conventions captured by downstream tests and real workflows.
80
+
81
+ Capture reference outputs and install evidence from released artifacts, not
82
+ only the source checkout: mature tools carry user trust and conventions a
83
+ line-by-line translation misses. Exit: reference outputs and the envelope
84
+ inventory are stored as artifacts.
85
+
86
+ ## Stage 4 — Deliver in independently valid stages
87
+
88
+ Split the work into the smallest stages that can each be checked against the
89
+ oracle. Every stage gets a before/after boundary, an acceptance command, a
90
+ retained artifact, and a rollback point. Prefer vertical slices that produce a
91
+ usable result over a big-bang rewrite.
92
+
93
+ Per stage:
94
+
95
+ 1. Capture the reference result on the pinned corpus and environment.
96
+ 2. Make one bounded change.
97
+ 3. Run compatibility and scientific-oracle checks.
98
+ 4. Preserve raw outputs, discrepancies, and provenance.
99
+ 5. Resolve or explicitly classify every mismatch before expanding scope.
100
+
101
+ An agent's confidence is not evidence. If a reviewer cannot reconstruct the
102
+ comparison from retained artifacts, the stage is unverified; say so. Exit per
103
+ stage: oracle checks pass or every mismatch is classified in writing.
104
+
105
+ ## Stage 5 — Budget explicitly for the last mile
106
+
107
+ Initial implementation is faster than convergence. Reserve work for: edge
108
+ cases, subtle numerical differences, nondeterminism, fresh environments,
109
+ large inputs, interrupted runs, packaging metadata, documentation, user
110
+ migration. Re-run the full oracle matrix after any optimization: correctness
111
+ proven before an optimization is not inherited by the optimized code.
112
+
113
+ Use `scientific-debugging` when a mismatch needs causal diagnosis. Never widen
114
+ tolerances or drop inconvenient corpus entries to obtain parity.
115
+
116
+ ## Stage 6 — Ship only with evidence and stewardship
117
+
118
+ A completion claim must include all of:
119
+
120
+ - the oracle and compatibility matrix that passed, plus retained artifacts;
121
+ - unresolved differences and their user-visible consequences;
122
+ - fresh-install and documented-workflow results;
123
+ - performance results, if claimed, under their registered protocol;
124
+ - upstream status, or the named owner and maintenance plan;
125
+ - migration, rollback, release, and deprecation instructions.
126
+
127
+ If scientific validity or durable ownership is unresolved, report the work as
128
+ a prototype. Do not call it a replacement, successor, or production release.
129
+
130
+ ## Red Flags
131
+
132
+ - A rewrite begins before maintainers or downstream users are consulted.
133
+ - Tests derived only from the new implementation.
134
+ - "The outputs look close" replacing declared tolerance semantics.
135
+ - Performance wins reported before scientific parity.
136
+ - One final comparison substituting for staged validation.
137
+ - A passing happy path hiding fresh-install, scale, or edge-case failures.
138
+ - A fork shipping without an accountable owner and maintenance horizon.