@iowarp/clio-coder 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (312) hide show
  1. package/.claude-plugin/marketplace.json +4 -4
  2. package/CHANGELOG.md +35 -0
  3. package/CONTRIBUTING.md +9 -0
  4. package/README.md +12 -9
  5. package/ROADMAP.md +30 -5
  6. package/dist/{acp-KS7ARGV6.js → acp-F6PGOFZ4.js} +2 -2
  7. package/dist/{agents-YOGJ2YEJ.js → agents-ZHP4R6EX.js} +22 -21
  8. package/dist/assets/codewiki.json +1 -1
  9. package/dist/{auth-CEDCO6O6.js → auth-N75U2H3Z.js} +4 -4
  10. package/dist/{chunk-D3KWFOUT.js → chunk-23UIXH6Z.js} +3 -3
  11. package/dist/{chunk-ZR5YA6ZF.js → chunk-36XVK5P5.js} +5 -5
  12. package/dist/{chunk-AVSR2HKI.js → chunk-3DZ4LPLS.js} +6 -6
  13. package/dist/{chunk-LYJJ6546.js → chunk-3GSPCWFX.js} +12 -12
  14. package/dist/{chunk-CRHPCPZP.js → chunk-4KC4CHKJ.js} +2 -2
  15. package/dist/{chunk-4VDHLCF5.js → chunk-4WJBWKMI.js} +2 -2
  16. package/dist/{chunk-KBUTSJSR.js → chunk-4XBYHXGK.js} +2 -2
  17. package/dist/{chunk-6W75D37F.js → chunk-53P775GE.js} +6 -3
  18. package/dist/{chunk-XZWSP67B.js → chunk-5APYAXCB.js} +2 -2
  19. package/dist/{chunk-D65E25JG.js → chunk-67HPYNTJ.js} +2 -2
  20. package/dist/{chunk-5JQR5B4D.js → chunk-6KV4KCQZ.js} +4 -4
  21. package/dist/{chunk-V4XJWDHR.js → chunk-6UKFRTXI.js} +2 -1
  22. package/dist/{chunk-NDPF2N2L.js → chunk-7DW7CJW2.js} +2 -2
  23. package/dist/{chunk-5KQEUIWM.js → chunk-7LUEQODV.js} +4 -4
  24. package/dist/{chunk-LNOXCGN2.js → chunk-7UFDIOBT.js} +3 -3
  25. package/dist/{chunk-DXSETN72.js → chunk-7WUGNN43.js} +2 -2
  26. package/dist/{chunk-V35KZDUZ.js → chunk-AA3GRZWS.js} +2 -2
  27. package/dist/{chunk-X4CEFOSG.js → chunk-ADO3OQIC.js} +4 -4
  28. package/dist/{chunk-UX2L5CUD.js → chunk-AI36V3ON.js} +3 -3
  29. package/dist/{chunk-B7GY6QER.js → chunk-B26OUMJW.js} +2 -2
  30. package/dist/{chunk-LAR7DX5E.js → chunk-B5TASTCK.js} +153 -62
  31. package/dist/{chunk-TGWZXJ7L.js → chunk-BEMQK5Q7.js} +92 -84
  32. package/dist/{chunk-RNDCROSG.js → chunk-CNXU7TR2.js} +3 -3
  33. package/dist/{chunk-3JLTRPN7.js → chunk-CVMUXT64.js} +35 -66
  34. package/dist/{chunk-733ZBT5P.js → chunk-E74MJF6G.js} +4 -4
  35. package/dist/{chunk-T2MDZES3.js → chunk-EBMTBE76.js} +3 -3
  36. package/dist/{chunk-OZ76TLGU.js → chunk-EEITU7S6.js} +2 -2
  37. package/dist/{chunk-GRHOXDKB.js → chunk-ELDJWJKM.js} +97 -14
  38. package/dist/{chunk-GCQFN7YI.js → chunk-EMYEQLWO.js} +2 -2
  39. package/dist/{chunk-LTWH2ACR.js → chunk-EQPFEDFV.js} +222 -127
  40. package/dist/{chunk-TDGYTWM6.js → chunk-F4I3H56H.js} +2 -2
  41. package/dist/chunk-F6KCJO3U.js +523 -0
  42. package/dist/{chunk-FGSQHUCK.js → chunk-F6VKWTTY.js} +5 -5
  43. package/dist/{chunk-UJD5MPP6.js → chunk-FLNXQQ5B.js} +2 -2
  44. package/dist/{chunk-EX57VVWW.js → chunk-FOSRJ2VZ.js} +4 -4
  45. package/dist/{chunk-ZARZ4POR.js → chunk-G5ZDQQE3.js} +3 -3
  46. package/dist/{chunk-HQVN7G4F.js → chunk-G62JBPXN.js} +7 -7
  47. package/dist/{chunk-ZJPORZTC.js → chunk-GIAI7A2K.js} +2 -2
  48. package/dist/chunk-GPTR6OTU.js +2551 -0
  49. package/dist/{chunk-EZD2BXGP.js → chunk-GUFQPTQK.js} +2 -2
  50. package/dist/{chunk-OPD7GL6F.js → chunk-H6FQVNZC.js} +8 -7
  51. package/dist/chunk-H7REJ7L5.js +80 -0
  52. package/dist/{chunk-EYTLV3W3.js → chunk-HTWTORF7.js} +2 -2
  53. package/dist/{chunk-DF6PI6GN.js → chunk-I6UKQPGS.js} +3 -3
  54. package/dist/{chunk-ISHUS7HC.js → chunk-JRW236JT.js} +3 -3
  55. package/dist/{chunk-DE2LU267.js → chunk-KKAJGT3B.js} +3 -2
  56. package/dist/{chunk-OV62D6KO.js → chunk-KQMYL5GR.js} +3 -3
  57. package/dist/{chunk-QDUUOZ3S.js → chunk-KWMFC7FN.js} +3 -3
  58. package/dist/{chunk-R7FGRZNF.js → chunk-LXYHCCIZ.js} +9 -9
  59. package/dist/{chunk-NFSK2VNU.js → chunk-LZMTVVN4.js} +2 -2
  60. package/dist/{chunk-62UK7AGW.js → chunk-M5PMRZSR.js} +397 -44
  61. package/dist/chunk-MMZMM6SW.js +131 -0
  62. package/dist/{chunk-FFYTC2KQ.js → chunk-MNABNEBX.js} +2 -2
  63. package/dist/{chunk-XMDBTYCH.js → chunk-N3CWOGVF.js} +3 -3
  64. package/dist/{chunk-VYRQHORQ.js → chunk-NMQSO6Z6.js} +4 -3
  65. package/dist/{chunk-R2UUNSNQ.js → chunk-O3ZAF7UY.js} +18 -5
  66. package/dist/{chunk-4SYHGVDE.js → chunk-OQNCLZI4.js} +4 -4
  67. package/dist/{chunk-EG4ARXPC.js → chunk-OX6QTA4N.js} +3 -3
  68. package/dist/{chunk-P4GNZPXO.js → chunk-OYUS3UZP.js} +2 -2
  69. package/dist/{chunk-OQ2EOAHA.js → chunk-PF36KOGI.js} +291 -112
  70. package/dist/{chunk-55RMDDFM.js → chunk-PGQG7ZMW.js} +2 -2
  71. package/dist/{chunk-CBAQPTDU.js → chunk-PIU6BXXW.js} +2 -2
  72. package/dist/{chunk-EKV4UBCM.js → chunk-PZBQUJ2F.js} +2 -2
  73. package/dist/{chunk-7RET77VB.js → chunk-Q2HKY32Y.js} +2 -2
  74. package/dist/{chunk-U6YTUVOX.js → chunk-Q4RMNWMZ.js} +4 -4
  75. package/dist/chunk-Q6FRR3AQ.js +48 -0
  76. package/dist/{chunk-6RJMYYIE.js → chunk-QJSIAJBY.js} +3 -3
  77. package/dist/{chunk-DOUISWEE.js → chunk-QOJ4ERMU.js} +5 -5
  78. package/dist/{chunk-IFPDAD7R.js → chunk-RWSRKXLT.js} +3 -3
  79. package/dist/{chunk-T33IYTZM.js → chunk-SAASVMIY.js} +3 -3
  80. package/dist/{chunk-2DTIWSAW.js → chunk-SEQDH6JC.js} +3 -3
  81. package/dist/{chunk-BRCOSE7O.js → chunk-SZUO5BSE.js} +2 -2
  82. package/dist/{chunk-HOEDA42N.js → chunk-T7HZ7KJF.js} +2 -2
  83. package/dist/{chunk-OJXOA6YU.js → chunk-TL6LNERH.js} +2 -2
  84. package/dist/{chunk-UCINIRHW.js → chunk-TPCZXWLT.js} +3 -3
  85. package/dist/{chunk-ZUKUCZYZ.js → chunk-TQ2KTH4A.js} +2 -2
  86. package/dist/{chunk-4WOOTKFD.js → chunk-U3FETNQB.js} +341 -20
  87. package/dist/{chunk-5B3RTYAK.js → chunk-UGDSB4AN.js} +2 -2
  88. package/dist/{chunk-VKHLUZNO.js → chunk-UZ7YBL43.js} +3 -3
  89. package/dist/{chunk-QNUQ7K7D.js → chunk-VKMSEO7Y.js} +4 -4
  90. package/dist/{chunk-IN7DGBVS.js → chunk-VLX5VZ35.js} +5 -5
  91. package/dist/{chunk-SJGS3GDI.js → chunk-WAGBMMNX.js} +5 -5
  92. package/dist/{chunk-BFOSV5EZ.js → chunk-WX2YCH7F.js} +2 -2
  93. package/dist/{chunk-3LT34CAM.js → chunk-XBIGUILU.js} +32 -17
  94. package/dist/{chunk-XDHUDE5K.js → chunk-XKYBFRWR.js} +5 -5
  95. package/dist/{chunk-U5QU5ZOD.js → chunk-XYPWFSU5.js} +3 -3
  96. package/dist/{chunk-YBUECLAF.js → chunk-YEJQDODA.js} +2 -2
  97. package/dist/{chunk-YQ6XEFVK.js → chunk-Z74OGONW.js} +39 -7
  98. package/dist/{chunk-I4ELN5BX.js → chunk-ZTNOEJRI.js} +2 -2
  99. package/dist/cli/index.js +26 -26
  100. package/dist/{clio-K2PBVCEM.js → clio-4IKFE25F.js} +2 -2
  101. package/dist/{clio-context-tools-GR4OHFSO.js → clio-context-tools-OHJXYX4R.js} +25 -23
  102. package/dist/{code-nav-4X3OI6DK.js → code-nav-XPM3MY5I.js} +7 -7
  103. package/dist/{config-Q5J7UBAB.js → config-343YS25T.js} +43 -41
  104. package/dist/{config-graph-P3555574.js → config-graph-WI263KDX.js} +43 -41
  105. package/dist/{configure-KIJ32ZO6.js → configure-GZDG4HER.js} +21 -21
  106. package/dist/{context-LFBQOSXT.js → context-4OFO7N7Y.js} +12 -12
  107. package/dist/{context-WBH3KSVR.js → context-JLK2RLOG.js} +42 -40
  108. package/dist/{context-NNLBES2Y.js → context-O6DIWCF6.js} +23 -21
  109. package/dist/{context-clear-YOHMQ6SQ.js → context-clear-MR7KOGQI.js} +42 -40
  110. package/dist/{context-working-set-MRAEVDNV.js → context-working-set-YZAD3KWO.js} +14 -13
  111. package/dist/{data-tool-JJPBGMQJ.js → data-tool-EKD7DFHG.js} +7 -7
  112. package/dist/{detail-RHJOTGN4.js → detail-YZPHFEVJ.js} +43 -41
  113. package/dist/{dispatch-runner-NC7PYNIO.js → dispatch-runner-VKVXVTWJ.js} +44 -42
  114. package/dist/{doctor-S5KZ2GTN.js → doctor-53AVQXAO.js} +17 -16
  115. package/dist/{doctor-deep-2EJQSHIZ.js → doctor-deep-BLRB2NCG.js} +5 -5
  116. package/dist/{eval-3D6M36N7.js → eval-P3BQSLQE.js} +21 -20
  117. package/dist/{evidence-XNVPPLVK.js → evidence-6LLI6SZH.js} +44 -42
  118. package/dist/{evidence-6IYN52ET.js → evidence-HTN3JLUO.js} +42 -40
  119. package/dist/{evidence-JVECPI3F.js → evidence-NJRE3R5H.js} +42 -40
  120. package/dist/{evolve-TLPUDJDM.js → evolve-WWZHJVJP.js} +42 -40
  121. package/dist/{fleet-WKVBF5SZ.js → fleet-GGZ4BVJE.js} +64 -62
  122. package/dist/{fleet-6RJWIOEO.js → fleet-QG3IIWWM.js} +44 -42
  123. package/dist/{fleet-commands-L6WFT7KZ.js → fleet-commands-J2ZIKISQ.js} +7 -7
  124. package/dist/{fleet-decisions-BTQMWU5A.js → fleet-decisions-Y2Q7375Q.js} +8 -8
  125. package/dist/{fleet-graph-553DZ67Q.js → fleet-graph-TCBCCCBP.js} +11 -11
  126. package/dist/{fleet-inspect-236FZPRB.js → fleet-inspect-KOLKBC56.js} +44 -42
  127. package/dist/{fleet-preflight-NZQHABWO.js → fleet-preflight-44UNCC4Y.js} +27 -25
  128. package/dist/{fleet-validate-7NLAE26R.js → fleet-validate-K2FXRZVA.js} +13 -13
  129. package/dist/{fleet-verify-E3RO6A5W.js → fleet-verify-AS444WDW.js} +42 -40
  130. package/dist/{fleet-view-KMZQY7YJ.js → fleet-view-X4JU4J6O.js} +43 -41
  131. package/dist/gui/ops-worker.js +9 -9
  132. package/dist/gui/reads-worker.js +9 -8
  133. package/dist/{init-DV5B6HQG.js → init-VSOWFHVL.js} +54 -52
  134. package/dist/{interactive-OXGYZYAS.js → interactive-JRCBENLZ.js} +1733 -577
  135. package/dist/{interop-J7UOBOSF.js → interop-JR6BCY7R.js} +10 -10
  136. package/dist/{inventory-OEGIL6PY.js → inventory-3JUGLVJI.js} +43 -41
  137. package/dist/{library-ROZAMUIL.js → library-7BOH3Y7N.js} +15 -15
  138. package/dist/{library-B3UPGY4V.js → library-7PFONXUI.js} +9 -9
  139. package/dist/{library-MXQA5DLT.js → library-TUYZMFIA.js} +7 -7
  140. package/dist/{library-import-2LTWOVBD.js → library-import-TJVXKLDK.js} +10 -10
  141. package/dist/{library-inventory-LJ3XHLWO.js → library-inventory-MD3AYTP4.js} +9 -9
  142. package/dist/{library-validation-DARHHUNW.js → library-validation-ZDLITEA3.js} +7 -7
  143. package/dist/{mcp-2GSPYQ54.js → mcp-EFCWUPUR.js} +3 -3
  144. package/dist/{memory-J3STUUIK.js → memory-Q2YVOKK5.js} +42 -40
  145. package/dist/{models-Y3GMFJ2V.js → models-M7BOCW7F.js} +17 -16
  146. package/dist/{monitor-LHNH577J.js → monitor-2HD3MC67.js} +47 -45
  147. package/dist/{orchestrator-DTAPXULA.js → orchestrator-OUW3ZN6Y.js} +1894 -226
  148. package/dist/{panes-A743ONZ5.js → panes-TTPPSYCL.js} +3 -3
  149. package/dist/{preload-OJBMAIXF.js → preload-A47C2NUF.js} +42 -40
  150. package/dist/{providers-CMETPZNS.js → providers-T43W7EL4.js} +4 -4
  151. package/dist/{resources-FY6XYFN2.js → resources-TP6X2V6W.js} +20 -11
  152. package/dist/{run-H64MEVTA.js → run-LDSBPMCS.js} +61 -59
  153. package/dist/{share-CAFMMQLM.js → share-URHKKMHR.js} +9 -9
  154. package/dist/{skills-ERKQUC7L.js → skills-E477MOOO.js} +13 -11
  155. package/dist/{skills-eval-CQ23MYQK.js → skills-eval-AIBR2L7H.js} +583 -93
  156. package/dist/{skills-inventory-TVUYJDVN.js → skills-inventory-HLMKY3R5.js} +13 -11
  157. package/dist/{slash-commands-ROG3KIFJ.js → slash-commands-LIIZKA5Y.js} +28 -26
  158. package/dist/{startup-background-KDWCQMWT.js → startup-background-JR2KDK6R.js} +43 -41
  159. package/dist/{steer-KKYXJSH4.js → steer-D7RTAPCM.js} +3 -3
  160. package/dist/{system-ZDW44ZGX.js → system-X2P5N4IY.js} +14 -13
  161. package/dist/{targets-AJWA6E7Z.js → targets-VFFAKUFM.js} +25 -24
  162. package/dist/{tasks-6L2JJRC5.js → tasks-ERAWZA22.js} +6 -6
  163. package/dist/{terminal-lease-YFUSDPAD.js → terminal-lease-V2N2BIKQ.js} +2 -2
  164. package/dist/{trace-LA5MVR4Y.js → trace-MA5BT7XY.js} +3 -3
  165. package/dist/{usage-XSILDHKF.js → usage-WYFY2L6M.js} +50 -48
  166. package/dist/{verifiers-ECA5GIHL.js → verifiers-TYIW2XKF.js} +7 -7
  167. package/dist/{verify-R6QQCCEB.js → verify-242UFYTP.js} +6 -6
  168. package/dist/{web-fetch-PVZNMBOB.js → web-fetch-LRAGYU6Y.js} +3 -3
  169. package/dist/{wiki-generate-RP22WP5K.js → wiki-generate-OFKOX2T6.js} +53 -51
  170. package/dist/worker/entry.js +31 -29
  171. package/docs/architecture/architecture.md +1 -0
  172. package/docs/architecture/context-engine.md +4 -2
  173. package/docs/architecture/observability.md +6 -4
  174. package/docs/architecture/prompt-envelope-and-tools.md +1 -1
  175. package/docs/architecture/session-lifecycle.md +1 -1
  176. package/docs/architecture/tui-design.md +12 -9
  177. package/docs/gui/parity/02-slash-and-surfaces.md +1 -1
  178. package/docs/guide/commands-and-modes.md +91 -5
  179. package/docs/guide/configuration-and-targets.md +2 -2
  180. package/docs/guide/context-continuity.md +49 -0
  181. package/docs/guide/proactive-memory.md +6 -3
  182. package/library/registry.yaml +292 -32
  183. package/library/skills/README.md +27 -2
  184. package/library/skills/context/context-handoff/SKILL.md +10 -4
  185. package/library/skills/meta/clio-coder-dev/SKILL.md +92 -88
  186. package/library/skills/meta/clio-coder-dev/evals.md +51 -46
  187. package/library/skills/meta/clio-coder-dev/plugin.json +2 -2
  188. package/library/skills/meta/clio-coder-dev/references/change-map.md +49 -0
  189. package/library/skills/meta/clio-coder-test/SKILL.md +89 -142
  190. package/library/skills/meta/clio-coder-test/evals.md +58 -62
  191. package/library/skills/meta/clio-coder-test/plugin.json +2 -2
  192. package/library/skills/meta/clio-coder-test/references/harness.md +3 -3
  193. package/library/skills/meta/clio-coder-test/references/lifecycle-validation.md +30 -0
  194. package/library/skills/meta/clio-coder-test/references/test-map.md +60 -89
  195. package/library/skills/registry.yaml +5 -5
  196. package/library/skills/skill-marketplace.json +12 -14
  197. package/package.json +1 -1
  198. package/src/cli/skills-eval.ts +1064 -54
  199. package/src/cli/usage.ts +2 -2
  200. package/src/core/clio-repo.ts +3 -0
  201. package/src/core/tool-names.ts +1 -0
  202. package/src/domains/context/budget/inspection.ts +9 -0
  203. package/src/domains/context/budget/live-view.ts +418 -0
  204. package/src/domains/context/budget/pressure.ts +393 -0
  205. package/src/domains/context/budget/request-fit.ts +15 -0
  206. package/src/domains/evidence/build.ts +24 -0
  207. package/src/domains/gateway/mcp/client.ts +1 -1
  208. package/src/domains/memory/commit-state.ts +160 -0
  209. package/src/domains/memory/operations.ts +18 -10
  210. package/src/domains/memory/prompt-cache.ts +95 -0
  211. package/src/domains/memory/prompt-section.ts +50 -7
  212. package/src/domains/memory/relevance.ts +93 -0
  213. package/src/domains/memory/restoration.ts +78 -0
  214. package/src/domains/memory/store.ts +39 -1
  215. package/src/domains/middleware/memory-intervention.ts +116 -24
  216. package/src/domains/observability/background-memory-usage.ts +1 -1
  217. package/src/domains/observability/cost.ts +3 -3
  218. package/src/domains/observability/extension.ts +1 -1
  219. package/src/domains/observability/metrics.ts +1 -1
  220. package/src/domains/prompts/compiler.ts +1 -1
  221. package/src/domains/prompts/extension.ts +38 -1
  222. package/src/domains/quota/anthropic-max-provider.ts +134 -0
  223. package/src/domains/quota/anthropic-usage.ts +220 -0
  224. package/src/domains/quota/antigravity-provider.ts +339 -0
  225. package/src/domains/quota/cache.ts +87 -0
  226. package/src/domains/quota/claude-code-provider.ts +169 -0
  227. package/src/domains/quota/codex-provider.ts +237 -0
  228. package/src/domains/quota/presentation.ts +240 -0
  229. package/src/domains/quota/registry.ts +23 -0
  230. package/src/domains/quota/service.ts +93 -0
  231. package/src/domains/quota/summary-feed.ts +86 -0
  232. package/src/domains/quota/types.ts +84 -0
  233. package/src/domains/resources/index.ts +11 -0
  234. package/src/domains/resources/skills/catalog-view.ts +571 -0
  235. package/src/domains/resources/skills/lexical-match.ts +136 -0
  236. package/src/domains/resources/skills/loader.ts +38 -0
  237. package/src/domains/resources/skills/promotion.ts +1 -55
  238. package/src/domains/resources/skills/provenance-pin.ts +50 -20
  239. package/src/domains/safety/action-classifier.ts +1 -0
  240. package/src/domains/session/compaction/branch-summary.ts +4 -1
  241. package/src/domains/session/compaction/compact.ts +18 -2
  242. package/src/domains/session/compaction/cut-point.ts +20 -1
  243. package/src/domains/session/compaction/tokens.ts +48 -3
  244. package/src/domains/session/context-accounting.ts +8 -1
  245. package/src/domains/session/continuity/carry.ts +59 -0
  246. package/src/domains/session/continuity/contract.ts +592 -0
  247. package/src/domains/session/continuity/evidence.ts +290 -0
  248. package/src/domains/session/continuity/fold.ts +1075 -0
  249. package/src/domains/session/continuity/note.ts +79 -0
  250. package/src/domains/session/continuity/operator-request.ts +104 -0
  251. package/src/domains/session/continuity/persistence.ts +408 -0
  252. package/src/domains/session/continuity/ports.ts +231 -0
  253. package/src/domains/session/continuity/projection.ts +538 -0
  254. package/src/domains/session/continuity/validate.ts +352 -0
  255. package/src/domains/session/entries.ts +50 -3
  256. package/src/domains/session/index.ts +48 -0
  257. package/src/domains/session/migrations/index.ts +7 -3
  258. package/src/domains/session/tree/fork.ts +15 -1
  259. package/src/domains/session/usage.ts +2 -2
  260. package/src/engine/acp/commands.ts +1 -1
  261. package/src/engine/agent.ts +133 -9
  262. package/src/engine/session.ts +13 -6
  263. package/src/entry/orchestrator.ts +116 -33
  264. package/src/interactive/chat-loop-messages.ts +2 -2
  265. package/src/interactive/chat-loop.ts +219 -63
  266. package/src/interactive/chat-renderer.ts +99 -8
  267. package/src/interactive/context-overlay.ts +24 -3
  268. package/src/interactive/continuity-controller.ts +526 -0
  269. package/src/interactive/dispatch-board.ts +21 -3
  270. package/src/interactive/footer/dashboard.ts +25 -1
  271. package/src/interactive/footer/key-hints.ts +2 -2
  272. package/src/interactive/footer/pages.ts +70 -27
  273. package/src/interactive/footer/widgets.ts +32 -7
  274. package/src/interactive/footer-panel.ts +1 -1
  275. package/src/interactive/interactive-application.ts +7 -2
  276. package/src/interactive/interactive-input-runtime.ts +2 -2
  277. package/src/interactive/interactive-presentation.ts +15 -0
  278. package/src/interactive/interactive-slash-runtime.ts +45 -8
  279. package/src/interactive/interactive-tickers.ts +7 -1
  280. package/src/interactive/model-session-replay.ts +130 -3
  281. package/src/interactive/output-reserve.ts +35 -0
  282. package/src/interactive/overlay-general-openers.ts +12 -8
  283. package/src/interactive/overlay-key-routing.ts +6 -9
  284. package/src/interactive/overlay-lifecycle.ts +10 -6
  285. package/src/interactive/overlay-session-lifecycle.ts +21 -14
  286. package/src/interactive/quota-view.ts +229 -0
  287. package/src/interactive/session-last-turn.ts +1 -1
  288. package/src/interactive/session-transcript.ts +2 -1
  289. package/src/interactive/session-usage-reseed.ts +2 -2
  290. package/src/interactive/side-question.ts +2 -2
  291. package/src/interactive/slash-commands.ts +28 -9
  292. package/src/interactive/turn-context.ts +686 -63
  293. package/src/interactive/turn-middleware.ts +39 -6
  294. package/src/interactive/turn-persistence.ts +15 -2
  295. package/src/interactive/turn-prewarm.ts +1 -1
  296. package/src/interactive/turn-runtime.ts +48 -5
  297. package/src/interactive/{cost-overlay.ts → usage-overlay.ts} +121 -33
  298. package/src/interactive/welcome-dashboard.ts +26 -3
  299. package/src/tools/agent-tools.ts +3 -1
  300. package/src/tools/bootstrap.ts +6 -0
  301. package/src/tools/builtin-tool-catalog.ts +9 -0
  302. package/src/tools/context/index.ts +148 -113
  303. package/src/tools/context/surface.ts +11 -7
  304. package/src/tools/core-bootstrap.ts +4 -0
  305. package/src/tools/observation.ts +8 -2
  306. package/src/tools/policy.ts +3 -0
  307. package/src/tools/self-compact.ts +31 -0
  308. package/src/tools/surface.ts +1 -0
  309. package/src/tools/tasks.ts +1 -1
  310. package/dist/chunk-CTFPFW3H.js +0 -44
  311. package/dist/chunk-EOJPDNUP.js +0 -1325
  312. package/dist/chunk-I4Y4WKZR.js +0 -386
@@ -1,9 +1,9 @@
1
1
  import { spawn } from "node:child_process";
2
2
  import { createHash, randomBytes } from "node:crypto";
3
- import { existsSync } from "node:fs";
4
- import { cp, mkdir, mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
3
+ import { existsSync, realpathSync } from "node:fs";
4
+ import { cp, mkdir, mkdtemp, readdir, readFile, readlink, realpath, rm, stat, writeFile } from "node:fs/promises";
5
5
  import { tmpdir } from "node:os";
6
- import { join, resolve } from "node:path";
6
+ import { isAbsolute, join, relative as relativePath, resolve } from "node:path";
7
7
  import { performance } from "node:perf_hooks";
8
8
  import { combineBashOutput, runBashCommand } from "../core/bash-exec.js";
9
9
  import { HEADLESS_PERMISSION_DENIED_MARKER } from "../core/headless-permission.js";
@@ -20,6 +20,7 @@ import {
20
20
  } from "../domains/eval/index.js";
21
21
  import { buildEvalEvidence } from "../domains/evidence/index.js";
22
22
  import {
23
+ checkSkillDrift,
23
24
  discoverMarketplaceSkills,
24
25
  loadSkills,
25
26
  parseSkillEvals,
@@ -44,6 +45,22 @@ import { formatColumns, printError } from "./shared.js";
44
45
  * command lists, receipt-backed token/cost totals when headless main-agent
45
46
  * receipts are present, and per-bullet detail in a `skill-eval.json` sidecar
46
47
  * registered in the bundle's `overview.json` files list.
48
+ *
49
+ * Two things the sidecar records beyond the rubric result.
50
+ *
51
+ * `subject` names the exact artifact the run measured: the resolved base
52
+ * directory, how it was resolved, both hashes of the SKILL.md, and whether that
53
+ * content still matches whatever hash was recorded for it. A bundle that names
54
+ * only a skill and an evals.md cannot be compared against another bundle for
55
+ * the same skill, because nothing in either says whether the skill changed.
56
+ *
57
+ * `attribution` answers "did the skill change anything?" by scoring the
58
+ * baseline arm too. That arm was always executed and always thrown away: the
59
+ * treatment judge is told the baseline exists only for context. A second,
60
+ * isolated judge scores it against the same bullets, and the two verdict sets
61
+ * are paired per bullet. It is advisory in the strict sense: `pass`,
62
+ * `exitCode` and `failureClass` keep their treatment-only meaning, and no
63
+ * attribution value reaches the exit code. `--no-attribution` skips it.
47
64
  */
48
65
 
49
66
  const DEFAULT_RUN_TIMEOUT_MS = 600_000;
@@ -82,6 +99,14 @@ export interface SkillsEvalOptions {
82
99
  * fetch the open web is measuring something else as well.
83
100
  */
84
101
  allowNetwork: boolean;
102
+ /**
103
+ * Skip the baseline judge, and with it the paired attribution.
104
+ *
105
+ * Attribution costs one extra judge run per scenario. Turning it off is a
106
+ * cost decision, not a result: a skipped comparison is recorded
107
+ * `not-attempted` with the reason, never `no-change`.
108
+ */
109
+ noAttribution: boolean;
85
110
  scenario?: string;
86
111
  target?: string;
87
112
  timeoutSeconds?: number;
@@ -104,6 +129,18 @@ export interface ScoredBullet {
104
129
  reason: string;
105
130
  }
106
131
 
132
+ /**
133
+ * One skill activation this run actually performed, as the activation contract
134
+ * reported it. `hash` is the sha256 of the bytes the child read.
135
+ *
136
+ * @internal Exported for contract tests.
137
+ */
138
+ export interface ObservedActivation {
139
+ name: string;
140
+ hash: string;
141
+ path: string;
142
+ }
143
+
107
144
  /** Exported for contracts tests. */
108
145
  export interface CapturedRun {
109
146
  sessionId: string | null;
@@ -113,6 +150,8 @@ export interface CapturedRun {
113
150
  timedOut: boolean;
114
151
  wallTimeMs: number;
115
152
  stderr: string;
153
+ /** Skill loads this run completed successfully; empty when none did. */
154
+ activations: ObservedActivation[];
116
155
  }
117
156
 
118
157
  interface ScenarioUsage {
@@ -121,12 +160,141 @@ interface ScenarioUsage {
121
160
  harness: EvalHarnessMetrics;
122
161
  }
123
162
 
163
+ /**
164
+ * The exact artifact a run measured.
165
+ *
166
+ * A bundle used to name only its `evals.md` by hash, so two runs of the same
167
+ * skill could not be told apart when the SKILL.md between them had changed.
168
+ * Everything here is already computed elsewhere in the run: the loader hashes
169
+ * the file, `resolveSkillBaseDir` resolves which copy activation would pick,
170
+ * and `checkSkillDrift` compares it against whatever recorded hash speaks for
171
+ * it. Discarding all of it was the whole gap.
172
+ *
173
+ * @internal Exported for contract tests.
174
+ */
175
+ export interface SkillEvalSubject {
176
+ name: string;
177
+ baseDir: string;
178
+ /** How the copy was resolved: `path`, `<source>/<scope>`, or `catalog`. */
179
+ origin: string;
180
+ /** sha256 of the SKILL.md exactly as read. */
181
+ sha256: string;
182
+ /** sha256 with install-lifecycle provenance stripped; this is what pins compare. */
183
+ normalizedHash: string;
184
+ /**
185
+ * Digest over every file under `baseDir`, not just SKILL.md.
186
+ *
187
+ * A skill is a directory: the body can point at `references/` and scripts
188
+ * the run will read. Naming the artifact by its SKILL.md hash alone would
189
+ * claim an identity for content that hash does not cover, and would miss an
190
+ * edit to a reference file entirely. Null when the tree could not be read.
191
+ */
192
+ treeSha256: string | null;
193
+ /**
194
+ * Whether the artifact could be snapshotted for the run. False when the body
195
+ * carries package references, which resolve against an owning package root
196
+ * above the skill directory: a copy of the directory alone would deliver
197
+ * different instructions from the live path.
198
+ */
199
+ pinnable: boolean;
200
+ evalsPath: string;
201
+ evalsSha256: string;
202
+ /** Null when nothing on this machine recorded a hash for the skill. */
203
+ drift: { verdict: "match" | "mismatch"; authority: string; expected: string } | null;
204
+ }
205
+
206
+ /**
207
+ * Whether the artifact on disk still matched {@link SkillEvalSubject} across
208
+ * one scenario's treatment arm.
209
+ *
210
+ * `verified` states exactly one thing: the tree digest taken immediately before
211
+ * the treatment arm and again immediately after both equalled the run-level
212
+ * subject. It is not proof of what the child activated. The treatment child
213
+ * loads the skill from the live source directory by path, so an edit landing
214
+ * between resolution and the arm, or between two scenarios, would otherwise be
215
+ * measured under the previous artifact's recorded identity. This check closes
216
+ * that window; it cannot see a change reverted entirely inside the arm, and the
217
+ * sidecar says so rather than implying activation was witnessed.
218
+ */
219
+ export type SubjectVerification =
220
+ | "verified"
221
+ | "mismatch"
222
+ | "unreadable"
223
+ | "not-pinned"
224
+ | "not-activated"
225
+ | "activation-mismatch"
226
+ | "not-checked";
227
+
228
+ /**
229
+ * Whether the skill changed the outcome for one bullet, relative to the same
230
+ * bullet in the baseline arm.
231
+ *
232
+ * `unmeasured` is not a comparison that came out even. It is the absence of a
233
+ * comparison, and it is returned whenever either arm failed to produce a
234
+ * verdict for that bullet.
235
+ *
236
+ * @internal Exported for contract tests.
237
+ */
238
+ export type BulletAttribution = "helped" | "regressed" | "no-change-pass" | "no-change-fail" | "unmeasured";
239
+
240
+ /**
241
+ * The scenario-level rollup.
242
+ *
243
+ * `not-attempted` and `unmeasured` are deliberately separate. The first means
244
+ * the comparison was never run, the second means it was impossible. Collapsing
245
+ * either into `no-change` would report "the skill made no difference" about a
246
+ * measurement that never happened.
247
+ *
248
+ * @internal Exported for contract tests.
249
+ */
250
+ export type ScenarioAttributionVerdict =
251
+ | "helped"
252
+ | "regressed"
253
+ | "mixed"
254
+ | "no-change"
255
+ | "unmeasured"
256
+ | "not-attempted";
257
+
258
+ /** @internal Exported for contract tests. */
259
+ export interface AttributedBullet {
260
+ index: number;
261
+ text: string;
262
+ baseline: BulletVerdict;
263
+ treatment: BulletVerdict;
264
+ attribution: BulletAttribution;
265
+ /** Why this bullet is unmeasured; absent once both arms supplied evidence. */
266
+ reason?: string;
267
+ }
268
+
269
+ /** @internal Exported for contract tests. */
270
+ export interface ScenarioAttribution {
271
+ verdict: ScenarioAttributionVerdict;
272
+ /** Why the verdict is `unmeasured` or `not-attempted`; null once a real comparison ran. */
273
+ reason: string | null;
274
+ bullets: AttributedBullet[];
275
+ counts: {
276
+ helped: number;
277
+ regressed: number;
278
+ noChangePass: number;
279
+ noChangeFail: number;
280
+ unmeasured: number;
281
+ };
282
+ }
283
+
124
284
  interface ScenarioOutcome {
125
285
  scenario: SkillEvalScenario;
126
286
  bullets: ScoredBullet[];
127
287
  baseline: CapturedRun | null;
128
288
  treatment: CapturedRun | null;
129
289
  judge: CapturedRun | null;
290
+ /** The isolated judge run that scored the baseline arm; null when none ran. */
291
+ baselineJudge: CapturedRun | null;
292
+ /** Whether the artifact on disk still matched the subject across this scenario. */
293
+ subjectVerification: SubjectVerification;
294
+ /** Tree digest observed after this scenario's treatment arm; null when unread. */
295
+ observedTreeSha256: string | null;
296
+ /** Advisory paired comparison. Never a gate, never folded into `pass`. */
297
+ attribution: ScenarioAttribution;
130
298
  /** Seed workspace cloned for both arms; removed after the run, kept as a record. */
131
299
  workspace: string;
132
300
  wallTimeMs: number;
@@ -134,6 +302,211 @@ interface ScenarioOutcome {
134
302
  usage: ScenarioUsage;
135
303
  }
136
304
 
305
+ /**
306
+ * Did the skill change this bullet's outcome?
307
+ *
308
+ * Either arm failing to produce a verdict makes the pair unmeasurable, and that
309
+ * check comes first: a bullet the baseline judge never scored carries no
310
+ * information about the treatment, whatever the treatment did.
311
+ *
312
+ * @internal Exported for contract tests.
313
+ */
314
+ export function deriveBulletAttribution(baseline: BulletVerdict, treatment: BulletVerdict): BulletAttribution {
315
+ if (baseline === "unmeasured" || baseline === "error") return "unmeasured";
316
+ if (treatment === "unmeasured" || treatment === "error") return "unmeasured";
317
+ if (baseline === "fail" && treatment === "pass") return "helped";
318
+ if (baseline === "pass" && treatment === "fail") return "regressed";
319
+ return baseline === "pass" ? "no-change-pass" : "no-change-fail";
320
+ }
321
+
322
+ /**
323
+ * Roll per-bullet attributions into one scenario verdict.
324
+ *
325
+ * Order matters and is deliberate. An unmeasured bullet poisons the scenario,
326
+ * because a rollup that ignored it would report a comparison over a subset
327
+ * while naming the whole. `mixed` outranks both single-direction verdicts, so a
328
+ * skill that fixed three bullets and broke one is never reported as simply
329
+ * having helped.
330
+ *
331
+ * @internal Exported for contract tests.
332
+ */
333
+ export function deriveScenarioAttributionVerdict(
334
+ bullets: ReadonlyArray<AttributedBullet>,
335
+ ): Exclude<ScenarioAttributionVerdict, "not-attempted"> {
336
+ if (bullets.length === 0) return "unmeasured";
337
+ if (bullets.some((bullet) => bullet.attribution === "unmeasured")) return "unmeasured";
338
+ const helped = bullets.some((bullet) => bullet.attribution === "helped");
339
+ const regressed = bullets.some((bullet) => bullet.attribution === "regressed");
340
+ if (helped && regressed) return "mixed";
341
+ if (regressed) return "regressed";
342
+ if (helped) return "helped";
343
+ return "no-change";
344
+ }
345
+
346
+ function attributionCounts(bullets: ReadonlyArray<AttributedBullet>): ScenarioAttribution["counts"] {
347
+ const counts = { helped: 0, regressed: 0, noChangePass: 0, noChangeFail: 0, unmeasured: 0 };
348
+ for (const bullet of bullets) {
349
+ if (bullet.attribution === "helped") counts.helped += 1;
350
+ else if (bullet.attribution === "regressed") counts.regressed += 1;
351
+ else if (bullet.attribution === "no-change-pass") counts.noChangePass += 1;
352
+ else if (bullet.attribution === "no-change-fail") counts.noChangeFail += 1;
353
+ else counts.unmeasured += 1;
354
+ }
355
+ return counts;
356
+ }
357
+
358
+ /**
359
+ * Pair two arms' strict verdicts into a per-bullet comparison.
360
+ *
361
+ * Both sides have to have supplied unambiguous evidence for the same bullet.
362
+ * When either did not, the bullet is `unmeasured` and carries the reason it was
363
+ * refused, so a reader can tell "the judge omitted it" from "the judge answered
364
+ * with a string" from "the judge contradicted itself". The reason used to be
365
+ * dropped during pairing, which left an unmeasured scenario with nothing saying
366
+ * why.
367
+ *
368
+ * @internal Exported for contract tests.
369
+ */
370
+ export function attributionFromStrict(
371
+ scenario: SkillEvalScenario,
372
+ baseline: StrictJudgeVerdicts,
373
+ treatment: StrictJudgeVerdicts,
374
+ ): ScenarioAttribution {
375
+ const verdictOf = (pass: boolean | undefined): BulletVerdict =>
376
+ pass === undefined ? "unmeasured" : pass ? "pass" : "fail";
377
+ const refusal = (side: string, strict: StrictJudgeVerdicts, index: number): string | null => {
378
+ if (strict.verdicts.has(index)) return null;
379
+ if (strict.absent !== null) return `${side} judge: ${strict.absent}`;
380
+ return `${side} judge: ${strict.rejected.get(index) ?? `no verdict for bullet ${index}`}`;
381
+ };
382
+ const bullets: AttributedBullet[] = scenario.expected.map((text, i) => {
383
+ const index = i + 1;
384
+ const baselinePass = baseline.verdicts.get(index);
385
+ const treatmentPass = treatment.verdicts.get(index);
386
+ const baselineVerdict = verdictOf(baselinePass);
387
+ const treatmentVerdict = verdictOf(treatmentPass);
388
+ const reason = refusal("baseline", baseline, index) ?? refusal("treatment", treatment, index);
389
+ return {
390
+ index,
391
+ text,
392
+ baseline: baselineVerdict,
393
+ treatment: treatmentVerdict,
394
+ attribution: deriveBulletAttribution(baselineVerdict, treatmentVerdict),
395
+ ...(reason !== null ? { reason } : {}),
396
+ };
397
+ });
398
+ const verdict = deriveScenarioAttributionVerdict(bullets);
399
+ const firstReason = bullets.find((bullet) => bullet.reason !== undefined)?.reason ?? null;
400
+ return {
401
+ verdict,
402
+ reason: verdict === "unmeasured" ? firstReason : null,
403
+ bullets,
404
+ counts: attributionCounts(bullets),
405
+ };
406
+ }
407
+
408
+ /**
409
+ * A digest of the whole skill directory, path-sensitive and order-independent.
410
+ *
411
+ * Sorted relative paths are hashed alongside their contents, so adding, moving
412
+ * or removing a reference file changes the digest as surely as editing one
413
+ * does. Null on any read failure rather than a partial digest, because a digest
414
+ * over some of a tree would compare unequal for a reason that is not a change.
415
+ *
416
+ * @internal Exported for contract tests.
417
+ */
418
+ export async function skillTreeDigest(baseDir: string): Promise<string | null> {
419
+ interface Entry {
420
+ relative: string;
421
+ kind: "file" | "symlink";
422
+ target?: string;
423
+ }
424
+ const root = resolve(baseDir);
425
+ const entries: Entry[] = [];
426
+ const seenDirs = new Set<string>();
427
+ const walk = async (dir: string): Promise<void> => {
428
+ // A directory symlink pointing at an ancestor would otherwise walk forever.
429
+ const real = await realpath(dir);
430
+ if (seenDirs.has(real)) throw new Error("cycle");
431
+ seenDirs.add(real);
432
+ const found = await readdir(dir, { withFileTypes: true });
433
+ for (const entry of found.sort((a, b) => a.name.localeCompare(b.name))) {
434
+ const full = join(dir, entry.name);
435
+ if (entry.isSymbolicLink()) {
436
+ // A link is part of the artifact's identity: retargeting one changes
437
+ // what the skill reads without changing any regular file. The link's
438
+ // own target string is hashed, and a link leaving the tree makes the
439
+ // artifact unverifiable rather than silently half-covered.
440
+ const target = await readlink(full);
441
+ const resolved = resolve(dir, target);
442
+ const relative = relativePath(root, resolved);
443
+ if (relative.startsWith("..") || isAbsolute(relative)) throw new Error("escaping symlink");
444
+ entries.push({ relative: full.slice(root.length), kind: "symlink", target });
445
+ continue;
446
+ }
447
+ if (entry.isDirectory()) await walk(full);
448
+ else if (entry.isFile()) entries.push({ relative: full.slice(root.length), kind: "file" });
449
+ }
450
+ };
451
+ try {
452
+ await walk(root);
453
+ const hash = createHash("sha256");
454
+ for (const entry of entries.sort((a, b) => a.relative.localeCompare(b.relative))) {
455
+ hash.update(entry.relative, "utf8");
456
+ hash.update("\0");
457
+ hash.update(entry.kind, "utf8");
458
+ hash.update("\0");
459
+ if (entry.kind === "symlink") hash.update(entry.target ?? "", "utf8");
460
+ else hash.update(await readFile(join(root, entry.relative)));
461
+ hash.update("\0");
462
+ }
463
+ return hash.digest("hex");
464
+ } catch {
465
+ return null;
466
+ }
467
+ }
468
+
469
+ /**
470
+ * Whether the artifact can be snapshotted for the run.
471
+ *
472
+ * The treatment arm runs against a private copy rather than the live source
473
+ * directory. Hashing the source before and after the arm catches an edit that
474
+ * persists and nothing else: an A to B to A change inside the arm leaves both
475
+ * observations equal while the child read B. A copy under a per-run temp root
476
+ * is not reachable from the source path, so an ordinary edit there cannot
477
+ * affect the run.
478
+ *
479
+ * What that is, precisely: an isolated snapshot, not an immutable pin. The
480
+ * copy sits in a writable temp directory, and the arm runs at full-auto, so a
481
+ * model that chose to edit the copy, read it, and restore it would not be
482
+ * detected. The harness has no mechanism that would make the copy read-only to
483
+ * its own child, and building one is out of scope here. These results are
484
+ * evidence about a cooperative model, which is the same caveat
485
+ * `materializeSkillEvalWorkspaces` already records about arm isolation.
486
+ *
487
+ * Returns false for the one case a copy would misrepresent: a body carrying
488
+ * package references. Those resolve against the owning package root, which
489
+ * sits above the skill directory and is not part of the copy, so a copied
490
+ * skill would deliver different instructions from the ones the live path
491
+ * delivers. A tree whose digest cannot be taken is a separate failure and is
492
+ * reported by `skillTreeDigest` returning null.
493
+ *
494
+ * @internal Exported for contract tests.
495
+ */
496
+ export function artifactIsPinnable(skillBody: string): boolean {
497
+ return !skillBody.includes("${pluginRoot}") && !skillBody.includes("${component:");
498
+ }
499
+
500
+ /** No comparison was attempted. Records why, and never reads as `no-change`. */
501
+ function notAttemptedAttribution(reason: string): ScenarioAttribution {
502
+ return { verdict: "not-attempted", reason, bullets: [], counts: attributionCounts([]) };
503
+ }
504
+
505
+ /** A comparison was impossible. Distinct from one that was never started. */
506
+ function unmeasurableAttribution(reason: string): ScenarioAttribution {
507
+ return { verdict: "unmeasured", reason, bullets: [], counts: attributionCounts([]) };
508
+ }
509
+
137
510
  async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptions): Promise<number> {
138
511
  const resolved = resolveSkillBaseDir(nameOrPath, process.cwd());
139
512
  if (resolved.baseDir === null) {
@@ -164,6 +537,42 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
164
537
  for (const diagnostic of parsed.diagnostics) {
165
538
  process.stderr.write(`clio-coder eval skill: ${diagnostic}\n`);
166
539
  }
540
+ // Coherence check on the subject itself. `skill` was loaded, and its hashes
541
+ // taken, before the tree digest; an edit landing in that gap would produce a
542
+ // subject naming one SKILL.md revision and a tree containing another. The
543
+ // SKILL.md is re-read here and compared, so the recorded identity describes
544
+ // one state of the directory or admits it could not.
545
+ const subjectTree = await skillTreeDigest(resolved.baseDir);
546
+ const reread = await readFile(join(resolved.baseDir, "SKILL.md"), "utf8").catch(() => null);
547
+ const coherent = reread !== null && createHash("sha256").update(reread, "utf8").digest("hex") === skill.hash;
548
+ const pinnable = reread !== null && artifactIsPinnable(reread);
549
+ const driftReport = checkSkillDrift(skill, process.cwd());
550
+ const subject: SkillEvalSubject = {
551
+ name: skill.name,
552
+ baseDir: resolved.baseDir,
553
+ origin: resolved.origin ?? "unknown",
554
+ sha256: skill.hash,
555
+ normalizedHash: skill.normalizedHash,
556
+ treeSha256: coherent ? subjectTree : null,
557
+ pinnable,
558
+ evalsPath,
559
+ evalsSha256: createHash("sha256").update(evalsRaw, "utf8").digest("hex"),
560
+ drift:
561
+ driftReport === null
562
+ ? null
563
+ : { verdict: driftReport.verdict, authority: driftReport.authority, expected: driftReport.expected },
564
+ };
565
+ // Said once, before any arm runs. A bundle measured against content that no
566
+ // longer matches its recorded form is evidence about a different artifact
567
+ // than the one it names, and the reader has to know that up front. It does
568
+ // not block: drift never gates activation either.
569
+ if (subject.drift?.verdict === "mismatch") {
570
+ process.stderr.write(
571
+ `clio-coder eval skill: WARNING skill_drift: ${skill.name} content (sha256 ${skill.normalizedHash.slice(0, 12)}…) ` +
572
+ `does not match the hash recorded for it by the ${subject.drift.authority} (expected ${subject.drift.expected.slice(0, 12)}…); ` +
573
+ "this run measures the copy on disk, not the recorded one\n",
574
+ );
575
+ }
167
576
  const matcher = options.scenario === undefined ? null : scenarioMatcher(options.scenario);
168
577
  if (options.scenario !== undefined && matcher === null) {
169
578
  printError(`invalid --scenario "${options.scenario}": use a scenario id like S1 or a bare number`);
@@ -189,24 +598,39 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
189
598
  }
190
599
  const startedAt = new Date().toISOString();
191
600
  const outcomes: ScenarioOutcome[] = [];
601
+ const attributionEnabled = !options.noAttribution;
192
602
  for (const scenario of scenarios) {
193
- process.stderr.write(`clio-coder eval skill: ${skill.name} ${scenario.id} baseline/treatment/judge...\n`);
603
+ process.stderr.write(
604
+ `clio-coder eval skill: ${skill.name} ${scenario.id} baseline/treatment/judge${attributionEnabled ? "/baseline-judge" : ""}...\n`,
605
+ );
194
606
  outcomes.push(
195
- await runScenario(
196
- skill.name,
197
- resolved.baseDir,
607
+ await runScenario({
608
+ skillName: skill.name,
609
+ skillBaseDir: resolved.baseDir,
198
610
  scenario,
199
- options.target,
611
+ target: options.target,
200
612
  timeoutMs,
201
613
  workspaceOverride,
202
- options.trustFixtures,
614
+ trustFixtures: options.trustFixtures,
203
615
  childEnv,
204
- ),
616
+ attributionEnabled,
617
+ subjectTreeSha256: subject.treeSha256,
618
+ subjectSha256: subject.sha256,
619
+ pinnable: subject.pinnable,
620
+ }),
621
+ );
622
+ }
623
+ const unverified = outcomes.filter((outcome) => outcome.subjectVerification === "mismatch");
624
+ if (unverified.length > 0) {
625
+ process.stderr.write(
626
+ `clio-coder eval skill: WARNING subject_changed: the skill tree at ${resolved.baseDir} changed during ${unverified
627
+ .map((outcome) => outcome.scenario.id)
628
+ .join(", ")}; those scenarios' rubric results stand but their baseline comparison is unmeasured\n`,
205
629
  );
206
630
  }
207
631
  const endedAt = new Date().toISOString();
208
632
 
209
- const artifact = synthesizeArtifact(skill.name, evalsPath, evalsRaw, startedAt, endedAt, outcomes, options.target);
633
+ const artifact = synthesizeArtifact(subject, evalsPath, evalsRaw, startedAt, endedAt, outcomes, options.target);
210
634
  let evidenceId: string | null = null;
211
635
  let evidenceDirectory: string | null = null;
212
636
  const evidenceErrors: string[] = [];
@@ -222,7 +646,7 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
222
646
  evidenceDirectory = built.directory;
223
647
  await writeFile(
224
648
  join(built.directory, SKILL_EVAL_SIDECAR),
225
- `${JSON.stringify(sidecar(skill.name, artifact.evalId, outcomes, options.allowNetwork), null, 2)}\n`,
649
+ `${JSON.stringify(sidecar(subject, artifact.evalId, outcomes, options.allowNetwork, attributionEnabled), null, 2)}\n`,
226
650
  "utf8",
227
651
  );
228
652
  } catch (error) {
@@ -234,23 +658,35 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
234
658
 
235
659
  if (options.json) {
236
660
  for (const outcome of outcomes) {
661
+ const attributed = new Map(outcome.attribution.bullets.map((item) => [item.index, item]));
237
662
  for (const bullet of outcome.bullets) {
663
+ const pair = attributed.get(bullet.index);
238
664
  process.stdout.write(
239
665
  `${JSON.stringify({
240
666
  schema: "experimental",
241
667
  kind: "skill-eval-bullet",
242
668
  skill: skill.name,
669
+ skillSha256: subject.normalizedHash,
670
+ skillOrigin: subject.origin,
671
+ skillDrift: subject.drift?.verdict ?? null,
243
672
  scenario: outcome.scenario.id,
244
673
  title: outcome.scenario.title,
245
674
  bullet: bullet.index,
246
675
  expected: bullet.text,
247
676
  verdict: bullet.verdict,
248
677
  reason: bullet.reason,
678
+ // Advisory, never a gate: `verdict` above is the rubric result and
679
+ // is unchanged by anything here.
680
+ baselineVerdict: pair?.baseline ?? null,
681
+ attribution: pair?.attribution ?? null,
682
+ scenarioAttribution: outcome.attribution.verdict,
683
+ attributionReason: outcome.attribution.reason,
249
684
  network: networkPolicyLabel(options.allowNetwork),
250
685
  autonomy: ARM_AUTONOMY,
251
686
  baselineSessionId: outcome.baseline?.sessionId ?? null,
252
687
  treatmentSessionId: outcome.treatment?.sessionId ?? null,
253
688
  judgeSessionId: outcome.judge?.sessionId ?? null,
689
+ baselineJudgeSessionId: outcome.baselineJudge?.sessionId ?? null,
254
690
  evalId: artifact.evalId,
255
691
  evidenceId,
256
692
  })}\n`,
@@ -258,7 +694,7 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
258
694
  }
259
695
  }
260
696
  } else {
261
- printHumanReport(skill.name, outcomes, artifact.evalId, evidenceId, evidenceDirectory, options.allowNetwork);
697
+ printHumanReport(subject, outcomes, artifact.evalId, evidenceId, evidenceDirectory, options.allowNetwork);
262
698
  }
263
699
  const anyFailure = outcomes.some((outcome) =>
264
700
  outcome.bullets.some((bullet) => bullet.verdict === "fail" || bullet.verdict === "error"),
@@ -300,7 +736,16 @@ function describeArmPolicyOutcome(allowNetwork: boolean): string {
300
736
  return `policy: the baseline and treatment arms ran at autonomy ${ARM_AUTONOMY}; ${network}`;
301
737
  }
302
738
 
303
- type EvalArm = "baseline" | "treatment" | "judge";
739
+ type EvalArm = "baseline" | "treatment" | "judge" | "baseline-judge";
740
+
741
+ /**
742
+ * A judge arm scores text and is told to call no tools, so granting it
743
+ * unattended write and exec buys nothing. Both judge arms are excluded from
744
+ * `--autonomy full-auto` for that reason.
745
+ */
746
+ function isJudgeArm(arm: EvalArm): boolean {
747
+ return arm === "judge" || arm === "baseline-judge";
748
+ }
304
749
 
305
750
  /**
306
751
  * The argv for one arm's child `clio-coder run`. Every arm streams the full JSON
@@ -318,7 +763,7 @@ function armRunArgs(
318
763
  options: { target?: string | undefined; skillBaseDir?: string | undefined } = {},
319
764
  ): string[] {
320
765
  const args = ["run", "--json", "--json-events", "full", "--no-skills"];
321
- if (arm !== "judge") args.push("--autonomy", ARM_AUTONOMY);
766
+ if (!isJudgeArm(arm)) args.push("--autonomy", ARM_AUTONOMY);
322
767
  if (options.skillBaseDir !== undefined) args.push("--skill", options.skillBaseDir);
323
768
  if (options.target !== undefined) args.push("--target", options.target);
324
769
  args.push(prompt);
@@ -432,22 +877,81 @@ async function resolveWorkspaceOverride(workspace: string | undefined): Promise<
432
877
  }
433
878
  }
434
879
 
435
- async function runScenario(
436
- skillName: string,
437
- skillBaseDir: string,
438
- scenario: SkillEvalScenario,
439
- target: string | undefined,
880
+ /**
881
+ * One arm's child process, behind a seam.
882
+ *
883
+ * The default is {@link captureHeadlessRun}, which spawns a real
884
+ * `clio-coder run`. Contract tests substitute a scripted runner so the arm
885
+ * sequencing, the skip gates and the recorded reasons are exercised through the
886
+ * production orchestration rather than by constructing already-correct typed
887
+ * data, and without paying for an inference.
888
+ *
889
+ * @internal
890
+ */
891
+ export type SkillEvalArmRunner = (
892
+ args: ReadonlyArray<string>,
893
+ cwd: string,
440
894
  timeoutMs: number,
441
- workspaceOverride: string | null,
442
- trustFixtures: boolean,
443
- childEnv: NodeJS.ProcessEnv,
444
- ): Promise<ScenarioOutcome> {
895
+ env: NodeJS.ProcessEnv,
896
+ ) => Promise<CapturedRun>;
897
+
898
+ /** @internal Exported for contract tests. */
899
+ export interface RunScenarioInput {
900
+ skillName: string;
901
+ skillBaseDir: string;
902
+ scenario: SkillEvalScenario;
903
+ target: string | undefined;
904
+ timeoutMs: number;
905
+ workspaceOverride: string | null;
906
+ trustFixtures: boolean;
907
+ childEnv: NodeJS.ProcessEnv;
908
+ attributionEnabled: boolean;
909
+ /** Tree digest recorded at resolution; the scenario re-checks against it. */
910
+ subjectTreeSha256: string | null;
911
+ /** sha256 of the subject's SKILL.md, compared against the activation receipt. */
912
+ subjectSha256: string;
913
+ /** False when the artifact cannot be snapshotted faithfully; see artifactIsPinnable. */
914
+ pinnable: boolean;
915
+ runner?: SkillEvalArmRunner;
916
+ }
917
+
918
+ /** @internal Exported for contract tests. */
919
+ export async function runScenario(input: RunScenarioInput): Promise<ScenarioOutcome> {
920
+ const {
921
+ skillName,
922
+ skillBaseDir,
923
+ scenario,
924
+ target,
925
+ timeoutMs,
926
+ workspaceOverride,
927
+ trustFixtures,
928
+ childEnv,
929
+ attributionEnabled,
930
+ subjectTreeSha256,
931
+ subjectSha256,
932
+ pinnable,
933
+ } = input;
934
+ const runArm = input.runner ?? captureHeadlessRun;
445
935
  // Published per-scenario figure: monotonic so a clock correction during a
446
936
  // long sweep cannot land in one row's wall time.
447
937
  const scenarioStart = performance.now();
448
938
  const workspace = await mkdtemp(join(tmpdir(), "clio-coder-skill-eval-seed-"));
449
939
  let runWorkspaces: MaterializedSkillEvalWorkspaces | null = null;
940
+ let subjectVerification: SubjectVerification = "not-checked";
941
+ let observedTreeSha256: string | null = null;
450
942
  try {
943
+ // A scenario that ended before the comparison could run records why. When
944
+ // attribution is off the operator's own choice is the operative reason,
945
+ // because the baseline judge would have been skipped either way.
946
+ const skipAttribution = (reason: string): ScenarioAttribution =>
947
+ attributionEnabled ? unmeasurableAttribution(reason) : notAttemptedAttribution(ATTRIBUTION_DISABLED_REASON);
948
+ const outcomeBase = () => ({
949
+ workspace,
950
+ subjectVerification,
951
+ observedTreeSha256,
952
+ wallTimeMs: Math.round(performance.now() - scenarioStart),
953
+ });
954
+
451
955
  if (workspaceOverride !== null) await copyWorkspace(workspaceOverride, workspace);
452
956
  const fixtureError = await runFixtureCommands(scenario, workspace, timeoutMs, trustFixtures);
453
957
  if (fixtureError !== null) {
@@ -457,24 +961,43 @@ async function runScenario(
457
961
  baseline: null,
458
962
  treatment: null,
459
963
  judge: null,
460
- workspace,
461
- wallTimeMs: Math.round(performance.now() - scenarioStart),
964
+ baselineJudge: null,
965
+ attribution: skipAttribution(fixtureError),
966
+ ...outcomeBase(),
462
967
  infraError: fixtureError,
463
968
  });
464
969
  }
465
- runWorkspaces = await materializeSkillEvalWorkspaces(workspace);
466
- const baseline = await captureHeadlessRun(
970
+ runWorkspaces = await materializeSkillEvalWorkspaces(workspace, pinnable ? skillBaseDir : null);
971
+ const baseline = await runArm(
467
972
  armRunArgs("baseline", scenario.setup, { target }),
468
973
  runWorkspaces.baseline,
469
974
  timeoutMs,
470
975
  childEnv,
471
976
  );
472
- const treatment = await captureHeadlessRun(
473
- armRunArgs("treatment", `/skill ${skillName} ${scenario.setup}`, { target, skillBaseDir }),
977
+ // The arm runs against a private copy, so an edit to the source tree
978
+ // cannot reach the child mid-run. The copy is still digested either side
979
+ // of the arm, so the claim rests on a measurement rather than on the
980
+ // assumption that nothing touched it.
981
+ const pinned = pinnable ? runWorkspaces.pin : null;
982
+ const armSkillDir = pinned ?? skillBaseDir;
983
+ const before = await skillTreeDigest(armSkillDir);
984
+ const treatment = await runArm(
985
+ armRunArgs("treatment", `/skill ${skillName} ${scenario.setup}`, { target, skillBaseDir: armSkillDir }),
474
986
  runWorkspaces.treatment,
475
987
  timeoutMs,
476
988
  childEnv,
477
989
  );
990
+ const after = await skillTreeDigest(armSkillDir);
991
+ observedTreeSha256 = after;
992
+ // Three conditions, in the order they can fail. The artifact must be
993
+ // pinnable, the pin must have held, and the child must have actually
994
+ // activated it. Disk equality alone is satisfied by a run that activated
995
+ // nothing, so it is necessary and not sufficient.
996
+ subjectVerification = !pinnable
997
+ ? "not-pinned"
998
+ : verifySubjectTree(subjectTreeSha256, before, after) === "verified"
999
+ ? verifyObservedActivation(skillName, subjectSha256, join(armSkillDir, "SKILL.md"), treatment.activations)
1000
+ : verifySubjectTree(subjectTreeSha256, before, after);
478
1001
  const infra = runInfraError("baseline", baseline) ?? runInfraError("treatment", treatment);
479
1002
  if (infra !== null) {
480
1003
  return await completeScenarioOutcome({
@@ -483,8 +1006,9 @@ async function runScenario(
483
1006
  baseline,
484
1007
  treatment,
485
1008
  judge: null,
486
- workspace,
487
- wallTimeMs: Math.round(performance.now() - scenarioStart),
1009
+ baselineJudge: null,
1010
+ attribution: skipAttribution(infra),
1011
+ ...outcomeBase(),
488
1012
  infraError: infra,
489
1013
  });
490
1014
  }
@@ -499,8 +1023,9 @@ async function runScenario(
499
1023
  baseline,
500
1024
  treatment,
501
1025
  judge: null,
502
- workspace,
503
- wallTimeMs: Math.round(performance.now() - scenarioStart),
1026
+ baselineJudge: null,
1027
+ attribution: skipAttribution(wall),
1028
+ ...outcomeBase(),
504
1029
  infraError: wall,
505
1030
  });
506
1031
  }
@@ -508,7 +1033,7 @@ async function runScenario(
508
1033
  // terminating tool (artifact plan/review/report) prints only that tool's
509
1034
  // result line in text mode, while the event stream carries the artifact
510
1035
  // content the verdict may live in.
511
- const judge = await captureHeadlessRun(
1036
+ const judge = await runArm(
512
1037
  armRunArgs("judge", judgePrompt(scenario, baseline.transcript, treatment.transcript), { target }),
513
1038
  runWorkspaces.judge,
514
1039
  timeoutMs,
@@ -522,20 +1047,38 @@ async function runScenario(
522
1047
  baseline,
523
1048
  treatment,
524
1049
  judge,
525
- workspace,
526
- wallTimeMs: Math.round(performance.now() - scenarioStart),
1050
+ baselineJudge: null,
1051
+ attribution: skipAttribution(judgeInfra),
1052
+ ...outcomeBase(),
527
1053
  infraError: judgeInfra,
528
1054
  });
529
1055
  }
530
1056
  const bullets = parseJudgeVerdicts(scenario, judge);
1057
+ // The rubric result above stands whatever the comparison decides. The
1058
+ // comparison, unlike the rubric, is refused when the artifact measured is
1059
+ // not demonstrably the artifact the subject names.
1060
+ const attributed = await attributeScenario({
1061
+ scenario,
1062
+ bullets,
1063
+ baseline,
1064
+ judge,
1065
+ target,
1066
+ timeoutMs,
1067
+ childEnv,
1068
+ attributionEnabled,
1069
+ subjectVerification,
1070
+ workspace: runWorkspaces.baselineJudge,
1071
+ runner: runArm,
1072
+ });
531
1073
  return await completeScenarioOutcome({
532
1074
  scenario,
533
1075
  bullets,
534
1076
  baseline,
535
1077
  treatment,
536
1078
  judge,
537
- workspace,
538
- wallTimeMs: Math.round(performance.now() - scenarioStart),
1079
+ baselineJudge: attributed.baselineJudge,
1080
+ attribution: attributed.attribution,
1081
+ ...outcomeBase(),
539
1082
  infraError: null,
540
1083
  });
541
1084
  } finally {
@@ -547,10 +1090,189 @@ async function runScenario(
547
1090
  }
548
1091
  }
549
1092
 
1093
+ /** The exact reason recorded when the operator turned the comparison off. */
1094
+ const ATTRIBUTION_DISABLED_REASON = "disabled by --no-attribution";
1095
+
1096
+ /**
1097
+ * Compare the tree digests taken either side of the treatment arm against the
1098
+ * one the subject was resolved from.
1099
+ *
1100
+ * All three must agree. `verified` therefore means "the artifact on disk was
1101
+ * the subject's artifact before the arm and still was after it", which is a
1102
+ * narrower claim than "the child activated the subject", and the sidecar states
1103
+ * that distinction rather than letting the word imply more.
1104
+ *
1105
+ * @internal Exported for contract tests.
1106
+ */
1107
+ export function verifySubjectTree(
1108
+ subject: string | null,
1109
+ before: string | null,
1110
+ after: string | null,
1111
+ ): SubjectVerification {
1112
+ if (subject === null || before === null || after === null) return "unreadable";
1113
+ return subject === before && before === after ? "verified" : "mismatch";
1114
+ }
1115
+
1116
+ /**
1117
+ * Did the treatment child actually activate the artifact the subject names?
1118
+ *
1119
+ * Disk equality answers a different question. It says the directory looked the
1120
+ * same either side of the arm, which is true of a run that activated nothing at
1121
+ * all, and true of a run that loaded a revision written and reverted inside the
1122
+ * arm. Neither is a measurement of the subject, and reporting `helped` about
1123
+ * either would attribute a difference to an artifact that was never read.
1124
+ *
1125
+ * So the positive evidence is the activation receipt: the child reports the
1126
+ * sha256 of the bytes it read, and that has to equal the pinned SKILL.md's.
1127
+ * Several activations of the same name are ambiguous and are refused rather
1128
+ * than resolved by picking one.
1129
+ *
1130
+ * @internal Exported for contract tests.
1131
+ */
1132
+ export function verifyObservedActivation(
1133
+ skillName: string,
1134
+ expectedSha256: string,
1135
+ expectedPath: string,
1136
+ activations: ReadonlyArray<ObservedActivation>,
1137
+ ): SubjectVerification {
1138
+ const matching = activations.filter((entry) => entry.name === skillName);
1139
+ if (matching.length === 0) return "not-activated";
1140
+ if (matching.length > 1 && new Set(matching.map((entry) => `${entry.hash}|${entry.path}`)).size > 1) {
1141
+ return "activation-mismatch";
1142
+ }
1143
+ const observed = matching[0];
1144
+ if (observed === undefined || observed.hash !== expectedSha256) return "activation-mismatch";
1145
+ // The hash says which bytes; the path says which copy. Two directories can
1146
+ // hold the same SKILL.md and different reference files beside it, and the
1147
+ // body's own pointers resolve against the directory it was loaded from, so a
1148
+ // matching hash from somewhere else is not the artifact under test.
1149
+ return samePath(observed.path, expectedPath) ? "verified" : "activation-mismatch";
1150
+ }
1151
+
1152
+ /** Compare two paths as the filesystem resolves them, falling back to lexical. */
1153
+ function samePath(left: string, right: string): boolean {
1154
+ if (left.length === 0 || right.length === 0) return false;
1155
+ try {
1156
+ return realpathSync.native(left) === realpathSync.native(right);
1157
+ } catch {
1158
+ return resolve(left) === resolve(right);
1159
+ }
1160
+ }
1161
+
1162
+ /** Why the comparison was refused, named after what was not established. */
1163
+ const SUBJECT_REFUSAL: Record<Exclude<SubjectVerification, "verified">, string> = {
1164
+ mismatch:
1165
+ "the measured skill tree changed on disk during this scenario, so its two arms did not run against one artifact",
1166
+ unreadable: "the measured skill tree could not be re-read, so the artifact under test is unverified",
1167
+ "not-pinned":
1168
+ "the skill body carries package references, which resolve against an owning package root outside the skill directory, so the artifact could not be copied faithfully for this run",
1169
+ "not-activated":
1170
+ "the treatment arm recorded no successful activation of this skill, so nothing establishes that the measured artifact ran",
1171
+ "activation-mismatch":
1172
+ "the treatment arm activated content whose hash is not the pinned artifact's, so the comparison would name the wrong revision",
1173
+ "not-checked": "the artifact identity was never checked for this scenario",
1174
+ };
1175
+
1176
+ interface AttributeScenarioInput {
1177
+ scenario: SkillEvalScenario;
1178
+ /** Treatment verdicts from the legacy rubric parser, used only as a skip gate. */
1179
+ bullets: ReadonlyArray<ScoredBullet>;
1180
+ baseline: CapturedRun;
1181
+ /** The treatment judge run, re-read under strict rules for the comparison. */
1182
+ judge: CapturedRun;
1183
+ target: string | undefined;
1184
+ timeoutMs: number;
1185
+ childEnv: NodeJS.ProcessEnv;
1186
+ attributionEnabled: boolean;
1187
+ subjectVerification: SubjectVerification;
1188
+ workspace: string;
1189
+ runner: SkillEvalArmRunner;
1190
+ }
1191
+
1192
+ /**
1193
+ * Score the baseline arm and pair it with the treatment, or say why not.
1194
+ *
1195
+ * The baseline run is already paid for by the time this is reached; only the
1196
+ * judge that reads it is new. Five states end the comparison before it produces
1197
+ * a verdict, and each is recorded with its own reason rather than collapsed
1198
+ * into one: the operator disabled it, the artifact measured was not
1199
+ * demonstrably the artifact the subject names, the treatment produced nothing
1200
+ * to compare against, the baseline judge failed to run, or the baseline judge
1201
+ * hit the harness's own permission wall.
1202
+ *
1203
+ * The verdicts it pairs come from {@link strictJudgeVerdicts}, not from the
1204
+ * rubric parser, so a judge that answered with a missing, null or string `pass`
1205
+ * contributes nothing instead of contributing an invented `fail`.
1206
+ */
1207
+ async function attributeScenario(
1208
+ input: AttributeScenarioInput,
1209
+ ): Promise<{ attribution: ScenarioAttribution; baselineJudge: CapturedRun | null }> {
1210
+ if (!input.attributionEnabled) {
1211
+ return { attribution: notAttemptedAttribution(ATTRIBUTION_DISABLED_REASON), baselineJudge: null };
1212
+ }
1213
+ // Nothing establishes which artifact ran, so nothing can be attributed to
1214
+ // one. Each refusal names what was not established rather than collapsing
1215
+ // into one verdict.
1216
+ if (input.subjectVerification !== "verified") {
1217
+ return { attribution: unmeasurableAttribution(SUBJECT_REFUSAL[input.subjectVerification]), baselineJudge: null };
1218
+ }
1219
+ // The same comparative eligibility the baseline judge gets. A treatment judge
1220
+ // that collected the harness's own denial and then answered was scoring under
1221
+ // a constraint the other arm did not have. The rubric result it produced
1222
+ // stands; only the comparison is refused.
1223
+ const treatmentWall = permissionWallReason("judge", input.judge);
1224
+ if (treatmentWall !== null) return { attribution: unmeasurableAttribution(treatmentWall), baselineJudge: null };
1225
+ // Nothing on the treatment side was scored, so no pair can be formed. Running
1226
+ // the baseline judge anyway would spend an inference to learn nothing.
1227
+ if (!input.bullets.some((bullet) => bullet.verdict === "pass" || bullet.verdict === "fail")) {
1228
+ return {
1229
+ attribution: unmeasurableAttribution(
1230
+ "the treatment arm produced no scored bullet, so there is nothing to compare a baseline against",
1231
+ ),
1232
+ baselineJudge: null,
1233
+ };
1234
+ }
1235
+ const baselineJudge = await input.runner(
1236
+ armRunArgs("baseline-judge", baselineJudgePrompt(input.scenario, input.baseline.transcript), {
1237
+ target: input.target,
1238
+ }),
1239
+ input.workspace,
1240
+ input.timeoutMs,
1241
+ input.childEnv,
1242
+ );
1243
+ const infra = runInfraError("baseline-judge", baselineJudge);
1244
+ if (infra !== null) {
1245
+ // The treatment verdicts stand: this failure is about the comparison, not
1246
+ // about the skill, so it never touches `bullets` or the exit code.
1247
+ return { attribution: unmeasurableAttribution(infra), baselineJudge };
1248
+ }
1249
+ // The prompt tells the judge to use no tools, which is an instruction and not
1250
+ // enforcement. A judge that tried anyway and collected the harness's denial
1251
+ // was scoring under a constraint the treatment judge did not have.
1252
+ const wall = permissionWallReason("baseline-judge", baselineJudge);
1253
+ if (wall !== null) return { attribution: unmeasurableAttribution(wall), baselineJudge };
1254
+ return {
1255
+ attribution: attributionFromStrict(
1256
+ input.scenario,
1257
+ strictJudgeVerdicts(input.scenario, baselineJudge),
1258
+ strictJudgeVerdicts(input.scenario, input.judge),
1259
+ ),
1260
+ baselineJudge,
1261
+ };
1262
+ }
1263
+
550
1264
  export interface MaterializedSkillEvalWorkspaces {
551
1265
  baseline: string;
552
1266
  treatment: string;
553
1267
  judge: string;
1268
+ /** The baseline judge gets its own root for the same reason the other arms do. */
1269
+ baselineJudge: string;
1270
+ /**
1271
+ * Private per-run copy of the artifact the treatment arm loads; null when the
1272
+ * artifact cannot be copied faithfully. Isolated from the source directory,
1273
+ * not read-only to the arm itself.
1274
+ */
1275
+ pin: string | null;
554
1276
  cleanup(): Promise<void>;
555
1277
  }
556
1278
 
@@ -581,17 +1303,33 @@ async function armWorkspace(created: string[]): Promise<string> {
581
1303
  }
582
1304
 
583
1305
  /** @internal Exported for contract tests. */
584
- async function materializeSkillEvalWorkspaces(seedWorkspace: string): Promise<MaterializedSkillEvalWorkspaces> {
1306
+ async function materializeSkillEvalWorkspaces(
1307
+ seedWorkspace: string,
1308
+ pinSource: string | null,
1309
+ ): Promise<MaterializedSkillEvalWorkspaces> {
585
1310
  const created: string[] = [];
586
1311
  try {
587
1312
  const baseline = await armWorkspace(created);
588
1313
  const treatment = await armWorkspace(created);
589
1314
  const judge = await armWorkspace(created);
1315
+ const baselineJudge = await armWorkspace(created);
1316
+ let pin: string | null = null;
1317
+ if (pinSource !== null) {
1318
+ const root = await mkdtemp(join(tmpdir(), "clio-coder-skill-eval-pin-"));
1319
+ created.push(root);
1320
+ pin = join(root, "skill");
1321
+ await cp(pinSource, pin, { recursive: true, preserveTimestamps: true, verbatimSymlinks: true });
1322
+ }
1323
+ // Only the acting arms get the fixture. A judge scores text and is told to
1324
+ // call no tools, so seeding its workspace would only give it the artifacts
1325
+ // it is supposed to read about.
590
1326
  await Promise.all([copyWorkspace(seedWorkspace, baseline), copyWorkspace(seedWorkspace, treatment)]);
591
1327
  return {
592
1328
  baseline,
593
1329
  treatment,
594
1330
  judge,
1331
+ baselineJudge,
1332
+ pin,
595
1333
  cleanup: async () => {
596
1334
  await Promise.all(created.map((path) => rm(path, { recursive: true, force: true })));
597
1335
  },
@@ -613,7 +1351,7 @@ async function copyWorkspace(source: string, destination: string): Promise<void>
613
1351
  async function completeScenarioOutcome(outcome: Omit<ScenarioOutcome, "usage">): Promise<ScenarioOutcome> {
614
1352
  return {
615
1353
  ...outcome,
616
- usage: await usageForCapturedRuns([outcome.baseline, outcome.treatment, outcome.judge]),
1354
+ usage: await usageForCapturedRuns([outcome.baseline, outcome.treatment, outcome.judge, outcome.baselineJudge]),
617
1355
  };
618
1356
  }
619
1357
 
@@ -816,12 +1554,36 @@ function loadsSkillBody(tool: string, args: unknown): boolean {
816
1554
  return args.scope === "skills" && typeof args.name === "string" && args.name.trim().length > 0;
817
1555
  }
818
1556
 
1557
+ /**
1558
+ * The activation contract inside a successful skill load's result details.
1559
+ *
1560
+ * `runSkillsScope` records `name`, `filePath` and `hash` on every activation,
1561
+ * and the agent loop puts the tool result's details on the execution-end event.
1562
+ * Anything missing either field is not an activation receipt and is ignored
1563
+ * rather than half-read.
1564
+ */
1565
+ function activationFromResult(result: unknown): ObservedActivation | null {
1566
+ if (!isRecord(result)) return null;
1567
+ const details = isRecord(result.details) ? result.details : null;
1568
+ if (details === null) return null;
1569
+ const name = readString(details.name);
1570
+ const hash = readString(details.hash);
1571
+ if (name === null || hash === null) return null;
1572
+ return { name, hash, path: readString(details.filePath) ?? readString(details.path) ?? "" };
1573
+ }
1574
+
819
1575
  /** @internal Exported for contract tests. */
820
- function parseRunStdout(stdout: string): { sessionId: string | null; transcript: string; finalText: string } {
1576
+ export function parseRunStdout(stdout: string): {
1577
+ sessionId: string | null;
1578
+ transcript: string;
1579
+ finalText: string;
1580
+ activations: ObservedActivation[];
1581
+ } {
821
1582
  let sessionId: string | null = null;
822
1583
  const lines: string[] = [];
823
1584
  let finalText = "";
824
1585
  let sawJson = false;
1586
+ const activations: ObservedActivation[] = [];
825
1587
  const streamedText = new Map<number, string>();
826
1588
  // Tool calls whose result is the skill's own SKILL.md. Correlated by
827
1589
  // toolCallId, which both the start and end events carry.
@@ -867,6 +1629,15 @@ function parseRunStdout(stdout: string): { sessionId: string | null; transcript:
867
1629
  const tool = readString(event.toolName) ?? readString(event.tool) ?? "tool";
868
1630
  const status = event.isError === true ? "error" : "ok";
869
1631
  const callId = readString(event.toolCallId);
1632
+ // A successful skill load carries the activation contract in its result
1633
+ // details: the name, the file and the sha256 of the bytes the child
1634
+ // actually read. That is the only evidence in this stream about which
1635
+ // artifact ran, and it is collected here and kept out of the transcript
1636
+ // so the judge still never sees the body or its identity.
1637
+ if (event.isError !== true && callId !== null && skillBodyCallIds.has(callId)) {
1638
+ const activation = activationFromResult(event.result);
1639
+ if (activation !== null) activations.push(activation);
1640
+ }
870
1641
  // The skill body is the instructions, not the behavior. Left in the
871
1642
  // transcript it is the easiest thing in the run for a judge to quote,
872
1643
  // and a 30B judge scored bullets as passing from SKILL.md prose that
@@ -903,9 +1674,9 @@ function parseRunStdout(stdout: string): { sessionId: string | null; transcript:
903
1674
  }
904
1675
  if (!sawJson) {
905
1676
  const text = stdout.trim();
906
- return { sessionId: null, transcript: text, finalText: text };
1677
+ return { sessionId: null, transcript: text, finalText: text, activations: [] };
907
1678
  }
908
- return { sessionId, transcript: elide(lines.join("\n")), finalText };
1679
+ return { sessionId, transcript: elide(lines.join("\n")), finalText, activations };
909
1680
  }
910
1681
 
911
1682
  function judgePrompt(scenario: SkillEvalScenario, baselineTranscript: string, treatmentTranscript: string): string {
@@ -932,6 +1703,41 @@ function judgePrompt(scenario: SkillEvalScenario, baselineTranscript: string, tr
932
1703
  ].join("\n");
933
1704
  }
934
1705
 
1706
+ /**
1707
+ * Score the baseline arm alone, against the same bullets.
1708
+ *
1709
+ * Isolated on purpose. The treatment judge above sees both transcripts and is
1710
+ * told to score only the treatment; giving the baseline judge the treatment
1711
+ * transcript as well would let a strong treatment run colour the baseline's
1712
+ * verdicts, and the comparison those verdicts feed would then be measuring the
1713
+ * judge. The bullet contract, the strict-JSON shape and the no-tools rule are
1714
+ * identical, so `parseJudgeVerdicts` reads both without branching.
1715
+ *
1716
+ * The treatment prompt is deliberately left byte-identical to what it was
1717
+ * before attribution existed, so treatment verdicts stay comparable with every
1718
+ * bundle already on disk.
1719
+ */
1720
+ function baselineJudgePrompt(scenario: SkillEvalScenario, baselineTranscript: string): string {
1721
+ const bullets = scenario.expected.map((text, index) => `${index + 1}. ${text}`).join("\n");
1722
+ return [
1723
+ "You are scoring an agent run. One transcript follows: it ran WITHOUT any skill loaded.",
1724
+ "Score each EXPECTED bullet strictly against the TRANSCRIPT.",
1725
+ "A bullet passes only if the transcript observably satisfies it; anything unverifiable from the transcript fails.",
1726
+ 'Reply with STRICT JSON only, no prose and no code fences, exactly: {"bullets":[{"index":1,"pass":true,"reason":"<= 25 words"}]}',
1727
+ `Include one entry per bullet, indexes 1 through ${scenario.expected.length} in order.`,
1728
+ "Do not use any tools. Respond with the JSON verdict directly.",
1729
+ "",
1730
+ `SCENARIO ${scenario.id} - ${scenario.title}`,
1731
+ `SETUP: ${scenario.setup}`,
1732
+ "",
1733
+ "EXPECTED BULLETS:",
1734
+ bullets,
1735
+ "",
1736
+ "TRANSCRIPT:",
1737
+ baselineTranscript.length > 0 ? baselineTranscript : "(empty)",
1738
+ ].join("\n");
1739
+ }
1740
+
935
1741
  /**
936
1742
  * Why a judge response carried no verdict, named after the thing that failed.
937
1743
  * A response that opened a bullets object and never closed it is the observed
@@ -984,6 +1790,85 @@ function parseJudgeVerdicts(scenario: SkillEvalScenario, judge: CapturedRun): Sc
984
1790
  });
985
1791
  }
986
1792
 
1793
+ /**
1794
+ * Verdicts strict enough to compare two arms against each other.
1795
+ *
1796
+ * {@link parseJudgeVerdicts} above is the legacy rubric gate and its coercions
1797
+ * are load-bearing for `pass` / `exitCode` / `failureClass`, so it is left
1798
+ * exactly as it is. It is, however, forgiving in ways that are fine for a
1799
+ * one-armed verdict and wrong for a comparison: `pass: item.pass === true`
1800
+ * turns a missing, null or string field into a genuine `fail`, and
1801
+ * `Number.parseInt(String(item.index))` reads `"1oops"` as bullet 1. Pairing a
1802
+ * real treatment `pass` against an invented baseline `fail` reports `helped`
1803
+ * about a measurement the judge never made.
1804
+ *
1805
+ * So comparison reads the raw judge output again under stricter rules, and
1806
+ * anything that does not clear them stays out of the comparison rather than
1807
+ * entering it as a verdict. Rejecting evidence here cannot change the rubric
1808
+ * result: the two parsers have separate callers on purpose.
1809
+ *
1810
+ * @internal Exported for contract tests.
1811
+ */
1812
+ export interface StrictJudgeVerdicts {
1813
+ /** Bullet index to its unambiguous boolean verdict. */
1814
+ verdicts: Map<number, boolean>;
1815
+ /** Bullet index to why its evidence was refused. */
1816
+ rejected: Map<number, string>;
1817
+ /** Set when the run produced no parseable verdict object at all. */
1818
+ absent: string | null;
1819
+ }
1820
+
1821
+ /** @internal Exported for contract tests. */
1822
+ export function strictJudgeVerdicts(scenario: SkillEvalScenario, judge: CapturedRun): StrictJudgeVerdicts {
1823
+ const verdicts = new Map<number, boolean>();
1824
+ const rejected = new Map<number, string>();
1825
+ const parsed = extractBulletsObject(judge.finalText) ?? extractBulletsObject(judge.transcript);
1826
+ if (parsed === null) return { verdicts, rejected, absent: judgeVerdictAbsenceReason(judge) };
1827
+ if (!Array.isArray(parsed.bullets)) {
1828
+ return { verdicts, rejected, absent: "judge verdict object carried no bullets array" };
1829
+ }
1830
+ const count = scenario.expected.length;
1831
+ const seen = new Map<number, boolean>();
1832
+ for (const item of parsed.bullets) {
1833
+ if (!isRecord(item)) continue;
1834
+ // A numeric index and nothing else. A string that happens to start with
1835
+ // digits is not an index the judge chose; it is a parse accident.
1836
+ const index = item.index;
1837
+ if (typeof index !== "number" || !Number.isInteger(index) || index < 1 || index > count) {
1838
+ continue;
1839
+ }
1840
+ // A rejection for an index is final and order-independent. An entry that
1841
+ // arrives after a valid one still poisons it: the judge emitted two
1842
+ // answers for one bullet and only one of them is usable, so which one it
1843
+ // "meant" is a guess, and a guess is not comparison evidence.
1844
+ if (typeof item.pass !== "boolean") {
1845
+ rejected.set(index, `judge gave no boolean pass for bullet ${index}: comparison evidence refused`);
1846
+ verdicts.delete(index);
1847
+ continue;
1848
+ }
1849
+ if (rejected.has(index)) {
1850
+ verdicts.delete(index);
1851
+ continue;
1852
+ }
1853
+ const previous = seen.get(index);
1854
+ if (previous !== undefined && previous !== item.pass) {
1855
+ // Two contradictory verdicts for one bullet. Taking either would pick a
1856
+ // winner the judge never picked.
1857
+ rejected.set(index, `judge gave contradictory verdicts for bullet ${index}: comparison evidence refused`);
1858
+ verdicts.delete(index);
1859
+ continue;
1860
+ }
1861
+ seen.set(index, item.pass);
1862
+ verdicts.set(index, item.pass);
1863
+ }
1864
+ for (let index = 1; index <= count; index += 1) {
1865
+ if (!verdicts.has(index) && !rejected.has(index)) {
1866
+ rejected.set(index, `judge omitted bullet ${index}: comparison evidence absent`);
1867
+ }
1868
+ }
1869
+ return { verdicts, rejected, absent: null };
1870
+ }
1871
+
987
1872
  /**
988
1873
  * Find the JSON object carrying the judge's `bullets` array anywhere in the
989
1874
  * text: models wrap verdicts in prose, code fences, or terminating tool
@@ -1043,7 +1928,7 @@ function balancedJsonSlice(text: string, start: number): string | null {
1043
1928
  }
1044
1929
 
1045
1930
  function synthesizeArtifact(
1046
- skillName: string,
1931
+ subject: SkillEvalSubject,
1047
1932
  evalsPath: string,
1048
1933
  evalsRaw: string,
1049
1934
  startedAt: string,
@@ -1051,6 +1936,7 @@ function synthesizeArtifact(
1051
1936
  outcomes: ReadonlyArray<ScenarioOutcome>,
1052
1937
  target: string | undefined,
1053
1938
  ): EvalRunArtifact {
1939
+ const skillName = subject.name;
1054
1940
  const contentHash = createHash("sha256").update(evalsRaw, "utf8").digest("hex");
1055
1941
  const stamp = startedAt.replace(/[-:.]/g, "");
1056
1942
  // Random suffix for the same reason createEvalId carries one: the stamp plus
@@ -1070,6 +1956,14 @@ function synthesizeArtifact(
1070
1956
  tags: [
1071
1957
  "skill-eval",
1072
1958
  `skill:${skillName}`,
1959
+ // The artifact identity, on every record. Two bundles for one skill
1960
+ // name are only comparable when both say which content they ran.
1961
+ `skill-sha:${subject.normalizedHash.slice(0, 12)}`,
1962
+ `skill-origin:${subject.origin}`,
1963
+ ...(subject.drift !== null ? [`skill-drift:${subject.drift.verdict}`] : []),
1964
+ // Advisory. `pass` and `exitCode` below keep their treatment-only
1965
+ // meaning; this tag is read by nothing that gates.
1966
+ `attribution:${outcome.attribution.verdict}`,
1073
1967
  ...(unmeasured ? ["scenario:unmeasured"] : []),
1074
1968
  ...outcome.bullets.map((bullet) => `bullet-${bullet.index}:${bullet.verdict}`),
1075
1969
  ],
@@ -1107,25 +2001,46 @@ function synthesizeArtifact(
1107
2001
  };
1108
2002
  }
1109
2003
 
1110
- function sidecar(
1111
- skillName: string,
2004
+ /** @internal Exported for contract tests. */
2005
+ export function sidecar(
2006
+ subject: SkillEvalSubject,
1112
2007
  evalId: string,
1113
2008
  outcomes: ReadonlyArray<ScenarioOutcome>,
1114
2009
  allowNetwork: boolean,
2010
+ attributionEnabled: boolean,
1115
2011
  ): unknown {
1116
2012
  return {
1117
- version: 1,
2013
+ version: 2,
1118
2014
  schema: "experimental",
1119
2015
  kind: "skill-eval",
1120
- skill: skillName,
2016
+ skill: subject.name,
2017
+ // What this run measured, by identity rather than by name. `version` moved
2018
+ // to 2 for this block; readers of version 1 keep working, they just never
2019
+ // learn which copy produced their numbers.
2020
+ subject,
1121
2021
  evalId,
1122
2022
  network: networkPolicyLabel(allowNetwork),
1123
2023
  autonomy: ARM_AUTONOMY,
2024
+ attributionEnabled,
2025
+ attributionSummary: attributionSummary(outcomes),
1124
2026
  deltas: [
1125
2027
  "bullet verdicts are judge-scored from run transcripts, not command exit codes; the evals-domain artifact carries scenario-level records with empty command lists",
1126
2028
  "a bullet the judge never scored is recorded unmeasured, not failed: its scenario record is pass:false with exitCode 3 and no failureClass",
1127
2029
  "an arm whose transcript carries the headless permission wall is recorded unmeasured with an infraError: the harness's own gate is not a verdict about the skill",
1128
2030
  "tokens and cost are rolled up from headless main-agent receipts when those receipts are present",
2031
+ "attribution is advisory and gates nothing: pass, exitCode and failureClass keep their treatment-only meaning",
2032
+ "attribution unmeasured means the comparison was impossible; not-attempted means it was never run. Neither is no-change",
2033
+ "the baseline judge scores the baseline transcript alone; the treatment judge prompt is unchanged from version 1 bundles",
2034
+ "comparison reads both judges' raw output under strict rules (boolean pass, integer in-range index, no contradictory duplicate); refused evidence stays unmeasured and never becomes a verdict",
2035
+ "the rubric bullets above keep the legacy parser's forgiving coercions: tightening comparison eligibility does not move the pass/exitCode gate",
2036
+ "the treatment arm runs against a private per-run copy of the skill directory, so an edit to the source during the arm cannot reach the child; subject.pinnable is false when the body carries package references, which resolve above the skill directory and cannot be snapshotted faithfully",
2037
+ "that copy is an isolated snapshot, not an immutable pin: it lives in a writable temp directory and the arm runs at full-auto, so a model that edited the copy, read it and restored it would not be detected. These are results about a cooperative model, the same caveat the arm workspaces already carry",
2038
+ "subjectVerification is verified only when the snapshot held (tree digest equal before and after the arm) AND the treatment arm reported a successful activation whose sha256 equals subject.sha256 and whose file is the snapshot's own SKILL.md by canonical path. not-activated, activation-mismatch, not-pinned, mismatch and unreadable each name what was not established",
2039
+ "a scenario whose subjectVerification is not verified keeps its rubric result and records its comparison unmeasured",
2040
+ "treeSha256 covers regular files and in-tree symlink targets; a symlink leaving the directory makes the tree unverifiable rather than partially hashed",
2041
+ "both judges are checked for the headless permission wall before their verdicts are eligible for comparison; the legacy rubric result is unaffected either way",
2042
+ "subject.drift records whether the measured content still matches its recorded hash; a mismatch never blocks the run",
2043
+ "subject hashes cover SKILL.md (sha256, normalizedHash) and the whole skill directory (treeSha256); neither covers resources the skill reads from outside its own directory",
1129
2044
  "this sidecar is additive and is registered in overview.json files[]",
1130
2045
  ],
1131
2046
  scenarios: outcomes.map((outcome) => ({
@@ -1136,13 +2051,31 @@ function sidecar(
1136
2051
  infraError: outcome.infraError,
1137
2052
  usage: outcome.usage,
1138
2053
  bullets: outcome.bullets,
2054
+ subjectVerification: outcome.subjectVerification,
2055
+ observedTreeSha256: outcome.observedTreeSha256,
2056
+ attribution: outcome.attribution,
1139
2057
  baseline: sidecarRun(outcome.baseline),
1140
2058
  treatment: sidecarRun(outcome.treatment),
1141
2059
  judge: sidecarRun(outcome.judge),
2060
+ baselineJudge: sidecarRun(outcome.baselineJudge),
1142
2061
  })),
1143
2062
  };
1144
2063
  }
1145
2064
 
2065
+ /** Scenario counts per attribution verdict. A rollup, never collapsed to one score. */
2066
+ function attributionSummary(outcomes: ReadonlyArray<ScenarioOutcome>): Record<ScenarioAttributionVerdict, number> {
2067
+ const summary: Record<ScenarioAttributionVerdict, number> = {
2068
+ helped: 0,
2069
+ regressed: 0,
2070
+ mixed: 0,
2071
+ "no-change": 0,
2072
+ unmeasured: 0,
2073
+ "not-attempted": 0,
2074
+ };
2075
+ for (const outcome of outcomes) summary[outcome.attribution.verdict] += 1;
2076
+ return summary;
2077
+ }
2078
+
1146
2079
  function sidecarRun(run: CapturedRun | null): unknown {
1147
2080
  if (run === null) return null;
1148
2081
  return {
@@ -1156,17 +2089,27 @@ function sidecarRun(run: CapturedRun | null): unknown {
1156
2089
  }
1157
2090
 
1158
2091
  function printHumanReport(
1159
- skillName: string,
2092
+ subject: SkillEvalSubject,
1160
2093
  outcomes: ReadonlyArray<ScenarioOutcome>,
1161
2094
  evalId: string,
1162
2095
  evidenceId: string | null,
1163
2096
  evidenceDirectory: string | null,
1164
2097
  allowNetwork: boolean,
1165
2098
  ): void {
1166
- const rows: string[][] = [["scenario", "bullet", "verdict", "expected"]];
2099
+ const skillName = subject.name;
2100
+ // `vs base` is the same bullet in the arm that ran without the skill. Kept in
2101
+ // its own column so a reader never mistakes it for part of the rubric verdict.
2102
+ const rows: string[][] = [["scenario", "bullet", "verdict", "vs base", "expected"]];
1167
2103
  for (const outcome of outcomes) {
2104
+ const attributed = new Map(outcome.attribution.bullets.map((item) => [item.index, item]));
1168
2105
  for (const bullet of outcome.bullets) {
1169
- rows.push([outcome.scenario.id, String(bullet.index), bullet.verdict, truncate(bullet.text, 76)]);
2106
+ rows.push([
2107
+ outcome.scenario.id,
2108
+ String(bullet.index),
2109
+ bullet.verdict,
2110
+ attributed.get(bullet.index)?.attribution ?? "-",
2111
+ truncate(bullet.text, 64),
2112
+ ]);
1170
2113
  }
1171
2114
  }
1172
2115
  process.stdout.write(formatColumns(rows));
@@ -1204,13 +2147,78 @@ function printHumanReport(
1204
2147
  );
1205
2148
  }
1206
2149
  }
2150
+ printAttributionReport(outcomes);
1207
2151
  process.stdout.write(`${describeArmPolicyOutcome(allowNetwork)}\n`);
2152
+ process.stdout.write(
2153
+ `subject: ${skillName} from ${subject.origin} at ${subject.baseDir} (sha256 ${subject.normalizedHash.slice(0, 12)}…)\n`,
2154
+ );
2155
+ if (subject.drift !== null) {
2156
+ process.stdout.write(
2157
+ subject.drift.verdict === "mismatch"
2158
+ ? `subject drift: MISMATCH against the ${subject.drift.authority} (expected ${subject.drift.expected.slice(0, 12)}…); this run measured the copy on disk\n`
2159
+ : `subject drift: matches the ${subject.drift.authority}\n`,
2160
+ );
2161
+ }
2162
+ const unverified = outcomes.filter((outcome) => outcome.subjectVerification !== "verified");
2163
+ if (unverified.length > 0) {
2164
+ process.stdout.write(
2165
+ `subject verification: ${unverified.map((o) => `${o.scenario.id}=${o.subjectVerification}`).join(", ")}; ` +
2166
+ "those scenarios kept their rubric result and recorded no baseline comparison\n",
2167
+ );
2168
+ }
1208
2169
  process.stdout.write(`eval artifact: ${evalId}\n`);
1209
2170
  if (evidenceId !== null && evidenceDirectory !== null) {
1210
2171
  process.stdout.write(`evidence: ${evidenceId} at ${evidenceDirectory} (per-bullet detail in skill-eval.json)\n`);
1211
2172
  }
1212
2173
  }
1213
2174
 
2175
+ /**
2176
+ * The paired comparison, stated as scenario counts and never as one score.
2177
+ *
2178
+ * Deliberately separated from the rubric block above it. A reader who takes
2179
+ * "3/4 bullets passed" and "1 scenario regressed" as the same measurement will
2180
+ * draw the wrong conclusion from both: the first says whether the skill met its
2181
+ * rubric, the second says whether it changed anything relative to no skill at
2182
+ * all. Nothing here influences the exit code.
2183
+ */
2184
+ function printAttributionReport(outcomes: ReadonlyArray<ScenarioOutcome>): void {
2185
+ if (outcomes.length === 0) return;
2186
+ const summary = attributionSummary(outcomes);
2187
+ const compared = summary.helped + summary.regressed + summary.mixed + summary["no-change"];
2188
+ if (compared === 0) {
2189
+ const reason = outcomes.find((outcome) => outcome.attribution.reason !== null)?.attribution.reason ?? null;
2190
+ process.stdout.write(
2191
+ `attribution: no scenario was compared against its baseline${reason !== null ? ` (${reason})` : ""}\n`,
2192
+ );
2193
+ return;
2194
+ }
2195
+ const parts = [
2196
+ `${summary.helped} helped`,
2197
+ `${summary.regressed} regressed`,
2198
+ `${summary.mixed} mixed`,
2199
+ `${summary["no-change"]} no-change`,
2200
+ ];
2201
+ if (summary.unmeasured > 0) parts.push(`${summary.unmeasured} unmeasured`);
2202
+ if (summary["not-attempted"] > 0) parts.push(`${summary["not-attempted"]} not-attempted`);
2203
+ process.stdout.write(
2204
+ `attribution (advisory, vs the no-skill baseline; gates nothing): ${parts.join(", ")} of ${outcomes.length} scenario${outcomes.length === 1 ? "" : "s"}\n`,
2205
+ );
2206
+ for (const outcome of outcomes) {
2207
+ if (outcome.attribution.verdict === "regressed" || outcome.attribution.verdict === "mixed") {
2208
+ const regressed = outcome.attribution.bullets
2209
+ .filter((bullet) => bullet.attribution === "regressed")
2210
+ .map((bullet) => String(bullet.index));
2211
+ process.stdout.write(
2212
+ `${outcome.scenario.id}: ${outcome.attribution.verdict}; the baseline passed bullet${regressed.length === 1 ? "" : "s"} ${regressed.join(", ")} and the treatment did not\n`,
2213
+ );
2214
+ } else if (outcome.attribution.reason !== null) {
2215
+ process.stdout.write(
2216
+ `${outcome.scenario.id}: attribution ${outcome.attribution.verdict}: ${outcome.attribution.reason}\n`,
2217
+ );
2218
+ }
2219
+ }
2220
+ }
2221
+
1214
2222
  function preview(value: unknown): string {
1215
2223
  if (value === undefined) return "";
1216
2224
  const text = typeof value === "string" ? value : safeStringify(value);
@@ -1249,7 +2257,7 @@ function isRecord(value: unknown): value is Record<string, unknown> {
1249
2257
 
1250
2258
  /** The experimental evals.md lane stays under eval, alongside package suite evals. */
1251
2259
  export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<number> {
1252
- const options: SkillsEvalOptions = { json: false, trustFixtures: false, allowNetwork: false };
2260
+ const options: SkillsEvalOptions = { json: false, trustFixtures: false, allowNetwork: false, noAttribution: false };
1253
2261
  let source: string | undefined;
1254
2262
  try {
1255
2263
  for (let i = 0; i < args.length; i++) {
@@ -1258,6 +2266,7 @@ export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<numb
1258
2266
  if (arg === "--json") options.json = true;
1259
2267
  else if (arg === "--trust-fixtures") options.trustFixtures = true;
1260
2268
  else if (arg === "--allow-network") options.allowNetwork = true;
2269
+ else if (arg === "--no-attribution") options.noAttribution = true;
1261
2270
  else if (["--scenario", "--target", "--workspace", "--timeout"].includes(arg)) {
1262
2271
  const value = args[++i];
1263
2272
  if (!value || value.startsWith("-")) throw new Error(`${arg} requires a value`);
@@ -1270,7 +2279,8 @@ export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<numb
1270
2279
  else options.workspace = value;
1271
2280
  } else if (arg === "--help" || arg === "-h") {
1272
2281
  process.stdout.write(
1273
- "clio-coder eval skill <name|path> [--scenario <id>] [--target <id>] [--workspace <path>] [--timeout <seconds>] [--trust-fixtures] [--allow-network] [--json]\n",
2282
+ "clio-coder eval skill <name|path> [--scenario <id>] [--target <id>] [--workspace <path>] [--timeout <seconds>] [--trust-fixtures] [--allow-network] [--no-attribution] [--json]\n" +
2283
+ " --no-attribution skip the baseline judge and the paired comparison; saves one judge run per scenario\n",
1274
2284
  );
1275
2285
  return 0;
1276
2286
  } else if (arg.startsWith("-") || source) throw new Error(`unexpected skill eval argument: ${arg}`);