@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,767 +0,0 @@
1
- import { o as NotFoundError } from "./errors-D-LKuDhb.js";
2
- import { i as CostLedger } from "./cost-ledger-DMFxsLKr.js";
3
- import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-DzvMUsS_.js";
4
- import { o as computeFindingId } from "./usage-receipt-t7vAzCRQ.js";
5
- import { a as clamp01, i as aggregateRunScore } from "./proposal-findings-2GIUo1et.js";
6
- import { f as Mutex } from "./ledger-core-DXZIqu17.js";
7
- import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, statSync } from "node:fs";
8
- import { dirname, join } from "node:path";
9
- //#region src/analyst/define.ts
10
- /**
11
- * Define a reusable trace-research question.
12
- *
13
- * The returned value contains no model, credentials, or execution state. Bind
14
- * it to any TraceAnalysisEngine with `runTraceAnalyst` or
15
- * `createTraceAnalyst`.
16
- */
17
- function defineTraceAnalyst(definition) {
18
- for (const [name, value] of [
19
- ["id", definition.id],
20
- ["description", definition.description],
21
- ["area", definition.area],
22
- ["version", definition.version],
23
- ["instructions", definition.instructions]
24
- ]) if (typeof value !== "string" || !value.trim()) throw new TypeError(`defineTraceAnalyst: ${name} must be a non-empty string`);
25
- return {
26
- ...definition,
27
- limits: definition.limits ? { ...definition.limits } : void 0
28
- };
29
- }
30
- function defineCustomAnalyst(options) {
31
- if (!options.id.trim()) throw new TypeError("defineCustomAnalyst: id must not be empty");
32
- if (!options.description.trim()) throw new TypeError("defineCustomAnalyst: description must not be empty");
33
- if (options.cost === void 0) throw new TypeError("defineCustomAnalyst: cost must be declared");
34
- if ("executionConfig" in options && (!options.executionConfig || typeof options.executionConfig !== "object" || Array.isArray(options.executionConfig))) throw new TypeError("defineCustomAnalyst: executionConfig must be an object");
35
- return {
36
- id: options.id,
37
- description: options.description,
38
- version: options.version ?? "1.0.0",
39
- inputKind: "trace-store",
40
- cost: options.cost,
41
- analyze: options.analyze,
42
- ..."executionConfig" in options ? { executionConfig: options.executionConfig } : {}
43
- };
44
- }
45
- //#endregion
46
- //#region src/locked-jsonl-appender.ts
47
- /**
48
- * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
49
- * payloads. The reference-replay store does the same thing for typed
50
- * `ReferenceReplayRun` rows; this is the generic version used by
51
- * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants
52
- * append-only durable telemetry without rolling its own lock.
53
- *
54
- * Locks are per absolute file path (process-local). Cross-process
55
- * concurrency is NOT addressed — that's an fcntl/flock problem.
56
- */
57
- const mutexes = /* @__PURE__ */ new Map();
58
- function getMutex(path) {
59
- let m = mutexes.get(path);
60
- if (!m) {
61
- m = new Mutex();
62
- mutexes.set(path, m);
63
- }
64
- return m;
65
- }
66
- var LockedJsonlAppender = class {
67
- path;
68
- mutex;
69
- constructor(path) {
70
- this.path = path;
71
- this.mutex = getMutex(path);
72
- if (!existsSync(dirname(path))) mkdirSync(dirname(path), { recursive: true });
73
- }
74
- async append(entry) {
75
- const line = `${JSON.stringify(entry)}\n`;
76
- await this.mutex.runExclusive(() => {
77
- appendFileSync(this.path, line);
78
- });
79
- }
80
- };
81
- //#endregion
82
- //#region src/analyst/findings-store.ts
83
- /**
84
- * FindingsStore — durable persistence for AnalystFinding rows + a diff
85
- * helper so we can answer "what changed since the last run?" without
86
- * recomputing analysts.
87
- *
88
- * On-disk shape is JSONL: one finding per line, append-only, locked via
89
- * LockedJsonlAppender. Operators get crash-safety (no partial JSON),
90
- * cheap reads (sequential parse), and trivial backup (rsync the file).
91
- *
92
- * Reads are non-locking: a reader sees a consistent snapshot of all
93
- * fully-written lines and skips an incomplete trailing line if the
94
- * writer is mid-append. Cross-process locking is intentionally out of
95
- * scope (see locked-jsonl-appender.ts).
96
- *
97
- * The store is run-scoped: callers pass `runId` on append and on load,
98
- * which keeps multi-run files cleanly partitioned. The `diffFindings`
99
- * helper compares two run-id sets using stable `finding_id` semantics —
100
- * the diff is the cross-run signal the regression dashboard renders.
101
- */
102
- var FindingsStore = class {
103
- path;
104
- appender;
105
- constructor(path) {
106
- this.path = path;
107
- this.appender = new LockedJsonlAppender(path);
108
- }
109
- async append(runId, findings) {
110
- for (const f of findings) {
111
- const row = {
112
- ...f,
113
- run_id: runId
114
- };
115
- await this.appender.append(row);
116
- }
117
- }
118
- /** Load every persisted finding. Discards malformed trailing lines silently. */
119
- loadAll() {
120
- if (!existsSync(this.path)) return [];
121
- const raw = readFileSync(this.path, "utf8");
122
- if (!raw) return [];
123
- const out = [];
124
- for (const line of raw.split("\n")) {
125
- if (!line) continue;
126
- try {
127
- out.push(JSON.parse(line));
128
- } catch {}
129
- }
130
- return out;
131
- }
132
- /** Filter to a single run. */
133
- loadRun(runId) {
134
- return this.loadAll().filter((r) => r.run_id === runId);
135
- }
136
- };
137
- /**
138
- * Default materiality test. Deliberately narrow so LLM-reword churn
139
- * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
140
- */
141
- function defaultIsMaterial(a, b) {
142
- if (a.severity !== b.severity) return true;
143
- if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > .05) return true;
144
- if (a.evidence_refs.length !== b.evidence_refs.length) return true;
145
- return false;
146
- }
147
- /**
148
- * Diff two findings sets by stable finding_id. Callers typically load
149
- * the two run-id slices from the same store and pass them in.
150
- */
151
- function diffFindings(previous, current, policy = {}) {
152
- const isMaterial = policy.isMaterial ?? defaultIsMaterial;
153
- const prevById = new Map(previous.map((f) => [f.finding_id, f]));
154
- const curById = new Map(current.map((f) => [f.finding_id, f]));
155
- const appeared = [];
156
- const disappeared = [];
157
- const persisted = [];
158
- const changed = [];
159
- for (const [id, cur] of curById) {
160
- const prev = prevById.get(id);
161
- if (!prev) {
162
- appeared.push(cur);
163
- continue;
164
- }
165
- if (isMaterial(prev, cur)) changed.push({
166
- previous: prev,
167
- current: cur
168
- });
169
- else persisted.push(cur);
170
- }
171
- for (const [id, prev] of prevById) if (!curById.has(id)) disappeared.push(prev);
172
- return {
173
- appeared,
174
- disappeared,
175
- persisted,
176
- changed
177
- };
178
- }
179
- //#endregion
180
- //#region src/analyst/kinds/skill-usage.ts
181
- /**
182
- * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill
183
- * library + its trace corpus. Unlike the trace-store kinds (failure-mode,
184
- * improvement, ...) this kind calls no LLM: it mines real usage and skill
185
- * structure and emits findings by rule.
186
- *
187
- * It exists because the naive "Skill-tool invocation count" lies low — it
188
- * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor
189
- * logs under the parent), slash-command entry, local-script bypass, and
190
- * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero
191
- * direct invocations, yet only one was a genuine cut: the rest were
192
- * measurement-invisible or discovery-limited. This analyst encodes that
193
- * lesson as a multi-signal usage model so a cheap repeatable pass can keep
194
- * the library honest, and so the expensive audit workflow's verdicts can
195
- * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).
196
- *
197
- * Report-building (`buildSkillUsageReport`, an fs scan) is separated from
198
- * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs
199
- * once at the registry boundary and the rule logic stays unit-testable.
200
- */
201
- /** Anthropic's authoring guidance keeps SKILL.md short; past this with no
202
- * `references/` split the body burns context budget every session. */
203
- const BLOAT_LINE_THRESHOLD = 300;
204
- const TANGLE_PRIVATE_RE = /\b(cli-bridge|tangletools|ops-board|drew-gtr-pro|@tangle-network\/|~\/company|tangle\.tools|gtm-agent)\b|\bkimi\b|\btcloud\b/gi;
205
- const TRIGGER_RE = /triggers?\s*[:-]/i;
206
- function listSkillDirs(root) {
207
- if (!existsSync(root)) return [];
208
- const out = [];
209
- for (const entry of readdirSync(root, { withFileTypes: true })) {
210
- if (!entry.isDirectory() && !entry.isSymbolicLink()) continue;
211
- const skillMd = join(root, entry.name, "SKILL.md");
212
- if (existsSync(skillMd)) out.push({
213
- name: entry.name,
214
- path: skillMd
215
- });
216
- }
217
- return out;
218
- }
219
- function walkJsonl(dir, cap) {
220
- if (!existsSync(dir)) return [];
221
- const files = [];
222
- const stack = [dir];
223
- while (stack.length) {
224
- const cur = stack.pop();
225
- let entries;
226
- try {
227
- entries = readdirSync(cur, { withFileTypes: true });
228
- } catch {
229
- continue;
230
- }
231
- for (const e of entries) {
232
- const full = join(cur, e.name);
233
- if (e.isDirectory()) stack.push(full);
234
- else if (e.name.endsWith(".jsonl")) {
235
- files.push(full);
236
- if (cap > 0 && files.length >= cap) return files;
237
- }
238
- }
239
- }
240
- return files;
241
- }
242
- function frontmatterDescription(body) {
243
- const block = /^---\n([\s\S]*?)\n---/.exec(body)?.[1] ?? "";
244
- return /description:\s*(.+)/i.exec(block)?.[1] ?? "";
245
- }
246
- function countArtifacts(roots, name, aliases) {
247
- let n = 0;
248
- for (const root of roots) {
249
- const candidates = [join(root, ".evolve", name), ...aliases.map((a) => join(root, a))];
250
- for (const dir of candidates) {
251
- if (!existsSync(dir)) continue;
252
- try {
253
- if (statSync(dir).isDirectory()) n += readdirSync(dir).length;
254
- else n += 1;
255
- } catch {}
256
- }
257
- }
258
- return n;
259
- }
260
- /** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */
261
- function buildSkillUsageReport(config) {
262
- const skills = config.skillRoots.flatMap(({ root, kind }) => listSkillDirs(root).map((s) => ({
263
- ...s,
264
- kind
265
- })));
266
- const names = skills.map((s) => s.name);
267
- const direct = new Map(names.map((n) => [n, 0]));
268
- const slash = new Map(names.map((n) => [n, 0]));
269
- const skillRe = /"skill"\s*:\s*"([a-z0-9_:-]+)"/g;
270
- const cmdRe = /<command-name>\/?([a-z0-9_:-]+)<\/command-name>/g;
271
- let transcripts = 0;
272
- for (const dir of config.transcriptDirs) for (const file of walkJsonl(dir, config.maxTranscriptsPerDir ?? 0)) {
273
- transcripts += 1;
274
- let data;
275
- try {
276
- data = readFileSync(file, "utf8");
277
- } catch {
278
- continue;
279
- }
280
- for (const m of data.matchAll(skillRe)) {
281
- const g = m[1];
282
- if (!g) continue;
283
- const n = g.split(":").pop() ?? g;
284
- const prev = direct.get(n);
285
- if (prev !== void 0) direct.set(n, prev + 1);
286
- }
287
- for (const m of data.matchAll(cmdRe)) {
288
- const g = m[1];
289
- if (g === void 0) continue;
290
- const prev = slash.get(g);
291
- if (prev !== void 0) slash.set(g, prev + 1);
292
- }
293
- }
294
- const bodies = /* @__PURE__ */ new Map();
295
- for (const s of skills) try {
296
- bodies.set(s.name, readFileSync(s.path, "utf8"));
297
- } catch {
298
- bodies.set(s.name, "");
299
- }
300
- const inbound = new Map(names.map((n) => [n, 0]));
301
- for (const target of names) {
302
- const ref = new RegExp(`/${target}\\b|\\[\\[${target}\\]\\]`);
303
- for (const s of skills) {
304
- if (s.name === target) continue;
305
- if (ref.test(bodies.get(s.name) ?? "")) inbound.set(target, inbound.get(target) + 1);
306
- }
307
- }
308
- const records = skills.map((s) => {
309
- const body = bodies.get(s.name) ?? "";
310
- const dir = s.path.replace(/\/SKILL\.md$/, "");
311
- return {
312
- name: s.name,
313
- kind: s.kind,
314
- path: s.path,
315
- lines: body ? body.split("\n").length : 0,
316
- directInvocations: direct.get(s.name) ?? 0,
317
- slashInvocations: slash.get(s.name) ?? 0,
318
- inboundRefs: inbound.get(s.name) ?? 0,
319
- artifactCount: countArtifacts(config.artifactRoots ?? [], s.name, config.artifactAliases?.[s.name] ?? []),
320
- tanglePrivateRefs: (body.match(TANGLE_PRIVATE_RE) ?? []).length,
321
- hasReferencesDir: existsSync(join(dir, "references")),
322
- hasEvalsDir: existsSync(join(dir, "evals")),
323
- logsRuns: body.includes("skill-runs.jsonl"),
324
- hasTriggerPhrases: TRIGGER_RE.test(frontmatterDescription(body) || body.slice(0, 600))
325
- };
326
- });
327
- return {
328
- generatedFromTraces: transcripts,
329
- records
330
- };
331
- }
332
- const ANALYST_ID = "skill-usage";
333
- function finding(area, subject, claim, severity, confidence, producedAt, recommended, evidenceUri, rationale) {
334
- return {
335
- schema_version: "1.0.0",
336
- finding_id: computeFindingId({
337
- analyst_id: ANALYST_ID,
338
- area,
339
- subject,
340
- claim
341
- }),
342
- analyst_id: ANALYST_ID,
343
- produced_at: producedAt,
344
- severity,
345
- area,
346
- claim,
347
- rationale,
348
- evidence_refs: [{
349
- kind: "artifact",
350
- uri: evidenceUri
351
- }],
352
- recommended_action: recommended,
353
- confidence,
354
- subject
355
- };
356
- }
357
- /** Pure rule pass over a report → findings. Exported for direct/unit use. */
358
- function emitSkillUsageFindings(report, producedAt) {
359
- const out = [];
360
- for (const r of report.records) {
361
- const directTotal = r.directInvocations + r.slashInvocations;
362
- if (directTotal + r.inboundRefs + r.artifactCount === 0) out.push(finding("skill-usage", r.name, `Skill '${r.name}' has zero usage across all signals (direct, slash, inbound-refs, artifacts)`, "high", .6, producedAt, "Confirm the skill covers a real recurring job; if not, deprecate. Zero true usage is the only deterministic deprecation candidate.", r.path, "No Skill-tool call, no slash invocation, no sibling dispatches to it, and no on-disk artifacts."));
363
- else if (directTotal === 0 && r.inboundRefs + r.artifactCount > 0) out.push(finding("skill-usage", r.name, `Skill '${r.name}' shows 0 direct invocations but is used via orchestration/artifacts (inbound=${r.inboundRefs}, artifacts=${r.artifactCount})`, "info", .8, producedAt, "Do NOT treat as unused — usage is real but logged under parent skills or on disk. Strengthen direct-invocation discovery only if direct use is desired.", r.path, "The Skill-tool counter undercounts orchestrated/chained leaf skills."));
364
- if (directTotal <= 2 && !r.hasTriggerPhrases) out.push(finding("discoverability", r.name, `Skill '${r.name}' is rarely invoked directly and its description has no explicit trigger phrases`, "medium", .7, producedAt, "Add a `Triggers:` clause with verbatim user phrases to the frontmatter description so the model auto-invokes it.", r.path));
365
- if (r.kind === "public" && r.tanglePrivateRefs > 0) out.push(finding("safety", r.name, `Public skill '${r.name}' carries ${r.tanglePrivateRefs} Tangle-private reference(s)`, "high", .75, producedAt, "Sanitize incidental internal refs (cli-bridge/kimi/tcloud/~company/private repos) or relocate to a private repo. Verify @tangle-network/* refs are to PUBLISHED packages before treating as a leak.", r.path));
366
- if (r.lines > BLOAT_LINE_THRESHOLD && !r.hasReferencesDir) out.push(finding("maintainability", r.name, `Skill '${r.name}' is ${r.lines} lines with no references/ split (progressive disclosure)`, "medium", .8, producedAt, `Split detail into references/ loaded on demand; keep SKILL.md a short overview. ${r.lines} lines load into every session's context budget.`, r.path));
367
- if (!r.hasEvalsDir) out.push(finding("data-quality", r.name, `Skill '${r.name}' ships no evals/`, "low", .6, producedAt, "Add evals/evals.json with >=3 scenarios proving the skill beats baseline; gives regression coverage.", r.path));
368
- if (!r.logsRuns) out.push(finding("observability", r.name, `Skill '${r.name}' never appends to .evolve/skill-runs.jsonl`, "low", .55, producedAt, "Append one run line to .evolve/skill-runs.jsonl on completion, or declare it a non-logging leaf, so the self-improvement loop can see it ran.", r.path));
369
- }
370
- return out;
371
- }
372
- var SkillUsageAnalyst = class {
373
- id = ANALYST_ID;
374
- description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
375
- inputKind = "custom";
376
- cost = {
377
- kind: "deterministic",
378
- est_usd_per_run: 0
379
- };
380
- version = "1.0.0";
381
- executionConfig = {
382
- kind: "skill-usage",
383
- bloat_line_threshold: BLOAT_LINE_THRESHOLD,
384
- produced_at_source: "tags.producedAt-or-system-clock"
385
- };
386
- async analyze(input, ctx) {
387
- const producedAt = ctx.tags?.producedAt ?? (/* @__PURE__ */ new Date()).toISOString();
388
- ctx.log?.(`skill-usage: ${input.records.length} skills over ${input.generatedFromTraces} transcripts`);
389
- return emitSkillUsageFindings(input, producedAt);
390
- }
391
- };
392
- const SKILL_USAGE_ANALYST = new SkillUsageAnalyst();
393
- //#endregion
394
- //#region src/run-critic.ts
395
- const DEFAULT_DRIFT_PATTERNS = [
396
- /https?:\/\//i,
397
- /\btitle:\s/i,
398
- /\bsummary:\s/i,
399
- /\burl:\s/i,
400
- /\bnpm package usage\b/i,
401
- /\bnews\b/i
402
- ];
403
- var RunCritic = class {
404
- weights;
405
- driftPatterns;
406
- constructor(options = {}) {
407
- this.weights = options.weights;
408
- this.driftPatterns = options.driftPatterns ?? DEFAULT_DRIFT_PATTERNS;
409
- }
410
- async score(store, runId) {
411
- const run = await store.getRun(runId);
412
- if (!run) throw new NotFoundError(`run ${runId} not found`);
413
- const [spans, events, artifacts, budget] = await Promise.all([
414
- store.spans({ runId }),
415
- store.events({ runId }),
416
- store.artifacts(runId),
417
- store.budget(runId)
418
- ]);
419
- return this.scoreTrace({
420
- run,
421
- spans,
422
- events,
423
- artifacts,
424
- budget
425
- });
426
- }
427
- scoreTrace(trace) {
428
- const notes = [];
429
- const llmSpans = trace.spans.filter((s) => s.kind === "llm");
430
- const toolSpans = trace.spans.filter((s) => s.kind === "tool");
431
- const judgeSpans = trace.spans.filter((s) => s.kind === "judge");
432
- const sandboxSpans = trace.spans.filter((s) => s.kind === "sandbox");
433
- const finalGateSpans = judgeSpans.filter((span) => span.dimension === "final_gate" || span.attributes?.finalGate === true);
434
- const success = trace.run.outcome?.pass === true ? 1 : trace.run.status === "completed" ? .5 : 0;
435
- if (!success) notes.push("run did not complete with pass=true");
436
- const judgeAverage = judgeSpans.length ? judgeSpans.reduce((sum, span) => sum + normalizeJudgeScore(span.score), 0) / judgeSpans.length : void 0;
437
- const goalProgress = (typeof trace.run.outcome?.score === "number" ? clamp01(trace.run.outcome.score > 1 ? trace.run.outcome.score / 100 : trace.run.outcome.score) : void 0) ?? judgeAverage ?? success;
438
- const successfulTools = toolSpans.filter((span) => span.status !== "error").length;
439
- const toolUseQuality = toolSpans.length === 0 ? 0 : successfulTools / toolSpans.length;
440
- if (toolSpans.length === 0) notes.push("no tool spans recorded");
441
- const patchEvidence = trace.artifacts.length + toolSpans.filter((span) => /write|edit|patch|apply/i.test(span.toolName)).length;
442
- const patchQuality = patchEvidence > 0 ? clamp01(patchEvidence / 4) : 0;
443
- if (!patchQuality) notes.push("no artifact or edit evidence recorded");
444
- const sandboxTests = sandboxSpans.filter((span) => typeof span.testsTotal === "number" && span.testsTotal > 0);
445
- const testReality = sandboxTests.length ? sandboxTests.reduce((sum, span) => sum + (span.testsPassed ?? 0) / Math.max(1, span.testsTotal ?? 1), 0) / sandboxTests.length : toolSpans.some((span) => /\btest|vitest|pytest|jest|build|tsc\b/i.test(JSON.stringify(span.args))) ? .4 : 0;
446
- if (!testReality) notes.push("no real test/build evidence recorded");
447
- const blockerSpans = judgeSpans.filter((span) => isBlockingJudge(span));
448
- const finalGateBlockers = finalGateSpans.filter((span) => isBlockingJudge(span));
449
- const finalGate = finalGateSpans.length ? finalGateBlockers.length ? 0 : 1 : success;
450
- if (finalGateBlockers.length) notes.push(`final gate blocked by ${finalGateBlockers.length} reviewer(s)`);
451
- else if (!finalGateSpans.length) notes.push("no final gate judgment recorded");
452
- const reviewerBlockers = judgeSpans.length ? blockerSpans.length / judgeSpans.length : 0;
453
- if (reviewerBlockers) notes.push(`detected ${blockerSpans.length} blocking reviewer signal(s)`);
454
- const positiveGroundingSignals = patchEvidence + sandboxSpans.length + llmSpans.filter((span) => looksRepoGrounded(span.output ?? "")).length;
455
- const driftSignals = llmSpans.filter((span) => this.isDrift(span.output ?? "")).length + trace.events.filter((event) => this.isDrift(JSON.stringify(event.payload))).length;
456
- const repoGroundedness = positiveGroundingSignals + driftSignals === 0 ? 0 : positiveGroundingSignals / (positiveGroundingSignals + driftSignals);
457
- const driftPenalty = positiveGroundingSignals + driftSignals === 0 ? 0 : driftSignals / (positiveGroundingSignals + driftSignals);
458
- if (driftSignals > 0) notes.push(`detected ${driftSignals} drift signal(s)`);
459
- return {
460
- success,
461
- goalProgress,
462
- repoGroundedness,
463
- driftPenalty,
464
- toolUseQuality,
465
- patchQuality,
466
- testReality,
467
- finalGate,
468
- reviewerBlockers,
469
- costUsd: trace.budget.length ? Math.max(...trace.budget.filter((entry) => entry.dimension === "usd").map((entry) => entry.consumed), 0) : llmSpans.reduce((sum, span) => sum + (span.costUsd ?? 0), 0),
470
- wallSeconds: trace.run.endedAt && trace.run.startedAt ? Math.max(0, (trace.run.endedAt - trace.run.startedAt) / 1e3) : 0,
471
- notes
472
- };
473
- }
474
- rank(score) {
475
- return aggregateRunScore(score, this.weights);
476
- }
477
- isDrift(text) {
478
- return this.driftPatterns.some((pattern) => pattern.test(text));
479
- }
480
- };
481
- function normalizeJudgeScore(score) {
482
- return score > 1 ? clamp01(score / 10) : clamp01(score);
483
- }
484
- function looksRepoGrounded(text) {
485
- return /(?:src\/|tests?\/|package\.json|tsconfig|\.ts\b|\.tsx\b|git status|pnpm |npm |vitest|pytest|jest)/i.test(text);
486
- }
487
- function isBlockingJudge(span) {
488
- return span.attributes?.blocking === true || span.attributes?.verdict === "BLOCKING" || positiveNumber(span.attributes?.blockingFindings) || positiveNumber(span.attributes?.highFindings) || span.score <= 2;
489
- }
490
- function positiveNumber(value) {
491
- return typeof value === "number" && value > 0;
492
- }
493
- //#endregion
494
- //#region src/semantic-concept-judge.ts
495
- /**
496
- * Semantic concept judge — "does the built artifact actually implement
497
- * the features the user asked for?"
498
- *
499
- * Distinct from the domain/code/coherence judges in `judges.ts`:
500
- * - those judges score free-form conversational agent outputs along
501
- * quality dimensions (accuracy, depth, etc.)
502
- * - this judge scores a *built artifact* (served HTML + source files)
503
- * against an explicit list of expected concepts, returning per-concept
504
- * {present, score 0-10, evidence, severity}.
505
- *
506
- * The judge is strict about distinguishing (a) a working implementation
507
- * from (b) a keyword-present stub. "// TODO: mint button" is NOT present.
508
- * Only real, functional, wired-up code counts.
509
- *
510
- * Use via {@link createSemanticConceptJudge} or directly via
511
- * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM
512
- * or JSON-parse errors so the caller can treat that as "layer skipped"
513
- * rather than "layer failed" in a multi-layer pipeline.
514
- */
515
- const DEFAULT_COMPLEXITY_WEIGHTS = {
516
- render: 1,
517
- integrate: 2,
518
- compute: 2.5
519
- };
520
- const SEMANTIC_CONCEPT_JUDGE_VERSION = "semantic-concept-judge-v1-2026-04-24";
521
- const DEFAULT_MAX_SOURCE = 45e3;
522
- const DEFAULT_MAX_HTML = 3e4;
523
- const DEFAULT_MAX_PER_FILE = 2e4;
524
- const DEFAULT_TIMEOUT = 3e5;
525
- const DEFAULT_MAX_TOKENS = 16e3;
526
- const DEFAULT_MODEL = "claude-sonnet-4-6";
527
- const SEMANTIC_SCHEMA = {
528
- type: "object",
529
- additionalProperties: false,
530
- required: ["summary", "concepts"],
531
- properties: {
532
- summary: {
533
- type: "string",
534
- minLength: 20,
535
- maxLength: 600
536
- },
537
- concepts: {
538
- type: "array",
539
- minItems: 1,
540
- items: {
541
- type: "object",
542
- additionalProperties: false,
543
- required: [
544
- "concept",
545
- "present",
546
- "score",
547
- "evidence",
548
- "severity"
549
- ],
550
- properties: {
551
- concept: {
552
- type: "string",
553
- minLength: 1,
554
- maxLength: 120
555
- },
556
- present: { type: "boolean" },
557
- score: {
558
- type: "number",
559
- minimum: 0,
560
- maximum: 10
561
- },
562
- evidence: {
563
- type: "string",
564
- minLength: 5,
565
- maxLength: 400
566
- },
567
- severity: {
568
- type: "string",
569
- enum: [
570
- "critical",
571
- "major",
572
- "minor",
573
- "info"
574
- ]
575
- }
576
- }
577
- }
578
- }
579
- }
580
- };
581
- function truncate(body, cap, label) {
582
- if (body.length <= cap) return body;
583
- return `${body.slice(0, cap)}\n… [truncated ${body.length - cap} chars of ${label}]`;
584
- }
585
- function buildPrompt(input, opts) {
586
- const sourceBlob = input.sourceFiles.filter((f) => f.content.length <= opts.maxPerFileChars).map((f) => `--- FILE: ${f.path} ---\n${f.content}`).join("\n\n");
587
- const html = input.servedHtml ?? "";
588
- return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.
589
-
590
- You MUST distinguish:
591
- (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),
592
- (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),
593
- (c) ABSENT (concept nowhere).
594
-
595
- A comment like "// TODO: add mint button" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.
596
-
597
- USER REQUEST (what the agent was asked to build):
598
- ${input.userRequest}
599
-
600
- ${input.artifactLabel ? `ARTIFACT METADATA:\n name: ${input.artifactLabel}\n description: ${input.artifactDescription ?? ""}\n\n` : ""}EXPECTED CONCEPTS (each must be graded independently):
601
- ${input.expectedConcepts.map((c, i) => ` ${i + 1}. "${c.name}"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(" | ")}]` : ""}`).join("\n")}
602
-
603
- ${html ? `SERVED HTML (what the preview returns when hit):\n${truncate(html, opts.maxHtmlChars, "HTML")}\n\n` : ""}SOURCE FILES (the agent's workdir):
604
- ${truncate(sourceBlob, opts.maxSourceChars, "source")}
605
-
606
- For EACH concept, return:
607
- - concept: the concept name as given (match exactly)
608
- - present: boolean — does a working implementation exist?
609
- - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent
610
- - evidence: cite "<file>:<line>" or "served-html:<selector>" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.
611
- - severity:
612
- "info" when present: true AND score >= 7
613
- "minor" when present: true AND 4 <= score < 7
614
- "major" when present: false OR score < 4
615
- "critical" when the concept is not only absent but a core user flow depends on it
616
-
617
- Also produce a "summary" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.
618
-
619
- BE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.
620
-
621
- Return STRICT JSON. No prose outside the JSON.`;
622
- }
623
- /**
624
- * Run the semantic concept judge. Soft-fails to available=false on
625
- * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat
626
- * that as "skip" rather than "fail."
627
- */
628
- async function runSemanticConceptJudge(input, options = {}) {
629
- const start = Date.now();
630
- const totalCount = input.expectedConcepts.length;
631
- if (totalCount === 0) return {
632
- kind: "semantic-concept",
633
- version: SEMANTIC_CONCEPT_JUDGE_VERSION,
634
- score: 0,
635
- presentCount: 0,
636
- totalCount: 0,
637
- findings: [],
638
- summary: "no expected concepts declared",
639
- durationMs: 0,
640
- costUsd: null,
641
- available: false,
642
- error: "no expected concepts declared"
643
- };
644
- const opts = {
645
- model: options.model ?? DEFAULT_MODEL,
646
- timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
647
- maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
648
- maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
649
- maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
650
- maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
651
- llm: options.llm ?? {},
652
- costLedger: options.costLedger ?? new CostLedger(),
653
- costPhase: options.costPhase ?? "judge.semantic-concept",
654
- costTags: options.costTags ?? {},
655
- signal: options.signal ?? new AbortController().signal,
656
- weightConcepts: options.weightConcepts ?? "mean",
657
- complexityWeights: {
658
- ...DEFAULT_COMPLEXITY_WEIGHTS,
659
- ...options.complexityWeights ?? {}
660
- }
661
- };
662
- const weightForConcept = (spec) => {
663
- if (opts.weightConcepts === "mean") return 1;
664
- if (spec.weight != null) return spec.weight;
665
- if (opts.weightConcepts === "complexity") return opts.complexityWeights[spec.complexity ?? "render"] ?? 1;
666
- return 1;
667
- };
668
- const weightByName = new Map(input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]));
669
- let receipt;
670
- try {
671
- const request = {
672
- model: opts.model,
673
- messages: [{
674
- role: "system",
675
- content: "You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation."
676
- }, {
677
- role: "user",
678
- content: buildPrompt(input, opts)
679
- }],
680
- jsonSchema: {
681
- name: "semantic_concept_judge",
682
- schema: SEMANTIC_SCHEMA
683
- },
684
- temperature: 0,
685
- maxTokens: opts.maxTokens,
686
- timeoutMs: opts.timeoutMs
687
- };
688
- const paid = await opts.costLedger.runPaidCall({
689
- channel: "judge",
690
- phase: opts.costPhase,
691
- actor: "semantic-concept",
692
- model: opts.model,
693
- ...Object.keys(opts.costTags).length > 0 ? { tags: opts.costTags } : {},
694
- maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
695
- signal: opts.signal,
696
- execute: (signal, callId) => callLlmJson(request, {
697
- ...opts.llm,
698
- signal,
699
- idempotencyKey: callId
700
- }),
701
- receipt: ({ result }) => costReceiptFromLlm(result),
702
- receiptFromError: costReceiptFromLlmError
703
- });
704
- receipt = paid.receipt;
705
- if (!paid.succeeded) throw paid.error;
706
- const { value } = paid.value;
707
- if (!value?.concepts || !Array.isArray(value.concepts)) throw new Error("judge returned malformed response — expected array under \"concepts\"");
708
- const findings = value.concepts.map((c) => ({
709
- concept: String(c.concept),
710
- present: Boolean(c.present),
711
- score: Math.max(0, Math.min(10, Number(c.score ?? 0))),
712
- evidence: String(c.evidence ?? ""),
713
- severity: [
714
- "critical",
715
- "major",
716
- "minor",
717
- "info"
718
- ].includes(c.severity) ? c.severity : "info"
719
- }));
720
- const presentCount = findings.filter((f) => f.present && f.score >= 7).length;
721
- let weightSum = 0;
722
- let weightedScoreSum = 0;
723
- for (const f of findings) {
724
- const w = weightByName.get(f.concept) ?? 1;
725
- weightSum += w;
726
- weightedScoreSum += w * f.score;
727
- }
728
- const scoreAvg = weightSum > 0 ? weightedScoreSum / weightSum : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length);
729
- return {
730
- kind: "semantic-concept",
731
- version: SEMANTIC_CONCEPT_JUDGE_VERSION,
732
- score: Number((scoreAvg / 10).toFixed(3)),
733
- presentCount,
734
- totalCount,
735
- findings,
736
- summary: String(value.summary ?? ""),
737
- durationMs: Date.now() - start,
738
- costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
739
- available: true
740
- };
741
- } catch (err) {
742
- return {
743
- kind: "semantic-concept",
744
- version: SEMANTIC_CONCEPT_JUDGE_VERSION,
745
- score: 0,
746
- presentCount: 0,
747
- totalCount,
748
- findings: [],
749
- summary: "",
750
- durationMs: Date.now() - start,
751
- costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
752
- available: false,
753
- error: err instanceof Error ? err.message : String(err)
754
- };
755
- }
756
- }
757
- /**
758
- * Factory: pin LLM options once, return a closure that accepts inputs.
759
- * Convenient for pipelines that want to share a single LlmClient config.
760
- */
761
- function createSemanticConceptJudge(options = {}) {
762
- return (input) => runSemanticConceptJudge(input, options);
763
- }
764
- //#endregion
765
- export { RunCritic as a, buildSkillUsageReport as c, defaultIsMaterial as d, diffFindings as f, defineTraceAnalyst as h, runSemanticConceptJudge as i, emitSkillUsageFindings as l, defineCustomAnalyst as m, SEMANTIC_CONCEPT_JUDGE_VERSION as n, SKILL_USAGE_ANALYST as o, LockedJsonlAppender as p, createSemanticConceptJudge as r, SkillUsageAnalyst as s, DEFAULT_COMPLEXITY_WEIGHTS as t, FindingsStore as u };
766
-
767
- //# sourceMappingURL=semantic-concept-judge-D1z-KepS.js.map