@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,388 +1,343 @@
1
- import {
2
- FindingsStore,
3
- RunCritic,
4
- SEMANTIC_CONCEPT_JUDGE_VERSION,
5
- SKILL_USAGE_ANALYST,
6
- SkillUsageAnalyst,
7
- buildSkillUsageReport,
8
- defaultIsMaterial,
9
- diffFindings,
10
- emitSkillUsageFindings,
11
- runSemanticConceptJudge
12
- } from "../chunk-ZUUWPZCV.js";
13
- import {
14
- ANALYST_SEVERITIES,
15
- AnalystRegistry,
16
- CanonicalRawAnalystFindingSchema,
17
- DEFAULT_TRACE_ANALYST_KINDS,
18
- FAILURE_MODE_KIND_SPEC,
19
- FINDING_SUBJECT_GRAMMAR_PROMPT,
20
- FINDING_SUBJECT_KINDS,
21
- FINDING_SUBJECT_SYNTAX,
22
- FindingSubjectStringSchema,
23
- IMPROVEMENT_KIND_SPEC,
24
- KIND_EXPECTED_SUBJECTS,
25
- KNOWLEDGE_GAP_KIND_SPEC,
26
- KNOWLEDGE_POISONING_KIND_SPEC,
27
- RAW_FINDING_SCHEMA_PROMPT,
28
- RawAnalystEvidenceSchema,
29
- RawAnalystFindingSchema,
30
- behavioralAnalyst,
31
- buildDefaultAnalystRegistry,
32
- buildTraceToolsForGroup,
33
- coerceJson,
34
- coerceToFindingRows,
35
- computeFindingId,
36
- createAnalystAi,
37
- createChatClient,
38
- createTraceAnalystKind,
39
- deriveEfficiencyFindings,
40
- evidenceRefsFromRawFinding,
41
- findingSubjectGrammarPromptFor,
42
- makeFinding,
43
- parseCanonicalRawFinding,
44
- parseFindingSubject,
45
- parseRawFinding,
46
- renderFindingSubject,
47
- renderPriorFindings,
48
- renderUpstreamFindings,
49
- settleUsageReceiptFromCostLedger,
50
- stripCodeFences,
51
- structureFindings,
52
- validateUsageSettlementTimeout
53
- } from "../chunk-DJKY2TSY.js";
54
- import "../chunk-HHWE3POT.js";
55
- import "../chunk-WGXIEX7P.js";
56
- import "../chunk-PBE2LOSS.js";
57
- import {
58
- CostLedger
59
- } from "../chunk-WS3NZZQQ.js";
60
- import "../chunk-VI2UW6B6.js";
61
- import "../chunk-P6FYH6K4.js";
62
- import "../chunk-PC4UYEBM.js";
63
- import "../chunk-ONWEPEDO.js";
64
- import "../chunk-K4DBDHLK.js";
65
- import "../chunk-PZ5AY32C.js";
66
-
67
- // src/analyst/adapters.ts
68
- var ADAPTER_REV = "1";
1
+ import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as makeFinding, L as createChatClient, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, P as computeFindingId, R as createAnalystAi, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-C-vFCSEc.js";
2
+ import { i as CostLedger } from "../cost-ledger-DIgQUFZZ.js";
3
+ import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-B6cWNJ2K.js";
4
+ //#region src/analyst/adapters.ts
5
+ /**
6
+ * Adapter factories — lift each existing agent-eval primitive into the
7
+ * Analyst contract without re-implementing it.
8
+ *
9
+ * Five primitives, five factories. Each one:
10
+ * - Builds an Analyst with a stable id (caller chooses; defaults
11
+ * given), a sensible default `inputKind`, a version derived from
12
+ * the wrapped primitive's version + an adapter revision, and an
13
+ * `analyze()` that calls the primitive and lifts its output to
14
+ * AnalystFinding[] using `makeFinding()`.
15
+ * - Maps severities: the existing `Severity` ('critical' | 'major' |
16
+ * 'minor' | 'info') projects onto AnalystSeverity ('critical' |
17
+ * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →
18
+ * 'medium'. Domain analysts that want finer-grained mapping override.
19
+ *
20
+ * Adapters never own state. Calling the same factory twice with the
21
+ * same primitive instance is safe.
22
+ */
23
+ const ADAPTER_REV = "1";
69
24
  function liftSeverity(s) {
70
- switch (s) {
71
- case "critical":
72
- return "critical";
73
- case "major":
74
- return "high";
75
- case "minor":
76
- return "medium";
77
- case "info":
78
- return "info";
79
- }
25
+ switch (s) {
26
+ case "critical": return "critical";
27
+ case "major": return "high";
28
+ case "minor": return "medium";
29
+ case "info": return "info";
30
+ }
80
31
  }
81
32
  function createVerifierAdapter(opts) {
82
- const id = opts.id ?? "multi-layer-verifier";
83
- const area = opts.area ?? "verification";
84
- return {
85
- id,
86
- description: "Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.",
87
- inputKind: "custom",
88
- cost: { kind: "deterministic" },
89
- version: `verifier-${ADAPTER_REV}`,
90
- async analyze(env, ctx) {
91
- const report = await opts.verifier.run({ env, ...opts.options });
92
- const out = [];
93
- for (const layer of report.layers) {
94
- for (const finding of layer.findings) {
95
- out.push(liftLayerFinding(id, area, layer.layer, finding));
96
- }
97
- if (layer.status === "fail" || layer.status === "error" || layer.status === "timeout") {
98
- out.push(
99
- makeFinding({
100
- analyst_id: id,
101
- area,
102
- subject: layer.layer,
103
- claim: `layer "${layer.layer}" ${layer.status}: ${layer.reason ?? "no reason given"}`,
104
- severity: layer.status === "error" ? "high" : layer.status === "timeout" ? "medium" : "high",
105
- confidence: 1,
106
- evidence_refs: [],
107
- metadata: {
108
- layer_status: layer.status,
109
- duration_ms: layer.durationMs,
110
- score: layer.score,
111
- diagnostics: layer.diagnostics
112
- }
113
- })
114
- );
115
- }
116
- }
117
- ctx.log?.("verifier complete", {
118
- layers: report.layers.length,
119
- blended: report.blendedScore,
120
- all_pass: report.allPass
121
- });
122
- return out;
123
- }
124
- };
33
+ const id = opts.id ?? "multi-layer-verifier";
34
+ const area = opts.area ?? "verification";
35
+ return {
36
+ id,
37
+ description: "Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.",
38
+ inputKind: "custom",
39
+ cost: { kind: "deterministic" },
40
+ version: `verifier-${ADAPTER_REV}`,
41
+ async analyze(env, ctx) {
42
+ const report = await opts.verifier.run({
43
+ env,
44
+ ...opts.options
45
+ });
46
+ const out = [];
47
+ for (const layer of report.layers) {
48
+ for (const finding of layer.findings) out.push(liftLayerFinding(id, area, layer.layer, finding));
49
+ if (layer.status === "fail" || layer.status === "error" || layer.status === "timeout") out.push(makeFinding({
50
+ analyst_id: id,
51
+ area,
52
+ subject: layer.layer,
53
+ claim: `layer "${layer.layer}" ${layer.status}: ${layer.reason ?? "no reason given"}`,
54
+ severity: layer.status === "error" ? "high" : layer.status === "timeout" ? "medium" : "high",
55
+ confidence: 1,
56
+ evidence_refs: [],
57
+ metadata: {
58
+ layer_status: layer.status,
59
+ duration_ms: layer.durationMs,
60
+ score: layer.score,
61
+ diagnostics: layer.diagnostics
62
+ }
63
+ }));
64
+ }
65
+ ctx.log?.("verifier complete", {
66
+ layers: report.layers.length,
67
+ blended: report.blendedScore,
68
+ all_pass: report.allPass
69
+ });
70
+ return out;
71
+ }
72
+ };
125
73
  }
126
74
  function liftLayerFinding(analyst_id, area, layer, f) {
127
- return makeFinding({
128
- analyst_id,
129
- area,
130
- subject: f.layer ?? layer,
131
- claim: f.message,
132
- severity: liftSeverity(f.severity),
133
- confidence: 0.85,
134
- evidence_refs: f.evidence ? [{ kind: "artifact", uri: "inline:evidence", excerpt: f.evidence }] : [],
135
- metadata: f.detail
136
- });
75
+ return makeFinding({
76
+ analyst_id,
77
+ area,
78
+ subject: f.layer ?? layer,
79
+ claim: f.message,
80
+ severity: liftSeverity(f.severity),
81
+ confidence: .85,
82
+ evidence_refs: f.evidence ? [{
83
+ kind: "artifact",
84
+ uri: "inline:evidence",
85
+ excerpt: f.evidence
86
+ }] : [],
87
+ metadata: f.detail
88
+ });
137
89
  }
138
90
  function createRunCriticAdapter(opts = {}) {
139
- const id = opts.id ?? "run-critic";
140
- const area = opts.area ?? "run-quality";
141
- const critic = opts.critic ?? new RunCritic();
142
- const threshold = opts.threshold ?? 0.5;
143
- return {
144
- id,
145
- description: "Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.",
146
- inputKind: "custom",
147
- cost: { kind: "deterministic" },
148
- version: `run-critic-${ADAPTER_REV}`,
149
- async analyze(trace) {
150
- const score = critic.scoreTrace(trace);
151
- const out = [];
152
- const dims = [
153
- ["success", "critical", "run did not complete successfully"],
154
- ["goalProgress", "high", "goal progress is low"],
155
- ["repoGroundedness", "high", "output is poorly grounded in the repository"],
156
- ["toolUseQuality", "medium", "tool use quality is low"],
157
- ["patchQuality", "medium", "no real patch/edit evidence"],
158
- ["testReality", "high", "no real test/build evidence"],
159
- ["finalGate", "critical", "final gate is blocking"]
160
- ];
161
- for (const [dim, sev, msg] of dims) {
162
- const value = score[dim];
163
- if (typeof value === "number" && value < threshold) {
164
- out.push(
165
- makeFinding({
166
- analyst_id: id,
167
- area,
168
- subject: dim,
169
- claim: msg,
170
- rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,
171
- severity: sev,
172
- confidence: 1,
173
- evidence_refs: [],
174
- metadata: { dimension: dim, value, threshold, run_id: trace.run.runId }
175
- })
176
- );
177
- }
178
- }
179
- if (score.driftPenalty > 1 - threshold) {
180
- out.push(
181
- makeFinding({
182
- analyst_id: id,
183
- area,
184
- subject: "drift",
185
- claim: "agent output drifted from repository signal",
186
- rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,
187
- severity: "medium",
188
- confidence: 0.9,
189
- evidence_refs: [],
190
- metadata: { drift_penalty: score.driftPenalty, notes: score.notes }
191
- })
192
- );
193
- }
194
- return out;
195
- }
196
- };
91
+ const id = opts.id ?? "run-critic";
92
+ const area = opts.area ?? "run-quality";
93
+ const critic = opts.critic ?? new RunCritic();
94
+ const threshold = opts.threshold ?? .5;
95
+ return {
96
+ id,
97
+ description: "Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.",
98
+ inputKind: "custom",
99
+ cost: { kind: "deterministic" },
100
+ version: `run-critic-${ADAPTER_REV}`,
101
+ async analyze(trace) {
102
+ const score = critic.scoreTrace(trace);
103
+ const out = [];
104
+ for (const [dim, sev, msg] of [
105
+ [
106
+ "success",
107
+ "critical",
108
+ "run did not complete successfully"
109
+ ],
110
+ [
111
+ "goalProgress",
112
+ "high",
113
+ "goal progress is low"
114
+ ],
115
+ [
116
+ "repoGroundedness",
117
+ "high",
118
+ "output is poorly grounded in the repository"
119
+ ],
120
+ [
121
+ "toolUseQuality",
122
+ "medium",
123
+ "tool use quality is low"
124
+ ],
125
+ [
126
+ "patchQuality",
127
+ "medium",
128
+ "no real patch/edit evidence"
129
+ ],
130
+ [
131
+ "testReality",
132
+ "high",
133
+ "no real test/build evidence"
134
+ ],
135
+ [
136
+ "finalGate",
137
+ "critical",
138
+ "final gate is blocking"
139
+ ]
140
+ ]) {
141
+ const value = score[dim];
142
+ if (typeof value === "number" && value < threshold) out.push(makeFinding({
143
+ analyst_id: id,
144
+ area,
145
+ subject: dim,
146
+ claim: msg,
147
+ rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,
148
+ severity: sev,
149
+ confidence: 1,
150
+ evidence_refs: [],
151
+ metadata: {
152
+ dimension: dim,
153
+ value,
154
+ threshold,
155
+ run_id: trace.run.runId
156
+ }
157
+ }));
158
+ }
159
+ if (score.driftPenalty > 1 - threshold) out.push(makeFinding({
160
+ analyst_id: id,
161
+ area,
162
+ subject: "drift",
163
+ claim: "agent output drifted from repository signal",
164
+ rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,
165
+ severity: "medium",
166
+ confidence: .9,
167
+ evidence_refs: [],
168
+ metadata: {
169
+ drift_penalty: score.driftPenalty,
170
+ notes: score.notes
171
+ }
172
+ }));
173
+ return out;
174
+ }
175
+ };
197
176
  }
198
177
  function createJudgeAdapter(opts) {
199
- const id = opts.id ?? "judge";
200
- const area = opts.area ?? "judge";
201
- const threshold = opts.threshold ?? 6;
202
- return {
203
- id,
204
- description: "Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.",
205
- inputKind: "judge-input",
206
- cost: opts.cost ?? { kind: "llm" },
207
- version: `judge-${ADAPTER_REV}`,
208
- async analyze(input) {
209
- const scores = await opts.judge(opts.tcloud, input);
210
- return scores.filter((s) => normalize10(s.score) < threshold).map((s) => liftJudgeScore(id, area, s));
211
- }
212
- };
178
+ const id = opts.id ?? "judge";
179
+ const area = opts.area ?? "judge";
180
+ const threshold = opts.threshold ?? 6;
181
+ return {
182
+ id,
183
+ description: "Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.",
184
+ inputKind: "judge-input",
185
+ cost: opts.cost ?? { kind: "llm" },
186
+ version: `judge-${ADAPTER_REV}`,
187
+ async analyze(input) {
188
+ return (await opts.judge(opts.chat, input)).filter((s) => normalize10(s.score) < threshold).map((s) => liftJudgeScore(id, area, s));
189
+ }
190
+ };
213
191
  }
214
192
  function normalize10(s) {
215
- return s <= 1 ? s * 10 : s;
193
+ return s <= 1 ? s * 10 : s;
216
194
  }
217
195
  function liftJudgeScore(analyst_id, area, s) {
218
- const score10 = normalize10(s.score);
219
- const severity = score10 < 3 ? "critical" : score10 < 5 ? "high" : score10 < 7 ? "medium" : "low";
220
- return makeFinding({
221
- analyst_id,
222
- area,
223
- subject: s.dimension,
224
- claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,
225
- rationale: s.reasoning,
226
- severity,
227
- confidence: 0.8,
228
- evidence_refs: s.evidence ? [{ kind: "artifact", uri: "inline:evidence", excerpt: s.evidence }] : [],
229
- // Provenance: this finding IS a judge verdict (an acceptance score), not an
230
- // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects
231
- // it from steering — even when it cites an artifact above — because letting a
232
- // verdict steer the next attempt is the held-out judge leaking into the loop.
233
- derived_from_judge: true,
234
- metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 }
235
- });
196
+ const score10 = normalize10(s.score);
197
+ const severity = score10 < 3 ? "critical" : score10 < 5 ? "high" : score10 < 7 ? "medium" : "low";
198
+ return makeFinding({
199
+ analyst_id,
200
+ area,
201
+ subject: s.dimension,
202
+ claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,
203
+ rationale: s.reasoning,
204
+ severity,
205
+ confidence: .8,
206
+ evidence_refs: s.evidence ? [{
207
+ kind: "artifact",
208
+ uri: "inline:evidence",
209
+ excerpt: s.evidence
210
+ }] : [],
211
+ derived_from_judge: true,
212
+ metadata: {
213
+ judge_name: s.judgeName,
214
+ dimension: s.dimension,
215
+ score_10: score10
216
+ }
217
+ });
236
218
  }
237
219
  function createSemanticConceptJudgeAdapter(opts = {}) {
238
- const id = opts.id ?? "semantic-concept-judge";
239
- const area = opts.area ?? "concept-coverage";
240
- const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs);
241
- return {
242
- id,
243
- description: "Runs the semantic-concept judge and surfaces missing / weak concepts as findings.",
244
- inputKind: "custom",
245
- cost: {
246
- kind: "llm",
247
- models: opts.options?.model ? [opts.options.model] : void 0,
248
- settlement_timeout_ms: settlementTimeoutMs
249
- },
250
- version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,
251
- async analyze(input, ctx) {
252
- const costLedger = new CostLedger(ctx.budgetUsd);
253
- let result;
254
- try {
255
- result = await runSemanticConceptJudge(input, {
256
- ...opts.options,
257
- costLedger,
258
- signal: ctx.signal
259
- });
260
- } finally {
261
- const usage = await settleUsageReceiptFromCostLedger(costLedger, {
262
- channel: "judge",
263
- timeoutMs: settlementTimeoutMs
264
- });
265
- if (!usage.settled) {
266
- ctx.log?.("semantic-concept judge provider settlement timed out", {
267
- pending_calls: usage.pendingCalls,
268
- timeout_ms: settlementTimeoutMs
269
- });
270
- }
271
- ctx.recordUsage?.(usage.receipt);
272
- }
273
- if (!result.available) {
274
- return [
275
- makeFinding({
276
- analyst_id: id,
277
- area,
278
- claim: "semantic-concept judge unavailable",
279
- rationale: result.error,
280
- severity: "info",
281
- confidence: 1,
282
- evidence_refs: [],
283
- metadata: { reason: result.error }
284
- })
285
- ];
286
- }
287
- const out = [];
288
- for (const f of result.findings) {
289
- if (f.present && f.score >= 7) continue;
290
- out.push(
291
- makeFinding({
292
- analyst_id: id,
293
- area,
294
- subject: f.concept,
295
- claim: f.present ? `concept "${f.concept}" is weak (${f.score}/10)` : `concept "${f.concept}" is missing`,
296
- rationale: f.evidence,
297
- severity: liftSeverity(f.severity),
298
- confidence: 0.85,
299
- evidence_refs: [{ kind: "artifact", uri: "inline:evidence", excerpt: f.evidence }],
300
- metadata: {
301
- concept: f.concept,
302
- present: f.present,
303
- score_10: f.score
304
- }
305
- })
306
- );
307
- }
308
- return out;
309
- }
310
- };
220
+ const id = opts.id ?? "semantic-concept-judge";
221
+ const area = opts.area ?? "concept-coverage";
222
+ const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs);
223
+ return {
224
+ id,
225
+ description: "Runs the semantic-concept judge and surfaces missing / weak concepts as findings.",
226
+ inputKind: "custom",
227
+ cost: {
228
+ kind: "llm",
229
+ models: opts.options?.model ? [opts.options.model] : void 0,
230
+ settlement_timeout_ms: settlementTimeoutMs
231
+ },
232
+ version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,
233
+ async analyze(input, ctx) {
234
+ const costLedger = new CostLedger(ctx.budgetUsd);
235
+ let result;
236
+ try {
237
+ result = await runSemanticConceptJudge(input, {
238
+ ...opts.options,
239
+ costLedger,
240
+ signal: ctx.signal
241
+ });
242
+ } finally {
243
+ const usage = await settleUsageReceiptFromCostLedger(costLedger, {
244
+ channel: "judge",
245
+ timeoutMs: settlementTimeoutMs
246
+ });
247
+ if (!usage.settled) ctx.log?.("semantic-concept judge provider settlement timed out", {
248
+ pending_calls: usage.pendingCalls,
249
+ timeout_ms: settlementTimeoutMs
250
+ });
251
+ ctx.recordUsage?.(usage.receipt);
252
+ }
253
+ if (!result.available) return [makeFinding({
254
+ analyst_id: id,
255
+ area,
256
+ claim: "semantic-concept judge unavailable",
257
+ rationale: result.error,
258
+ severity: "info",
259
+ confidence: 1,
260
+ evidence_refs: [],
261
+ metadata: { reason: result.error }
262
+ })];
263
+ const out = [];
264
+ for (const f of result.findings) {
265
+ if (f.present && f.score >= 7) continue;
266
+ out.push(makeFinding({
267
+ analyst_id: id,
268
+ area,
269
+ subject: f.concept,
270
+ claim: f.present ? `concept "${f.concept}" is weak (${f.score}/10)` : `concept "${f.concept}" is missing`,
271
+ rationale: f.evidence,
272
+ severity: liftSeverity(f.severity),
273
+ confidence: .85,
274
+ evidence_refs: [{
275
+ kind: "artifact",
276
+ uri: "inline:evidence",
277
+ excerpt: f.evidence
278
+ }],
279
+ metadata: {
280
+ concept: f.concept,
281
+ present: f.present,
282
+ score_10: f.score
283
+ }
284
+ }));
285
+ }
286
+ return out;
287
+ }
288
+ };
311
289
  }
312
-
313
- // src/analyst/steer-firewall.ts
314
- var OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
315
- "span",
316
- "event",
317
- "artifact"
290
+ //#endregion
291
+ //#region src/analyst/steer-firewall.ts
292
+ /** Evidence grounded in the agent's OWN execution: OTLP trace elements
293
+ * (`span`/`event`) or the artifact it produced (`artifact`). */
294
+ const OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
295
+ "span",
296
+ "event",
297
+ "artifact"
318
298
  ]);
299
+ /** DESCRIPTIVE predicate: does the finding cite at least one observable
300
+ * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
301
+ * rendering — it is NOT the steer gate. Evidence presence is the WRONG
302
+ * discriminator for steering: a legitimate trace-analyst observation may cite
303
+ * nothing (it would be wrongly rejected), and a judge verdict may cite an
304
+ * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
305
+ * steering; use this only where "is this grounded in observable evidence" is the
306
+ * literal question. */
319
307
  function isTraceObservable(finding) {
320
- return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
308
+ return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
321
309
  }
310
+ /** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
311
+ * finding), identified by provenance set at the lift site — independent of
312
+ * whatever evidence it cites. */
322
313
  function isJudgeVerdict(finding) {
323
- return finding.derived_from_judge === true;
314
+ return finding.derived_from_judge === true;
324
315
  }
316
+ /**
317
+ * THE steer firewall. Fail-loud guard for any path that admits analyst findings
318
+ * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
319
+ * finding whose provenance is a judge verdict, rather than let `J` leak into the
320
+ * loop. Returns the findings unchanged for chaining.
321
+ *
322
+ * Call this at the chokepoint where a detector that ALSO scores/gates has its
323
+ * findings turned into a steer (the judge-and-steer dual-role case). It keys on
324
+ * provenance, so it correctly admits evidence-less trace-analyst observations and
325
+ * correctly rejects an artifact-citing judge verdict — the cases an evidence
326
+ * check gets backwards.
327
+ *
328
+ * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
329
+ * whose output is laundered through a hand-built finding with no provenance flag
330
+ * is out of its reach — provenance must be honestly set at every judge→finding
331
+ * lift (today: createJudgeAdapter). That is why the integrity rule lives at the
332
+ * lift site, and why ProposeContext.judgeScores?: never is the complementary
333
+ * compile-time tripwire on the obvious direct channel.
334
+ */
325
335
  function assertNoJudgeVerdict(findings, context = "steer") {
326
- const leaks = findings.filter(isJudgeVerdict);
327
- if (leaks.length > 0) {
328
- throw new Error(
329
- `${context}: a judge verdict cannot be admitted as steering input \u2014 that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`
330
- );
331
- }
332
- return findings;
336
+ const leaks = findings.filter(isJudgeVerdict);
337
+ if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
338
+ return findings;
333
339
  }
334
- export {
335
- ANALYST_SEVERITIES,
336
- AnalystRegistry,
337
- CanonicalRawAnalystFindingSchema,
338
- DEFAULT_TRACE_ANALYST_KINDS,
339
- FAILURE_MODE_KIND_SPEC,
340
- FINDING_SUBJECT_GRAMMAR_PROMPT,
341
- FINDING_SUBJECT_KINDS,
342
- FINDING_SUBJECT_SYNTAX,
343
- FindingSubjectStringSchema,
344
- FindingsStore,
345
- IMPROVEMENT_KIND_SPEC,
346
- KIND_EXPECTED_SUBJECTS,
347
- KNOWLEDGE_GAP_KIND_SPEC,
348
- KNOWLEDGE_POISONING_KIND_SPEC,
349
- RAW_FINDING_SCHEMA_PROMPT,
350
- RawAnalystEvidenceSchema,
351
- RawAnalystFindingSchema,
352
- SKILL_USAGE_ANALYST,
353
- SkillUsageAnalyst,
354
- assertNoJudgeVerdict,
355
- behavioralAnalyst,
356
- buildDefaultAnalystRegistry,
357
- buildSkillUsageReport,
358
- buildTraceToolsForGroup,
359
- coerceJson,
360
- coerceToFindingRows,
361
- computeFindingId,
362
- createAnalystAi,
363
- createChatClient,
364
- createJudgeAdapter,
365
- createRunCriticAdapter,
366
- createSemanticConceptJudgeAdapter,
367
- createTraceAnalystKind,
368
- createVerifierAdapter,
369
- defaultIsMaterial,
370
- deriveEfficiencyFindings,
371
- diffFindings,
372
- emitSkillUsageFindings,
373
- evidenceRefsFromRawFinding,
374
- findingSubjectGrammarPromptFor,
375
- isJudgeVerdict,
376
- isTraceObservable,
377
- liftSeverity,
378
- makeFinding,
379
- parseCanonicalRawFinding,
380
- parseFindingSubject,
381
- parseRawFinding,
382
- renderFindingSubject,
383
- renderPriorFindings,
384
- renderUpstreamFindings,
385
- stripCodeFences,
386
- structureFindings
387
- };
340
+ //#endregion
341
+ export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
342
+
388
343
  //# sourceMappingURL=index.js.map