@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -1,1027 +1,4 @@
1
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
2
- interface BudgetSpec {
3
- tokens?: number;
4
- wallMs?: number;
5
- calls?: number;
6
- usd?: number;
7
- }
8
- interface RunOutcome$1 {
9
- score?: number;
10
- pass?: boolean;
11
- failureClass?: FailureClass;
12
- notes?: string;
13
- }
14
- /**
15
- * Layer — optional classification in a nested build workflow.
16
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
17
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
18
- * `app-runtime`: a run of the generated agent against a domain scenario.
19
- * `meta`: any meta-eval (judge replay, correlation analysis).
20
- */
21
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
22
- interface Run {
23
- runId: string;
24
- /**
25
- * Stable identifier of the scenario being executed.
26
- *
27
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
28
- * input WITHOUT this field, substituting a sensible default
29
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
30
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
31
- * keeps the persisted shape unambiguous for downstream filters + aggregations
32
- * while removing the boilerplate of inventing placeholder ids at the call site.
33
- */
34
- scenarioId: string;
35
- variantId?: string;
36
- datasetVersion?: string;
37
- /** Git SHA of agent code at run time. */
38
- codeSha?: string;
39
- /** Hash of the prompt template + any system prompt. */
40
- promptSha?: string;
41
- /** Model id + date + system-prompt hash, concatenated. */
42
- modelFingerprint?: string;
43
- seed?: number;
44
- /** Arbitrary environment markers (shell, docker version, tz). */
45
- envFingerprint?: Record<string, string>;
46
- /** Version of the redaction rules applied to this run. */
47
- redactionVersion?: string;
48
- /** Parent run in a nested build workflow. A builder run's children are
49
- * app-build runs; those children are app-runtime runs. */
50
- parentRunId?: string;
51
- /** Stable project identifier — groups runs across chats + sessions. */
52
- projectId?: string;
53
- /** Chat/conversation identifier within a project. */
54
- chatId?: string;
55
- /** Layer classification — hint for aggregation; not enforced. */
56
- layer?: RunLayer;
57
- startedAt: number;
58
- endedAt?: number;
59
- status: RunStatus;
60
- outcome?: RunOutcome$1;
61
- budget?: BudgetSpec;
62
- /** Free-form labels for downstream grouping. */
63
- tags?: Record<string, string>;
64
- }
65
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
66
- type SpanStatus = 'ok' | 'error';
67
- interface SpanBase {
68
- spanId: string;
69
- parentSpanId?: string;
70
- runId: string;
71
- kind: SpanKind;
72
- name: string;
73
- startedAt: number;
74
- endedAt?: number;
75
- status?: SpanStatus;
76
- error?: string;
77
- /** Anything not covered by typed fields. Kept deliberately free-form. */
78
- attributes?: Record<string, unknown>;
79
- }
80
- interface Message {
81
- role: 'system' | 'user' | 'assistant' | 'tool';
82
- content: string;
83
- tokens?: number;
84
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
85
- images?: Array<{
86
- artifactId?: string;
87
- url?: string;
88
- mime?: string;
89
- }>;
90
- }
91
- interface LlmSpan extends SpanBase {
92
- kind: 'llm';
93
- model: string;
94
- messages: Message[];
95
- output?: string;
96
- inputTokens?: number;
97
- /** All generated tokens, including the reasoning subset when present. */
98
- outputTokens?: number;
99
- cachedTokens?: number;
100
- cacheWriteTokens?: number;
101
- /** Reasoning-token subset of `outputTokens`. */
102
- reasoningTokens?: number;
103
- costUsd?: number;
104
- finishReason?: string;
105
- }
106
- interface ToolSpan extends SpanBase {
107
- kind: 'tool';
108
- toolName: string;
109
- args: unknown;
110
- /** False when the source observed the call but did not capture its arguments. */
111
- argsCaptured?: boolean;
112
- result?: unknown;
113
- latencyMs?: number;
114
- }
115
- interface RetrievalSpan extends SpanBase {
116
- kind: 'retrieval';
117
- query: string;
118
- hits: Array<{
119
- docId: string;
120
- score: number;
121
- content?: string;
122
- }>;
123
- }
124
- interface JudgeSpan extends SpanBase {
125
- kind: 'judge';
126
- judgeId: string;
127
- /** Span this judgment applies to. */
128
- targetSpanId: string;
129
- dimension: string;
130
- /** Numeric score (free-range; interpretation up to the judge). */
131
- score: number;
132
- rationale?: string;
133
- evidence?: string;
134
- }
135
- interface SandboxSpan extends SpanBase {
136
- kind: 'sandbox';
137
- image?: string;
138
- command?: string;
139
- exitCode?: number;
140
- testsTotal?: number;
141
- testsPassed?: number;
142
- stdoutHash?: string;
143
- stderrHash?: string;
144
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
145
- wallMs?: number;
146
- }
147
- interface GenericSpan extends SpanBase {
148
- kind: 'agent' | 'custom';
149
- }
150
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
151
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
152
- interface TraceEvent {
153
- eventId: string;
154
- runId: string;
155
- spanId?: string;
156
- kind: EventKind;
157
- timestamp: number;
158
- payload: Record<string, unknown>;
159
- }
160
- interface BudgetLedgerEntry {
161
- runId: string;
162
- dimension: keyof BudgetSpec;
163
- limit: number;
164
- consumed: number;
165
- remaining: number;
166
- timestamp: number;
167
- breached: boolean;
168
- /** Span that triggered this entry, if any. */
169
- spanId?: string;
170
- }
171
- interface Artifact {
172
- artifactId: string;
173
- runId: string;
174
- spanId?: string;
175
- contentType: string;
176
- sizeBytes: number;
177
- /** sha256 in hex. */
178
- hash: string;
179
- /** External storage URL (R2, S3, filesystem path). */
180
- storageUrl?: string;
181
- /** Inline content for small blobs — keep under ~64KB. */
182
- inlineContent?: string;
183
- }
184
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
185
-
186
- interface RunFilter {
187
- scenarioId?: string;
188
- variantId?: string;
189
- status?: RunStatus;
190
- since?: number;
191
- until?: number;
192
- tag?: {
193
- key: string;
194
- value: string;
195
- };
196
- parentRunId?: string;
197
- projectId?: string;
198
- chatId?: string;
199
- layer?: RunLayer;
200
- }
201
- interface SpanFilter {
202
- runId?: string;
203
- parentSpanId?: string;
204
- kind?: SpanKind;
205
- name?: string;
206
- toolName?: string;
207
- judgeId?: string;
208
- since?: number;
209
- until?: number;
210
- }
211
- interface EventFilter {
212
- runId?: string;
213
- spanId?: string;
214
- kind?: EventKind;
215
- since?: number;
216
- until?: number;
217
- }
218
- interface TraceStore {
219
- appendRun(run: Run): Promise<void>;
220
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
221
- appendSpan(span: Span): Promise<void>;
222
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
223
- appendEvent(event: TraceEvent): Promise<void>;
224
- appendArtifact(artifact: Artifact): Promise<void>;
225
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
226
- getRun(runId: string): Promise<Run | undefined>;
227
- listRuns(filter?: RunFilter): Promise<Run[]>;
228
- spans(filter?: SpanFilter): Promise<Span[]>;
229
- events(filter?: EventFilter): Promise<TraceEvent[]>;
230
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
231
- artifacts(runId: string): Promise<Artifact[]>;
232
- }
233
-
234
- /**
235
- * OutcomeStore — deployment outcomes attached to Run IDs.
236
- *
237
- * Outcomes arrive asynchronously from production telemetry after the
238
- * eval run completed: user ratings, retention flags, conversion events,
239
- * revenue, support-ticket rate, anything a product team can measure.
240
- * The store is a peer to TraceStore — separate lifecycle, same runId
241
- * foreign key.
242
- *
243
- * The whole point of this module is to make the meta-eval correlation
244
- * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
245
- */
246
- interface DeploymentOutcome {
247
- runId: string;
248
- capturedAt: number;
249
- /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
250
- metrics: Record<string, number>;
251
- /** Dimensions for stratified analysis — cohort, region, user_segment. */
252
- labels?: Record<string, string>;
253
- /** Free-form provenance (source system, pipeline version). */
254
- source?: string;
255
- }
256
- interface OutcomeFilter {
257
- runIds?: string[];
258
- since?: number;
259
- until?: number;
260
- label?: {
261
- key: string;
262
- value: string;
263
- };
264
- source?: string;
265
- }
266
- interface OutcomeStore {
267
- append(outcome: DeploymentOutcome): Promise<void>;
268
- /** All outcomes attached to this run (a single run can have many — multiple
269
- * capture windows over deployment time). */
270
- forRun(runId: string): Promise<DeploymentOutcome[]>;
271
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
272
- }
273
- declare class InMemoryOutcomeStore implements OutcomeStore {
274
- private items;
275
- append(outcome: DeploymentOutcome): Promise<void>;
276
- forRun(runId: string): Promise<DeploymentOutcome[]>;
277
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
278
- }
279
- interface FileSystemOutcomeStoreOptions {
280
- dir: string;
281
- maxBytes?: number;
282
- }
283
- declare class FileSystemOutcomeStore implements OutcomeStore {
284
- private dir;
285
- private maxBytes;
286
- private memo?;
287
- private loaded;
288
- constructor(options: FileSystemOutcomeStoreOptions);
289
- private ensureDir;
290
- append(outcome: DeploymentOutcome): Promise<void>;
291
- private load;
292
- forRun(runId: string): Promise<DeploymentOutcome[]>;
293
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
294
- }
295
-
296
- /**
297
- * Correlation study — "does our eval score predict real-world outcomes?"
298
- *
299
- * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,
300
- * joins on runId, computes Pearson + Spearman + bootstrap CI for every
301
- * (evalMetric, outcomeMetric) pair the caller declares.
302
- *
303
- * Without this number the framework is ornamental. With it and r > 0.6
304
- * the framework is a moat — no other agent-eval tool publishes one.
305
- */
306
-
307
- interface EvalMetricSpec {
308
- id: string;
309
- /** Extract a scalar from a run (defaults cover score/pass/durationMs/costUsd/tokens). */
310
- extract?: (run: Run, store: TraceStore) => Promise<number | null>;
311
- }
312
- interface OutcomePair {
313
- evalMetric: string;
314
- outcomeMetric: string;
315
- }
316
- interface CorrelationResult {
317
- evalMetric: string;
318
- outcomeMetric: string;
319
- n: number;
320
- pearson: number;
321
- spearman: number;
322
- /** 95% bootstrap CI for Pearson. */
323
- pearsonCi95: {
324
- lower: number;
325
- upper: number;
326
- };
327
- /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */
328
- verdict: 'strong' | 'moderate' | 'weak';
329
- }
330
- interface CorrelationStudyResult {
331
- pairs: CorrelationResult[];
332
- joinedSamples: number;
333
- skippedRuns: number;
334
- }
335
- interface CorrelationStudyOptions {
336
- /** Only join outcomes captured within this window after run.startedAt. */
337
- maxCaptureLagMs?: number;
338
- /** Restrict to a subset of outcomes (cohort, region, source). */
339
- outcomeFilter?: OutcomeFilter;
340
- /** Which outcome per run to use when multiple exist. Default 'latest'. */
341
- reduction?: 'latest' | 'mean' | 'max';
342
- /** Bootstrap iterations for the CI. Default 500. */
343
- bootstrapIterations?: number;
344
- }
345
- declare function correlationStudy(traceStore: TraceStore, outcomeStore: OutcomeStore, evalMetrics: EvalMetricSpec[], outcomeMetricNames: string[], options?: CorrelationStudyOptions): Promise<CorrelationStudyResult>;
346
-
347
- /**
348
- * Calibration curve — binned "if eval says X, what does reality show?"
349
- *
350
- * Companion to correlationStudy. Raw correlation is a single number;
351
- * the calibration curve shows *where* the eval is well-calibrated vs
352
- * overconfident / underconfident. Buckets the eval metric, computes
353
- * mean outcome per bucket, reports expected-calibration-error (ECE).
354
- */
355
-
356
- interface CalibrationBin {
357
- lower: number;
358
- upper: number;
359
- n: number;
360
- evalMean: number;
361
- outcomeMean: number;
362
- /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */
363
- gap: number;
364
- }
365
- interface CalibrationReport {
366
- evalMetric: string;
367
- outcomeMetric: string;
368
- n: number;
369
- bins: CalibrationBin[];
370
- /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */
371
- ece: number;
372
- /** Max bin gap — upper bound on miscalibration. */
373
- maxGap: number;
374
- }
375
- interface CalibrationOptions {
376
- bins?: number;
377
- /** Equal-width (fixed bin edges) or equal-frequency (quantile bins). */
378
- binning?: 'equal-width' | 'equal-frequency';
379
- /** Clip eval values to [lo, hi] before binning. */
380
- range?: {
381
- lo: number;
382
- hi: number;
383
- };
384
- }
385
- interface CalibrationPair {
386
- evalScore: number;
387
- outcome: number;
388
- }
389
- declare function calibrationCurve(traceStore: TraceStore, outcomeStore: OutcomeStore, evalMetric: EvalMetricSpec, outcomeMetric: string, options?: CalibrationOptions): Promise<CalibrationReport | null>;
390
- declare function calibrationFromPairs(inputPairs: CalibrationPair[], evalMetric: string, outcomeMetric: string, options?: CalibrationOptions): CalibrationReport | null;
391
-
392
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
393
- type AgentProfileDimensionValue = string | number | boolean | null;
394
- interface AgentProfileSource {
395
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
396
- kind: string;
397
- /** sha256 over the canonical source profile object. */
398
- hash: string;
399
- }
400
- interface AgentProfileHarness {
401
- id: string;
402
- version?: string;
403
- hash?: string;
404
- }
405
- interface AgentProfileCell {
406
- schemaVersion: AgentProfileCellSchemaVersion;
407
- cellId: string;
408
- profileId: string;
409
- sourceProfile: AgentProfileSource;
410
- harness?: AgentProfileHarness;
411
- model?: string;
412
- promptHash?: string;
413
- dimensions?: Record<string, AgentProfileDimensionValue>;
414
- }
415
-
416
- /**
417
- * Paper-grade RunRecord schema + runtime validator.
418
- *
419
- * Every run that participates in a promotion gate, paper table, or
420
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
421
- * fields are exactly those the paper "Two Loops, Three Roles" requires
422
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
423
- * holdout split tag. A task score is optional because execution-only records
424
- * must preserve missing labels instead of converting errors into zero quality.
425
- *
426
- * This is intentionally NOT a replacement for the rich `Run` /
427
- * `ProposeReviewReport` / `ScenarioResult` types already in the
428
- * package. Those are runtime structures with full provenance. A
429
- * `RunRecord` is the analysis-time projection — the JSON-friendly
430
- * row you'd put in a parquet file or paste into a notebook.
431
- *
432
- * Validate at the boundary:
433
- *
434
- * const rec = validateRunRecord(rawJson) // throws on missing
435
- * const ok = isRunRecord(rawJson) // boolean check
436
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
437
- *
438
- * The validator runs in pure TS — zod is intentionally NOT a
439
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
440
- */
441
-
442
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
443
- * combined train+test pool that the optimizer is allowed to read. */
444
- type RunSplitTag = 'search' | 'dev' | 'holdout';
445
- /**
446
- * Explicit execution-lifecycle result for a run.
447
- *
448
- * This is separate from task quality (`outcome`) and failure classification.
449
- * Producers set it only from root-run or process evidence.
450
- */
451
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
452
- interface RunTokenUsage {
453
- input: number;
454
- /** All generated tokens charged as output, including reasoning tokens. */
455
- output: number;
456
- /** Reasoning-token subset of `output`, when the provider reports it. */
457
- reasoning?: number;
458
- /** Prompt tokens served from a provider cache. */
459
- cached?: number;
460
- /** Prompt tokens written into a provider cache. */
461
- cacheWrite?: number;
462
- }
463
- /**
464
- * How a run's USD amount was obtained.
465
- */
466
- type RunCostProvenance = {
467
- kind: 'observed';
468
- usd: number;
469
- } | {
470
- kind: 'estimated';
471
- usd: number;
472
- } | {
473
- kind: 'uncaptured';
474
- usd: null;
475
- };
476
- interface RunJudgeMetadata {
477
- model: string;
478
- promptVersion: string;
479
- /** [0,1] confidence the judge declared. Constant judge confidence
480
- * across many runs is a fallback signal (see `canary.ts`). */
481
- confidence: number;
482
- /** True if the judge degraded to a fallback path (rules-only,
483
- * prior-call cache, etc.). The canary uses this to alert. */
484
- fallback: boolean;
485
- }
486
- /**
487
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
488
- * judges over a multi-dimensional rubric.
489
- *
490
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
491
- * composite the gate uses. The full breakdown belongs here so consumers
492
- * can answer "which judge disagreed?", "which dimension dragged the
493
- * composite down?", and "did half the panel fail?" without re-running.
494
- *
495
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
496
- * `composite` are convenience projections — derivable but precomputed so
497
- * downstream IRR primitives (`interRaterReliability`,
498
- * `corpusInterRaterAgreement`) and reporters don't pay the same
499
- * aggregation twice.
500
- *
501
- * Fail-loud discipline: judges that errored out land in `failedJudges`
502
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
503
- * run); the explicit list makes a partial-failure recorded as such.
504
- */
505
- interface JudgeScoresRecord {
506
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
507
- perJudge: Record<string, Record<string, number>>;
508
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
509
- perDimMean: Record<string, number>;
510
- /** Composite mean across successful judges. Mirrors the task score only
511
- * when `failedJudges` is empty. */
512
- composite: number;
513
- /** Judges that errored or returned an unparseable verdict. Recorded
514
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
515
- * not inferred from missing keys in `perJudge`. */
516
- failedJudges?: string[];
517
- /** Free-form notes the judges emitted (joined across judges or
518
- * first-judge only — consumer's choice). */
519
- notes?: string;
520
- }
521
- interface RunOutcome {
522
- /** Score on the search/optimization split. Optional for holdout-only and
523
- * execution-only records. */
524
- searchScore?: number;
525
- /** Score on the held-out split. Optional for search-only and execution-only
526
- * records. When both scores are absent, the run is explicitly unlabeled. */
527
- holdoutScore?: number;
528
- /** Bag of any other metric the run produced — judge dimensions,
529
- * pass/fail counters, latency stats, etc. Numeric only — keeps
530
- * reporters honest. */
531
- raw: Record<string, number>;
532
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
533
- * judgements populate this; substrate primitives like
534
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
535
- * these records as input. Optional — single-judge or scalar-only
536
- * runs leave it unset. */
537
- judgeScores?: JudgeScoresRecord;
538
- /** Authenticity / realness verdict — did the run build the REAL thing on the
539
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
540
- * with an authenticity config populate it. Carried in the corpus so the
541
- * flywheel / off-policy learning can optimize for real completion, not gamed
542
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
543
- * must not count as a real success regardless of `score`. */
544
- realness?: {
545
- score: number;
546
- gated: boolean;
547
- reason?: string;
548
- };
549
- }
550
- /**
551
- * Mandatory paper-grade fields for a single evaluation run. Optional
552
- * fields are extension points; mandatory fields throw if missing.
553
- *
554
- * Hash discipline:
555
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
556
- * model (after any steering bundle merge).
557
- * - `configHash` is the sha256 of the effective run config (model,
558
- * temperature, tools, judges, splits). The pair (promptHash,
559
- * configHash) uniquely identifies an experiment cell.
560
- *
561
- * Model snapshot discipline:
562
- * - `model` MUST encode a snapshot version. Bare aliases like
563
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
564
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
565
- */
566
- interface RunRecord {
567
- /** UUID for the run. */
568
- runId: string;
569
- /** Logical experiment grouping (a treatment vs a baseline within
570
- * the same sweep should share `experimentId`). */
571
- experimentId: string;
572
- /** Stable identifier for the candidate (variant) being run. The
573
- * promotion gate compares two `candidateId`s on matched items. */
574
- candidateId: string;
575
- /** RNG seed for the run. Always recorded — silent re-seeding is
576
- * the most common cause of non-reproducible numbers. */
577
- seed: number;
578
- /** Model identifier WITH snapshot version. */
579
- model: string;
580
- /** sha256 of the effective prompt (post-steering). */
581
- promptHash: string;
582
- /** sha256 of the effective config. */
583
- configHash: string;
584
- /** Git SHA the harness was run from. */
585
- commitSha: string;
586
- /** End-to-end wall-clock duration in milliseconds. */
587
- wallMs: number;
588
- /** Time spent queued before execution started, if known. */
589
- queueMs?: number;
590
- /** Total USD cost, or null when the producer could not capture one. */
591
- costUsd: number | null;
592
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
593
- costProvenance: RunCostProvenance;
594
- /** Token usage breakdown. */
595
- tokenUsage: RunTokenUsage;
596
- /** Root-run or process terminal result. Never inferred from a child span. */
597
- terminalOutcome: RunTerminalOutcome;
598
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
599
- * or incomplete terminal result; never populated from a child span. */
600
- terminalFailureReason?: string;
601
- /** Judge-side metadata, if a judge was used. */
602
- judgeMetadata?: RunJudgeMetadata;
603
- /** Per-split scores + raw bag. */
604
- outcome: RunOutcome;
605
- /** Canonical task-failure class drawn from the shared
606
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
607
- * evidence. Execution errors belong in
608
- * `outcome.raw.execution_error_count`. */
609
- failureClass?: FailureClass;
610
- /** Free-form task-failure detail scoped under a non-success
611
- * `failureClass`. It is invalid without that class. */
612
- failureMode?: string;
613
- /** Which split this run was drawn from. */
614
- splitTag: RunSplitTag;
615
- /**
616
- * Stable scenario identifier the run observed or was scored against.
617
- * Comparison primitives match this identity rather than input order.
618
- */
619
- scenarioId: string;
620
- /**
621
- * Canonical identity for the agent profile cell that produced this row:
622
- * profile artifact hash plus optional harness/model/prompt/reporting
623
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
624
- * longitudinal reports by the complete source profile, not by a loose
625
- * candidate label or opaque config hash.
626
- */
627
- agentProfile?: AgentProfileCell;
628
- }
629
-
630
- /**
631
- * Rubric predictive validity — does our eval rubric predict deployment
632
- * outcomes?
633
- *
634
- * `correlationStudy` (already in this package) joins a `TraceStore` to an
635
- * `OutcomeStore` and computes Pearson + Spearman + bootstrap CI for each
636
- * (eval-metric, outcome-metric) pair. That answers "does X correlate with
637
- * Y at all." `rubricPredictiveValidity` is the campaign-shaped wrapper
638
- * around it: take a sequence of `RunRecord`s (the canonical campaign
639
- * artifact) and a `DeploymentOutcomeStore`, join on `runId`, return a
640
- * ranked verdict on every rubric whose dimension scores were captured in
641
- * `outcome.raw`.
642
- *
643
- * The point — quoting the methodology doc — is that **without this loop
644
- * every rubric is faith-based**. Once it's wired, you know which rubrics
645
- * have earned their promotion power and which ones are decoration.
646
- *
647
- * const validity = await rubricPredictiveValidity({
648
- * runs: lastQuarter,
649
- * outcomes: shipFlagOutcomeStore,
650
- * outcomeMetrics: ['revenue_lift', 'retention_30d', 'csat'],
651
- * rubrics: ['anti_slop', 'semantic_concept', 'tool_recovery'],
652
- * })
653
- * for (const r of validity.ranked) {
654
- * console.log(`${r.rubric} → ${r.bestOutcome}: ρ=${r.spearman.toFixed(2)}`)
655
- * }
656
- *
657
- * The function is intentionally read-only. Use the verdict to deprecate
658
- * decorative rubrics, re-weight composite scores, or trigger a
659
- * recalibration sweep when predictive validity drops below a threshold.
660
- */
661
-
662
- interface RubricPredictiveValidityInput {
663
- /**
664
- * Canonical campaign output. Each record's `outcome.raw[<rubricId>]`
665
- * provides the eval score; missing keys are silently skipped per pair.
666
- */
667
- runs: RunRecord[];
668
- outcomes: OutcomeStore;
669
- /**
670
- * Outcome metric names to evaluate against. Each must appear in at
671
- * least one `DeploymentOutcome.metrics` keyspace; pairs with too few
672
- * joined samples are excluded from the result.
673
- */
674
- outcomeMetrics: string[];
675
- /**
676
- * Rubric ids to evaluate. Must appear as keys in `RunRecord.outcome.raw`.
677
- * If omitted, every numeric key in `outcome.raw` across the run set is
678
- * treated as a rubric.
679
- */
680
- rubrics?: string[];
681
- /** Minimum joined-sample count before a pair is reported. Default 8. */
682
- minSamples?: number;
683
- /** Bootstrap resamples for CI. Default 500. */
684
- bootstrapResamples?: number;
685
- /** Random seed for the bootstrap (mulberry32). Default unset (Math.random). */
686
- seed?: number;
687
- /**
688
- * Reduction when multiple outcomes attach to one runId. Default `'latest'`
689
- * (most recently captured).
690
- */
691
- reduction?: 'latest' | 'mean' | 'max';
692
- }
693
- interface RubricOutcomePair {
694
- rubric: string;
695
- outcome: string;
696
- n: number;
697
- pearson: number;
698
- spearman: number;
699
- ci95: {
700
- low: number;
701
- high: number;
702
- };
703
- /**
704
- * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,
705
- * `decorative` < 0.4 in absolute correlation. A negative correlation
706
- * with a desired outcome is also `decorative` — actively misleading
707
- * is worse than uninformative.
708
- */
709
- verdict: 'load_bearing' | 'informative' | 'decorative';
710
- }
711
- interface RubricRanking {
712
- rubric: string;
713
- /** Outcome metric this rubric correlated best with. */
714
- bestOutcome: string;
715
- spearman: number;
716
- pearson: number;
717
- n: number;
718
- verdict: RubricOutcomePair['verdict'];
719
- }
720
- interface RubricPredictiveValidityReport {
721
- pairs: RubricOutcomePair[];
722
- /** Per-rubric best pair, sorted descending by |spearman|. */
723
- ranked: RubricRanking[];
724
- joinedSamples: number;
725
- skippedRuns: number;
726
- /** Rubrics that were declared but never produced a usable score. */
727
- rubricsWithoutData: string[];
728
- }
729
- declare function rubricPredictiveValidity(input: RubricPredictiveValidityInput): Promise<RubricPredictiveValidityReport>;
730
-
731
- /**
732
- * Judge calibration — measure judge quality against human gold + bias.
733
- *
734
- * Workflow:
735
- * 1. Build a golden set: {itemId, humanScore}[].
736
- * 2. Run candidate judges; each produces {itemId, score}.
737
- * 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
738
- * 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
739
- * κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
740
- * and bootstrap CIs — use this for fine-grained judges where rounding
741
- * to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
742
- * look "perfectly agreed" to integer κ).
743
- * 5. Run bias probes (positional, verbosity, self-preference) to
744
- * detect systematic score inflation.
745
- * 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
746
- * reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
747
- *
748
- * Returns actionable diagnostics, not a single number. Consumers then
749
- * decide whether to trust the judge, retrain it, or add a tie-breaker.
750
- */
751
- interface GoldenItem {
752
- itemId: string;
753
- humanScore: number;
754
- /** Optional group used for per-group bias audits (e.g. model-of-output family). */
755
- group?: string;
756
- }
757
- interface CandidateScore {
758
- itemId: string;
759
- score: number;
760
- /** Optional — enables positional-bias analysis (did order matter?). */
761
- positionOfAInput?: 'first' | 'second';
762
- }
763
- interface CalibrationResult {
764
- n: number;
765
- pearson: number;
766
- /** Cohen's κ with quadratic weights over integer-rounded scores. */
767
- kappa: number;
768
- /** Mean absolute error vs human. */
769
- mae: number;
770
- /** Worst-5 miscalibrations (largest |judge - human|). */
771
- worstItems: Array<{
772
- itemId: string;
773
- judge: number;
774
- human: number;
775
- delta: number;
776
- }>;
777
- }
778
- interface ContinuousAgreement {
779
- /** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
780
- weightedKappa: number;
781
- /** ICC(2,1): two-way random effects, absolute agreement, single rater. */
782
- icc: number;
783
- /** Pearson product-moment correlation (averaged over rater pairs if N>2). */
784
- pearson: number;
785
- /** Spearman rank correlation (averaged over rater pairs if N>2). */
786
- spearman: number;
787
- /** 95% bootstrap percentile CIs over items. */
788
- ci: {
789
- icc: [number, number];
790
- weightedKappa: [number, number];
791
- };
792
- /** Number of complete items (no NaN across raters). */
793
- n: number;
794
- /** Number of raters. */
795
- raters: number;
796
- }
797
- interface ContinuousCalibrationResult extends CalibrationResult {
798
- /** Cohen's κ_w computed on raw (un-rounded) scores. */
799
- weightedKappaContinuous: number;
800
- /** ICC(2,1) treating golden + candidate as two raters. */
801
- icc: number;
802
- spearman: number;
803
- ci: {
804
- icc: [number, number];
805
- weightedKappa: [number, number];
806
- };
807
- }
808
-
809
- /**
810
- * Series convergence — detects whether a sequence of scalar measurements
811
- * is stabilizing, drifting, or noisy.
812
- *
813
- * Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
814
- * about progress *within* a single run; this module is about drift
815
- * *across* runs (e.g. "are my nightly eval scores stabilizing?").
816
- *
817
- * Three signals:
818
- * - stabilized: last K values have low variance (< epsilon) — done
819
- * - drifting: recent trend is monotonic and beyond noise — regressing or improving
820
- * - noisy: neither — keep iterating, but flag as untrustworthy for gating
821
- */
822
- interface SeriesConvergenceOptions {
823
- /** Window size for "recent" analysis (default 5). */
824
- window?: number;
825
- /** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
826
- stableCv?: number;
827
- /** Minimum monotone run length to call drift (default 3). */
828
- driftRun?: number;
829
- }
830
- interface SeriesConvergenceResult {
831
- state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
832
- windowMean: number;
833
- windowCv: number;
834
- /** Longest monotonic run at the tail of the series (positive for up, negative for down). */
835
- tailRun: number;
836
- /** True when n ≥ window AND windowCv ≤ stableCv. */
837
- stable: boolean;
838
- }
839
-
840
- interface CorpusAgreementPerDimension extends ContinuousAgreement {
841
- dimension: string;
842
- /** Item IDs that contributed to this dimension's matrix (every judge scored them). */
843
- itemIds: string[];
844
- /** Judge IDs that contributed to this dimension's matrix. */
845
- judgeIds: string[];
846
- }
847
- interface CorpusAgreementReport {
848
- /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
849
- perDimension: CorpusAgreementPerDimension[];
850
- /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
851
- overallIcc: number;
852
- /** Mean weighted κ across dimensions (NaN if none finite). */
853
- overallWeightedKappa: number;
854
- /** Dimensions evaluated (sorted). */
855
- dimensions: string[];
856
- /** Judges seen across the corpus (sorted). */
857
- judgeIds: string[];
858
- }
859
-
860
- /**
861
- * Judge sentinel — eval trustworthiness as a continuously measured,
862
- * alarmed trend.
863
- *
864
- * Judges are models; models change underneath us; calibration decays.
865
- * This module composes three existing instruments into one loop:
866
- *
867
- * snapshot — adapters turn real instrument outputs (`calibrateJudge` /
868
- * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /
869
- * `continuousAgreement`, a rerun sentinel golden set) into
870
- * `SentinelSnapshot`s
871
- * series — `analyzeSeries` (src/series-convergence) classifies each
872
- * judge×metric history as stabilized / drifting / noisy
873
- * alarm — `judgeSentinelReport` turns trends + thresholds into named
874
- * alarms; `evalHealthStamp` is the tiny object a campaign
875
- * attaches to its verdicts
876
- *
877
- * Alarm conditions:
878
- * - convergence state `drifting-down` on any tracked metric
879
- * - irr below `minIrr`
880
- * - calibrationKappa / sentinelPassRate dropped vs the series baseline
881
- * beyond `maxKappaDrop` / `maxSentinelDrop`
882
- * - judge model changed with no post-change golden-grounded snapshot
883
- * (the silent-upgrade trap: judges agreeing with each other after a
884
- * model swap proves nothing — only re-measuring against gold does)
885
- * - newest snapshot older than `staleAfterDays` relative to `asOf`
886
- *
887
- * Wire-in: run the snapshot adapters from a post-campaign hook or a
888
- * nightly job, append to a `SentinelStore`, then compute
889
- * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =
890
- * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign
891
- * verdicts: an alarmed sentinel means every downstream verdict carries
892
- * an integrity warning until the judge is recalibrated.
893
- *
894
- * The module is clock-free — every timestamp (`at`, `asOf`) is
895
- * caller-supplied ISO-8601, so reports are reproducible.
896
- */
897
-
898
- declare const SENTINEL_METRIC_NAMES: readonly ["irr", "calibrationKappa", "sentinelPassRate"];
899
- type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number];
900
- interface SentinelMetrics {
901
- /** Inter-rater reliability — ICC(2,1) from agreement instruments. */
902
- irr?: number;
903
- /** Weighted κ vs the human golden set (`calibrateJudge*`). */
904
- calibrationKappa?: number;
905
- /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */
906
- sentinelPassRate?: number;
907
- }
908
- interface SentinelSnapshot {
909
- /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */
910
- at: string;
911
- judgeId: string;
912
- /** Exact model identity behind the judge (pin the full version string). */
913
- judgeModel: string;
914
- /**
915
- * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a
916
- * model-change marker — it records the new `judgeModel` immediately and
917
- * the silent-upgrade alarm stays raised until a golden-grounded snapshot
918
- * (calibrationKappa or sentinelPassRate) follows.
919
- */
920
- metrics: SentinelMetrics;
921
- }
922
- /** Identity + timestamp the caller supplies alongside an instrument output. */
923
- interface SnapshotMeta {
924
- at: string;
925
- judgeId: string;
926
- judgeModel: string;
927
- }
928
- /** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */
929
- declare function validateSentinelSnapshot(snapshot: SentinelSnapshot, source?: string): void;
930
- /**
931
- * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the
932
- * un-rounded continuous κ_w when the report carries one (integer κ
933
- * discards information for [0,1] judges).
934
- */
935
- declare function snapshotFromCalibration(report: CalibrationResult | ContinuousCalibrationResult, meta: SnapshotMeta): SentinelSnapshot;
936
- /**
937
- * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`
938
- * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability
939
- * coefficient both instruments compute. For ensemble-level agreement,
940
- * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its
941
- * composition (e.g. a joined list of member model strings).
942
- */
943
- declare function snapshotFromAgreement(report: CorpusAgreementReport | ContinuousAgreement, meta: SnapshotMeta): SentinelSnapshot;
944
- interface SentinelSetOptions {
945
- /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */
946
- tolerance?: number;
947
- }
948
- /**
949
- * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the
950
- * judge was originally calibrated on, re-scored periodically. Pass rate =
951
- * fraction of joined items where the judge stays within `tolerance` of the
952
- * human score. Items the judge didn't score are excluded from the
953
- * denominator; zero joined items is an error, not a 0% pass rate.
954
- */
955
- declare function snapshotFromSentinelSet(scores: CandidateScore[], golden: GoldenItem[], meta: SnapshotMeta, options?: SentinelSetOptions): SentinelSnapshot;
956
- interface SentinelStore {
957
- append(snapshot: SentinelSnapshot): Promise<void>;
958
- /** All snapshots (optionally for one judge), in append order. */
959
- history(judgeId?: string): Promise<SentinelSnapshot[]>;
960
- }
961
- declare function inMemorySentinelStore(initial?: SentinelSnapshot[]): SentinelStore;
962
- /**
963
- * JSONL store — one snapshot per line, appended atomically per call.
964
- * A corrupt or shape-invalid line is a loud error naming the file and
965
- * line number: a sentinel history that silently drops records would
966
- * defeat the drift detection it exists to provide.
967
- */
968
- declare function fileSentinelStore(path: string): SentinelStore;
969
- interface SentinelThresholds {
970
- /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */
971
- minIrr?: number;
972
- /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */
973
- maxKappaDrop?: number;
974
- /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */
975
- maxSentinelDrop?: number;
976
- /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */
977
- staleAfterDays?: number;
978
- }
979
- interface JudgeSentinelOptions {
980
- /** Caller-supplied ISO timestamp staleness is measured against. */
981
- asOf: string;
982
- thresholds?: SentinelThresholds;
983
- /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */
984
- convergence?: SeriesConvergenceOptions;
985
- }
986
- interface SentinelTrend {
987
- judgeId: string;
988
- metric: SentinelMetricName;
989
- /** Verbatim `analyzeSeries` state for this judge×metric history. */
990
- state: SeriesConvergenceResult['state'];
991
- /** Newest value in the series. */
992
- current: number;
993
- /** Oldest value in the series — the reference the drop thresholds compare against. */
994
- baseline: number;
995
- /** current − baseline (signed; negative = decayed). */
996
- drift: number;
997
- alarmed: boolean;
998
- reason?: string;
999
- }
1000
- interface SentinelReport {
1001
- perJudge: SentinelTrend[];
1002
- alarms: string[];
1003
- healthy: boolean;
1004
- /**
1005
- * `judgeId:metric` pairs with too few snapshots for the convergence
1006
- * machine. Named, not counted — a blind spot you can't see is a blind
1007
- * spot you won't fix. Floor/drop checks still apply to thin series, so
1008
- * insufficient history alone never masks an alarm.
1009
- */
1010
- insufficientHistory: string[];
1011
- }
1012
- declare function judgeSentinelReport(history: SentinelSnapshot[], opts: JudgeSentinelOptions): SentinelReport;
1013
- interface EvalHealthStamp {
1014
- healthy: boolean;
1015
- alarms: string[];
1016
- }
1017
- /**
1018
- * The object a campaign attaches to its verdicts. Compute it from the
1019
- * sentinel report in a post-campaign hook (or a nightly job feeding the
1020
- * next day's campaigns) and store it alongside the verdict payload.
1021
- * `healthy: false` means the judges that produced those verdicts have an
1022
- * unresolved drift / decay / silent-upgrade / staleness alarm — treat the
1023
- * verdicts as carrying an integrity warning until recalibration clears it.
1024
- */
1025
- declare function evalHealthStamp(report: SentinelReport): EvalHealthStamp;
1026
-
1027
- export { type CalibrationBin, type CalibrationOptions, type CalibrationPair, type CalibrationReport, type CorrelationResult, type CorrelationStudyOptions, type CorrelationStudyResult, type DeploymentOutcome, type EvalHealthStamp, type EvalMetricSpec, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, type JudgeSentinelOptions, type OutcomeFilter, type OutcomePair, type OutcomeStore, type RubricOutcomePair, type RubricPredictiveValidityInput, type RubricPredictiveValidityReport, type RubricRanking, SENTINEL_METRIC_NAMES, type SentinelMetricName, type SentinelMetrics, type SentinelReport, type SentinelSetOptions, type SentinelSnapshot, type SentinelStore, type SentinelThresholds, type SentinelTrend, type SnapshotMeta, calibrationCurve, calibrationFromPairs, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
1
+ import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
2
+ import { A as EvalMetricSpec, C as CalibrationPair, D as CorrelationResult, E as calibrationFromPairs, M as correlationStudy, O as CorrelationStudyOptions, S as CalibrationOptions, T as calibrationCurve, _ as snapshotFromAgreement, a as SentinelMetrics, b as validateSentinelSnapshot, c as SentinelSnapshot, d as SentinelTrend, f as SnapshotMeta, g as judgeSentinelReport, h as inMemorySentinelStore, i as SentinelMetricName, j as OutcomePair, k as CorrelationStudyResult, l as SentinelStore, m as fileSentinelStore, n as JudgeSentinelOptions, o as SentinelReport, p as evalHealthStamp, r as SENTINEL_METRIC_NAMES, s as SentinelSetOptions, t as EvalHealthStamp, u as SentinelThresholds, v as snapshotFromCalibration, w as CalibrationReport, x as CalibrationBin, y as snapshotFromSentinelSet } from "../index-6N0aYmpW.js";
3
+ import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-Ku_clp_1.js";
4
+ export { CalibrationBin, CalibrationOptions, CalibrationPair, CalibrationReport, CorrelationResult, CorrelationStudyOptions, CorrelationStudyResult, DeploymentOutcome, EvalHealthStamp, EvalMetricSpec, FileSystemOutcomeStore, FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, JudgeSentinelOptions, OutcomeFilter, OutcomePair, OutcomeStore, RubricOutcomePair, RubricPredictiveValidityInput, RubricPredictiveValidityReport, RubricRanking, SENTINEL_METRIC_NAMES, SentinelMetricName, SentinelMetrics, SentinelReport, SentinelSetOptions, SentinelSnapshot, SentinelStore, SentinelThresholds, SentinelTrend, SnapshotMeta, calibrationCurve, calibrationFromPairs, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };