@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,581 @@
1
+ import { l as RunTerminalOutcome } from "./run-record-CnZu_gjl.js";
2
+ import { o as FailureClass } from "./schema-BtVldJ3T.js";
3
+ import { r as ContinuousAgreement } from "./judge-calibration-DFtEMlde.js";
4
+ import { _ as GateDecision, j as MutableSurface } from "./types-k9tZGKUg.js";
5
+ import { i as ParetoFigureSpec, t as GainDistributionBin } from "./summary-report-Cj9gdw4i.js";
6
+ //#region src/contract/insight-report.d.ts
7
+ interface InsightReport {
8
+ /** Number of runs analyzed. */
9
+ n: number;
10
+ /** Runtime facts carried by the run records. These describe execution,
11
+ * not task quality: duration, queueing, token categories, models,
12
+ * execution errors, and terminal outcomes. */
13
+ execution: ExecutionInsight;
14
+ /** Composite-score distribution across all runs. Always present. */
15
+ composite: ScalarDistribution;
16
+ /** Per-dimension distributions for every dimension that appeared in any
17
+ * run's judge scores. Empty when no judge scores were recorded. */
18
+ perDimension: Record<string, ScalarDistribution>;
19
+ /** Cost/quality distribution and Pareto frontier. */
20
+ costQuality: {
21
+ cost: ScalarDistribution;
22
+ pareto: ParetoFigureSpec;
23
+ /** Cost source coverage. `uncaptured` rows are excluded from the USD
24
+ * distribution and Pareto chart; observed and estimated totals remain
25
+ * separate so reports never present estimates as billed spend. */
26
+ provenance?: CostProvenanceSummary;
27
+ /** Set when the cost/quality view is degraded because the input data
28
+ * doesn't fully support it — e.g. all `costUsd` were zero, or only a
29
+ * single candidate appears (so the Pareto is a single point). The
30
+ * named fields name the degraded sub-view, free-text the reason. */
31
+ degraded?: {
32
+ cost?: string;
33
+ pareto?: string;
34
+ };
35
+ };
36
+ /** Per-judge calibration + bias detection. Populated for every judge name
37
+ * that appears in `outcome.judgeScores`. Bias fields require either a
38
+ * gold reference or multi-rater data. */
39
+ judges: Record<string, JudgeInsight>;
40
+ /** Inter-rater agreement when multiple judges scored the same runs.
41
+ * Includes pairwise kappa and the specific run ids where raters
42
+ * disagree — the cases worth a human meeting. */
43
+ interRater?: InterRaterInsight;
44
+ /** Pairwise lift (baseline → candidate) with bootstrap CI. Present when
45
+ * `RunRecord.splitTag` includes both `holdout` and search/dev splits,
46
+ * or when caller passes an explicit baseline/candidate split. */
47
+ lift?: LiftInsight;
48
+ /** Failure clusters with exemplars. Populated when an AnalystRegistry
49
+ * is wired in `analyzeRuns({ analyst })`. */
50
+ failureClusters?: FailureClusterInsight;
51
+ /** Canary leak count + holdout audit status. Populated when canary
52
+ * scenarios are passed in. */
53
+ contamination?: ContaminationInsight;
54
+ /** Correlation between judge composite and a downstream outcome the
55
+ * caller supplies (engagement, revenue, downstream pass rate, etc.).
56
+ * When present, the optional reward model is the model that maps
57
+ * judge scores → predicted outcome. */
58
+ outcomeCorrelation?: OutcomeCorrelationInsight;
59
+ /** Aggregate release-readiness summary. A consumer needing the full
60
+ * substrate `ReleaseConfidenceScorecard` (SLO-axis evaluation,
61
+ * ActionableSideInfo bag) calls `evaluateReleaseConfidence()` directly;
62
+ * this summary captures the analyzeRuns-derived axes. */
63
+ release: ReleaseSummary;
64
+ /** Delta vs a prior period when `baselineRuns` is passed. Per-metric
65
+ * current vs baseline with Welch CI + Cohen's d + significance flag.
66
+ * Answers "did my last change help?" — the customer-conversion question.
67
+ * Surfaced metrics: composite, cost, duration, tokenUsage, plus any
68
+ * per-dimension judge metric present in both windows. */
69
+ priorPeriodComparison?: PriorPeriodComparison;
70
+ /** Model-free task-failure breakdown from `RunRecord.failureClass`, ranked
71
+ * by count descending. Domain-specific `failureMode` detail is retained on
72
+ * each record but never creates a second aggregation vocabulary. */
73
+ failureClasses?: FailureClassTally[];
74
+ /** Top-N actionable recommendations, ranked by priority. The packet's
75
+ * human-readable layer; the numeric sections are the evidence. */
76
+ recommendations: Recommendation[];
77
+ }
78
+ interface CostProvenanceSummary {
79
+ observed: {
80
+ n: number;
81
+ totalUsd: number;
82
+ };
83
+ estimated: {
84
+ n: number;
85
+ totalUsd: number;
86
+ };
87
+ uncaptured: {
88
+ n: number;
89
+ };
90
+ knownFraction: number;
91
+ }
92
+ interface ExecutionInsight {
93
+ /** End-to-end wall time for every run. */
94
+ durationMs: ScalarDistribution;
95
+ /** Queue time for the subset of runs that recorded it. */
96
+ queueMs: ScalarDistribution;
97
+ /** Token distributions plus corpus totals. Optional token categories use
98
+ * distribution `n` to disclose how many runs recorded that category. */
99
+ tokenUsage: TokenUsageInsight;
100
+ /** Usage reported only by orchestration or agent aggregate spans.
101
+ * Kept separate because it may duplicate model-call telemetry in other traces. */
102
+ aggregateUsage: {
103
+ runs: number;
104
+ tokenUsage: TokenUsageInsight;
105
+ costUsd: ScalarDistribution;
106
+ totalCostUsd: number;
107
+ };
108
+ /** Stable model counts, largest cohort first. */
109
+ models: Array<{
110
+ model: string;
111
+ runs: number;
112
+ }>;
113
+ /** Model-call coverage. `events` is available only from producers that
114
+ * record `outcome.raw.llm_span_count`; `runs` also recognizes non-zero
115
+ * token usage from other producers. */
116
+ modelCalls: {
117
+ runs: number;
118
+ events: number;
119
+ reportingRuns: number;
120
+ };
121
+ /** Runs with explicit execution-error telemetry. This is independent of
122
+ * whether the root run ultimately succeeded, failed, or has no terminal
123
+ * evidence. */
124
+ executionErrors: {
125
+ runs: number;
126
+ /** Share among runs that supplied an execution-error count.
127
+ * `null` when no run supplied error telemetry. */
128
+ fraction: number | null;
129
+ /** Execution-error events reported through the canonical count. */
130
+ events: number;
131
+ /** Runs that supplied an execution-error count, including explicit zeroes. */
132
+ reportingRuns: number;
133
+ /** Exact sum of `outcome.raw.error_span_count`, kept separate from other errors. */
134
+ errorSpanEvents: number;
135
+ /** Runs that supplied `outcome.raw.error_span_count`, including explicit zeroes. */
136
+ errorSpanReportingRuns: number;
137
+ /**
138
+ * Error-telemetry coverage crossed with independently reported terminal
139
+ * outcomes. `unreported` is distinct from a reported zero.
140
+ */
141
+ byTerminalOutcome: Record<RunTerminalOutcome, ExecutionErrorOutcomeCell>;
142
+ };
143
+ /** Root-run or process outcomes. Missing `RunRecord.terminalOutcome` values
144
+ * count as `unknown`; child-span status never changes these counts. */
145
+ terminalOutcomes: {
146
+ succeeded: number;
147
+ failed: number;
148
+ cancelled: number;
149
+ incomplete: number;
150
+ unknown: number;
151
+ };
152
+ }
153
+ interface ExecutionErrorOutcomeCell {
154
+ /** Runs that explicitly reported one or more execution errors. */
155
+ withErrors: number;
156
+ /** Runs that explicitly reported zero execution errors. */
157
+ withoutErrors: number;
158
+ /** Runs with no execution-error count from the producer. */
159
+ unreported: number;
160
+ }
161
+ interface TokenUsageInsight {
162
+ input: ScalarDistribution;
163
+ output: ScalarDistribution;
164
+ reasoning: ScalarDistribution;
165
+ cached: ScalarDistribution;
166
+ cacheWrite: ScalarDistribution;
167
+ totals: {
168
+ input: number;
169
+ output: number;
170
+ reasoning: number;
171
+ cached: number;
172
+ cacheWrite: number;
173
+ };
174
+ }
175
+ /** Distributional summary of a scalar-valued metric. */
176
+ interface ScalarDistribution {
177
+ /** Sample count after dropping non-finite values. */
178
+ n: number;
179
+ /** Null when `n` is zero. */
180
+ mean: number | null;
181
+ /** Null when `n` is zero. */
182
+ p50: number | null;
183
+ /** Null when `n` is zero. */
184
+ p95: number | null;
185
+ /** Null when `n` is zero. */
186
+ stddev: number | null;
187
+ /** Null when `n` is zero. */
188
+ min: number | null;
189
+ /** Null when `n` is zero. */
190
+ max: number | null;
191
+ /** Histogram bins using `agent-eval`'s `gainHistogram` primitive. */
192
+ histogram: GainDistributionBin[];
193
+ /** Worst-N runs by score, ascending. Populated for the composite
194
+ * distribution so the report names the runs a customer should
195
+ * inspect first. Undefined when the distribution was computed from a
196
+ * raw value list with no run identity (e.g. cost). */
197
+ tailRuns?: Array<{
198
+ runId: string;
199
+ score: number;
200
+ }>;
201
+ }
202
+ interface JudgeInsight {
203
+ /** Number of times this judge scored a run. */
204
+ n: number;
205
+ /** Mean composite over this judge's runs. */
206
+ meanScore: number;
207
+ /** Calibration against a gold reference, when provided. Cohen's κ for
208
+ * binary thresholding + continuous agreement metrics. */
209
+ calibration?: ContinuousAgreement;
210
+ /** Positional bias — when the judge sees options in different orders,
211
+ * do its preferences track the content or the position? */
212
+ positionalBias?: number;
213
+ /** Self-preference — when the judge sees its own model's output vs a
214
+ * competitor, does it over-pick its own? */
215
+ selfPreference?: number;
216
+ /** Verbosity bias — does the judge reward longer outputs regardless of
217
+ * quality? */
218
+ verbosityBias?: number;
219
+ }
220
+ interface InterRaterInsight {
221
+ /** Number of raters whose scores were aggregated. */
222
+ raters: number;
223
+ /** Number of runs every rater scored. */
224
+ jointlyRated: number;
225
+ /** Multi-rater weighted kappa over the jointly rated runs. */
226
+ kappa: number;
227
+ /** Absolute agreement across raters, using ICC(2,1). */
228
+ icc: number;
229
+ /** Mean pairwise Pearson correlation. Correlation is not agreement. */
230
+ pearson: number;
231
+ /** Mean pairwise Spearman rank correlation. */
232
+ spearman: number;
233
+ /** Pairwise weighted kappa per rater pair (key = `"raterA::raterB"`). */
234
+ perPair: Record<string, number>;
235
+ /** Run ids where raters disagree the most — the high-value triage list. */
236
+ disagreementCases: Array<{
237
+ runId: string;
238
+ ratings: Array<{
239
+ rater: string;
240
+ score: number;
241
+ }>;
242
+ range: number;
243
+ }>;
244
+ }
245
+ interface LiftInsight {
246
+ baselineMean: number;
247
+ candidateMean: number;
248
+ /** Candidate − baseline. */
249
+ delta: number;
250
+ /** Lower / upper bound of bootstrap CI on the delta. */
251
+ ci95: [number, number];
252
+ /** Paired-t-test p-value. */
253
+ pValue: number;
254
+ /** Number of paired observations. */
255
+ n: number;
256
+ /** Scored baseline observations without a candidate match. */
257
+ unpairedBaseline: number;
258
+ /** Scored candidate observations without a baseline match. */
259
+ unpairedCandidate: number;
260
+ /** Cohen's dz for paired deltas; null when the observed delta variance is zero. */
261
+ cohensD: number | null;
262
+ /** Minimum detectable effect at current n, 80% power. */
263
+ mde: number;
264
+ /** Paired sample size needed to detect the standardized effect at 80% power. */
265
+ requiredN: number | null;
266
+ }
267
+ interface FailureClusterInsight {
268
+ /** All clusters identified by the registry, ranked by share descending. */
269
+ clusters: Array<{
270
+ id: string;
271
+ name: string;
272
+ /** Fraction of failed runs in this cluster, 0..1. */
273
+ share: number;
274
+ /** Exemplar `runId`s (≤ 5) the consumer can drill into. */
275
+ exemplars: string[];
276
+ /** Short LLM-generated suggested fix when the registry supports it. */
277
+ suggestedFix?: string;
278
+ }>;
279
+ totalFailures: number;
280
+ }
281
+ /** Model-free task-failure breakdown over canonical `RunRecord.failureClass`
282
+ * values. Unlike semantic failure clusters, this is computed directly from
283
+ * run records and does not require a model analyst. */
284
+ interface FailureClassTally {
285
+ /** Canonical task-failure class. */
286
+ failureClass: FailureClass;
287
+ /** Number of failed runs carrying this class. */
288
+ count: number;
289
+ /** Share of the whole corpus, 0..1. */
290
+ share: number;
291
+ }
292
+ interface ContaminationInsight {
293
+ /** Canary phrases that leaked into outputs. */
294
+ leaks: number;
295
+ /** Holdout audit verdict — did any holdout-tagged run end up in the
296
+ * search/dev pool, or vice versa? */
297
+ holdoutAuditPassed: boolean;
298
+ details?: Array<{
299
+ runId: string;
300
+ canary: string;
301
+ matched: string;
302
+ }>;
303
+ }
304
+ interface OutcomeCorrelationInsight {
305
+ /** What outcome the consumer is correlating against (e.g.
306
+ * `'engagement_rate'`, `'approval_rate'`, `'downstream_pass'`). */
307
+ metric: string;
308
+ /** Number of (run, outcome) pairs used. */
309
+ n: number;
310
+ /** Pearson correlation between composite score and outcome. */
311
+ pearson: number;
312
+ /** Spearman rank correlation — robust to monotonic non-linearity. */
313
+ spearman: number;
314
+ /** When present, the simple linear reward model fit to the data. */
315
+ rewardModel?: {
316
+ intercept: number;
317
+ slope: number;
318
+ r2: number;
319
+ };
320
+ }
321
+ interface ReleaseSummary {
322
+ /** Overall verdict across axes — fail if any axis fails, else warn if any
323
+ * warns, else pass. */
324
+ status: 'pass' | 'warn' | 'fail';
325
+ axes: Array<{
326
+ name: 'quality-lift' | 'contamination' | 'composite-distribution';
327
+ status: 'pass' | 'warn' | 'fail' | 'not_evaluated';
328
+ detail: string;
329
+ }>;
330
+ /** Free-form issues surfaced beyond the standard axes. Empty by default;
331
+ * consumers can post-process to populate. */
332
+ issues: string[];
333
+ }
334
+ interface MetricDelta {
335
+ /** Current-period mean. */
336
+ current: number;
337
+ /** Baseline-period mean. */
338
+ baseline: number;
339
+ /** current - baseline. Positive means improved (or, for cost/duration,
340
+ * the consumer-side interpretation: "higher current" — semantic
341
+ * direction depends on the metric). */
342
+ delta: number;
343
+ /** Welch 95% confidence interval on the delta. Two-sample, unpaired —
344
+ * the baseline and current run sets may have different scenarios. */
345
+ ci95: [number, number];
346
+ /** Welch t-test p-value (two-sided). */
347
+ pValue: number;
348
+ /** Cohen's d (pooled stddev). Effect size, signed. */
349
+ cohensD: number;
350
+ /** Sample sizes. */
351
+ baselineN: number;
352
+ currentN: number;
353
+ /** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
354
+ * conjunction prevents large-effect-but-noisy and significant-but-
355
+ * tiny from triggering recommendations. */
356
+ significant: boolean;
357
+ }
358
+ interface PriorPeriodComparison {
359
+ /** Sample counts. */
360
+ baselineN: number;
361
+ currentN: number;
362
+ /** Optional human-readable label — "vs prior 7 days", "vs v3 release". */
363
+ windowLabel?: string;
364
+ /** Every metric we could compare. Keys: 'composite', 'cost', 'duration',
365
+ * 'tokenUsage' for always-present ones; per-dimension keys when both
366
+ * windows have judge scores on the same dimension. */
367
+ metrics: Record<string, MetricDelta>;
368
+ /** Metric names where current is significantly WORSE than baseline.
369
+ * Direction-aware: for cost/duration, higher current = worse. */
370
+ regressedMetrics: string[];
371
+ /** Metric names where current is significantly BETTER than baseline. */
372
+ improvedMetrics: string[];
373
+ }
374
+ interface Recommendation {
375
+ priority: 'critical' | 'high' | 'medium' | 'low';
376
+ kind: 'ship' | 'hold' | 'investigate' | 'fix' | 'recalibrate' | 'expand-corpus';
377
+ title: string;
378
+ detail: string;
379
+ /** Optional pointer back into the report for the evidence. */
380
+ evidencePath?: string;
381
+ }
382
+ //#endregion
383
+ //#region src/hosted/types.d.ts
384
+ declare const HOSTED_WIRE_VERSION: '2026-07-24.v1';
385
+ type HostedWireVersion = typeof HOSTED_WIRE_VERSION;
386
+ /** Every ingest request carries these. */
387
+ interface HostedIngestHeaders {
388
+ /** Bearer token. The orchestrator validates against the tenant key. */
389
+ authorization: `Bearer ${string}`;
390
+ /** Stable tenant id (the orchestrator-side primary key for the tenant). */
391
+ 'x-tangle-tenant-id': string;
392
+ /** Wire-version pin so the server can reject incompatible payloads. */
393
+ 'x-tangle-wire-version': HostedWireVersion;
394
+ /** Stable request key generated once and reused across retries. */
395
+ 'idempotency-key': string;
396
+ }
397
+ /** Lifecycle stages of an eval-run as the substrate reports them. */
398
+ type EvalRunStatus = 'started' | 'baseline-complete' | 'generation-complete' | 'gate-decided' | 'finished' | 'errored';
399
+ interface EvalRunCellScore {
400
+ /** Stable scenario id from the consumer's scenario set. */
401
+ scenarioId: string;
402
+ /** Repetition index when reps > 1; 0 for the default. */
403
+ rep: number;
404
+ /** Composite score across successful judges, or null when unscored. */
405
+ compositeMean: number | null;
406
+ /** Per-judge and per-dimension scores; failed or missing judges are absent. */
407
+ dimensions: Record<string, Record<string, number>>;
408
+ /** Root execution result, kept separate from task quality. */
409
+ terminalOutcome: RunTerminalOutcome;
410
+ /** Canonical execution-error count, or null when the producer did not measure it. */
411
+ executionErrorCount: number | null;
412
+ /** Per-cell dispatch or judge error. Missing on success. */
413
+ errorMessage?: string;
414
+ }
415
+ interface EvalRunGenerationSnapshot {
416
+ /** Generation index. 0 is baseline. */
417
+ index: number;
418
+ /** Candidate surface fingerprint (stable hash) — pivot key into the
419
+ * trace stream to fetch the underlying execution. */
420
+ surfaceHash: string;
421
+ /** The candidate surface itself. May be omitted to avoid PII when the
422
+ * consumer prefers not to ship verbatim prompts. */
423
+ surface?: MutableSurface;
424
+ /** Per-cell scores for this generation. */
425
+ cells: EvalRunCellScore[];
426
+ /** Mean across scored cells, or null when no cell has a task-quality label. */
427
+ compositeMean: number | null;
428
+ /** Total $ spent across this generation. */
429
+ costUsd: number;
430
+ /** Wall-clock duration of this generation. */
431
+ durationMs: number;
432
+ }
433
+ /**
434
+ * The top-level eval-run event. One ingest call per logical eval-run;
435
+ * generations stream in incrementally via repeated calls with the same
436
+ * `runId`. The orchestrator deduplicates by `(runId, generation.index)`.
437
+ */
438
+ interface EvalRunEvent {
439
+ /** Stable run id (the substrate's `runId`). UUID or substrate-generated. */
440
+ runId: string;
441
+ /** Where this run was happening — derived from `RunCampaignOptions.runDir`. */
442
+ runDir: string;
443
+ /** ISO-8601 timestamp the substrate recorded the event. */
444
+ timestamp: string;
445
+ /** Lifecycle stage this event represents. */
446
+ status: EvalRunStatus;
447
+ /** Free-form consumer tags (env, branch, model id, etc.). Searchable. */
448
+ labels: Record<string, string>;
449
+ /** Baseline campaign snapshot. Present when status >= baseline-complete. */
450
+ baseline?: EvalRunGenerationSnapshot;
451
+ /** Per-generation snapshots. Streams in; orchestrator appends. */
452
+ generations: EvalRunGenerationSnapshot[];
453
+ /** Final gate decision. Present when status >= gate-decided. */
454
+ gateDecision?: GateDecision;
455
+ /** Held-out lift = winner-on-holdout - baseline-on-holdout. */
456
+ holdoutLift?: number;
457
+ /** Total $ spent across baseline + every generation. */
458
+ totalCostUsd: number;
459
+ /** Total wall-clock duration. */
460
+ totalDurationMs: number;
461
+ /** Error message if status === 'errored'. */
462
+ errorMessage?: string;
463
+ /** Rigor packet emitted alongside the run — distributional summary,
464
+ * paired-bootstrap lift CI, judge stats, inter-rater agreement,
465
+ * contamination check, failure clusters (when an analyst is wired),
466
+ * outcome correlation (when downstream signal is supplied), and the
467
+ * recommendations the dashboard surfaces verbatim. */
468
+ insightReport?: InsightReport;
469
+ }
470
+ /**
471
+ * Canonical unsigned 64-bit integer encoded as a base-10 string.
472
+ * JSON numbers cannot represent OTLP nanosecond timestamps exactly.
473
+ */
474
+ type UnixNanoTimestamp = string;
475
+ /**
476
+ * OTel-shape span with a few additional attributes for eval-run pivoting.
477
+ * Compatible with any OTLP collector — `name`, `traceId`, `spanId`,
478
+ * `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.
479
+ */
480
+ interface TraceSpanEvent {
481
+ traceId: string;
482
+ spanId: string;
483
+ parentSpanId?: string;
484
+ name: string;
485
+ startTimeUnixNano: UnixNanoTimestamp;
486
+ endTimeUnixNano: UnixNanoTimestamp;
487
+ attributes: Record<string, string | number | boolean>;
488
+ events?: Array<{
489
+ timeUnixNano: UnixNanoTimestamp;
490
+ name: string;
491
+ attributes?: Record<string, string | number | boolean>;
492
+ }>;
493
+ status?: {
494
+ code: 'OK' | 'ERROR' | 'UNSET';
495
+ message?: string;
496
+ };
497
+ /** Pivot back into the eval-run stream. */
498
+ 'tangle.runId'?: string;
499
+ /** Pivot to the specific generation. */
500
+ 'tangle.generation'?: number;
501
+ /** Pivot to the specific cell. */
502
+ 'tangle.cellId'?: string;
503
+ /** Pivot to the specific scenario. */
504
+ 'tangle.scenarioId'?: string;
505
+ }
506
+ interface IngestEvalRunsRequest {
507
+ wireVersion: HostedWireVersion;
508
+ events: EvalRunEvent[];
509
+ }
510
+ interface IngestTracesRequest {
511
+ wireVersion: HostedWireVersion;
512
+ spans: TraceSpanEvent[];
513
+ }
514
+ interface IngestResponse {
515
+ /** Accepted events / spans count. */
516
+ accepted: number;
517
+ /** Rejected events with reasons (validation failures, dup idempotency key, etc.). */
518
+ rejected: Array<{
519
+ index: number;
520
+ reason: string;
521
+ }>;
522
+ }
523
+ //#endregion
524
+ //#region src/hosted/client.d.ts
525
+ interface HostedTenant {
526
+ /** Orchestrator endpoint base URL (no trailing slash). Required. */
527
+ endpoint: string;
528
+ /** Bearer token issued by the orchestrator. Required. */
529
+ apiKey: string;
530
+ /** Tenant id — the orchestrator's primary key for this consumer. Required. */
531
+ tenantId: string;
532
+ /** Optional `fetch` override (auth wrappers, custom agent, test mocks). */
533
+ fetchImpl?: typeof fetch;
534
+ /** Per-call timeout in ms. Default 30s. */
535
+ timeoutMs?: number;
536
+ /** Retries on 5xx / network errors. Default 2. */
537
+ retries?: number;
538
+ }
539
+ interface HostedClient {
540
+ ingestEvalRun(event: EvalRunEvent, idempotencyKey?: string): Promise<IngestResponse>;
541
+ ingestEvalRuns(events: EvalRunEvent[], idempotencyKey?: string): Promise<IngestResponse>;
542
+ ingestTraces(spans: TraceSpanEvent[], idempotencyKey?: string): Promise<IngestResponse>;
543
+ readonly tenant: HostedTenant;
544
+ readonly wireVersion: HostedWireVersion;
545
+ }
546
+ declare function createHostedClient(tenant: HostedTenant): HostedClient;
547
+ /**
548
+ * Build a `HostedClient` from environment, or `undefined` when ingest is not
549
+ * configured — the canonical, fail-soft wiring every product uses so eval-run +
550
+ * trace provenance lands in the Intelligence dashboard with ONE call:
551
+ *
552
+ * const hosted = hostedClientFromEnv()
553
+ * // ...run the loop...
554
+ * await emitLoopProvenance({ ..., hostedClient: hosted }) // no-op if undefined
555
+ *
556
+ * Returns `undefined` (NOT an error) when any of endpoint / apiKey / tenantId is
557
+ * missing — so a product wires the ship call unconditionally and it stays a
558
+ * no-op until the env is set. Env precedence:
559
+ * - endpoint: `TANGLE_INGEST_URL` → `TANGLE_ORCHESTRATOR_URL`
560
+ * - apiKey: `TANGLE_INGEST_API_KEY` → `TANGLE_API_KEY`
561
+ * - tenantId: `TANGLE_TENANT_ID`
562
+ * A trailing slash on the endpoint is stripped. Pass `overrides` to supply any
563
+ * field directly (e.g. a fixed `tenantId` per product) — overrides win over env.
564
+ */
565
+ /**
566
+ * Build a {@link HostedTenant} config from env — the input `selfImprove`'s
567
+ * `hostedTenant` and `emitLoopProvenance` take. Same env precedence + overrides
568
+ * as {@link hostedClientFromEnv}; returns `undefined` (not an error) when any of
569
+ * endpoint / apiKey / tenantId is missing, so a product wires
570
+ * `hostedTenant: hostedTenantFromEnv({ tenantId: 'my-agent' })` unconditionally
571
+ * and it stays off until the env is set.
572
+ */
573
+ declare function hostedTenantFromEnv(overrides?: Partial<HostedTenant> & {
574
+ env?: Record<string, string | undefined>;
575
+ }): HostedTenant | undefined;
576
+ declare function hostedClientFromEnv(overrides?: Partial<HostedTenant> & {
577
+ env?: Record<string, string | undefined>;
578
+ }): HostedClient | undefined;
579
+ //#endregion
580
+ export { ScalarDistribution as A, InsightReport as C, OutcomeCorrelationInsight as D, LiftInsight as E, Recommendation as O, FailureClusterInsight as S, JudgeInsight as T, UnixNanoTimestamp as _, hostedTenantFromEnv as a, ExecutionInsight as b, EvalRunGenerationSnapshot as c, HostedIngestHeaders as d, HostedWireVersion as f, TraceSpanEvent as g, IngestTracesRequest as h, hostedClientFromEnv as i, TokenUsageInsight as j, ReleaseSummary as k, EvalRunStatus as l, IngestResponse as m, HostedTenant as n, EvalRunCellScore as o, IngestEvalRunsRequest as p, createHostedClient as r, EvalRunEvent as s, HostedClient as t, HOSTED_WIRE_VERSION as u, CostProvenanceSummary as v, InterRaterInsight as w, FailureClassTally as x, ExecutionErrorOutcomeCell as y };
581
+ //# sourceMappingURL=client-C97NMzqi.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"client-C97NMzqi.d.ts","names":[],"sources":["../src/contract/insight-report.ts","../src/hosted/types.ts","../src/hosted/client.ts"],"mappings":";;;;;;UAqCiB;;EAEf;;;;EAKA,WAAW;;EAGX,WAAW;;;EAIX,cAAc,eAAe;;EAG7B;IACE,MAAM;IACN,QAAQ;;;;IAIR,aAAa;;;;;IAKb;MAAa;MAAe;;;;;;EAM9B,QAAQ,eAAe;;;;EAKvB,aAAa;;;;EAKb,OAAO;;;EAIP,kBAAkB;;;EAIlB,gBAAgB;;;;;EAMhB,qBAAqB;;;;;EAMrB,SAAS;;;;;;EAOT,wBAAwB;;;;EAKxB,iBAAiB;;;EAIjB,iBAAiB;;UAGF;EACf;IAAY;IAAW;;EACvB;IAAa;IAAW;;EACxB;IAAc;;EACd;;UAGe;;EAEf,YAAY;;EAEZ,SAAS;;;EAGT,YAAY;;;EAGZ;IACE;IACA,YAAY;IACZ,SAAS;IACT;;;EAGF,QAAQ;IAAQ;IAAe;;;;;EAI/B;IACE;IACA;IACA;;;;;EAKF;IACE;;;IAGA;;IAEA;;IAEA;;IAEA;;IAEA;;;;;IAKA,mBAAmB,OAAO,oBAAoB;;;;EAIhD;IACE;IACA;IACA;IACA;IACA;;;UAIa;;EAEf;;EAEA;;EAEA;;UAGe;EACf,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;EACR,YAAY;EACZ;IACE;IACA;IACA;IACA;IACA;;;;UAOa;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,WAAW;;;;;EAKX,WAAW;IAAQ;IAAe;;;UAGnB;;EAEf;;EAEA;;;EAGA,cAAc;;;EAGd;;;EAGA;;;EAGA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,SAAS;;EAET,mBAAmB;IACjB;IACA,SAAS;MAAQ;MAAe;;IAChC;;;UAIa;EACf;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf,UAAU;IACR;IACA;;IAEA;;IAEA;;IAEA;;EAEF;;;;;UAMe;;EAEf,cAAc;;EAEd;;EAEA;;UAGe;;EAEf;;;EAGA;EACA,UAAU;IAAQ;IAAe;IAAgB;;;UAGlC;;;EAGf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;IACA;;;UAIa;;;EAGf;EACA,MAAM;IACJ;IACA;IACA;;;;EAIF;;UAGe;;EAEf;;EAEA;;;;EAIA;;;EAGA;;EAEA;;EAEA;;EAEA;EACA;;;;EAIA;;UAGe;;EAEf;EACA;;EAEA;;;;EAIA,SAAS,eAAe;;;EAGxB;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA;;;;cClYW;KACD,2BAA2B;;UAKtB;;EAEf;;EAEA;;EAEA,yBAAyB;;EAEzB;;;KAMU;UAQK;;EAEf;;EAEA;;EAEA;;EAEA,YAAY,eAAe;;EAE3B,iBAAiB;;EAEjB;;EAEA;;UAGe;;EAEf;;;EAGA;;;EAGA,UAAU;;EAEV,OAAO;;EAEP;;EAEA;;EAEA;;;;;;;UAQe;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,QAAQ;;EAER,WAAW;;EAEX,aAAa;;EAEb,eAAe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;EAMA,gBAAgB;;;;;;KASN;;;;;;UAOK;EACf;EACA;EACA;EACA;EACA,mBAAmB;EACnB,iBAAiB;EACjB,YAAY;EACZ,SAAS;IACP,cAAc;IACd;IACA,aAAa;;EAEf;IAAW;IAAgC;;;EAE3C;;EAEA;;EAEA;;EAEA;;UAKe;EACf,aAAa;EACb,QAAQ;;UAGO;EACf,aAAa;EACb,OAAO;;UAGQ;;EAEf;;EAEA,UAAU;IAAQ;IAAe;;;;;UC1JlB;;EAEf;;EAEA;;EAEA;;EAEA,mBAAmB;;EAEnB;;EAEA;;UAGe;EACf,cAAc,OAAO,cAAc,0BAA0B,QAAQ;EACrE,eAAe,QAAQ,gBAAgB,0BAA0B,QAAQ;EACzE,aAAa,OAAO,kBAAkB,0BAA0B,QAAQ;WAC/D,QAAQ;WACR,aAAa;;iBAyGR,mBAAmB,QAAQ,eAAe;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsE1C,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C;iBAiBa,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C"}