@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,\n * joins on runId, computes Pearson + Spearman + bootstrap CI for every\n * (evalMetric, outcomeMetric) pair the caller declares.\n *\n * Without this number the framework is ornamental. With it and r > 0.6\n * the framework is a moat — no other agent-eval tool publishes one.\n */\n\nimport { pearsonR, spearmanR } from '../statistics'\nimport { aggregateLlm, llmSpans } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run (defaults cover score/pass/durationMs/costUsd/tokens). */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number }\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const reduction = options.reduction ?? 'latest'\n const maxLag = options.maxCaptureLagMs ?? Infinity\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of evalMetrics) {\n for (const om of outcomeMetricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag)\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n for (const em of evalMetrics) {\n const extract = em.extract ?? defaultExtract(em.id)\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of outcomeMetricNames) {\n const values = eligible\n .map((o) => o.metrics[om])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, reduction, eligible)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n }\n }\n joined++\n }\n\n const results: CorrelationResult[] = pairs\n .filter((p) => p.xs.length >= 3)\n .map((p) => {\n const pearson = pearsonR(p.xs, p.ys)\n const spearman = spearmanR(p.xs, p.ys)\n const pearsonCi95 = bootstrapPearsonCi(p.xs, p.ys, options.bootstrapIterations ?? 500)\n const verdict: CorrelationResult['verdict'] =\n Math.abs(pearson) >= 0.7 ? 'strong' : Math.abs(pearson) >= 0.4 ? 'moderate' : 'weak'\n return {\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n pearson,\n spearman,\n pearsonCi95,\n verdict,\n }\n })\n\n return { pairs: results, joinedSamples: joined, skippedRuns: skipped }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n kind: 'latest' | 'mean' | 'max',\n outcomes: DeploymentOutcome[],\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((a, b) => a + b, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest': pick the outcome captured last, then lookup its metric\n const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0]\n if (!latest) return null\n const latestKey = Object.keys(latest.metrics)[0]\n const v = latestKey !== undefined ? latest.metrics[latestKey] : undefined\n // For 'latest' we already have `values` aligned; use the last-captured one\n const paired = outcomes\n .map((o) => {\n const k = Object.keys(o.metrics)[0]\n return {\n at: o.capturedAt,\n v: k !== undefined ? values.find((x) => o.metrics[k] === x) : undefined,\n }\n })\n .filter((p) => p.v !== undefined)\n if (paired.length === 0) return v ?? null\n return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null\n}\n\nfunction bootstrapPearsonCi(\n xs: number[],\n ys: number[],\n iterations: number,\n): { lower: number; upper: number } {\n const n = xs.length\n if (n < 3) return { lower: NaN, upper: NaN }\n const rs: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = new Array(n)\n const ry: number[] = new Array(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(Math.random() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) rs.push(r)\n }\n rs.sort((a, b) => a - b)\n if (rs.length === 0) return { lower: NaN, upper: NaN }\n return {\n lower: rs[Math.floor(0.025 * rs.length)]!,\n upper: rs[Math.min(rs.length - 1, Math.floor(0.975 * rs.length))]!,\n }\n}\n\nfunction defaultExtract(metric: string): (run: Run, store: TraceStore) => Promise<number | null> {\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).costUsd\n }\n case 'inputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).inputTokens\n }\n default:\n return null\n }\n }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nexport const SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAyDA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;AACjC,QAAM,OAAO,MAAM,WAAW,SAAS;AACvC,QAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;AAC9D,QAAM,gBAAgB,oBAAI,IAAiC;AAC3D,aAAW,KAAK,UAAU;AACxB,UAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;AAC3C,QAAI,KAAK,CAAC;AACV,kBAAc,IAAI,EAAE,OAAO,GAAG;AAAA,EAChC;AAEA,QAAM,YAAY,QAAQ,aAAa;AACvC,QAAM,SAAS,QAAQ,mBAAmB;AAE1C,QAAM,QAA0F,CAAC;AACjG,aAAW,MAAM,aAAa;AAC5B,eAAW,MAAM,oBAAoB;AACnC,YAAM,KAAK,EAAE,YAAY,GAAG,IAAI,eAAe,IAAI,IAAI,CAAC,GAAG,IAAI,CAAC,EAAE,CAAC;AAAA,IACrE;AAAA,EACF;AAEA,MAAI,SAAS;AACb,MAAI,UAAU;AACd,aAAW,OAAO,MAAM;AACtB,UAAM,KAAK,cAAc,IAAI,IAAI,KAAK;AACtC,QAAI,CAAC,MAAM,GAAG,WAAW,GAAG;AAC1B;AACA;AAAA,IACF;AACA,UAAM,WAAW,GAAG,OAAO,CAAC,MAAM,EAAE,aAAa,IAAI,aAAa,MAAM;AACxE,QAAI,SAAS,WAAW,GAAG;AACzB;AACA;AAAA,IACF;AAEA,eAAW,MAAM,aAAa;AAC5B,YAAM,UAAU,GAAG,WAAW,eAAe,GAAG,EAAE;AAClD,YAAM,IAAI,MAAM,QAAQ,KAAK,UAAU;AACvC,UAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,EAAG;AAEvC,iBAAW,MAAM,oBAAoB;AACnC,cAAM,SAAS,SACZ,IAAI,CAAC,MAAM,EAAE,QAAQ,EAAE,CAAC,EACxB,OAAO,CAAC,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;AACzE,YAAI,OAAO,WAAW,EAAG;AACzB,cAAM,IAAI,OAAO,QAAQ,WAAW,QAAQ;AAC5C,YAAI,MAAM,KAAM;AAChB,cAAM,OAAO,MAAM,KAAK,CAAC,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;AAC/E,aAAK,GAAG,KAAK,CAAC;AACd,aAAK,GAAG,KAAK,CAAC;AAAA,MAChB;AAAA,IACF;AACA;AAAA,EACF;AAEA,QAAM,UAA+B,MAClC,OAAO,CAAC,MAAM,EAAE,GAAG,UAAU,CAAC,EAC9B,IAAI,CAAC,MAAM;AACV,UAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;AACnC,UAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;AACrC,UAAM,cAAc,mBAAmB,EAAE,IAAI,EAAE,IAAI,QAAQ,uBAAuB,GAAG;AACrF,UAAM,UACJ,KAAK,IAAI,OAAO,KAAK,MAAM,WAAW,KAAK,IAAI,OAAO,KAAK,MAAM,aAAa;AAChF,WAAO;AAAA,MACL,YAAY,EAAE;AAAA,MACd,eAAe,EAAE;AAAA,MACjB,GAAG,EAAE,GAAG;AAAA,MACR;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,IACF;AAAA,EACF,CAAC;AAEH,SAAO,EAAE,OAAO,SAAS,eAAe,QAAQ,aAAa,QAAQ;AACvE;AAIA,SAAS,OACP,QACA,MACA,UACe;AACf,MAAI,OAAO,WAAW,EAAG,QAAO;AAChC,MAAI,SAAS,OAAQ,QAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AACvE,MAAI,SAAS,MAAO,QAAO,KAAK,IAAI,GAAG,MAAM;AAE7C,QAAM,SAAS,CAAC,GAAG,QAAQ,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,EAAE,CAAC;AAC1E,MAAI,CAAC,OAAQ,QAAO;AACpB,QAAM,YAAY,OAAO,KAAK,OAAO,OAAO,EAAE,CAAC;AAC/C,QAAM,IAAI,cAAc,SAAY,OAAO,QAAQ,SAAS,IAAI;AAEhE,QAAM,SAAS,SACZ,IAAI,CAAC,MAAM;AACV,UAAM,IAAI,OAAO,KAAK,EAAE,OAAO,EAAE,CAAC;AAClC,WAAO;AAAA,MACL,IAAI,EAAE;AAAA,MACN,GAAG,MAAM,SAAY,OAAO,KAAK,CAAC,MAAM,EAAE,QAAQ,CAAC,MAAM,CAAC,IAAI;AAAA,IAChE;AAAA,EACF,CAAC,EACA,OAAO,CAAC,MAAM,EAAE,MAAM,MAAS;AAClC,MAAI,OAAO,WAAW,EAAG,QAAO,KAAK;AACrC,SAAO,OAAO,KAAK,CAAC,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,EAAE,CAAC,GAAG,KAAK;AACrD;AAEA,SAAS,mBACP,IACA,IACA,YACkC;AAClC,QAAM,IAAI,GAAG;AACb,MAAI,IAAI,EAAG,QAAO,EAAE,OAAO,KAAK,OAAO,IAAI;AAC3C,QAAM,KAAe,CAAC;AACtB,WAAS,IAAI,GAAG,IAAI,YAAY,KAAK;AACnC,UAAM,KAAe,IAAI,MAAM,CAAC;AAChC,UAAM,KAAe,IAAI,MAAM,CAAC;AAChC,aAAS,IAAI,GAAG,IAAI,GAAG,KAAK;AAC1B,YAAM,MAAM,KAAK,MAAM,KAAK,OAAO,IAAI,CAAC;AACxC,SAAG,CAAC,IAAI,GAAG,GAAG;AACd,SAAG,CAAC,IAAI,GAAG,GAAG;AAAA,IAChB;AACA,UAAM,IAAI,SAAS,IAAI,EAAE;AACzB,QAAI,OAAO,SAAS,CAAC,EAAG,IAAG,KAAK,CAAC;AAAA,EACnC;AACA,KAAG,KAAK,CAAC,GAAG,MAAM,IAAI,CAAC;AACvB,MAAI,GAAG,WAAW,EAAG,QAAO,EAAE,OAAO,KAAK,OAAO,IAAI;AACrD,SAAO;AAAA,IACL,OAAO,GAAG,KAAK,MAAM,QAAQ,GAAG,MAAM,CAAC;AAAA,IACvC,OAAO,GAAG,KAAK,IAAI,GAAG,SAAS,GAAG,KAAK,MAAM,QAAQ,GAAG,MAAM,CAAC,CAAC;AAAA,EAClE;AACF;AAEA,SAAS,eAAe,QAAyE;AAC/F,SAAO,OAAO,KAAK,UAAU;AAC3B,YAAQ,QAAQ;AAAA,MACd,KAAK;AAAA,MACL,KAAK;AACH,eAAO,IAAI,SAAS,SAAS;AAAA,MAC/B,KAAK;AACH,eAAO,IAAI,SAAS,SAAS,OAAO,IAAI;AAAA,MAC1C,KAAK;AACH,eAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;AAAA,MACtE,KAAK,WAAW;AACd,cAAM,MAAM,MAAM,SAAS,OAAO,IAAI,KAAK;AAC3C,eAAO,aAAa,GAAG,EAAE;AAAA,MAC3B;AAAA,MACA,KAAK,eAAe;AAClB,cAAM,MAAM,MAAM,SAAS,OAAO,IAAI,KAAK;AAC3C,eAAO,aAAa,GAAG,EAAE;AAAA,MAC3B;AAAA,MACA;AACE,eAAO;AAAA,IACX;AAAA,EACF;AACF;;;ACpKO,IAAM,wBAAwB,CAAC,OAAO,oBAAoB,kBAAkB;AAmCnF,SAAS,SAAS,OAAe,OAAuB;AACtD,QAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;AAC3D,MAAI,CAAC,OAAO,SAAS,EAAE,GAAG;AACxB,UAAM,IAAI;AAAA,MACR,mBAAmB,KAAK,sCAAsC,KAAK,UAAU,KAAK,CAAC;AAAA,IACrF;AAAA,EACF;AACA,SAAO;AACT;AAGO,SAAS,yBAAyB,UAA4B,SAAS,YAAkB;AAC9F,WAAS,SAAS,IAAI,GAAG,MAAM,KAAK;AACpC,MAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GAAG;AACzE,UAAM,IAAI,gBAAgB,mBAAmB,MAAM,qCAAqC;AAAA,EAC1F;AACA,MAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAAG;AAC/E,UAAM,IAAI,gBAAgB,mBAAmB,MAAM,wCAAwC;AAAA,EAC7F;AACA,MAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAAU;AACrE,UAAM,IAAI,gBAAgB,mBAAmB,MAAM,2CAA2C;AAAA,EAChG;AACA,aAAW,CAAC,KAAK,KAAK,KAAK,OAAO,QAAQ,SAAS,OAAO,GAAG;AAC3D,QAAI,CAAE,sBAA4C,SAAS,GAAG,GAAG;AAE/D,YAAM,IAAI;AAAA,QACR,mBAAmB,MAAM,iCAAiC,GAAG,mBAAc,sBAAsB,KAAK,IAAI,CAAC;AAAA,MAC7G;AAAA,IACF;AACA,QAAI,UAAU,UAAa,CAAC,OAAO,SAAS,KAAK,GAAG;AAClD,YAAM,IAAI;AAAA,QACR,mBAAmB,MAAM,YAAY,GAAG,4BAA4B,OAAO,KAAK,CAAC;AAAA,MACnF;AAAA,IACF;AAAA,EACF;AACF;AASO,SAAS,wBACd,QACA,MACkB;AAClB,QAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;AAC5F,MAAI,CAAC,OAAO,SAAS,KAAK,GAAG;AAC3B,UAAM,IAAI;AAAA,MACR,oDAA+C,OAAO,CAAC;AAAA,IACzD;AAAA,EACF;AACA,QAAM,WAA6B,EAAE,GAAG,MAAM,SAAS,EAAE,kBAAkB,MAAM,EAAE;AACnF,2BAAyB,QAAQ;AACjC,SAAO;AACT;AASO,SAAS,sBACd,QACA,MACkB;AAClB,QAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;AAChE,MAAI,CAAC,OAAO,SAAS,GAAG,GAAG;AACzB,UAAM,IAAI;AAAA,MACR;AAAA,IACF;AAAA,EACF;AACA,QAAM,WAA6B,EAAE,GAAG,MAAM,SAAS,EAAE,IAAI,EAAE;AAC/D,2BAAyB,QAAQ;AACjC,SAAO;AACT;AAcO,SAAS,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;AAClB,QAAM,YAAY,QAAQ,aAAa;AACvC,MAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAAG;AAChD,UAAM,IAAI;AAAA,MACR,4EAAuE,SAAS;AAAA,IAClF;AAAA,EACF;AACA,QAAM,aAAa,oBAAI,IAAoB;AAC3C,aAAW,QAAQ,QAAQ;AACzB,QAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAAG;AACrC,YAAM,IAAI;AAAA,QACR,yCAAyC,KAAK,MAAM;AAAA,MACtD;AAAA,IACF;AACA,QAAI,WAAW,IAAI,KAAK,MAAM,GAAG;AAC/B,YAAM,IAAI,gBAAgB,qDAAqD,KAAK,MAAM,GAAG;AAAA,IAC/F;AACA,eAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;AAAA,EAC7C;AACA,MAAI,SAAS;AACb,MAAI,SAAS;AACb,aAAW,KAAK,QAAQ;AACtB,UAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;AACrC,QAAI,UAAU,OAAW;AACzB,QAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAAG;AAC7B,YAAM,IAAI;AAAA,QACR,kDAAkD,EAAE,MAAM;AAAA,MAC5D;AAAA,IACF;AACA;AACA,QAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,UAAW;AAAA,EAC9C;AACA,MAAI,WAAW,GAAG;AAChB,UAAM,IAAI;AAAA,MACR;AAAA,IACF;AAAA,EACF;AACA,QAAM,WAA6B,EAAE,GAAG,MAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO,EAAE;AAC7F,2BAAyB,QAAQ;AACjC,SAAO;AACT;AAUO,SAAS,sBAAsB,UAA8B,CAAC,GAAkB;AACrF,aAAW,KAAK,QAAS,0BAAyB,CAAC;AACnD,QAAM,QAAQ,QAAQ,IAAI,CAAC,OAAO,EAAE,GAAG,GAAG,SAAS,EAAE,GAAG,EAAE,QAAQ,EAAE,EAAE;AACtE,SAAO;AAAA,IACL,MAAM,OAAO,UAAU;AACrB,+BAAyB,QAAQ;AACjC,YAAM,KAAK,EAAE,GAAG,UAAU,SAAS,EAAE,GAAG,SAAS,QAAQ,EAAE,CAAC;AAAA,IAC9D;AAAA,IACA,MAAM,QAAQ,SAAS;AACrB,YAAM,MAAM,MAAM,IAAI,CAAC,OAAO,EAAE,GAAG,GAAG,SAAS,EAAE,GAAG,EAAE,QAAQ,EAAE,EAAE;AAClE,aAAO,YAAY,SAAY,MAAM,IAAI,OAAO,CAAC,MAAM,EAAE,YAAY,OAAO;AAAA,IAC9E;AAAA,EACF;AACF;AAQO,SAAS,kBAAkB,MAA6B;AAC7D,SAAO;AAAA,IACL,MAAM,OAAO,UAAU;AACrB,+BAAyB,QAAQ;AACjC,YAAM,KAAK,MAAM,OAAO,aAAkB;AAC1C,YAAM,UAAU,MAAM,OAAO,MAAW;AACxC,YAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;AACzD,YAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,CAAC;AAAA,GAAM,MAAM;AAAA,IACnE;AAAA,IACA,MAAM,QAAQ,SAAS;AACrB,YAAM,KAAK,MAAM,OAAO,aAAkB;AAC1C,UAAI;AACJ,UAAI;AACF,cAAM,MAAM,GAAG,SAAS,MAAM,MAAM;AAAA,MACtC,SAAS,KAAK;AACZ,YAAK,IAA8B,SAAS,SAAU,QAAO,CAAC;AAC9D,cAAM;AAAA,MACR;AACA,YAAM,QAAQ,IAAI,MAAM,IAAI;AAC5B,YAAM,YAAgC,CAAC;AACvC,eAAS,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;AACrC,cAAM,OAAO,MAAM,CAAC;AACpB,YAAI,KAAK,KAAK,EAAE,WAAW,EAAG;AAC9B,YAAI;AACJ,YAAI;AACF,mBAAS,KAAK,MAAM,IAAI;AAAA,QAC1B,SAAS,KAAK;AACZ,gBAAM,IAAI;AAAA,YACR,4BAA4B,IAAI,6BAA6B,IAAI,CAAC,KAAM,IAAc,OAAO;AAAA,UAC/F;AAAA,QACF;AACA,cAAM,WAAW;AACjB,iCAAyB,UAAU,GAAG,IAAI,IAAI,IAAI,CAAC,EAAE;AACrD,kBAAU,KAAK,QAAQ;AAAA,MACzB;AACA,aAAO,YAAY,SAAY,YAAY,UAAU,OAAO,CAAC,MAAM,EAAE,YAAY,OAAO;AAAA,IAC1F;AAAA,EACF;AACF;AAmDA,IAAM,aAAa;AAEZ,SAAS,oBACd,SACA,MACgB;AAChB,QAAM,SAAS,SAAS,KAAK,MAAM,MAAM;AACzC,QAAM,SAAS,KAAK,YAAY,UAAU;AAC1C,QAAM,eAAe,KAAK,YAAY,gBAAgB;AACtD,QAAM,kBAAkB,KAAK,YAAY,mBAAmB;AAC5D,QAAM,iBAAiB,KAAK,YAAY,kBAAkB;AAE1D,QAAM,UAAU,oBAAI,IAAwD;AAC5E,aAAW,YAAY,SAAS;AAC9B,6BAAyB,QAAQ;AACjC,UAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,OAAO,IAAI;AAC9E,UAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;AAC9C,QAAI,KAAK,EAAE,GAAG,UAAU,KAAK,CAAC;AAC9B,YAAQ,IAAI,SAAS,SAAS,GAAG;AAAA,EACnC;AAEA,QAAM,WAA4B,CAAC;AACnC,QAAM,SAAmB,CAAC;AAC1B,QAAM,sBAAgC,CAAC;AAEvC,aAAW,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,EAAE,KAAK,GAAG;AAChD,UAAM,QAAQ,QAAQ,IAAI,OAAO,EAAG,KAAK,CAAC,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;AAClE,UAAM,SAAS,MAAM,MAAM,SAAS,CAAC;AAErC,UAAM,WAAW,SAAS,OAAO,QAAQ;AACzC,QAAI,UAAU,gBAAgB;AAC5B,aAAO;AAAA,QACL,UAAU,OAAO,mCAA8B,OAAO,EAAE,OAAO,QAAQ,QAAQ,CAAC,CAAC,wBAAwB,cAAc;AAAA,MACzH;AAAA,IACF;AAKA,QAAI,YAAY;AAChB,aAAS,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;AACrC,UAAI,MAAM,CAAC,EAAG,eAAe,MAAM,IAAI,CAAC,EAAG,WAAY,aAAY;AAAA,IACrE;AACA,QAAI,aAAa,GAAG;AAClB,YAAM,eAAe,MAClB,MAAM,SAAS,EACf;AAAA,QACC,CAAC,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB;AAAA,MAC9C;AACF,UAAI,CAAC,cAAc;AACjB,eAAO;AAAA,UACL,UAAU,OAAO,qBAAqB,MAAM,YAAY,CAAC,EAAG,UAAU,aAAQ,MAAM,SAAS,EAAG,UAAU,QAAQ,MAAM,SAAS,EAAG,EAAE;AAAA,QACxI;AAAA,MACF;AAAA,IACF;AAEA,eAAW,UAAU,uBAAuB;AAC1C,YAAM,SAAS,MACZ,OAAO,CAAC,MAAM,OAAO,EAAE,QAAQ,MAAM,MAAM,QAAQ,EACnD,IAAI,CAAC,MAAM,EAAE,QAAQ,MAAM,CAAE;AAChC,UAAI,OAAO,WAAW,EAAG;AAEzB,YAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;AAC1D,UAAI,YAAY,UAAU;AACxB,4BAAoB,KAAK,GAAG,OAAO,IAAI,MAAM,EAAE;AAEjD,YAAM,UAAU,OAAO,OAAO,SAAS,CAAC;AACxC,YAAM,WAAW,OAAO,CAAC;AACzB,YAAM,QAAQ,UAAU;AACxB,YAAM,UAAoB,CAAC;AAC3B,UAAI,YAAY,UAAU,iBAAiB;AACzC,gBAAQ;AAAA,UACN,6CAA6C,YAAY,OAAO,iBAAiB,YAAY,WAAW,QAAQ,CAAC,CAAC;AAAA,QACpH;AAAA,MACF;AACA,UAAI,WAAW,SAAS,UAAU,QAAQ;AACxC,gBAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,CAAC,gBAAgB,MAAM,EAAE;AAAA,MAChE;AACA,UAAI,WAAW,sBAAsB,WAAW,UAAU,cAAc;AACtE,gBAAQ;AAAA,UACN,6BAA6B,WAAW,SAAS,QAAQ,CAAC,CAAC,kBAAkB,SAAS,QAAQ,CAAC,CAAC,WAAW,YAAY;AAAA,QACzH;AAAA,MACF;AACA,UAAI,WAAW,sBAAsB,WAAW,UAAU,iBAAiB;AACzE,gBAAQ;AAAA,UACN,6BAA6B,WAAW,SAAS,QAAQ,CAAC,CAAC,kBAAkB,SAAS,QAAQ,CAAC,CAAC,WAAW,eAAe;AAAA,QAC5H;AAAA,MACF;AAEA,YAAM,UAAU,QAAQ,SAAS;AACjC,YAAM,QAAuB;AAAA,QAC3B;AAAA,QACA;AAAA,QACA,OAAO,YAAY;AAAA,QACnB;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,MACF;AACA,UAAI,SAAS;AACX,cAAM,SAAS,QAAQ,KAAK,IAAI;AAChC,eAAO,KAAK,UAAU,OAAO,KAAK,MAAM,KAAK,MAAM,MAAM,EAAE;AAAA,MAC7D;AACA,eAAS,KAAK,KAAK;AAAA,IACrB;AAAA,EACF;AAEA,SAAO,EAAE,UAAU,QAAQ,SAAS,OAAO,WAAW,GAAG,oBAAoB;AAC/E;AAiBO,SAAS,gBAAgB,QAAyC;AACvE,SAAO,EAAE,SAAS,OAAO,SAAS,QAAQ,CAAC,GAAG,OAAO,MAAM,EAAE;AAC/D;","names":[]}
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,\n * joins on runId, computes Pearson + Spearman + bootstrap CI for every\n * (evalMetric, outcomeMetric) pair the caller declares.\n *\n * Without this number the framework is ornamental. With it and r > 0.6\n * the framework is a moat — no other agent-eval tool publishes one.\n */\n\nimport { pearsonR, spearmanR } from '../statistics'\nimport { aggregateLlm, llmSpans } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run (defaults cover score/pass/durationMs/costUsd/tokens). */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number }\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const reduction = options.reduction ?? 'latest'\n const maxLag = options.maxCaptureLagMs ?? Infinity\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of evalMetrics) {\n for (const om of outcomeMetricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag)\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n for (const em of evalMetrics) {\n const extract = em.extract ?? defaultExtract(em.id)\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of outcomeMetricNames) {\n const values = eligible\n .map((o) => o.metrics[om])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, reduction, eligible)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n }\n }\n joined++\n }\n\n const results: CorrelationResult[] = pairs\n .filter((p) => p.xs.length >= 3)\n .map((p) => {\n const pearson = pearsonR(p.xs, p.ys)\n const spearman = spearmanR(p.xs, p.ys)\n const pearsonCi95 = bootstrapPearsonCi(p.xs, p.ys, options.bootstrapIterations ?? 500)\n const verdict: CorrelationResult['verdict'] =\n Math.abs(pearson) >= 0.7 ? 'strong' : Math.abs(pearson) >= 0.4 ? 'moderate' : 'weak'\n return {\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n pearson,\n spearman,\n pearsonCi95,\n verdict,\n }\n })\n\n return { pairs: results, joinedSamples: joined, skippedRuns: skipped }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n kind: 'latest' | 'mean' | 'max',\n outcomes: DeploymentOutcome[],\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((a, b) => a + b, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest': pick the outcome captured last, then lookup its metric\n const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0]\n if (!latest) return null\n const latestKey = Object.keys(latest.metrics)[0]\n const v = latestKey !== undefined ? latest.metrics[latestKey] : undefined\n // For 'latest' we already have `values` aligned; use the last-captured one\n const paired = outcomes\n .map((o) => {\n const k = Object.keys(o.metrics)[0]\n return {\n at: o.capturedAt,\n v: k !== undefined ? values.find((x) => o.metrics[k] === x) : undefined,\n }\n })\n .filter((p) => p.v !== undefined)\n if (paired.length === 0) return v ?? null\n return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null\n}\n\nfunction bootstrapPearsonCi(\n xs: number[],\n ys: number[],\n iterations: number,\n): { lower: number; upper: number } {\n const n = xs.length\n if (n < 3) return { lower: NaN, upper: NaN }\n const rs: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = new Array(n)\n const ry: number[] = new Array(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(Math.random() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) rs.push(r)\n }\n rs.sort((a, b) => a - b)\n if (rs.length === 0) return { lower: NaN, upper: NaN }\n return {\n lower: rs[Math.floor(0.025 * rs.length)]!,\n upper: rs[Math.min(rs.length - 1, Math.floor(0.975 * rs.length))]!,\n }\n}\n\nfunction defaultExtract(metric: string): (run: Run, store: TraceStore) => Promise<number | null> {\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).costUsd\n }\n case 'inputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).inputTokens\n }\n default:\n return null\n }\n }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nexport const SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAyDA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;CACjC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;CAC9D,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAEA,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,SAAS,QAAQ,mBAAmB;CAE1C,MAAM,QAA0F,CAAC;CACjG,KAAK,MAAM,MAAM,aACf,KAAK,MAAM,MAAM,oBACf,MAAM,KAAK;EAAE,YAAY,GAAG;EAAI,eAAe;EAAI,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAIvE,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,MAAM,WAAW,GAAG,QAAQ,MAAM,EAAE,aAAa,IAAI,aAAa,MAAM;EACxE,IAAI,SAAS,WAAW,GAAG;GACzB;GACA;EACF;EAEA,KAAK,MAAM,MAAM,aAAa;GAE5B,MAAM,IAAI,OADM,GAAG,WAAW,eAAe,GAAG,EAAE,EAAA,CAC1B,KAAK,UAAU;GACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;GAEvC,KAAK,MAAM,MAAM,oBAAoB;IACnC,MAAM,SAAS,SACZ,KAAK,MAAM,EAAE,QAAQ,GAAG,CAAC,CACzB,QAAQ,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;IACzE,IAAI,OAAO,WAAW,GAAG;IACzB,MAAM,IAAI,OAAO,QAAQ,WAAW,QAAQ;IAC5C,IAAI,MAAM,MAAM;IAChB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;IAC/E,KAAK,GAAG,KAAK,CAAC;IACd,KAAK,GAAG,KAAK,CAAC;GAChB;EACF;EACA;CACF;CAqBA,OAAO;EAAE,OAnB4B,MAClC,QAAQ,MAAM,EAAE,GAAG,UAAU,CAAC,CAAC,CAC/B,KAAK,MAAM;GACV,MAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;GACnC,MAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;GACrC,MAAM,cAAc,mBAAmB,EAAE,IAAI,EAAE,IAAI,QAAQ,uBAAuB,GAAG;GACrF,MAAM,UACJ,KAAK,IAAI,OAAO,KAAK,KAAM,WAAW,KAAK,IAAI,OAAO,KAAK,KAAM,aAAa;GAChF,OAAO;IACL,YAAY,EAAE;IACd,eAAe,EAAE;IACjB,GAAG,EAAE,GAAG;IACR;IACA;IACA;IACA;GACF;EACF,CAEoB;EAAG,eAAe;EAAQ,aAAa;CAAQ;AACvE;AAIA,SAAS,OACP,QACA,MACA,UACe;CACf,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,SAAS,QAAQ,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACvE,IAAI,SAAS,OAAO,OAAO,KAAK,IAAI,GAAG,MAAM;CAE7C,MAAM,SAAS,CAAC,GAAG,QAAQ,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC;CACzE,IAAI,CAAC,QAAQ,OAAO;CACpB,MAAM,YAAY,OAAO,KAAK,OAAO,OAAO,CAAC,CAAC;CAC9C,MAAM,IAAI,cAAc,KAAA,IAAY,OAAO,QAAQ,aAAa,KAAA;CAEhE,MAAM,SAAS,SACZ,KAAK,MAAM;EACV,MAAM,IAAI,OAAO,KAAK,EAAE,OAAO,CAAC,CAAC;EACjC,OAAO;GACL,IAAI,EAAE;GACN,GAAG,MAAM,KAAA,IAAY,OAAO,MAAM,MAAM,EAAE,QAAQ,OAAO,CAAC,IAAI,KAAA;EAChE;CACF,CAAC,CAAC,CACD,QAAQ,MAAM,EAAE,MAAM,KAAA,CAAS;CAClC,IAAI,OAAO,WAAW,GAAG,OAAO,KAAK;CACrC,OAAO,OAAO,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,KAAK;AACrD;AAEA,SAAS,mBACP,IACA,IACA,YACkC;CAClC,MAAM,IAAI,GAAG;CACb,IAAI,IAAI,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CAC3C,MAAM,KAAe,CAAC;CACtB,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,MAAM,KAAK,MAAM,KAAK,OAAO,IAAI,CAAC;GACxC,GAAG,KAAK,GAAG;GACX,GAAG,KAAK,GAAG;EACb;EACA,MAAM,IAAI,SAAS,IAAI,EAAE;EACzB,IAAI,OAAO,SAAS,CAAC,GAAG,GAAG,KAAK,CAAC;CACnC;CACA,GAAG,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,IAAI,GAAG,WAAW,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CACrD,OAAO;EACL,OAAO,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM;EACtC,OAAO,GAAG,KAAK,IAAI,GAAG,SAAS,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM,CAAC;CACjE;AACF;AAEA,SAAS,eAAe,QAAyE;CAC/F,OAAO,OAAO,KAAK,UAAU;EAC3B,QAAQ,QAAR;GACE,KAAK;GACL,KAAK,gBACH,OAAO,IAAI,SAAS,SAAS;GAC/B,KAAK,QACH,OAAO,IAAI,SAAS,SAAS,OAAO,IAAI;GAC1C,KAAK,cACH,OAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;GACtE,KAAK,WAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,eAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,SACE,OAAO;EACX;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACpKA,MAAa,wBAAwB;CAAC;CAAO;CAAoB;AAAkB;AAmCnF,SAAS,SAAS,OAAe,OAAuB;CACtD,MAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;CAC3D,IAAI,CAAC,OAAO,SAAS,EAAE,GACrB,MAAM,IAAI,gBACR,mBAAmB,MAAM,qCAAqC,KAAK,UAAU,KAAK,GACpF;CAEF,OAAO;AACT;;AAGA,SAAgB,yBAAyB,UAA4B,SAAS,YAAkB;CAC9F,SAAS,SAAS,IAAI,GAAG,OAAO,IAAI;CACpC,IAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GACtE,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,oCAAoC;CAE1F,IAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAC5E,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,uCAAuC;CAE7F,IAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAC3D,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,0CAA0C;CAEhG,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,SAAS,OAAO,GAAG;EAC3D,IAAI,CAAE,sBAA4C,SAAS,GAAG,GAE5D,MAAM,IAAI,gBACR,mBAAmB,OAAO,gCAAgC,IAAI,aAAa,sBAAsB,KAAK,IAAI,GAC5G;EAEF,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,GAC/C,MAAM,IAAI,gBACR,mBAAmB,OAAO,WAAW,IAAI,2BAA2B,OAAO,KAAK,GAClF;CAEJ;AACF;;;;;;AASA,SAAgB,wBACd,QACA,MACkB;CAClB,MAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;CAC5F,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gBACR,+CAA+C,OAAO,EAAE,iEAC1D;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,MAAM;CAAE;CACnF,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AASA,SAAgB,sBACd,QACA,MACkB;CAClB,MAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;CAChE,IAAI,CAAC,OAAO,SAAS,GAAG,GACtB,MAAM,IAAI,gBACR,4FACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,IAAI;CAAE;CAC/D,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AAcA,SAAgB,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;CAClB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAC7C,MAAM,IAAI,gBACR,uEAAuE,WACzE;CAEF,MAAM,6BAAa,IAAI,IAAoB;CAC3C,KAAK,MAAM,QAAQ,QAAQ;EACzB,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,yCAAyC,KAAK,OAAO,8BACvD;EAEF,IAAI,WAAW,IAAI,KAAK,MAAM,GAC5B,MAAM,IAAI,gBAAgB,qDAAqD,KAAK,OAAO,EAAE;EAE/F,WAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;CAC7C;CACA,IAAI,SAAS;CACb,IAAI,SAAS;CACb,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;EACrC,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,kDAAkD,EAAE,OAAO,gBAC7D;EAEF;EACA,IAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,WAAW;CAC9C;CACA,IAAI,WAAW,GACb,MAAM,IAAI,gBACR,yGACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO;CAAE;CAC7F,yBAAyB,QAAQ;CACjC,OAAO;AACT;AAUA,SAAgB,sBAAsB,UAA8B,CAAC,GAAkB;CACrF,KAAK,MAAM,KAAK,SAAS,yBAAyB,CAAC;CACnD,MAAM,QAAQ,QAAQ,KAAK,OAAO;EAAE,GAAG;EAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;CAAE,EAAE;CACtE,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK;IAAE,GAAG;IAAU,SAAS,EAAE,GAAG,SAAS,QAAQ;GAAE,CAAC;EAC9D;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,MAAM,MAAM,KAAK,OAAO;IAAE,GAAG;IAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;GAAE,EAAE;GAClE,OAAO,YAAY,KAAA,IAAY,MAAM,IAAI,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC9E;CACF;AACF;;;;;;;AAQA,SAAgB,kBAAkB,MAA6B;CAC7D,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,EAAE,KAAK,MAAM;EACnE;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;GACJ,IAAI;IACF,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;GACtC,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;GACA,MAAM,QAAQ,IAAI,MAAM,IAAI;GAC5B,MAAM,YAAgC,CAAC;GACvC,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;IACrC,MAAM,OAAO,MAAM;IACnB,IAAI,KAAK,KAAK,CAAC,CAAC,WAAW,GAAG;IAC9B,IAAI;IACJ,IAAI;KACF,SAAS,KAAK,MAAM,IAAI;IAC1B,SAAS,KAAK;KACZ,MAAM,IAAI,gBACR,4BAA4B,KAAK,4BAA4B,IAAI,EAAE,IAAK,IAAc,SACxF;IACF;IACA,MAAM,WAAW;IACjB,yBAAyB,UAAU,GAAG,KAAK,GAAG,IAAI,GAAG;IACrD,UAAU,KAAK,QAAQ;GACzB;GACA,OAAO,YAAY,KAAA,IAAY,YAAY,UAAU,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC1F;CACF;AACF;AAmDA,MAAM,aAAa;AAEnB,SAAgB,oBACd,SACA,MACgB;CAChB,MAAM,SAAS,SAAS,KAAK,MAAM,MAAM;CACzC,MAAM,SAAS,KAAK,YAAY,UAAU;CAC1C,MAAM,eAAe,KAAK,YAAY,gBAAgB;CACtD,MAAM,kBAAkB,KAAK,YAAY,mBAAmB;CAC5D,MAAM,iBAAiB,KAAK,YAAY,kBAAkB;CAE1D,MAAM,0BAAU,IAAI,IAAwD;CAC5E,KAAK,MAAM,YAAY,SAAS;EAC9B,yBAAyB,QAAQ;EACjC,MAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,QAAQ,GAAG;EAC9E,MAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;EAC9C,IAAI,KAAK;GAAE,GAAG;GAAU;EAAK,CAAC;EAC9B,QAAQ,IAAI,SAAS,SAAS,GAAG;CACnC;CAEA,MAAM,WAA4B,CAAC;CACnC,MAAM,SAAmB,CAAC;CAC1B,MAAM,sBAAgC,CAAC;CAEvC,KAAK,MAAM,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,KAAK,GAAG;EAChD,MAAM,QAAQ,QAAQ,IAAI,OAAO,CAAC,CAAE,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;EAClE,MAAM,SAAS,MAAM,MAAM,SAAS;EAEpC,MAAM,WAAW,SAAS,OAAO,QAAQ;EACzC,IAAI,UAAU,gBACZ,OAAO,KACL,UAAU,QAAQ,6BAA6B,OAAO,GAAG,MAAM,QAAQ,QAAQ,CAAC,EAAE,uBAAuB,eAAe,GAC1H;EAMF,IAAI,YAAY;EAChB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAChC,IAAI,MAAM,EAAE,CAAE,eAAe,MAAM,IAAI,EAAE,CAAE,YAAY,YAAY;EAErE,IAAI,aAAa,GAQX;OAAA,CAPiB,MAClB,MAAM,SAAS,CAAC,CAChB,MACE,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB,CAEhC,GACd,OAAO,KACL,UAAU,QAAQ,oBAAoB,MAAM,YAAY,EAAE,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,GAAG,gIACzI;EAAA;EAIJ,KAAK,MAAM,UAAU,uBAAuB;GAC1C,MAAM,SAAS,MACZ,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,KAAK,MAAM,EAAE,QAAQ,OAAQ;GAChC,IAAI,OAAO,WAAW,GAAG;GAEzB,MAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;GAC1D,IAAI,YAAY,UAAU,qBACxB,oBAAoB,KAAK,GAAG,QAAQ,GAAG,QAAQ;GAEjD,MAAM,UAAU,OAAO,OAAO,SAAS;GACvC,MAAM,WAAW,OAAO;GACxB,MAAM,QAAQ,UAAU;GACxB,MAAM,UAAoB,CAAC;GAC3B,IAAI,YAAY,UAAU,iBACxB,QAAQ,KACN,6CAA6C,YAAY,QAAQ,gBAAgB,YAAY,WAAW,QAAQ,CAAC,EAAE,EACrH;GAEF,IAAI,WAAW,SAAS,UAAU,QAChC,QAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,EAAE,eAAe,QAAQ;GAEhE,IAAI,WAAW,sBAAsB,WAAW,UAAU,cACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,aAAa,EAC1H;GAEF,IAAI,WAAW,sBAAsB,WAAW,UAAU,iBACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,gBAAgB,EAC7H;GAGF,MAAM,UAAU,QAAQ,SAAS;GACjC,MAAM,QAAuB;IAC3B;IACA;IACA,OAAO,YAAY;IACnB;IACA;IACA;IACA;GACF;GACA,IAAI,SAAS;IACX,MAAM,SAAS,QAAQ,KAAK,IAAI;IAChC,OAAO,KAAK,UAAU,QAAQ,IAAI,OAAO,IAAI,MAAM,QAAQ;GAC7D;GACA,SAAS,KAAK,KAAK;EACrB;CACF;CAEA,OAAO;EAAE;EAAU;EAAQ,SAAS,OAAO,WAAW;EAAG;CAAoB;AAC/E;;;;;;;;;AAiBA,SAAgB,gBAAgB,QAAyC;CACvE,OAAO;EAAE,SAAS,OAAO;EAAS,QAAQ,CAAC,GAAG,OAAO,MAAM;CAAE;AAC/D"}
@@ -0,0 +1,239 @@
1
+ //#region src/metrics.ts
2
+ /** Per-1K token pricing for exact model ids. */
3
+ const MODEL_PRICING = {
4
+ "gpt-4o": {
5
+ input: .0025,
6
+ output: .01
7
+ },
8
+ "gpt-4o-mini": {
9
+ input: 15e-5,
10
+ output: 6e-4
11
+ },
12
+ "gpt-4-turbo": {
13
+ input: .01,
14
+ output: .03
15
+ },
16
+ "claude-sonnet-4-20250514": {
17
+ input: .003,
18
+ output: .015
19
+ },
20
+ "claude-opus-4-20250514": {
21
+ input: .015,
22
+ output: .075
23
+ },
24
+ "claude-3-haiku-20240307": {
25
+ input: 25e-5,
26
+ output: .00125
27
+ }
28
+ };
29
+ /** Family-level pricing fallbacks (per-1K), matched against a normalized id
30
+ * after exact lookup misses. Ordered — first match wins. Covers the model
31
+ * ids actually used through the Tangle router + cli-bridge harnesses
32
+ * (`claude-code/sonnet`, `opencode/zai-coding-plan/glm-5.1`,
33
+ * `kimi-code/kimi-k2.6`, `deepseek-v4-pro`, `anthropic/claude-sonnet-4-6`, …),
34
+ * none of which appear in the exact table above — without this they priced
35
+ * to a silent $0, blanking every cost/Pareto axis downstream. */
36
+ const FAMILY_PRICING = [
37
+ [/claude.*opus/, {
38
+ input: .015,
39
+ output: .075
40
+ }],
41
+ [/claude.*haiku/, {
42
+ input: 8e-4,
43
+ output: .004
44
+ }],
45
+ [/claude.*sonnet|claude-code|claude-sonnet/, {
46
+ input: .003,
47
+ output: .015
48
+ }],
49
+ [/gpt-4o-mini/, {
50
+ input: 15e-5,
51
+ output: 6e-4
52
+ }],
53
+ [/gpt-5|gpt-4\.1|o[134]\b/, {
54
+ input: .00125,
55
+ output: .01
56
+ }],
57
+ [/gpt-4o|gpt-4/, {
58
+ input: .0025,
59
+ output: .01
60
+ }],
61
+ [/deepseek/, {
62
+ input: 3e-4,
63
+ output: .0011
64
+ }],
65
+ [/glm|zhipu|zai/, {
66
+ input: 6e-4,
67
+ output: .0022
68
+ }],
69
+ [/kimi|moonshot/, {
70
+ input: 6e-4,
71
+ output: .0025
72
+ }],
73
+ [/qwen/, {
74
+ input: 4e-4,
75
+ output: .0012
76
+ }],
77
+ [/gemini.*flash/, {
78
+ input: 1e-4,
79
+ output: 4e-4
80
+ }],
81
+ [/gemini/, {
82
+ input: .00125,
83
+ output: .005
84
+ }],
85
+ [/llama/, {
86
+ input: 2e-4,
87
+ output: 6e-4
88
+ }]
89
+ ];
90
+ /** Normalize a model id for pricing: drop a `@snapshot` suffix, lowercase,
91
+ * and keep the final harness/provider-prefixed segment so family regexes
92
+ * match (`opencode/zai-coding-plan/glm-5.1` → `glm-5.1`). */
93
+ function normalizeModelId(model) {
94
+ return (model.split("@")[0] ?? model).trim().toLowerCase();
95
+ }
96
+ /** Resolve pricing for a model id: exact table, then family fallback.
97
+ * Returns null when the id matches nothing (caller decides — never a
98
+ * silent-zero masquerading as a real $0 cost). */
99
+ function resolveModelPricing(model) {
100
+ if (MODEL_PRICING[model]) return MODEL_PRICING[model];
101
+ const id = normalizeModelId(model);
102
+ if (MODEL_PRICING[id]) return MODEL_PRICING[id];
103
+ for (const [pattern, price] of FAMILY_PRICING) if (pattern.test(id)) return price;
104
+ return null;
105
+ }
106
+ /** True when `model` has known pricing (exact or family). Lets cost-aware
107
+ * callers distinguish a real $0 from an unpriced model. */
108
+ function isModelPriced(model) {
109
+ return resolveModelPricing(model) !== null;
110
+ }
111
+ const warnedUnpricedModels = /* @__PURE__ */ new Set();
112
+ /** Estimate token count from string length (chars / 4 approximation) */
113
+ function estimateTokens(text) {
114
+ return Math.ceil(text.length / 4);
115
+ }
116
+ /** Calculate cost in USD from token counts and model. Unknown models warn
117
+ * once (not a silent zero) and return 0 so callers that ignore pricing keep
118
+ * working; cost-sensitive callers should gate on {@link isModelPriced}. */
119
+ function estimateCost(inputTokens, outputTokens, model) {
120
+ const pricing = resolveModelPricing(model);
121
+ if (!pricing) {
122
+ if (!warnedUnpricedModels.has(model)) {
123
+ warnedUnpricedModels.add(model);
124
+ console.warn(`estimateCost: no pricing for model "${model}" — returning 0; add it to MODEL_PRICING/FAMILY_PRICING (cost/Pareto axes will be blank until then)`);
125
+ }
126
+ return 0;
127
+ }
128
+ return inputTokens / 1e3 * pricing.input + outputTokens / 1e3 * pricing.output;
129
+ }
130
+ /**
131
+ * TokenCounter — accumulates token usage and cost across turns.
132
+ */
133
+ var TokenCounter = class {
134
+ totalInput = 0;
135
+ totalOutput = 0;
136
+ totalCost = 0;
137
+ model;
138
+ constructor(model = "gpt-4o") {
139
+ this.model = model;
140
+ }
141
+ /** Record tokens for a turn, returns per-turn cost */
142
+ record(inputTokens, outputTokens) {
143
+ this.totalInput += inputTokens;
144
+ this.totalOutput += outputTokens;
145
+ const cost = estimateCost(inputTokens, outputTokens, this.model);
146
+ this.totalCost += cost;
147
+ return cost;
148
+ }
149
+ /** Estimate and record from raw text */
150
+ recordFromText(inputText, outputText) {
151
+ const inputTokens = estimateTokens(inputText);
152
+ const outputTokens = estimateTokens(outputText);
153
+ return {
154
+ inputTokens,
155
+ outputTokens,
156
+ cost: this.record(inputTokens, outputTokens)
157
+ };
158
+ }
159
+ getTotalInput() {
160
+ return this.totalInput;
161
+ }
162
+ getTotalOutput() {
163
+ return this.totalOutput;
164
+ }
165
+ getTotalCost() {
166
+ return this.totalCost;
167
+ }
168
+ };
169
+ /**
170
+ * MetricsCollector — collects per-turn metrics from the product.
171
+ *
172
+ * After each turn, queries the product's APIs to measure state changes.
173
+ */
174
+ var MetricsCollector = class {
175
+ client;
176
+ workspaceId;
177
+ metrics = [];
178
+ constructor(client, workspaceId) {
179
+ this.client = client;
180
+ this.workspaceId = workspaceId;
181
+ }
182
+ /** Collect metrics after a turn completes */
183
+ async collect(turn, responseLatencyMs, responseChars, codeBlocksProduced, blocksExtracted, completionCriteriaMet, completionCriteriaTotal, qualityScore, inputTokens = 0, outputTokens = 0, estimatedCostUsd = 0) {
184
+ const state = await this.getState();
185
+ const m = {
186
+ turn,
187
+ timestamp: (/* @__PURE__ */ new Date()).toISOString(),
188
+ tasks: state.tasks,
189
+ events: state.events,
190
+ proposals: state.proposals,
191
+ vaultFiles: state.vaultFiles.length,
192
+ responseLatencyMs,
193
+ responseChars,
194
+ codeBlocksProduced,
195
+ blocksExtracted,
196
+ qualityScore,
197
+ inputTokens,
198
+ outputTokens,
199
+ estimatedCostUsd,
200
+ totalCostUsd: estimatedCostUsd,
201
+ completionPercent: completionCriteriaTotal > 0 ? completionCriteriaMet / completionCriteriaTotal * 100 : 0
202
+ };
203
+ this.metrics.push(m);
204
+ return m;
205
+ }
206
+ /** Get current product state */
207
+ async getState() {
208
+ const [tasks, events, approvals, vaultFiles] = await Promise.all([
209
+ this.client.getTasks(this.workspaceId),
210
+ this.client.getEvents(this.workspaceId),
211
+ this.client.getApprovals(this.workspaceId),
212
+ this.client.getVaultTree(this.workspaceId)
213
+ ]);
214
+ return {
215
+ tasks: tasks.length,
216
+ events: events.length,
217
+ proposals: {
218
+ pending: approvals.filter((a) => a.status === "pending").length,
219
+ approved: approvals.filter((a) => a.status === "approved").length,
220
+ rejected: approvals.filter((a) => a.status === "rejected").length
221
+ },
222
+ vaultFiles,
223
+ codeBlocks: 0,
224
+ generations: 0
225
+ };
226
+ }
227
+ /** Get all collected metrics */
228
+ getMetrics() {
229
+ return [...this.metrics];
230
+ }
231
+ /** Get convergence curve (completion% over turns) */
232
+ getConvergenceCurve() {
233
+ return this.metrics.map((m) => m.completionPercent);
234
+ }
235
+ };
236
+ //#endregion
237
+ export { estimateTokens as a, estimateCost as i, MetricsCollector as n, isModelPriced as o, TokenCounter as r, resolveModelPricing as s, MODEL_PRICING as t };
238
+
239
+ //# sourceMappingURL=metrics-C9YY1OcL.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"metrics-C9YY1OcL.js","names":[],"sources":["../src/metrics.ts"],"sourcesContent":["import type { ProductClient } from './client'\nimport type { DriverState, TurnMetrics } from './types'\n\ninterface TokenPrice {\n input: number\n output: number\n}\n\n/** Per-1K token pricing for exact model ids. */\nexport const MODEL_PRICING: Record<string, TokenPrice> = {\n 'gpt-4o': { input: 0.0025, output: 0.01 },\n 'gpt-4o-mini': { input: 0.00015, output: 0.0006 },\n 'gpt-4-turbo': { input: 0.01, output: 0.03 },\n 'claude-sonnet-4-20250514': { input: 0.003, output: 0.015 },\n 'claude-opus-4-20250514': { input: 0.015, output: 0.075 },\n 'claude-3-haiku-20240307': { input: 0.00025, output: 0.00125 },\n}\n\n/** Family-level pricing fallbacks (per-1K), matched against a normalized id\n * after exact lookup misses. Ordered — first match wins. Covers the model\n * ids actually used through the Tangle router + cli-bridge harnesses\n * (`claude-code/sonnet`, `opencode/zai-coding-plan/glm-5.1`,\n * `kimi-code/kimi-k2.6`, `deepseek-v4-pro`, `anthropic/claude-sonnet-4-6`, …),\n * none of which appear in the exact table above — without this they priced\n * to a silent $0, blanking every cost/Pareto axis downstream. */\nconst FAMILY_PRICING: Array<[RegExp, TokenPrice]> = [\n [/claude.*opus/, { input: 0.015, output: 0.075 }],\n [/claude.*haiku/, { input: 0.0008, output: 0.004 }],\n [/claude.*sonnet|claude-code|claude-sonnet/, { input: 0.003, output: 0.015 }],\n [/gpt-4o-mini/, { input: 0.00015, output: 0.0006 }],\n [/gpt-5|gpt-4\\.1|o[134]\\b/, { input: 0.00125, output: 0.01 }],\n [/gpt-4o|gpt-4/, { input: 0.0025, output: 0.01 }],\n [/deepseek/, { input: 0.0003, output: 0.0011 }],\n [/glm|zhipu|zai/, { input: 0.0006, output: 0.0022 }],\n [/kimi|moonshot/, { input: 0.0006, output: 0.0025 }],\n [/qwen/, { input: 0.0004, output: 0.0012 }],\n [/gemini.*flash/, { input: 0.0001, output: 0.0004 }],\n [/gemini/, { input: 0.00125, output: 0.005 }],\n [/llama/, { input: 0.0002, output: 0.0006 }],\n]\n\n/** Normalize a model id for pricing: drop a `@snapshot` suffix, lowercase,\n * and keep the final harness/provider-prefixed segment so family regexes\n * match (`opencode/zai-coding-plan/glm-5.1` → `glm-5.1`). */\nfunction normalizeModelId(model: string): string {\n return (model.split('@')[0] ?? model).trim().toLowerCase()\n}\n\n/** Resolve pricing for a model id: exact table, then family fallback.\n * Returns null when the id matches nothing (caller decides — never a\n * silent-zero masquerading as a real $0 cost). */\nexport function resolveModelPricing(model: string): TokenPrice | null {\n if (MODEL_PRICING[model]) return MODEL_PRICING[model]\n const id = normalizeModelId(model)\n if (MODEL_PRICING[id]) return MODEL_PRICING[id]\n for (const [pattern, price] of FAMILY_PRICING) {\n if (pattern.test(id)) return price\n }\n return null\n}\n\n/** True when `model` has known pricing (exact or family). Lets cost-aware\n * callers distinguish a real $0 from an unpriced model. */\nexport function isModelPriced(model: string): boolean {\n return resolveModelPricing(model) !== null\n}\n\nconst warnedUnpricedModels = new Set<string>()\n\n/** Estimate token count from string length (chars / 4 approximation) */\nexport function estimateTokens(text: string): number {\n return Math.ceil(text.length / 4)\n}\n\n/** Calculate cost in USD from token counts and model. Unknown models warn\n * once (not a silent zero) and return 0 so callers that ignore pricing keep\n * working; cost-sensitive callers should gate on {@link isModelPriced}. */\nexport function estimateCost(inputTokens: number, outputTokens: number, model: string): number {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n if (!warnedUnpricedModels.has(model)) {\n warnedUnpricedModels.add(model)\n console.warn(\n `estimateCost: no pricing for model \"${model}\" — returning 0; add it to ` +\n 'MODEL_PRICING/FAMILY_PRICING (cost/Pareto axes will be blank until then)',\n )\n }\n return 0\n }\n return (inputTokens / 1000) * pricing.input + (outputTokens / 1000) * pricing.output\n}\n\n/**\n * TokenCounter — accumulates token usage and cost across turns.\n */\nexport class TokenCounter {\n private totalInput = 0\n private totalOutput = 0\n private totalCost = 0\n private model: string\n\n constructor(model = 'gpt-4o') {\n this.model = model\n }\n\n /** Record tokens for a turn, returns per-turn cost */\n record(inputTokens: number, outputTokens: number): number {\n this.totalInput += inputTokens\n this.totalOutput += outputTokens\n const cost = estimateCost(inputTokens, outputTokens, this.model)\n this.totalCost += cost\n return cost\n }\n\n /** Estimate and record from raw text */\n recordFromText(\n inputText: string,\n outputText: string,\n ): { inputTokens: number; outputTokens: number; cost: number } {\n const inputTokens = estimateTokens(inputText)\n const outputTokens = estimateTokens(outputText)\n const cost = this.record(inputTokens, outputTokens)\n return { inputTokens, outputTokens, cost }\n }\n\n getTotalInput(): number {\n return this.totalInput\n }\n getTotalOutput(): number {\n return this.totalOutput\n }\n getTotalCost(): number {\n return this.totalCost\n }\n}\n\n/**\n * MetricsCollector — collects per-turn metrics from the product.\n *\n * After each turn, queries the product's APIs to measure state changes.\n */\nexport class MetricsCollector {\n private client: ProductClient\n private workspaceId: string\n private metrics: TurnMetrics[] = []\n constructor(client: ProductClient, workspaceId: string) {\n this.client = client\n this.workspaceId = workspaceId\n }\n\n /** Collect metrics after a turn completes */\n async collect(\n turn: number,\n responseLatencyMs: number,\n responseChars: number,\n codeBlocksProduced: number,\n blocksExtracted: number,\n completionCriteriaMet: number,\n completionCriteriaTotal: number,\n qualityScore?: number,\n inputTokens = 0,\n outputTokens = 0,\n estimatedCostUsd = 0,\n ): Promise<TurnMetrics> {\n const state = await this.getState()\n\n const m: TurnMetrics = {\n turn,\n timestamp: new Date().toISOString(),\n tasks: state.tasks,\n events: state.events,\n proposals: state.proposals,\n vaultFiles: state.vaultFiles.length,\n responseLatencyMs,\n responseChars,\n codeBlocksProduced,\n blocksExtracted,\n qualityScore,\n inputTokens,\n outputTokens,\n estimatedCostUsd,\n totalCostUsd: estimatedCostUsd,\n completionPercent:\n completionCriteriaTotal > 0 ? (completionCriteriaMet / completionCriteriaTotal) * 100 : 0,\n }\n\n this.metrics.push(m)\n return m\n }\n\n /** Get current product state */\n async getState(): Promise<DriverState> {\n const [tasks, events, approvals, vaultFiles] = await Promise.all([\n this.client.getTasks(this.workspaceId),\n this.client.getEvents(this.workspaceId),\n this.client.getApprovals(this.workspaceId),\n this.client.getVaultTree(this.workspaceId),\n ])\n\n return {\n tasks: tasks.length,\n events: events.length,\n proposals: {\n pending: approvals.filter((a) => a.status === 'pending').length,\n approved: approvals.filter((a) => a.status === 'approved').length,\n rejected: approvals.filter((a) => a.status === 'rejected').length,\n },\n vaultFiles,\n codeBlocks: 0,\n generations: 0,\n }\n }\n\n /** Get all collected metrics */\n getMetrics(): TurnMetrics[] {\n return [...this.metrics]\n }\n\n /** Get convergence curve (completion% over turns) */\n getConvergenceCurve(): number[] {\n return this.metrics.map((m) => m.completionPercent)\n }\n}\n"],"mappings":";;AASA,MAAa,gBAA4C;CACvD,UAAU;EAAE,OAAO;EAAQ,QAAQ;CAAK;CACxC,eAAe;EAAE,OAAO;EAAS,QAAQ;CAAO;CAChD,eAAe;EAAE,OAAO;EAAM,QAAQ;CAAK;CAC3C,4BAA4B;EAAE,OAAO;EAAO,QAAQ;CAAM;CAC1D,0BAA0B;EAAE,OAAO;EAAO,QAAQ;CAAM;CACxD,2BAA2B;EAAE,OAAO;EAAS,QAAQ;CAAQ;AAC/D;;;;;;;;AASA,MAAM,iBAA8C;CAClD,CAAC,gBAAgB;EAAE,OAAO;EAAO,QAAQ;CAAM,CAAC;CAChD,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAM,CAAC;CAClD,CAAC,4CAA4C;EAAE,OAAO;EAAO,QAAQ;CAAM,CAAC;CAC5E,CAAC,eAAe;EAAE,OAAO;EAAS,QAAQ;CAAO,CAAC;CAClD,CAAC,2BAA2B;EAAE,OAAO;EAAS,QAAQ;CAAK,CAAC;CAC5D,CAAC,gBAAgB;EAAE,OAAO;EAAQ,QAAQ;CAAK,CAAC;CAChD,CAAC,YAAY;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CAC9C,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,QAAQ;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CAC1C,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,UAAU;EAAE,OAAO;EAAS,QAAQ;CAAM,CAAC;CAC5C,CAAC,SAAS;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;AAC7C;;;;AAKA,SAAS,iBAAiB,OAAuB;CAC/C,QAAQ,MAAM,MAAM,GAAG,CAAC,CAAC,MAAM,MAAA,CAAO,KAAK,CAAC,CAAC,YAAY;AAC3D;;;;AAKA,SAAgB,oBAAoB,OAAkC;CACpE,IAAI,cAAc,QAAQ,OAAO,cAAc;CAC/C,MAAM,KAAK,iBAAiB,KAAK;CACjC,IAAI,cAAc,KAAK,OAAO,cAAc;CAC5C,KAAK,MAAM,CAAC,SAAS,UAAU,gBAC7B,IAAI,QAAQ,KAAK,EAAE,GAAG,OAAO;CAE/B,OAAO;AACT;;;AAIA,SAAgB,cAAc,OAAwB;CACpD,OAAO,oBAAoB,KAAK,MAAM;AACxC;AAEA,MAAM,uCAAuB,IAAI,IAAY;;AAG7C,SAAgB,eAAe,MAAsB;CACnD,OAAO,KAAK,KAAK,KAAK,SAAS,CAAC;AAClC;;;;AAKA,SAAgB,aAAa,aAAqB,cAAsB,OAAuB;CAC7F,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SAAS;EACZ,IAAI,CAAC,qBAAqB,IAAI,KAAK,GAAG;GACpC,qBAAqB,IAAI,KAAK;GAC9B,QAAQ,KACN,uCAAuC,MAAM,oGAE/C;EACF;EACA,OAAO;CACT;CACA,OAAQ,cAAc,MAAQ,QAAQ,QAAS,eAAe,MAAQ,QAAQ;AAChF;;;;AAKA,IAAa,eAAb,MAA0B;CACxB,aAAqB;CACrB,cAAsB;CACtB,YAAoB;CACpB;CAEA,YAAY,QAAQ,UAAU;EAC5B,KAAK,QAAQ;CACf;;CAGA,OAAO,aAAqB,cAA8B;EACxD,KAAK,cAAc;EACnB,KAAK,eAAe;EACpB,MAAM,OAAO,aAAa,aAAa,cAAc,KAAK,KAAK;EAC/D,KAAK,aAAa;EAClB,OAAO;CACT;;CAGA,eACE,WACA,YAC6D;EAC7D,MAAM,cAAc,eAAe,SAAS;EAC5C,MAAM,eAAe,eAAe,UAAU;EAE9C,OAAO;GAAE;GAAa;GAAc,MADvB,KAAK,OAAO,aAAa,YACC;EAAE;CAC3C;CAEA,gBAAwB;EACtB,OAAO,KAAK;CACd;CACA,iBAAyB;EACvB,OAAO,KAAK;CACd;CACA,eAAuB;EACrB,OAAO,KAAK;CACd;AACF;;;;;;AAOA,IAAa,mBAAb,MAA8B;CAC5B;CACA;CACA,UAAiC,CAAC;CAClC,YAAY,QAAuB,aAAqB;EACtD,KAAK,SAAS;EACd,KAAK,cAAc;CACrB;;CAGA,MAAM,QACJ,MACA,mBACA,eACA,oBACA,iBACA,uBACA,yBACA,cACA,cAAc,GACd,eAAe,GACf,mBAAmB,GACG;EACtB,MAAM,QAAQ,MAAM,KAAK,SAAS;EAElC,MAAM,IAAiB;GACrB;GACA,4BAAW,IAAI,KAAK,EAAA,CAAE,YAAY;GAClC,OAAO,MAAM;GACb,QAAQ,MAAM;GACd,WAAW,MAAM;GACjB,YAAY,MAAM,WAAW;GAC7B;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA,cAAc;GACd,mBACE,0BAA0B,IAAK,wBAAwB,0BAA2B,MAAM;EAC5F;EAEA,KAAK,QAAQ,KAAK,CAAC;EACnB,OAAO;CACT;;CAGA,MAAM,WAAiC;EACrC,MAAM,CAAC,OAAO,QAAQ,WAAW,cAAc,MAAM,QAAQ,IAAI;GAC/D,KAAK,OAAO,SAAS,KAAK,WAAW;GACrC,KAAK,OAAO,UAAU,KAAK,WAAW;GACtC,KAAK,OAAO,aAAa,KAAK,WAAW;GACzC,KAAK,OAAO,aAAa,KAAK,WAAW;EAC3C,CAAC;EAED,OAAO;GACL,OAAO,MAAM;GACb,QAAQ,OAAO;GACf,WAAW;IACT,SAAS,UAAU,QAAQ,MAAM,EAAE,WAAW,SAAS,CAAC,CAAC;IACzD,UAAU,UAAU,QAAQ,MAAM,EAAE,WAAW,UAAU,CAAC,CAAC;IAC3D,UAAU,UAAU,QAAQ,MAAM,EAAE,WAAW,UAAU,CAAC,CAAC;GAC7D;GACA;GACA,YAAY;GACZ,aAAa;EACf;CACF;;CAGA,aAA4B;EAC1B,OAAO,CAAC,GAAG,KAAK,OAAO;CACzB;;CAGA,sBAAgC;EAC9B,OAAO,KAAK,QAAQ,KAAK,MAAM,EAAE,iBAAiB;CACpD;AACF"}
@@ -0,0 +1,201 @@
1
+ import { s as ValidationError } from "./errors-8YnH8WlF.js";
2
+ import { o as runTaskScore } from "./run-record-BuoE80Dq.js";
3
+ import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
4
+ import { i as ROLLOUT_SCHEMA, s as assertMinted } from "./schema-C6DW4ZHR.js";
5
+ import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
6
+ //#region src/rollout/mint.ts
7
+ /**
8
+ * Rollout minting — `tangle.rollout.v1` lines joined from the records the
9
+ * substrate ALREADY keeps. There is no separate rollout store: a rollout
10
+ * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with
11
+ * its trace (spans share `runId`), projected into the canonical line.
12
+ *
13
+ * Composition, not duplication:
14
+ * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)
15
+ * - step structure → `buildTrajectory` over the shared TraceStore
16
+ * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
17
+ * - PRM / reward-model → `reward-model-export.ts`
18
+ *
19
+ * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
20
+ * never exported with a positive reward OR with any of the numbers that reward
21
+ * was computed from. The gate travels into the training data (`reward` forced
22
+ * to 0, `realness_gated: true`) and the whole outcome is transformed by
23
+ * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
24
+ * `verdict` to `provenance.gated_evidence`. Mint returns
25
+ * `MintedRolloutLine[]`: the brand the training exporters require, which only
26
+ * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
27
+ *
28
+ * A record carrying NEITHER split score is REJECTED (`ValidationError`), never
29
+ * minted at 0 — "nobody graded this" is not the same claim as "graded a total
30
+ * failure", and a trainer reading 0 learns the second. Lines that already
31
+ * carry `reward: null` (interchange imports, existing ledgers) remain valid on
32
+ * the wire; only the RunRecord→line door refuses.
33
+ *
34
+ * Records without spans become labeled GAP LINES (messages: [],
35
+ * provenance.gap) — present in the output AND surfaced in
36
+ * `missingTraces`; a capture gap is a finding, never a silent omission.
37
+ */
38
+ const asText = (v, scrub) => {
39
+ return scrub((typeof v === "string" ? v : JSON.stringify(v)) ?? "");
40
+ };
41
+ function projectStep(span, scrub) {
42
+ const base = {
43
+ kind: span.kind,
44
+ name: scrub(span.name),
45
+ status: span.status,
46
+ durationMs: span.endedAt !== void 0 ? span.endedAt - span.startedAt : void 0
47
+ };
48
+ if (span.kind === "llm") {
49
+ const llm = span;
50
+ const last = llm.messages[llm.messages.length - 1];
51
+ if (last) base.input = scrub(last.content);
52
+ if (llm.output !== void 0) base.output = scrub(llm.output);
53
+ } else if (span.kind === "tool") {
54
+ const tool = span;
55
+ base.input = asText(tool.args, scrub);
56
+ if (tool.result !== void 0) base.output = asText(tool.result, scrub);
57
+ }
58
+ return base;
59
+ }
60
+ /** The final llm span's history + output is the completed conversation. */
61
+ function finalConversation(spans, scrub) {
62
+ const llms = spans.filter((s) => s.kind === "llm");
63
+ const last = llms[llms.length - 1];
64
+ if (!last) return [];
65
+ const messages = last.messages.map((m) => ({
66
+ role: m.role,
67
+ content: scrub(m.content)
68
+ }));
69
+ if (last.output !== void 0 && last.output !== "") messages.push({
70
+ role: "assistant",
71
+ content: scrub(last.output)
72
+ });
73
+ return messages;
74
+ }
75
+ const REWARD_SOURCE = {
76
+ holdout: "run-record/holdout-score",
77
+ search: "run-record/search-score",
78
+ unscored: "run-record/unscored"
79
+ };
80
+ /**
81
+ * The mint door refuses an execution-only record: a missing training label is
82
+ * not a zero reward, and not a mintable line either. Lines that already carry
83
+ * `reward: null` — interchange imports, existing ledgers — stay valid on the
84
+ * wire and keep their labeled gap; this guard is only about the
85
+ * RunRecord→line door, where the producer can still be told to go score the
86
+ * run instead of shipping an unlabeled row.
87
+ */
88
+ function requireTaskScore(record) {
89
+ if (runTaskScore(record) === void 0) throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`);
90
+ }
91
+ const SPLIT_FROM_TAG = {
92
+ search: "search",
93
+ dev: "dev",
94
+ holdout: "holdout"
95
+ };
96
+ function mintLine(record, steps, messages, options, capturedAt, gap) {
97
+ requireTaskScore(record);
98
+ const rewardFields = rolloutRewardFields(record);
99
+ const uncaptured = record.costProvenance.kind === "uncaptured";
100
+ const terminalOutcome = record.terminalOutcome;
101
+ const isCompleted = terminalOutcome === "succeeded" || terminalOutcome === "failed";
102
+ const isTruncated = terminalOutcome === "cancelled" || terminalOutcome === "incomplete";
103
+ const terminalError = terminalOutcome === "failed" || terminalOutcome === "cancelled" || terminalOutcome === "incomplete" ? record.terminalFailureReason ?? `run ended ${terminalOutcome}` : null;
104
+ return assertMinted({
105
+ schema: ROLLOUT_SCHEMA,
106
+ rollout_id: record.runId,
107
+ parent_rollout_id: null,
108
+ run_id: record.runId,
109
+ experiment_id: record.experimentId,
110
+ candidate_id: record.candidateId,
111
+ generation: null,
112
+ candidate_index: null,
113
+ role: options.role ?? "agent",
114
+ task: {
115
+ suite: options.suite ?? record.experimentId,
116
+ instance_id: record.scenarioId,
117
+ split: SPLIT_FROM_TAG[record.splitTag],
118
+ seed: record.seed,
119
+ rep: 0
120
+ },
121
+ policy: {
122
+ harness: null,
123
+ harness_version: null,
124
+ model: record.model,
125
+ provider: null,
126
+ profile_commit: record.commitSha,
127
+ prompt_hash: record.promptHash,
128
+ config_hash: record.configHash,
129
+ agent_profile_cell_id: record.agentProfile?.cellId ?? null,
130
+ sampling: null
131
+ },
132
+ messages,
133
+ tool_defs: [],
134
+ ...steps.length > 0 ? { steps } : {},
135
+ outcome: {
136
+ ...rewardFields,
137
+ reward_source: REWARD_SOURCE[scoreOrigin(record)],
138
+ verdict: null,
139
+ metrics: { ...record.outcome.raw },
140
+ is_completed: isCompleted,
141
+ is_truncated: isTruncated,
142
+ error: terminalError
143
+ },
144
+ cost: {
145
+ usd: uncaptured ? null : record.costUsd,
146
+ tokens_in: record.tokenUsage.input,
147
+ tokens_out: record.tokenUsage.output,
148
+ tokens_reasoning: record.tokenUsage.reasoning ?? null,
149
+ cache_read: record.tokenUsage.cached ?? null,
150
+ cache_write: record.tokenUsage.cacheWrite ?? null,
151
+ wall_s: Math.round(record.wallMs / 1e3)
152
+ },
153
+ artifacts: {
154
+ patch_path: null,
155
+ run_dir: null,
156
+ transcript_ref: null
157
+ },
158
+ provenance: {
159
+ captured_at: capturedAt,
160
+ capture: "mint",
161
+ ...gap !== void 0 ? { gap } : {}
162
+ }
163
+ }, `minted rollout line for run ${record.runId}`);
164
+ }
165
+ /**
166
+ * Join RunRecords with their traces into canonical rollout lines. Records
167
+ * without spans are emitted as labeled gap lines and reported in
168
+ * `missingTraces`. Execution-only records without a task score are rejected
169
+ * because a missing training label is not a zero reward.
170
+ */
171
+ async function mintRolloutRows(records, store, options = {}) {
172
+ const scrub = options.scrub ?? ((t) => t);
173
+ const capturedAt = (options.now?.() ?? /* @__PURE__ */ new Date()).toISOString();
174
+ const rows = [];
175
+ const missingTraces = [];
176
+ for (const record of records) {
177
+ const trajectory = await buildTrajectory(store, record.runId);
178
+ if (trajectory.steps.length === 0) {
179
+ missingTraces.push(record.runId);
180
+ rows.push(mintLine(record, [], [], options, capturedAt, "no trace spans recorded for this runId"));
181
+ continue;
182
+ }
183
+ let steps = trajectory.steps.map((s) => projectStep(s.span, scrub));
184
+ if (options.maxSteps !== void 0 && steps.length > options.maxSteps) {
185
+ const head = Math.ceil(options.maxSteps / 2);
186
+ const tail = options.maxSteps - head;
187
+ steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)];
188
+ }
189
+ const conversation = finalConversation(trajectory.steps.map((s) => s.span), scrub);
190
+ const gap = conversation.length === 0 ? "trace has no llm spans — no conversation to inline" : void 0;
191
+ rows.push(mintLine(record, steps, conversation, options, capturedAt, gap));
192
+ }
193
+ return {
194
+ rows,
195
+ missingTraces
196
+ };
197
+ }
198
+ //#endregion
199
+ export { mintRolloutRows as t };
200
+
201
+ //# sourceMappingURL=mint-yN2M2eh0.js.map