@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -0,0 +1,182 @@
1
+ # Connect rubric scores to deployment outcomes
2
+
3
+ `rubricPredictiveValidity()` measures associations between rubric scores and observations from deployment.
4
+ Declare the desired direction for each outcome before reading the results.
5
+ Define rubric scores so higher values mean better evaluated behavior.
6
+
7
+ ```ts
8
+ import {
9
+ InMemoryOutcomeStore,
10
+ rubricPredictiveValidity,
11
+ type DeploymentOutcome,
12
+ type OutcomeMetricSpec,
13
+ } from '@tangle-network/agent-eval/meta-eval'
14
+ import type { RunRecord } from '@tangle-network/agent-eval'
15
+
16
+ async function assessOutcomes(runs: RunRecord[], observations: DeploymentOutcome[]) {
17
+ const outcomes = new InMemoryOutcomeStore()
18
+ for (const observation of observations) await outcomes.append(observation)
19
+ const metrics: OutcomeMetricSpec[] = [
20
+ { id: 'success_rate', direction: 'higher-is-better' },
21
+ { id: 'failure_rate', direction: 'lower-is-better' },
22
+ ]
23
+
24
+ return rubricPredictiveValidity({
25
+ runs,
26
+ outcomes,
27
+ outcomeMetrics: metrics,
28
+ rubrics: ['task_quality'],
29
+ })
30
+ }
31
+ ```
32
+
33
+ Populate `RunRecord.outcome.raw.task_quality` with the captured rubric score.
34
+ Append deployment observations with the same `runId` and exact outcome metric keys.
35
+ The store accepts finite numbers, including zero.
36
+ Omit an unmeasured metric instead of replacing it with zero.
37
+
38
+ An observation records when and where its outcomes were captured:
39
+
40
+ ```ts
41
+ const observation: DeploymentOutcome = {
42
+ runId: 'support-run-42',
43
+ capturedAt: Date.now(),
44
+ metrics: { success_rate: 1, failure_rate: 0 },
45
+ labels: { cohort: 'support' },
46
+ source: 'resolution-events-v1',
47
+ }
48
+ ```
49
+
50
+ `capturedAt` uses epoch milliseconds.
51
+ The matching run supplies the rubric score; this observation supplies the measured deployment outcome.
52
+
53
+ The default reduction selects the latest finite observation of each requested metric.
54
+ A newer row containing another metric cannot supply its value or erase an older observation.
55
+ The `mean` and `max` reductions operate on that metric alone.
56
+
57
+ ## Read the report
58
+
59
+ | Field | Interpretation |
60
+ |---|---|
61
+ | `pearson`, `spearman` | Association with the recorded outcome, retaining its original sign. |
62
+ | `alignedPearson`, `alignedSpearman` | Positive means higher rubric scores associate with better outcomes. |
63
+ | `pearsonCi95`, `spearmanCi95` | Bootstrap intervals for the raw associations. |
64
+ | `alignedSpearmanCi95` | Spearman interval after applying the declared outcome direction. |
65
+ | `verdict` | `aligned` at aligned Spearman ≥ 0.4; `inverse` at ≤ −0.4; otherwise `weak`. |
66
+ | `n` | Finite joined run observations for this rubric and outcome. |
67
+ | `excludedPairs` | Unestimated pairs, their observation counts, and the reason for exclusion. |
68
+ | `rubricsWithoutData` | Declared rubrics with no finite score in the supplied runs. |
69
+
70
+ A rubric can correlate negatively with failure rate and still receive `aligned`.
71
+ The same negative correlation with success rate receives `inverse`.
72
+ These labels describe association strength and direction.
73
+ They do not grant release authority or establish that changing a rubric weight will improve outcomes.
74
+
75
+ Pairs require eight observations by default.
76
+ `minSamples` can set another integer of at least three.
77
+ Insufficient observations and constant scores or outcomes remain explicit exclusions.
78
+ Constant observations cannot establish a perfect predictor.
79
+ Intervals remain `null` when no bootstrap resample has an estimable correlation.
80
+
81
+ `joinedSamples + skippedRuns` equals the supplied run count.
82
+ A run is joined when at least one finite score and outcome pair exists.
83
+ It remains joined even if that pair has too few observations for estimation.
84
+ Duplicate run IDs are rejected.
85
+
86
+ Each run is an independent bootstrap observation.
87
+ Repeated observations from the same person or task can violate that assumption.
88
+ Aggregate at the independent unit or use a study with an appropriate grouped estimator.
89
+
90
+ `ranked` selects each rubric's highest direction-aligned association across the declared outcomes.
91
+ This ordering is exploratory and can conceal conflicts among outcomes.
92
+ Inspect all `pairs` and use a target chosen before analysis for an automated recommendation.
93
+ Confirm any proposed change with fresh evidence and fixed scoring rules.
94
+
95
+ ## Propose an experiment against one target
96
+
97
+ ```ts
98
+ import { PredictiveValidityResearcher } from '@tangle-network/agent-eval/rl'
99
+ import { InMemoryOutcomeStore } from '@tangle-network/agent-eval/meta-eval'
100
+
101
+ const outcomes = new InMemoryOutcomeStore()
102
+ const researcher = new PredictiveValidityResearcher({
103
+ outcomes,
104
+ targetOutcome: { id: 'success_rate', direction: 'higher-is-better' },
105
+ rubrics: ['task_quality'],
106
+ })
107
+ ```
108
+
109
+ Supply observed outcomes before calling `runValidityCheck(runs)`.
110
+ Pass the resulting failure groups to `proposeChange(failures)`.
111
+ The researcher uses its declared target even if another outcome has a stronger association.
112
+ It proposes increased-weight experiments for aligned associations and reversal or replacement experiments for inverse associations.
113
+ Both require an aligned Spearman interval that excludes zero.
114
+ Weak or uncertain associations produce requests for calibration evidence.
115
+ Missing estimates produce requests for more outcome observations.
116
+
117
+ The proposals contain their association, interval, sample count, and target direction.
118
+ They contain no predicted improvement because a correlation does not identify a causal treatment effect.
119
+ `applyChange()` appends proposals to a plan.
120
+ `evaluateChange()` declines promotion because the caller owns experiment execution.
121
+
122
+ `runRLCampaign()` accepts the same outcome specifications through `outcomeMetrics` when `outcomeStore` is supplied.
123
+ Supply both options together; incomplete or empty declarations fail before campaign execution.
124
+ Its summary reports direction-aligned association and preserves missing estimates.
125
+ The target declaration remains fixed while the campaign runs.
126
+
127
+ The researcher copies its target declaration, rubric list, and cached reports.
128
+ Changing callback arguments, returned reports, or proposal payloads cannot rewrite its cached evidence.
129
+
130
+ ## Store observations without losing failures
131
+
132
+ `InMemoryOutcomeStore` and `FileSystemOutcomeStore` copy observations at ingestion and retrieval.
133
+ Changing a caller's metric or label object cannot rewrite stored evidence.
134
+ The filesystem store rereads observations so another instance's later writes remain visible.
135
+ Use one writer per directory; operations on that writer are serialized.
136
+
137
+ A nonexistent directory is an empty store.
138
+ An unreadable file, malformed JSON, or invalid outcome record throws `OutcomeStoreError`.
139
+ The error carries its operation, path, source line when available, and original cause.
140
+ Read errors never become empty studies or partial successful results.
141
+ Repair the source and retry the read.
142
+
143
+ For trace data, `correlationStudy()` accepts outcome names and reports descriptive associations without a desired direction.
144
+ It shares the metric reduction, bootstrap, and exclusion behavior.
145
+ It excludes observations captured before the run started.
146
+ `maxCaptureLagMs` optionally bounds the elapsed time from run start through outcome capture, including both endpoints.
147
+
148
+ ## Inspect calibration by score range
149
+
150
+ Use `calibrationFromPairs()` when scores and outcomes are already joined.
151
+ It accepts readonly observations directly.
152
+
153
+ ```ts
154
+ import { calibrationFromPairs } from '@tangle-network/agent-eval/meta-eval'
155
+
156
+ const calibration = calibrationFromPairs([
157
+ { evalScore: 1, outcome: 0 },
158
+ { evalScore: 1, outcome: 0 },
159
+ ], 'predicted-success', 'observed-success')
160
+
161
+ console.log(calibration?.ece) // 1: confident predictions, observed failures.
162
+ ```
163
+
164
+ Direct input rejects nonfinite pairs with the offending index.
165
+ It does not silently discard them from the denominator.
166
+ `calibrationCurve()` performs the join for trace and outcome stores using the latest finite observation of the named outcome metric.
167
+ It reports each bin's count, mean score, mean outcome, and absolute gap.
168
+ `ece` weights each bin's gap by its share of the finite joined observations.
169
+ Both quantities must use comparable numerical scales for this difference to measure calibration.
170
+
171
+ The `range` option clips scores before binning and retains every joined observation.
172
+ The reported calibration error then describes the clipped scores; retain that transformation with the result.
173
+ Constant scores form one bin, so a consistently wrong predictor still receives a measured calibration error.
174
+ Equal-frequency binning produces the requested number of bins, capped by the observation count.
175
+ Bin counts differ by at most one, except when all scores share one value.
176
+ Equal-width binning omits empty bins.
177
+ On either path, bin counts sum to the reported `n`.
178
+
179
+ Fewer than two finite pairs returns `null`.
180
+ Both entry points validate metric identities, binning options, and range bounds.
181
+ Invalid requests fail even when no evidence is available.
182
+ The store entry point validates before reading evidence.
@@ -152,8 +152,7 @@ set with a signed note.
152
152
 
153
153
  ## Optimization
154
154
 
155
- Use `runImprovementLoop()` when the system is a multi-step agent, not a
156
- single prompt.
155
+ Use `runImprovementLoop()` when a `SurfaceProposer` supplies candidates for Agent Eval's search loop.
157
156
 
158
157
  Good optimization targets:
159
158
 
@@ -60,14 +60,14 @@ In order: first match wins:
60
60
 
61
61
  | Quantity | Function | Source file |
62
62
  |---|---|---|
63
- | Marginal CI on score mean | `confidenceInterval` | `statistics.ts` |
64
- | Paired Cohen's dz vs comparator | `pairedCohensDz` | `statistics.ts` |
65
- | Wilcoxon signed-rank (paired), exact at n ≤ 20 | `wilcoxonSignedRank` | `statistics.ts` |
66
- | BH-FDR q-values | `benjaminiHochberg` | `statistics.ts` |
67
- | Paired bootstrap CI on median delta | `pairedBootstrap` | `statistics.ts` |
68
- | Smallest p a rank-test design can produce | `pFloor` on the result | `statistics.ts` |
63
+ | Marginal CI on score mean | `confidenceInterval` | [`descriptive.ts`](../src/statistics/descriptive.ts) |
64
+ | Paired Cohen's dz vs comparator | `pairedCohensDz` | [`effect-sizes.ts`](../src/statistics/effect-sizes.ts) |
65
+ | Wilcoxon signed-rank (paired), exact at n ≤ 20 | `wilcoxonSignedRank` | [`rank-tests.ts`](../src/statistics/rank-tests.ts) |
66
+ | BH-FDR q-values | `benjaminiHochberg` | [`multiplicity.ts`](../src/statistics/multiplicity.ts) |
67
+ | Paired bootstrap CI on median delta | `pairedBootstrap` | [`paired-tests.ts`](../src/statistics/paired-tests.ts) |
68
+ | Smallest p a rank-test design can produce | `pFloor` on the result | [`rank-tests.ts`](../src/statistics/rank-tests.ts) |
69
69
  | Bayesian-bootstrap Pr(Δ>0), Pr(Δ∈ROPE) | `bayesianBootstrapMeanSamples` | `summary-report.ts` (private) |
70
- | Minimum detectable paired effect | `pairedMde` | `statistics.ts` |
70
+ | Minimum detectable paired effect | `pairedMde` | [`power-and-mde.ts`](../src/statistics/power-and-mde.ts) |
71
71
  | Run fingerprint | `hashJson(...)` | `pre-registration.ts` |
72
72
 
73
73
  The Pr(Δ>0) and Pr(Δ∈ROPE) summaries use Rubin's Bayesian bootstrap.
@@ -222,3 +222,11 @@ Receipts describe measurements; they do not override the release gate or replace
222
222
  `createCampaignEvidenceReceipt` on `/experiment` provides the same binding for other complete campaign consumers.
223
223
  Changing an output changes its output digest; changing a judge result changes its measurement digest.
224
224
  The final receipt retains caller authority, including `candidate-self-report`, without upgrading it.
225
+
226
+ ## Implementation ownership
227
+
228
+ `src/campaign/search-ledger.ts` owns validation, canonical event normalization, and composition with the shared ledger journal.
229
+ `search-ledger-projector.ts` owns campaign replay invariants and audit projection.
230
+ `search-ledger-types.ts` defines the shared data contracts, reexported through the existing facade.
231
+ `search-ledger-ordering.ts` shares artifact identity and deterministic string ordering between normalization and replay.
232
+ The journal in `src/ledger-core` owns hashing, locking, durable appends, and chain verification.
@@ -0,0 +1,129 @@
1
+ # Statistical evidence
2
+
3
+ Repeated runs measure execution variation on the cases you supplied.
4
+ A claim about new cases also needs independent observations from its target population.
5
+ Repeating one incident 100 times supplies one incident, even when every execution has a distinct identifier.
6
+
7
+ ## Declare the independent unit
8
+
9
+ Pass `independentUnitByScenarioId` to `defaultProductionGate`, `heldoutSignificance`, or `dimensionRegressions` when scenarios share a source.
10
+ The map assigns each scenario to the incident, document, task family, or other unit sampled independently.
11
+ Choose this assignment before examining scores.
12
+
13
+ ```ts
14
+ import { defaultProductionGate } from '@tangle-network/agent-eval/campaign'
15
+
16
+ const holdoutScenarios = [
17
+ { id: 'incident:17:original', kind: 'support' },
18
+ { id: 'incident:17:paraphrase', kind: 'support' },
19
+ { id: 'incident:28:original', kind: 'support' },
20
+ ]
21
+
22
+ const gate = defaultProductionGate({
23
+ holdoutScenarios,
24
+ independentUnitByScenarioId: new Map([
25
+ ['incident:17:original', 'incident-17'],
26
+ ['incident:17:paraphrase', 'incident-17'],
27
+ ['incident:28:original', 'incident-28'],
28
+ ]),
29
+ deltaThreshold: 0.05,
30
+ criticalDimensions: ['factualAccuracy'],
31
+ })
32
+ ```
33
+
34
+ This example has two independent units and cannot meet its observation minimum.
35
+ More repetitions of these scenarios will not change that count.
36
+ The default gate copies the map when it is constructed.
37
+ Later mutations cannot change the grouping used by that gate.
38
+
39
+ Pairing precedes aggregation.
40
+ Candidate and baseline must contain the same full cell identifiers and the same selected judges within each cell.
41
+ The last numeric suffix identifies a repetition, so colons within scenario identifiers remain valid.
42
+ The implementation averages matched cells within each unit, then gives every unit equal weight.
43
+ It refuses asymmetric cells, asymmetric selected judges, duplicate cells, non-finite scores, and missing unit assignments.
44
+
45
+ Reports distinguish the following counts:
46
+
47
+ | Field | Meaning |
48
+ |---|---|
49
+ | `n` | Paired observation units used for inference |
50
+ | `pairedCellN` | Matched execution cells before grouping |
51
+ | `observationUnit` | `registered` when a supplied map defines units; `cell` on an ungrouped fixed-roster path |
52
+ | `unitIds` | The units represented in a held-out significance result |
53
+
54
+ Without a map, fixed-roster significance uses execution cells as observations.
55
+ Its uncertainty concerns independently sampled execution outcomes conditional on that roster.
56
+ It does not establish generalization to new incidents or task families.
57
+ Identifiers and aggregation do not establish independence; the sampling design must justify it.
58
+
59
+ ## Read the interval that decided
60
+
61
+ `heldoutSignificance().decision` contains the test, interval, observation minimum, and promotion decision.
62
+ Pass/fail scores use the shared paired risk-difference rule.
63
+ Continuous scores use the shared paired bootstrap or its small-sample test.
64
+ `bootstrap` and `medianBootstrap` remain diagnostics when another test decides.
65
+
66
+ Continuous mean targets require 20 observations for bootstrap eligibility.
67
+ Below that count, a reported exact sign-test diagnostic cannot establish a mean effect.
68
+ Binary outcomes and explicitly requested median targets use their actual confidence-dependent observation minimum.
69
+ The observation minimum establishes estimator eligibility; it does not establish statistical power or representative sampling.
70
+ A zero-width bootstrap interval cannot establish improvement under the shared decision rule.
71
+
72
+ Required dimensions also need sufficient observations and complete coverage.
73
+ The default gate reports `not_evaluated` when a required dimension lacks enough units or omits matched cells or scenarios.
74
+ `fewRuns`, `missingCellIds`, and `missingScenarioIds` preserve the missing evidence.
75
+ An observed regression can still hold the gate while evidence remains incomplete.
76
+ Passing the regression guard does not certify every safety property of the candidate.
77
+
78
+ ## Sequential decisions require a conditional-mean assumption
79
+
80
+ `sequentialPairedGate` consumes one observation per scenario by default in `decide()`.
81
+ Its `independentUnitByScenarioId` option groups related scenarios before testing and copies the mapping at construction.
82
+ `maxN` counts these independent units.
83
+ The report preserves both the consumed count `n` and the available counts `pairedN` and `pairedCellN`.
84
+
85
+ The statistical guarantee requires each next delta's conditional expectation to remain below the registered null boundary.
86
+ Independent sampling with that bound is sufficient.
87
+ Shuffling an exchangeable sequence does not establish the condition.
88
+ One random sign repeated 100 times has only one independent draw.
89
+ After its first observation, later signs reveal no new evidence.
90
+
91
+ Direct `observe(delta)` callers must perform the required aggregation themselves.
92
+ They must also justify the sampling assumption.
93
+ `sequentialDecide` stops candidate exploration heuristically; candidate selection and reused incumbent scores prevent a general type-I error guarantee.
94
+ The selected candidate still requires fresh held-out evidence.
95
+
96
+ ## Adaptation comparisons pair whole scenarios
97
+
98
+ `runAdaptationCurve` and `compareAdaptationCurves` are exported from `@tangle-network/agent-eval/rl`.
99
+ Scenarios require explicit, unique `scenarioId` values.
100
+ The runner validates repetitions, the demonstration grid, and finite scores in `[0,1]`.
101
+
102
+ Comparisons require identical scenario cohorts and identical demonstration grids.
103
+ Missing pairs, duplicate identities, and changing cohorts are errors.
104
+ The comparison computes one area per scenario before applying the existing paired estimators.
105
+ This preserves dependence across demonstration counts.
106
+ Repeated executions improve each scenario mean without increasing the number of independent scenarios.
107
+
108
+ `perK` intervals describe the curve.
109
+ The paired area decisions determine `a_better`, `b_better`, `inconclusive`, or `insufficient_evidence`.
110
+ Continuous area outcomes require the bootstrap minimum of 20 paired scenarios.
111
+ Binary area outcomes use the shared score-interval rule and can support a decision with fewer observations.
112
+ The deciding interval must remain nondegenerate.
113
+ `inconclusive` does not establish equivalence.
114
+ `firstPassK` describes the first observed crossing and carries no reliability guarantee.
115
+
116
+ ## Perturbation sensitivity is a diagnostic
117
+
118
+ `runContaminationProbe` reports observed score differences and one global Wilcoxon paired test.
119
+ Use `alpha` for that test's significance threshold.
120
+ Per-item differences carry no p-values or q-values because the probe defines no calibrated item-level sampling null.
121
+
122
+ The report retains every observed pair and lists `excludedScenarioIds` when a score floor excludes evidence.
123
+ Summaries describe the included population.
124
+ Fewer than four included pairs produce `pairedTest: null` while preserving measured means and medians.
125
+ Zero included pairs produce null summaries.
126
+
127
+ A significant drop can reflect changed task difficulty, broken perturbations, or contamination.
128
+ `contaminationSuspected` therefore requests investigation; it does not identify the cause.
129
+ The global test also relies on independent pairs and the Wilcoxon assumptions for paired differences.
package/docs/verdicts.md CHANGED
@@ -1,67 +1,94 @@
1
- # One verdict vocabulary
1
+ # Verdicts and certifications
2
2
 
3
- Every verification path in this package lands in one type: `DefaultVerdict` (`src/verdict.ts`).
4
- `valid` answers "did it pass", `score` answers "how well" in [0, 1], `scores` carries the per-dimension breakdown, and `certification` says WHO certified.
3
+ `DefaultVerdict` is the shared base type for validator results.
4
+ Campaign judges return `JudgeScore`, and release gates return `GateResult`; their fields and score scales differ.
5
+ See [release check results](./concepts.md#release-check-results) when interpreting a campaign decision.
5
6
 
6
- A certification is the epistemics a bare `valid` + `score` pair cannot carry.
7
- A kernel-checked proof and an LLM judge can produce the same `{ valid: true, score: 1 }`; the certification is what tells them apart:
7
+ In `DefaultVerdict`, `valid` reports whether the validator's pass criteria were met, and `score` is its aggregate in [0, 1].
8
+ Optional `scores` and `notes` carry dimensions and explanation.
9
+ Optional `certification` records the verification strategy, checker identity, assumptions, and evidence digest.
10
+ A certification can accompany a failed or incomplete check.
11
+ It does not establish that the result passed or that every required measurement exists.
8
12
 
9
- - `strategy` which verification-strategy member vouches (the 10-member family with per-member failure modes: [docs/verification-strategies.md](./verification-strategies.md));
10
- - `checker` — the exact identity that ran, with version and content pins, so the check is re-runnable;
11
- - `assumptions` — every step the certificate rests on that the checker did NOT verify, named one by one;
12
- - `evidenceDigest` — sha-256 of the evidence artifact (`certificationEvidenceDigest`).
13
+ The certification fields are:
13
14
 
14
- An absent certification is itself a statement: scored, but nothing vouches.
15
- No producer fakes one — a closed gate, an unexecuted proof, or an unattested checker yields an uncertified verdict, never an invented certificate.
15
+ - `strategy`: the [verification strategy](./verification-strategies.md), with its documented failure mode.
16
+ - `checker`: a name, version, and optional dependency pins.
17
+ - `assumptions`: the producer's list of steps the checker did not verify.
18
+ - `evidenceDigest`: the evidence identity; `certificationEvidenceDigest()` hashes its JSON-serialized form with canonical JSON and SHA-256.
19
+
20
+ These fields record the producer's claims.
21
+ Reproduction also requires access to the evidence, checker, dependencies, and execution environment.
22
+ An absent certification means no verification strategy is recorded for that verdict.
16
23
 
17
24
  ## Producers
18
25
 
19
- Every verifier below returns a `DefaultVerdict` (usually a richer extension of it) with a produced certification.
26
+ These result types extend `DefaultVerdict`.
27
+ Their additional fields distinguish failed checks from incomplete or unmeasured work.
20
28
 
21
- | Verifier | Verdict type | Strategy | Certifies | Assumptions it names |
29
+ | Producer | Result type | Strategy | Scope | Fields to inspect |
22
30
  | --- | --- | --- | --- | --- |
23
- | `MultiLayerVerifier.run` (`src/multi-layer-verifier.ts`) | `VerificationReport` | `composite` | ordered layer pipeline blend | each skipped / errored / timed-out layer |
24
- | `verifyCompletion` (`src/completion-verifier.ts`) | `CompletionVerdict` | the checker's own (`judge` for the LLM checker, `schema` for token recall) | task completion over produced state | lexical structural stage + the checker's attestation |
25
- | `evaluateTraceContract` (`src/trace-contracts.ts`) | `ContractVerdict` | `invariant` | LTLf rules over a span sequence | array ordering without timestamps; custom predicate functions |
26
- | `evaluateOracles` (`src/oracle.ts`) | `OracleReport` | `test` | declarative expected-outcome assertions | (the oracle set is its own answer key) |
27
- | `replayVerify` (`src/trajectory-replay/verify.ts`) | `ReplayVerdict` | `replication` | recorded failure reproduced (and fix vanished) under re-execution | unadjudicated prefix steps; truncated prefix; returncode-only signature |
28
- | `verifyFindings` (`src/trajectory-replay/findings.ts`) | `VerifyFindingsRun` | `replication` | a batch of analyst findings under executed replay | not-replayable findings leave the denominator |
29
- | `gradeRepairRow` (`src/trace-repair/grade.ts`) | `RepairRowResult` | `test` | a proposed repair against the row's held-out suite (pins: suite + policy digests) | vacuous reproduction gate; prefix divergence |
30
- | `runEquivalenceCheck` + `equivalenceVerdict` (`src/verification-strategy.ts`, `src/verdict.ts`) | `DefaultVerdict` | the spec's member (`proof-kernel` in the pilot) | two blind formal statements are equivalent | arms self-declare blindness |
31
-
32
- Two producers certify conditionally, on purpose:
33
-
34
- - `gradeRepairRow` certifies only a `measured` outcome — a funnel gate that closed before the suite ran has nothing to vouch for.
35
- - `verifyFindings` certifies only when at least one proof executed — a batch where nothing was replayable measured nothing.
36
-
37
- ## Reading a score out of a judge
38
-
39
- A model judge emits one grade per dimension. Discrete grades tie: two candidates that both score `8` carry no ranking signal between them, and a best-of-N selection then picks arbitrarily.
40
-
41
- `llmJudge({ scoring })` chooses how the number is read:
42
-
43
- | `scoring` | What it reads | Requires |
31
+ | `MultiLayerVerifier.run()` | `VerificationReport` | `composite` | Ordered verification layers | `layers`, `allPass`, and optional `taskScore` |
32
+ | `verifyCompletion()` | `CompletionVerdict` | The supplied checker's strategy | Completion requirements matched against produced state | Requirement evidence, `correct`, and `unmeasured` |
33
+ | `evaluateTraceContract()` | `ContractVerdict` | `invariant` | Temporal rules over recorded spans | Rule results and assumptions about ordering and predicates |
34
+ | `evaluateOracles()` | `OracleReport` | `test` | Declared expected-outcome assertions | `results`, `passCount`, and `failCount` |
35
+ | `replayVerify()` | `ReplayVerdict` | `replication` | Failure reproduction and an optional fix under re-execution | Prefix fidelity, signature matches, and both execution arms |
36
+ | `verifyFindings()` | `VerifyFindingsRun` | `replication` | Analyst findings checked through replay | `executions`, `counts`, and individual verifications |
37
+ | `gradeRepairRow()` | `RepairRowResult` | `test` | A repair against the admitted row's held-out suite | The grade's outcome and funnel evidence |
38
+ | `equivalenceVerdict(record)` | `DefaultVerdict` | The record's strategy | Formal-statement equivalence checked by `runEquivalenceCheck()` | Whether the obligation was proved, refuted, or unresolved |
39
+
40
+ Certification is conditional for several producers:
41
+
42
+ - `verifyCompletion()` includes it when the supplied checker provides an attestation.
43
+ Inspect requirement evidence to see which checks were assessed.
44
+ - `equivalenceVerdict()` includes it for a proved or refuted obligation with an evidence digest.
45
+ A certified refutation has `valid: false`.
46
+ - `gradeRepairRow()` includes it only for a `measured` grade.
47
+ - `verifyFindings()` includes it only when at least one replay execution ran.
48
+ Non-replayable findings remain in `counts` and are excluded from the score's denominator.
49
+ A score of 1 can therefore coexist with `valid: false` when some findings were not replayable.
50
+
51
+ Some result shapes use `score: 0` when no task measurement is available.
52
+ For `VerificationReport`, use the presence of `taskScore` to identify a complete task measurement; `blendedScore` can describe a partial panel.
53
+ An empty oracle set has certification metadata but returns `valid: false` and no executed oracle results.
54
+ Read these completeness fields before using scores as task labels or release evidence.
55
+
56
+ ## Reading a score from a model judge
57
+
58
+ A tied dimension score supplies no ordering between candidates.
59
+ `llmJudge({ scoring })` controls how the dimension score is read:
60
+
61
+ | `scoring` | Measurement | Requirement |
44
62
  |---|---|---|
45
- | `{ method: 'sampled' }` (default) | the grade the model emitted | nothing |
46
- | `{ method: 'expectation', whenUnavailable }` | the expected grade over the integer grades the model considered at the score token | `scale: 'ten'` and a provider that returns log probabilities |
47
-
48
- Expectation scoring asks the provider for `logprobs` with `top_logprobs`, finds the token that carried each dimension's grade, and averages the integer grades in that token's probability window, weighted by probability. Two answers that both sample `8` separate by how much mass sat on `7` versus `9`.
63
+ | `{ method: 'sampled' }` (default) | The emitted grade | A valid grade response |
64
+ | `{ method: 'expectation', whenUnavailable }` | A probability-weighted grade from returned token alternatives | `scale: 'ten'` and provider log probabilities |
49
65
 
50
- It needs one integer in one token, which is why `scale: 'ten'` is required: a `[0,1]` float is several tokens, and no single position carries its distribution. A grade that did not land in exactly one token — a two-token `10` is refused rather than approximated.
66
+ With `scale: 'ten'`, the model emits grades from 0 to 10; `llmJudge()` divides them by 10 before returning dimensions and composite.
67
+ Expectation scoring finds each grade's token, keeps valid integer alternatives, and renormalizes their returned probabilities before averaging.
68
+ Its distribution is limited to those returned alternatives.
69
+ Additional precision alone does not establish better calibration or ranking accuracy.
51
70
 
52
- `whenUnavailable` decides what happens when the provider returns no log probabilities, or the grade spans tokens:
71
+ Each emitted grade must occupy one integer token for expectation scoring.
72
+ A split `10`, missing grade token, or unavailable log probabilities invokes `whenUnavailable`:
53
73
 
54
- - `'fail'` throws, and the campaign records a failed cell.
55
- - `'sampled'` reads the emitted grade instead.
74
+ - `'fail'` throws; the campaign records a judge failure.
75
+ - `'sampled'` uses the emitted grades for that judge result.
56
76
 
57
- `JudgeScore.scoringMethod` reports what actually produced the number, so a declared expectation run that fell back reads `'sampled'` and stays auditable. `JudgeScore.distribution` carries the probability mass per grade, and is present only for an expectation score. Panels are unchanged: `ensembleJudge` consumes the composite either way.
77
+ `JudgeScore.scoringMethod` is present when `scoring` was explicitly configured and records the method used.
78
+ An expectation request that falls back reports `'sampled'`.
79
+ When `scoring` is omitted, sampled scoring is the default and this metadata field is absent.
80
+ `JudgeScore.distribution` is present only for expectation scoring and contains the normalized probabilities over returned integer alternatives.
81
+ `ensembleJudge()` consumes the resulting composite through its usual interface.
58
82
 
59
- Whether a given endpoint returns `logprobs.content` is a property of that provider and model, not of this package. `LlmCallResult.logprobs` is `null` when the provider returned none — never an inferred distribution. See `evidence/records/judge-logprob-wire-support.json` for the current verification state of that wire behavior.
83
+ Provider and model support determines whether log probabilities are available.
84
+ `LlmCallResult.logprobs` is `null` when none were returned.
85
+ The [recorded wire check](../evidence/records/judge-logprob-wire-support.json) documents the endpoints and conditions that were inspected.
60
86
 
61
- ## Consuming a certification
87
+ ## Interpreting certification limits
62
88
 
63
- Read `certification.strategy`, then weigh the member's documented failure mode — `VERIFICATION_STRATEGIES[strategy].failureMode` carries it at runtime.
64
- "Certified" is never one bit: a `judge` certificate is Goodhart-gameable, a `test` certificate covers only its suite, a `composite` certificate can hide which member carried the score.
65
- The assumptions list is the honest remainder; an empty list is the producer's explicit claim that nothing was left unverified, not a default.
89
+ Read `VERIFICATION_STRATEGIES[strategy].failureMode` alongside the evidence and assumptions.
90
+ A judge can reward misleading output; tests cover their suite; a proof can establish the wrong formal statement for the intended task.
91
+ A composite certification requires inspection of its component results.
92
+ An empty assumptions list is the producer's declaration, not independent verification that no assumptions remain.
66
93
 
67
- Related docs: [verification-strategies.md](./verification-strategies.md) (the family and the equivalence protocol), [trace-repair-grader.md](./trace-repair-grader.md), [trajectory-replay.md](./trajectory-replay.md).
94
+ See [verification strategies](./verification-strategies.md), [repair grading](./trace-repair-grader.md), and [trajectory replay](./trajectory-replay.md) for the corresponding execution contracts.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.179.0",
3
+ "version": "0.181.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -1 +0,0 @@
1
- {"version":3,"file":"agent-profile-cell-0gSi5ffD.js","names":["sortKeysDeep"],"sources":["../src/pre-registration.ts","../src/agent-profile-cell.ts"],"sourcesContent":["/**\n * Pre-registered hypotheses — declare what you're testing BEFORE the\n * run, check it AFTER. Prevents p-hacking, optional stopping, and the\n * \"we ran until it looked good\" failure mode.\n *\n * Manifest is a plain JSON-friendly object. Sign it with a content hash\n * + timestamp; the registered record becomes immutable. Post-run,\n * evaluate the manifest against observed results — the library refuses\n * to let you re-interpret a different metric as the declared one.\n *\n * A signed manifest is a portable record: it is written once and verified\n * later, possibly by a different release. `algo` names the digest scheme it\n * was signed under, and verification selects the encoder by that field, so a\n * manifest signed by an earlier release still verifies.\n */\n\nimport { createHash } from 'node:crypto'\nimport { canonicalString, hashCanonical } from './ledger-core/canonical'\n\nexport interface HypothesisManifest {\n id: string\n /** Human prose — goes into the audit trail. */\n hypothesis: string\n /** Metric the hypothesis claims to move. */\n metric: string\n /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */\n direction: 'increase' | 'decrease'\n /** Minimum effect size to count (same units as the metric). */\n minEffect: number\n /** Alpha threshold. */\n alpha: number\n /** Target statistical power at which sample size was pre-computed. */\n power: number\n /** Declared N per arm before running. */\n preRegisteredN: number\n /** ISO8601 timestamp the manifest was registered. */\n registeredAt: string\n /** Optional identifiers to tie into the trace corpus. */\n baselineLabel?: string\n candidateLabel?: string\n}\n\n/**\n * Identifier for the hashing scheme used to produce `contentHash`.\n *\n * Both schemes are sha256 hex over the manifest with `contentHash` and `algo`\n * stripped, and differ only in how that manifest is serialized:\n *\n * - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}\n * emits.\n * - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests\n * signed by an earlier release carry it, or carry no `algo` at all, and\n * {@link verifyManifest} still verifies them.\n */\nexport type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785'\n\nexport interface SignedManifest extends HypothesisManifest {\n /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */\n contentHash: string\n /**\n * Algorithm string describing how `contentHash` was produced.\n *\n * Optional on the type so serialized manifests without it still parse,\n * but ALWAYS populated by {@link signManifest}. Consumers that want to\n * enforce a known algorithm should reject manifests where this field\n * is missing or unrecognized.\n */\n algo?: SignedManifestAlgo\n}\n\nexport interface HypothesisResult {\n manifest: SignedManifest\n observedN: number\n observedEffect: number\n observedPValue: number\n /** True iff the observed effect hits the pre-declared direction with\n * magnitude ≥ minEffect AND p < alpha. */\n confirmed: boolean\n /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */\n rejectionReasons: Array<\n 'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'\n >\n notes?: string\n}\n\n/**\n * SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of\n * `obj` — the package's one identity scheme, shared with `ledger-core`.\n *\n * Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,\n * class instances, cycles — are refused rather than coerced, because a\n * coercion maps two distinct records onto one digest.\n *\n * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,\n * which takes a string input and returns a truncated 12-char prompt id.\n *\n * @example\n * const hash = await hashJson({ id: '1', kind: 'spec' })\n * // 'a3f1...' (64 hex chars)\n */\nexport async function hashJson<T>(obj: T): Promise<string> {\n return hashCanonical(obj).slice('sha256:'.length)\n}\n\n/**\n * Key-sorted `JSON.stringify` digest. Private and read-only: it exists so a\n * manifest signed under `'sha256-content'` still verifies, and nothing that\n * WRITES a digest may call it.\n */\nfunction legacyContentDigest(value: unknown): string {\n return createHash('sha256')\n .update(JSON.stringify(sortKeysDeep(value)), 'utf8')\n .digest('hex')\n}\n\nfunction sortKeysDeep(value: unknown): unknown {\n if (value === null || typeof value !== 'object') return value\n if (Array.isArray(value)) return value.map(sortKeysDeep)\n const out: Record<string, unknown> = {}\n for (const key of Object.keys(value as Record<string, unknown>).sort()) {\n out[key] = sortKeysDeep((value as Record<string, unknown>)[key])\n }\n return out\n}\n\n/**\n * Digest of a manifest under its own declared scheme, with `contentHash` and\n * `algo` stripped. Synchronous, so a caller that must fail before consuming an\n * observation does not have to await. Throws on an `algo` this release does\n * not know — an unverifiable manifest must not read as a valid one.\n */\nexport function manifestContentDigest(manifest: SignedManifest): string {\n const { contentHash: _contentHash, algo, ...rest } = manifest\n void _contentHash\n if (algo === undefined || algo === 'sha256-content') return legacyContentDigest(rest)\n if (algo === 'sha256-rfc8785') {\n return createHash('sha256').update(canonicalString(rest), 'utf8').digest('hex')\n }\n throw new Error(`pre-registration: unrecognized manifest hash algo '${String(algo)}'`)\n}\n\n/**\n * Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical\n * JSON, with `contentHash` and `algo` stripped, and stamp the scheme in\n * `algo` so a later reader knows which encoder to verify with.\n */\nexport async function signManifest(m: HypothesisManifest): Promise<SignedManifest> {\n const signed: SignedManifest = { ...m, contentHash: '', algo: 'sha256-rfc8785' }\n return { ...signed, contentHash: manifestContentDigest(signed) }\n}\n\n/**\n * Verify that a signed manifest has not been tampered with, under the scheme\n * the manifest itself declares.\n */\nexport async function verifyManifest(m: SignedManifest): Promise<boolean> {\n return manifestContentDigest(m) === m.contentHash\n}\n\n/**\n * Evaluate a pre-registered hypothesis against observed results.\n * Mechanical — no re-interpretation permitted.\n */\nexport async function evaluateHypothesis(\n manifest: SignedManifest,\n observed: { n: number; effect: number; pValue: number },\n): Promise<HypothesisResult> {\n if (!(await verifyManifest(manifest))) {\n throw new Error('evaluateHypothesis: manifest content hash mismatch (tampered)')\n }\n const reasons: HypothesisResult['rejectionReasons'] = []\n const directionOk = manifest.direction === 'increase' ? observed.effect > 0 : observed.effect < 0\n if (!directionOk) reasons.push('wrong_direction')\n if (Math.abs(observed.effect) < manifest.minEffect) reasons.push('effect_too_small')\n if (observed.pValue >= manifest.alpha) reasons.push('not_significant')\n if (observed.n < manifest.preRegisteredN) reasons.push('undersampled')\n return {\n manifest,\n observedN: observed.n,\n observedEffect: observed.effect,\n observedPValue: observed.pValue,\n confirmed: reasons.length === 0,\n rejectionReasons: reasons,\n }\n}\n","import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { hashJson } from './pre-registration'\n\nexport type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1'\n\nexport type AgentProfileJsonObject = { [key: string]: AgentProfileJson }\n\nexport type AgentProfileJson =\n | string\n | number\n | boolean\n | null\n | AgentProfileJson[]\n | AgentProfileJsonObject\n\nexport type AgentProfileDimensionValue = string | number | boolean | null\n\nexport interface AgentProfileSource {\n /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */\n kind: string\n /** sha256 over the canonical source profile object. */\n hash: string\n}\n\nexport interface AgentProfileSourceInput {\n kind: string\n /** Precomputed sha256 for callers that already sign their profile artifact. */\n hash?: string\n /** Full canonical runtime profile; hashed and then discarded from the cell. */\n profile?: AgentProfileJson\n}\n\nexport interface AgentProfileHarness {\n id: string\n version?: string\n hash?: string\n}\n\nexport interface AgentProfileCellInput {\n profileId: string\n sourceProfile: AgentProfileSourceInput\n harness?: AgentProfileHarness\n model?: string\n promptHash?: string\n dimensions?: Record<string, AgentProfileDimensionValue>\n}\n\nexport interface AgentProfileCell {\n schemaVersion: AgentProfileCellSchemaVersion\n cellId: string\n profileId: string\n sourceProfile: AgentProfileSource\n harness?: AgentProfileHarness\n model?: string\n promptHash?: string\n dimensions?: Record<string, AgentProfileDimensionValue>\n}\n\nexport class AgentProfileCellValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\nconst SHA256_HEX = /^[0-9a-f]{64}$/\n/**\n * A cell id names the digest scheme that produced it. `sha256-rfc8785` is what\n * {@link buildAgentProfileCell} mints; the bare `sha256` form is read-only,\n * carried by cells built under an earlier release, and still verifies.\n */\nconst CELL_ID = /^agent-profile-cell:sha256(?:-rfc8785)?:[0-9a-f]{64}$/\nconst CELL_ID_PREFIX = 'agent-profile-cell:sha256-rfc8785:'\nconst LEGACY_CELL_ID_PREFIX = 'agent-profile-cell:sha256:'\n\nexport async function buildAgentProfileCell(\n input: AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n const material = await normalizeAgentProfileCellInput(input)\n const cellId = `${CELL_ID_PREFIX}${await hashJson(material)}`\n return { ...material, cellId }\n}\n\nexport function agentProfileCellHashMaterial(\n cell: AgentProfileCell,\n): Omit<AgentProfileCell, 'cellId'> {\n const { cellId: _cellId, ...material } = cell\n void _cellId\n return normalizeAgentProfileCell(material)\n}\n\n/**\n * Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material\n * fields, confirming the record has not been tampered with. The id names its own\n * digest scheme, so a cell minted by an earlier release verifies under that scheme.\n */\nexport async function verifyAgentProfileCell(cell: AgentProfileCell): Promise<boolean> {\n validateAgentProfileCell(cell)\n const material = agentProfileCellHashMaterial(cell)\n if (cell.cellId.startsWith(CELL_ID_PREFIX)) {\n return cell.cellId === `${CELL_ID_PREFIX}${await hashJson(material)}`\n }\n return cell.cellId === `${LEGACY_CELL_ID_PREFIX}${legacyCellDigest(material)}`\n}\n\n/**\n * Key-sorted `JSON.stringify` digest. Private and read-only: it verifies a cell\n * id minted before the RFC 8785 scheme, and no path that MINTS an id calls it.\n */\nfunction legacyCellDigest(value: unknown): string {\n return createHash('sha256')\n .update(JSON.stringify(sortKeysDeep(value)), 'utf8')\n .digest('hex')\n}\n\nfunction sortKeysDeep(value: unknown): unknown {\n if (value === null || typeof value !== 'object') return value\n if (Array.isArray(value)) return value.map(sortKeysDeep)\n const out: Record<string, unknown> = {}\n for (const key of Object.keys(value as Record<string, unknown>).sort()) {\n out[key] = sortKeysDeep((value as Record<string, unknown>)[key])\n }\n return out\n}\n\nexport function validateAgentProfileCell(input: unknown): AgentProfileCell {\n if (input === null || typeof input !== 'object') {\n throw new AgentProfileCellValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n expectLiteral(obj.schemaVersion, 'agent-profile-cell/v1', 'schemaVersion')\n if (typeof obj.cellId !== 'string' || !CELL_ID.test(obj.cellId)) {\n throw new AgentProfileCellValidationError(\n 'cellId must match agent-profile-cell:sha256:<64 lowercase hex chars>',\n 'cellId',\n )\n }\n expectString(obj.profileId, 'profileId')\n validateSource(obj.sourceProfile, 'sourceProfile')\n if (obj.harness !== undefined) validateHarness(obj.harness, 'harness')\n if (obj.model !== undefined) expectString(obj.model, 'model')\n if (obj.promptHash !== undefined) expectString(obj.promptHash, 'promptHash')\n if (obj.dimensions !== undefined) validateDimensions(obj.dimensions, 'dimensions')\n return input as AgentProfileCell\n}\n\nexport function requireAgentProfileCell(record: {\n runId: string\n agentProfile?: AgentProfileCell\n}): AgentProfileCell {\n if (!record.agentProfile) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" is missing agentProfile; profile-cell grouping requires explicit profile identity`,\n 'agentProfile',\n )\n }\n return validateAgentProfileCell(record.agentProfile)\n}\n\nexport function agentProfileCellKey(record: {\n runId: string\n agentProfile?: AgentProfileCell\n}): string {\n return requireAgentProfileCell(record).cellId\n}\n\nexport async function assertRunAgentProfileCell(record: {\n runId: string\n model: string\n promptHash: string\n agentProfile?: AgentProfileCell\n}): Promise<AgentProfileCell> {\n const profile = requireAgentProfileCell(record)\n if (!(await verifyAgentProfileCell(profile))) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" has an agentProfile.cellId that does not match its content`,\n 'agentProfile.cellId',\n )\n }\n if (profile.model !== undefined && profile.model !== record.model) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" agentProfile.model \"${profile.model}\" does not match model \"${record.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== record.promptHash) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${record.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n return profile\n}\n\nexport function groupRunsByAgentProfileCell<\n T extends { runId: string; agentProfile?: AgentProfileCell },\n>(records: readonly T[]): Map<string, T[]> {\n const groups = new Map<string, T[]>()\n for (const record of records) {\n const key = agentProfileCellKey(record)\n const bucket = groups.get(key)\n if (bucket) bucket.push(record)\n else groups.set(key, [record])\n }\n return groups\n}\n\nasync function normalizeAgentProfileCellInput(\n input: AgentProfileCellInput,\n): Promise<Omit<AgentProfileCell, 'cellId'>> {\n return normalizeAgentProfileCell({\n schemaVersion: 'agent-profile-cell/v1',\n profileId: input.profileId,\n sourceProfile: await normalizeSourceInput(input.sourceProfile),\n harness: input.harness,\n model: input.model,\n promptHash: input.promptHash,\n dimensions: input.dimensions,\n })\n}\n\nfunction normalizeAgentProfileCell(\n input: Omit<AgentProfileCell, 'cellId'>,\n): Omit<AgentProfileCell, 'cellId'> {\n return compactObject({\n schemaVersion: 'agent-profile-cell/v1' as const,\n profileId: requireNonEmpty(input.profileId, 'profileId'),\n sourceProfile: normalizeSource(input.sourceProfile),\n harness: input.harness ? normalizeHarness(input.harness, 'harness') : undefined,\n model: optionalNonEmpty(input.model, 'model'),\n promptHash: optionalNonEmpty(input.promptHash, 'promptHash'),\n dimensions: input.dimensions\n ? nonEmptyRecord(normalizeDimensions(input.dimensions))\n : undefined,\n })\n}\n\nasync function normalizeSourceInput(input: AgentProfileSourceInput): Promise<AgentProfileSource> {\n const kind = requireNonEmpty(input.kind, 'sourceProfile.kind')\n if (input.hash !== undefined && input.profile !== undefined) {\n throw new AgentProfileCellValidationError(\n 'sourceProfile must provide either hash or profile, not both',\n 'sourceProfile',\n )\n }\n if (input.hash !== undefined) {\n return { kind, hash: requireSha256Hex(input.hash, 'sourceProfile.hash') }\n }\n if (input.profile === undefined) {\n throw new AgentProfileCellValidationError(\n 'sourceProfile must provide hash or profile',\n 'sourceProfile',\n )\n }\n assertJson(input.profile, 'sourceProfile.profile')\n return { kind, hash: await hashJson(input.profile) }\n}\n\nfunction normalizeSource(input: AgentProfileSource): AgentProfileSource {\n return {\n kind: requireNonEmpty(input.kind, 'sourceProfile.kind'),\n hash: requireSha256Hex(input.hash, 'sourceProfile.hash'),\n }\n}\n\nfunction normalizeHarness(input: AgentProfileHarness, path: string): AgentProfileHarness {\n return compactObject({\n id: requireNonEmpty(input.id, `${path}.id`),\n version: optionalNonEmpty(input.version, `${path}.version`),\n hash: optionalNonEmpty(input.hash, `${path}.hash`),\n })\n}\n\nfunction normalizeDimensions(\n input: Record<string, AgentProfileDimensionValue>,\n): Record<string, AgentProfileDimensionValue> {\n const out: Record<string, AgentProfileDimensionValue> = {}\n for (const key of Object.keys(input).sort()) {\n const value = input[key]\n requireNonEmpty(key, 'dimensions.<key>')\n if (\n value !== null &&\n typeof value !== 'string' &&\n typeof value !== 'number' &&\n typeof value !== 'boolean'\n ) {\n throw new AgentProfileCellValidationError(\n 'expected primitive dimension value',\n `dimensions.${key}`,\n )\n }\n if (typeof value === 'number' && !Number.isFinite(value)) {\n throw new AgentProfileCellValidationError('expected finite number', `dimensions.${key}`)\n }\n out[key] = value\n }\n return out\n}\n\nfunction compactObject<T extends Record<string, unknown>>(input: T): T {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as T\n}\n\nfunction nonEmptyRecord<T extends Record<string, unknown>>(input: T): T | undefined {\n return Object.keys(input).length > 0 ? input : undefined\n}\n\nfunction validateSource(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n const rec = value as Record<string, unknown>\n expectString(rec.kind, `${path}.kind`)\n requireSha256Hex(rec.hash, `${path}.hash`)\n}\n\nfunction validateHarness(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n const rec = value as Record<string, unknown>\n expectString(rec.id, `${path}.id`)\n if (rec.version !== undefined) expectString(rec.version, `${path}.version`)\n if (rec.hash !== undefined) expectString(rec.hash, `${path}.hash`)\n}\n\nfunction validateDimensions(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n normalizeDimensions(value as Record<string, AgentProfileDimensionValue>)\n}\n\nfunction assertJson(value: AgentProfileJson, path: string): void {\n if (value === null) return\n const type = typeof value\n if (type === 'string' || type === 'boolean') return\n if (type === 'number') {\n if (!Number.isFinite(value)) {\n throw new AgentProfileCellValidationError('expected finite number', path)\n }\n return\n }\n if (Array.isArray(value)) {\n value.forEach((item, index) => {\n assertJson(item, `${path}[${index}]`)\n })\n return\n }\n if (type === 'object') {\n for (const [key, nested] of Object.entries(value)) {\n requireNonEmpty(key, `${path}.<key>`)\n assertJson(nested, `${path}.${key}`)\n }\n return\n }\n throw new AgentProfileCellValidationError('expected JSON-compatible value', path)\n}\n\nfunction expectLiteral(value: unknown, expected: string, path: string): void {\n if (value !== expected) {\n throw new AgentProfileCellValidationError(`expected ${expected}`, path)\n }\n}\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new AgentProfileCellValidationError('expected non-empty string', path)\n }\n}\n\nfunction requireNonEmpty(value: string, path: string): string {\n if (typeof value !== 'string' || value.length === 0) {\n throw new AgentProfileCellValidationError('expected non-empty string', path)\n }\n return value\n}\n\nfunction optionalNonEmpty(value: string | undefined, path: string): string | undefined {\n if (value === undefined) return undefined\n return requireNonEmpty(value, path)\n}\n\nfunction requireSha256Hex(value: unknown, path: string): string {\n if (typeof value !== 'string' || !SHA256_HEX.test(value)) {\n throw new AgentProfileCellValidationError('expected 64 lowercase sha256 hex chars', path)\n }\n return value\n}\n\n// ── Consumer helpers ─────────────────────────────────────────────────\n//\n// Boilerplate every product consuming `buildAgentProfileCell` used to duplicate:\n//\n// 1. A `JSON.parse(JSON.stringify(value))` helper that canonicalizes an\n// arbitrary `@tangle-network/agent-interface` `AgentProfile` into the recursive\n// `AgentProfileJson` shape, with a fail-loud error when the profile\n// is not JSON-serializable.\n//\n// 2. The magic string `'agent-interface-profile'` for `sourceProfile.kind`.\n//\n// Both belong here so the cross-product cell join (same canonical profile\n// hashes to the same `sourceProfile.hash` across products) is enforced by\n// the type system, not by every consumer remembering to do it right.\n// See blueprint-agent issue tangle-network/agent-eval#82.\n\n/** Canonical `sourceProfile.kind` values. Two products fingerprinting the\n * same canonical profile MUST use the same kind for their cells to share\n * `sourceProfile.hash`. Extend rather than create new strings — adding a\n * new kind is a deliberate cross-product schema change. */\nexport const AGENT_PROFILE_KINDS = {\n /** A profile declared via `defineAgentProfile(...)` from\n * `@tangle-network/agent-interface`. The default kind for router-backed\n * and sandbox-backed products. */\n AGENT_INTERFACE_PROFILE: 'agent-interface-profile',\n} as const\n\nexport type AgentProfileKind = (typeof AGENT_PROFILE_KINDS)[keyof typeof AGENT_PROFILE_KINDS]\n\n/** Canonicalize an arbitrary value into `AgentProfileJson` by JSON\n * round-trip. Throws when the value contains anything not representable\n * as JSON (functions, BigInt, cycles) — non-portable profiles fail loud\n * rather than silently dropping fields. */\nexport function toAgentProfileJson(value: unknown): AgentProfileJson {\n let serialized: string | undefined\n try {\n serialized = JSON.stringify(value)\n } catch (err) {\n throw new AgentProfileCellValidationError(\n `agent profile must be JSON-serializable: ${err instanceof Error ? err.message : String(err)}`,\n 'sourceProfile.profile',\n )\n }\n if (serialized === undefined) {\n throw new AgentProfileCellValidationError(\n 'agent profile must be JSON-serializable (got undefined after JSON.stringify)',\n 'sourceProfile.profile',\n )\n }\n return JSON.parse(serialized) as AgentProfileJson\n}\n\n/** Canonical AgentProfile shape required when deriving a stable cell id. */\nexport type AgentInterfaceProfileLike = AgentProfile & { name: string; version: string }\n\n/** Higher-level helper that hard-codes the canonical\n * `agent-interface-profile` kind plus the JSON canonicalization. Equivalent\n * to calling `buildAgentProfileCell` with `profileId = \\`${name}@${version}\\``\n * and `sourceProfile = { kind: AGENT_INTERFACE_PROFILE, profile: <round-tripped> }`.\n *\n * Use this from any product consuming an agent-interface `AgentProfile`; the\n * manual `buildAgentProfileCell` call is reserved for advanced cases\n * (custom kinds, pre-computed source hashes, alternate profileId\n * conventions). */\nexport async function buildAgentInterfaceProfileCell(\n profile: AgentInterfaceProfileLike,\n input: Omit<AgentProfileCellInput, 'profileId' | 'sourceProfile'>,\n): Promise<AgentProfileCell> {\n if (!profile || typeof profile !== 'object') {\n throw new AgentProfileCellValidationError('AgentProfile must be an object', 'profile')\n }\n if (typeof profile.name !== 'string' || profile.name.length === 0) {\n throw new AgentProfileCellValidationError(\n 'AgentProfile must have a non-empty `name`',\n 'profile.name',\n )\n }\n if (typeof profile.version !== 'string' || profile.version.length === 0) {\n throw new AgentProfileCellValidationError(\n 'AgentProfile must have a non-empty `version`',\n 'profile.version',\n )\n }\n return buildAgentProfileCell({\n ...input,\n profileId: `${profile.name}@${profile.version}`,\n sourceProfile: {\n kind: AGENT_PROFILE_KINDS.AGENT_INTERFACE_PROFILE,\n profile: toAgentProfileJson(profile),\n },\n })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoGA,eAAsB,SAAY,KAAyB;CACzD,OAAO,cAAc,GAAG,CAAC,CAAC,MAAM,CAAgB;AAClD;;;;;;AAOA,SAAS,oBAAoB,OAAwB;CACnD,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAUA,eAAa,KAAK,CAAC,GAAG,MAAM,CAAC,CACnD,OAAO,KAAK;AACjB;AAEA,SAASA,eAAa,OAAyB;CAC7C,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU,OAAO;CACxD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,IAAIA,cAAY;CACvD,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,OAAO,OAAO,KAAK,KAAgC,CAAC,CAAC,KAAK,GACnE,IAAI,OAAOA,eAAc,MAAkC,IAAI;CAEjE,OAAO;AACT;;;;;;;AAQA,SAAgB,sBAAsB,UAAkC;CACtE,MAAM,EAAE,aAAa,cAAc,MAAM,GAAG,SAAS;CAErD,IAAI,SAAS,KAAA,KAAa,SAAS,kBAAkB,OAAO,oBAAoB,IAAI;CACpF,IAAI,SAAS,kBACX,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,IAAI,GAAG,MAAM,CAAC,CAAC,OAAO,KAAK;CAEhF,MAAM,IAAI,MAAM,sDAAsD,OAAO,IAAI,EAAE,EAAE;AACvF;;;;;;AAOA,eAAsB,aAAa,GAAgD;CACjF,MAAM,SAAyB;EAAE,GAAG;EAAG,aAAa;EAAI,MAAM;CAAiB;CAC/E,OAAO;EAAE,GAAG;EAAQ,aAAa,sBAAsB,MAAM;CAAE;AACjE;;;;;AAMA,eAAsB,eAAe,GAAqC;CACxE,OAAO,sBAAsB,CAAC,MAAM,EAAE;AACxC;;;;;AAMA,eAAsB,mBACpB,UACA,UAC2B;CAC3B,IAAI,CAAE,MAAM,eAAe,QAAQ,GACjC,MAAM,IAAI,MAAM,+DAA+D;CAEjF,MAAM,UAAgD,CAAC;CAEvD,IAAI,EADgB,SAAS,cAAc,aAAa,SAAS,SAAS,IAAI,SAAS,SAAS,IAC9E,QAAQ,KAAK,iBAAiB;CAChD,IAAI,KAAK,IAAI,SAAS,MAAM,IAAI,SAAS,WAAW,QAAQ,KAAK,kBAAkB;CACnF,IAAI,SAAS,UAAU,SAAS,OAAO,QAAQ,KAAK,iBAAiB;CACrE,IAAI,SAAS,IAAI,SAAS,gBAAgB,QAAQ,KAAK,cAAc;CACrE,OAAO;EACL;EACA,WAAW,SAAS;EACpB,gBAAgB,SAAS;EACzB,gBAAgB,SAAS;EACzB,WAAW,QAAQ,WAAW;EAC9B,kBAAkB;CACpB;AACF;;;AC5HA,IAAa,kCAAb,cAAqD,gBAAgB;CACnE;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;AAEA,MAAM,aAAa;;;;;;AAMnB,MAAM,UAAU;AAChB,MAAM,iBAAiB;AACvB,MAAM,wBAAwB;AAE9B,eAAsB,sBACpB,OAC2B;CAC3B,MAAM,WAAW,MAAM,+BAA+B,KAAK;CAC3D,MAAM,SAAS,GAAG,iBAAiB,MAAM,SAAS,QAAQ;CAC1D,OAAO;EAAE,GAAG;EAAU;CAAO;AAC/B;AAEA,SAAgB,6BACd,MACkC;CAClC,MAAM,EAAE,QAAQ,SAAS,GAAG,aAAa;CAEzC,OAAO,0BAA0B,QAAQ;AAC3C;;;;;;AAOA,eAAsB,uBAAuB,MAA0C;CACrF,yBAAyB,IAAI;CAC7B,MAAM,WAAW,6BAA6B,IAAI;CAClD,IAAI,KAAK,OAAO,WAAW,cAAc,GACvC,OAAO,KAAK,WAAW,GAAG,iBAAiB,MAAM,SAAS,QAAQ;CAEpE,OAAO,KAAK,WAAW,GAAG,wBAAwB,iBAAiB,QAAQ;AAC7E;;;;;AAMA,SAAS,iBAAiB,OAAwB;CAChD,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAU,aAAa,KAAK,CAAC,GAAG,MAAM,CAAC,CACnD,OAAO,KAAK;AACjB;AAEA,SAAS,aAAa,OAAyB;CAC7C,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU,OAAO;CACxD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,IAAI,YAAY;CACvD,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,OAAO,OAAO,KAAK,KAAgC,CAAC,CAAC,KAAK,GACnE,IAAI,OAAO,aAAc,MAAkC,IAAI;CAEjE,OAAO;AACT;AAEA,SAAgB,yBAAyB,OAAkC;CACzE,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,gCAAgC,iBAAiB;CAE7D,MAAM,MAAM;CACZ,cAAc,IAAI,eAAe,yBAAyB,eAAe;CACzE,IAAI,OAAO,IAAI,WAAW,YAAY,CAAC,QAAQ,KAAK,IAAI,MAAM,GAC5D,MAAM,IAAI,gCACR,wEACA,QACF;CAEF,aAAa,IAAI,WAAW,WAAW;CACvC,eAAe,IAAI,eAAe,eAAe;CACjD,IAAI,IAAI,YAAY,KAAA,GAAW,gBAAgB,IAAI,SAAS,SAAS;CACrE,IAAI,IAAI,UAAU,KAAA,GAAW,aAAa,IAAI,OAAO,OAAO;CAC5D,IAAI,IAAI,eAAe,KAAA,GAAW,aAAa,IAAI,YAAY,YAAY;CAC3E,IAAI,IAAI,eAAe,KAAA,GAAW,mBAAmB,IAAI,YAAY,YAAY;CACjF,OAAO;AACT;AAEA,SAAgB,wBAAwB,QAGnB;CACnB,IAAI,CAAC,OAAO,cACV,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,sFACrB,cACF;CAEF,OAAO,yBAAyB,OAAO,YAAY;AACrD;AAEA,SAAgB,oBAAoB,QAGzB;CACT,OAAO,wBAAwB,MAAM,CAAC,CAAC;AACzC;AAEA,eAAsB,0BAA0B,QAKlB;CAC5B,MAAM,UAAU,wBAAwB,MAAM;CAC9C,IAAI,CAAE,MAAM,uBAAuB,OAAO,GACxC,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,+DACrB,qBACF;CAEF,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OAAO,OAC1D,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,wBAAwB,QAAQ,MAAM,0BAA0B,OAAO,MAAM,IAClG,oBACF;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,OAAO,YACpE,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,6BAA6B,QAAQ,WAAW,+BAA+B,OAAO,WAAW,IACtH,yBACF;CAEF,OAAO;AACT;AAEA,SAAgB,4BAEd,SAAyC;CACzC,MAAM,yBAAS,IAAI,IAAiB;CACpC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,MAAM,oBAAoB,MAAM;EACtC,MAAM,SAAS,OAAO,IAAI,GAAG;EAC7B,IAAI,QAAQ,OAAO,KAAK,MAAM;OACzB,OAAO,IAAI,KAAK,CAAC,MAAM,CAAC;CAC/B;CACA,OAAO;AACT;AAEA,eAAe,+BACb,OAC2C;CAC3C,OAAO,0BAA0B;EAC/B,eAAe;EACf,WAAW,MAAM;EACjB,eAAe,MAAM,qBAAqB,MAAM,aAAa;EAC7D,SAAS,MAAM;EACf,OAAO,MAAM;EACb,YAAY,MAAM;EAClB,YAAY,MAAM;CACpB,CAAC;AACH;AAEA,SAAS,0BACP,OACkC;CAClC,OAAO,cAAc;EACnB,eAAe;EACf,WAAW,gBAAgB,MAAM,WAAW,WAAW;EACvD,eAAe,gBAAgB,MAAM,aAAa;EAClD,SAAS,MAAM,UAAU,iBAAiB,MAAM,SAAS,SAAS,IAAI,KAAA;EACtE,OAAO,iBAAiB,MAAM,OAAO,OAAO;EAC5C,YAAY,iBAAiB,MAAM,YAAY,YAAY;EAC3D,YAAY,MAAM,aACd,eAAe,oBAAoB,MAAM,UAAU,CAAC,IACpD,KAAA;CACN,CAAC;AACH;AAEA,eAAe,qBAAqB,OAA6D;CAC/F,MAAM,OAAO,gBAAgB,MAAM,MAAM,oBAAoB;CAC7D,IAAI,MAAM,SAAS,KAAA,KAAa,MAAM,YAAY,KAAA,GAChD,MAAM,IAAI,gCACR,+DACA,eACF;CAEF,IAAI,MAAM,SAAS,KAAA,GACjB,OAAO;EAAE;EAAM,MAAM,iBAAiB,MAAM,MAAM,oBAAoB;CAAE;CAE1E,IAAI,MAAM,YAAY,KAAA,GACpB,MAAM,IAAI,gCACR,8CACA,eACF;CAEF,WAAW,MAAM,SAAS,uBAAuB;CACjD,OAAO;EAAE;EAAM,MAAM,MAAM,SAAS,MAAM,OAAO;CAAE;AACrD;AAEA,SAAS,gBAAgB,OAA+C;CACtE,OAAO;EACL,MAAM,gBAAgB,MAAM,MAAM,oBAAoB;EACtD,MAAM,iBAAiB,MAAM,MAAM,oBAAoB;CACzD;AACF;AAEA,SAAS,iBAAiB,OAA4B,MAAmC;CACvF,OAAO,cAAc;EACnB,IAAI,gBAAgB,MAAM,IAAI,GAAG,KAAK,IAAI;EAC1C,SAAS,iBAAiB,MAAM,SAAS,GAAG,KAAK,SAAS;EAC1D,MAAM,iBAAiB,MAAM,MAAM,GAAG,KAAK,MAAM;CACnD,CAAC;AACH;AAEA,SAAS,oBACP,OAC4C;CAC5C,MAAM,MAAkD,CAAC;CACzD,KAAK,MAAM,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,KAAK,GAAG;EAC3C,MAAM,QAAQ,MAAM;EACpB,gBAAgB,KAAK,kBAAkB;EACvC,IACE,UAAU,QACV,OAAO,UAAU,YACjB,OAAO,UAAU,YACjB,OAAO,UAAU,WAEjB,MAAM,IAAI,gCACR,sCACA,cAAc,KAChB;EAEF,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,gCAAgC,0BAA0B,cAAc,KAAK;EAEzF,IAAI,OAAO;CACb;CACA,OAAO;AACT;AAEA,SAAS,cAAiD,OAAa;CACrE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;AAEA,SAAS,eAAkD,OAAyB;CAClF,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,SAAS,IAAI,QAAQ,KAAA;AACjD;AAEA,SAAS,eAAe,OAAgB,MAAoB;CAC1D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,MAAM,MAAM;CACZ,aAAa,IAAI,MAAM,GAAG,KAAK,MAAM;CACrC,iBAAiB,IAAI,MAAM,GAAG,KAAK,MAAM;AAC3C;AAEA,SAAS,gBAAgB,OAAgB,MAAoB;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,MAAM,MAAM;CACZ,aAAa,IAAI,IAAI,GAAG,KAAK,IAAI;CACjC,IAAI,IAAI,YAAY,KAAA,GAAW,aAAa,IAAI,SAAS,GAAG,KAAK,SAAS;CAC1E,IAAI,IAAI,SAAS,KAAA,GAAW,aAAa,IAAI,MAAM,GAAG,KAAK,MAAM;AACnE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,oBAAoB,KAAmD;AACzE;AAEA,SAAS,WAAW,OAAyB,MAAoB;CAC/D,IAAI,UAAU,MAAM;CACpB,MAAM,OAAO,OAAO;CACpB,IAAI,SAAS,YAAY,SAAS,WAAW;CAC7C,IAAI,SAAS,UAAU;EACrB,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gCAAgC,0BAA0B,IAAI;EAE1E;CACF;CACA,IAAI,MAAM,QAAQ,KAAK,GAAG;EACxB,MAAM,SAAS,MAAM,UAAU;GAC7B,WAAW,MAAM,GAAG,KAAK,GAAG,MAAM,EAAE;EACtC,CAAC;EACD;CACF;CACA,IAAI,SAAS,UAAU;EACrB,KAAK,MAAM,CAAC,KAAK,WAAW,OAAO,QAAQ,KAAK,GAAG;GACjD,gBAAgB,KAAK,GAAG,KAAK,OAAO;GACpC,WAAW,QAAQ,GAAG,KAAK,GAAG,KAAK;EACrC;EACA;CACF;CACA,MAAM,IAAI,gCAAgC,kCAAkC,IAAI;AAClF;AAEA,SAAS,cAAc,OAAgB,UAAkB,MAAoB;CAC3E,IAAI,UAAU,UACZ,MAAM,IAAI,gCAAgC,YAAY,YAAY,IAAI;AAE1E;AAEA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,gCAAgC,6BAA6B,IAAI;AAE/E;AAEA,SAAS,gBAAgB,OAAe,MAAsB;CAC5D,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,gCAAgC,6BAA6B,IAAI;CAE7E,OAAO;AACT;AAEA,SAAS,iBAAiB,OAA2B,MAAkC;CACrF,IAAI,UAAU,KAAA,GAAW,OAAO,KAAA;CAChC,OAAO,gBAAgB,OAAO,IAAI;AACpC;AAEA,SAAS,iBAAiB,OAAgB,MAAsB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,WAAW,KAAK,KAAK,GACrD,MAAM,IAAI,gCAAgC,0CAA0C,IAAI;CAE1F,OAAO;AACT;;;;;AAsBA,MAAa,sBAAsB;;;;AAIjC,yBAAyB,0BAC3B;;;;;AAQA,SAAgB,mBAAmB,OAAkC;CACnE,IAAI;CACJ,IAAI;EACF,aAAa,KAAK,UAAU,KAAK;CACnC,SAAS,KAAK;EACZ,MAAM,IAAI,gCACR,4CAA4C,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,KAC3F,uBACF;CACF;CACA,IAAI,eAAe,KAAA,GACjB,MAAM,IAAI,gCACR,gFACA,uBACF;CAEF,OAAO,KAAK,MAAM,UAAU;AAC9B;;;;;;;;;;AAcA,eAAsB,+BACpB,SACA,OAC2B;CAC3B,IAAI,CAAC,WAAW,OAAO,YAAY,UACjC,MAAM,IAAI,gCAAgC,kCAAkC,SAAS;CAEvF,IAAI,OAAO,QAAQ,SAAS,YAAY,QAAQ,KAAK,WAAW,GAC9D,MAAM,IAAI,gCACR,6CACA,cACF;CAEF,IAAI,OAAO,QAAQ,YAAY,YAAY,QAAQ,QAAQ,WAAW,GACpE,MAAM,IAAI,gCACR,gDACA,iBACF;CAEF,OAAO,sBAAsB;EAC3B,GAAG;EACH,WAAW,GAAG,QAAQ,KAAK,GAAG,QAAQ;EACtC,eAAe;GACb,MAAM,oBAAoB;GAC1B,SAAS,mBAAmB,OAAO;EACrC;CACF,CAAC;AACH"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"agent-profile-cell-CTOZJUuE.d.ts","names":[],"sources":["../src/agent-profile-cell.ts"],"mappings":";;;KAKY;KAEA;GAA4B,cAAc;;KAE1C,sDAKR,qBACA;KAEQ;UAEK;;EAEf;;EAEA;;UAGe;EACf;;EAEA;;EAEA,UAAU;;UAGK;EACf;EACA;EACA;;UAGe;EACf;EACA,eAAe;EACf,UAAU;EACV;EACA;EACA,aAAa,eAAe;;UAGb;EACf,eAAe;EACf;EACA;EACA,eAAe;EACf,UAAU;EACV;EACA;EACA,aAAa,eAAe;;cAGjB,wCAAwC;WAC1C;EACT,YAAY,iBAAiB;;iBAgBT,sBACpB,OAAO,wBACN,QAAQ;iBAMK,6BACd,MAAM,mBACL,KAAK;;;;;;iBAWc,uBAAuB,MAAM,mBAAmB;iBA6BtD,yBAAyB,iBAAiB;iBAqB1C,wBAAwB;EACtC;EACA,eAAe;IACb;iBAUY,oBAAoB;EAClC;EACA,eAAe;;iBAKK,0BAA0B;EAC9C;EACA;EACA;EACA,eAAe;IACb,QAAQ;iBAuBI,4BACd;EAAY;EAAe,eAAe;GAC1C,kBAAkB,MAAM,YAAY;;;;;cA0NzB;;;;WAIX;;KAGU,2BAA2B,kCAAkC;;;;;iBAMzD,mBAAmB,iBAAiB;;KAoBxC,4BAA4B;EAAiB;EAAc;;;;;;;;;;;iBAWjD,+BACpB,SAAS,2BACT,OAAO,KAAK,wDACX,QAAQ"}