@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,476 +1,208 @@
1
- # `InsightReport`: the report
1
+ # Read an InsightReport
2
2
 
3
- The single shape every analysis call returns. `selfImprove()` embeds it in `SelfImproveResult.insight`; `analyzeRuns()` returns it directly. The hosted-tier wire format carries it on `EvalRunEvent.insightReport?`.
4
-
5
- Use `summarizeExecution({ runs })` when observed traces have no task-quality labels.
6
- It returns only `execution` and `costProvenance`, so callers do not need to fabricate a quality score to report runtime facts.
7
-
8
- Every section is **opt-in based on what your data supports**: the function never invents signal. If your runs don't carry judge scores, `judges` is empty. If there's no baseline/candidate split, `lift` is undefined. The shape is consistent; population is honest.
9
-
10
- This page walks every section with a real (synthetic) example and explains how to act on it.
11
-
12
- ---
13
-
14
- ## At a glance
3
+ `analyzeRuns()` summarizes captured `RunRecord` evidence and returns an `InsightReport`.
4
+ `selfImprove()` includes the same report in `result.insight`.
5
+ Analysis makes no model calls unless you supply an analyst that uses one.
15
6
 
16
7
  ```ts
17
- interface InsightReport {
18
- n: number // runs analyzed
19
- execution: ExecutionInsight // duration, tokens, errors, terminal outcomes
20
- composite: ScalarDistribution // always
21
- perDimension: Record<string, ScalarDistribution> // when judgeScores carry dimensions
22
- costQuality: { cost: ScalarDistribution; pareto: ParetoFigureSpec } // always
23
- judges: Record<string, JudgeInsight> // when runs carry judge scores
24
- interRater?: InterRaterInsight // when raterScores supplied
25
- lift?: LiftInsight // when baseline + candidate present
26
- failureClasses?: FailureClassTally[] // canonical task-failure counts
27
- failureClusters?: FailureClusterInsight // when AnalystRegistry wired
28
- contamination?: ContaminationInsight // when canaryScenarios supplied
29
- outcomeCorrelation?: OutcomeCorrelationInsight // when outcomeSignal supplied
30
- release: ReleaseSummary // always
31
- recommendations: Recommendation[] // always: read this FIRST
32
- }
33
- ```
34
-
35
- ---
36
-
37
- ## `execution`: runtime facts, separate from quality
38
-
39
- Always present.
40
- It reports duration, optional queue time, direct input, output, reasoning, cache-read, and cache-write tokens, model-call coverage, model cohorts, execution errors, terminal outcomes, and separately reported orchestration aggregates.
41
- These fields describe what ran; they do not claim whether the task succeeded.
42
- `executionErrors` counts child or internal errors reported by the producer.
43
- `terminalOutcomes` reads only `RunRecord.terminalOutcome`, which must come from root-run or process evidence.
44
- A child tool error can therefore appear in a run whose terminal outcome is `succeeded`.
45
- Current OTel and code-agent adapters also preserve process, guardrail, judge, propagated-parent, and unknown error counts in `RunRecord.outcome.raw`.
46
- These counters are diagnostic and never become task-quality scores.
47
-
48
- ```jsonc
49
- {
50
- "execution": {
51
- "durationMs": { "n": 30, "p50": 5400, "p95": 82000, "min": 900, "max": 190000 },
52
- "queueMs": {
53
- "n": 0,
54
- "mean": null,
55
- "p50": null,
56
- "p95": null,
57
- "stddev": null,
58
- "min": null,
59
- "max": null,
60
- "histogram": []
61
- },
62
- "tokenUsage": {
63
- "totals": { "input": 50132, "output": 471783, "reasoning": 12000, "cached": 60489565, "cacheWrite": 3032227 },
64
- "input": { "n": 30, "p50": 25, "p95": 56 },
65
- "output": { "n": 30, "p50": 230, "p95": 2651 },
66
- "reasoning": { "n": 12, "p50": 800, "p95": 2400 },
67
- "cached": { "n": 20, "p50": 94193, "p95": 310178 },
68
- "cacheWrite": { "n": 20, "p50": 3070, "p95": 11148 }
69
- },
70
- "aggregateUsage": {
71
- "runs": 2,
72
- "tokenUsage": {
73
- "totals": { "input": 5000, "output": 176829, "reasoning": 0, "cached": 0, "cacheWrite": 0 }
74
- },
75
- "costUsd": { "n": 0 },
76
- "totalCostUsd": 0
77
- },
78
- "modelCalls": { "runs": 20, "events": 42, "reportingRuns": 30 },
79
- "models": [{ "model": "claude-opus@2026-07-01", "runs": 20 }],
80
- "executionErrors": {
81
- "runs": 2,
82
- "fraction": 0.067,
83
- "events": 3,
84
- "reportingRuns": 30,
85
- "errorSpanEvents": 3,
86
- "errorSpanReportingRuns": 30,
87
- "byTerminalOutcome": {
88
- "succeeded": { "withErrors": 1, "withoutErrors": 26, "unreported": 0 },
89
- "failed": { "withErrors": 0, "withoutErrors": 1, "unreported": 0 },
90
- "cancelled": { "withErrors": 0, "withoutErrors": 0, "unreported": 0 },
91
- "incomplete": { "withErrors": 0, "withoutErrors": 0, "unreported": 0 },
92
- "unknown": { "withErrors": 1, "withoutErrors": 1, "unreported": 0 }
93
- }
94
- },
95
- "terminalOutcomes": {
96
- "succeeded": 27,
97
- "failed": 1,
98
- "cancelled": 0,
99
- "incomplete": 0,
100
- "unknown": 2
101
- }
102
- }
103
- }
104
- ```
105
-
106
- Use `distribution.n` for optional fields to distinguish an uncaptured category from a recorded zero.
107
- When `distribution.n` is zero, `mean`, percentiles, standard deviation, minimum, and maximum are `null`.
108
- Use `executionErrors.reportingRuns` to assess error-telemetry coverage.
109
- `errorSpanEvents` preserves the exact child-span error count separately from other reported execution errors.
110
- The error fraction uses `reportingRuns` as its denominator and is `null` when no run reported error telemetry, so missing telemetry is not treated as a clean run.
111
- `byTerminalOutcome` is a cross-tab, not a causal recovery claim.
112
- It keeps reported errors, reported zeroes, and missing error telemetry separate for every terminal result.
113
- Missing terminal evidence counts as `unknown`, not `failed`.
114
- Never add `aggregateUsage` to direct `tokenUsage`: orchestration spans may repeat model-call usage from other traces.
115
- Cost remains in `costQuality`, where observed, estimated, and uncaptured USD stay separate.
116
-
117
- ---
118
-
119
- ## `n` + `composite` + `perDimension`: distributional summary
120
-
121
- Always present. The basic "where are my numbers" view.
122
-
123
- ```jsonc
124
- {
125
- "n": 30,
126
- "composite": {
127
- "n": 30,
128
- "mean": 0.683, "p50": 0.667, "p95": 1.000, "stddev": 0.231,
129
- "min": 0.0, "max": 1.0,
130
- "histogram": [
131
- { "lo": 0.0, "hi": 0.083, "count": 5 },
132
- { "lo": 0.083, "hi": 0.167, "count": 0 },
133
- // ...12 bins by default
134
- ]
135
- },
136
- "perDimension": {
137
- "clarity": { "mean": 0.72, "p50": 0.75, "p95": 0.95, "stddev": 0.18, /* ... */ },
138
- "concision": { "mean": 0.65, "p50": 0.68, "p95": 0.88, "stddev": 0.21, /* ... */ }
139
- }
140
- }
141
- ```
142
-
143
- Read `composite.mean` only when `composite.n > 0`.
144
- A `null` mean means task quality was not measured, not that quality was zero.
145
- When a measured mean is below 0.5, inspect the lowest-scoring runs before tuning.
146
-
147
- **Read next:** `perDimension`. If `clarity` is high but `concision` is low, your prompts get the right ideas in too many words: different fix than "wrong ideas."
148
-
149
- **Use the histogram for:** finding bimodal failure modes. A bin with `count > 0` near zero and another > 0 near 1 means your agent has two distinct behaviors, not one noisy one.
150
-
151
- ### Why this is not the same shape as a campaign aggregate
152
-
153
- This report uses `ScalarDistribution`.
154
- A campaign aggregate (`CampaignResult.aggregates`) uses `SeriesDistribution`, the value `summarizeNumberSeries` returns.
155
- The two shapes stay separate for three reasons, and none of them is an accident of history.
156
-
157
- 1. `ScalarDistribution` is a wire contract.
158
- `ScalarDistributionSchema` in `src/hosted/schemas.ts:45` is a strict Zod object.
159
- It is embedded in `InsightReportSchema`, which is embedded in `EvalRunEventSchema`, which the hosted client validates every event against before it ships (`src/hosted/client.ts:182`).
160
- A strict object rejects an unknown key, so adding or renaming a field breaks every event a consumer already sends.
161
- 2. `ScalarDistribution` reports what a report needs and a series summary does not have: a histogram, the worst-N `tailRuns` by score, `p95` for a latency question, and `mean` plus `stddev` beside the order statistics.
162
- `SeriesDistribution` is the in-memory summary of a plain number series with no run identity attached.
163
- 3. The two answer at different `n = 0` boundaries.
164
- `ScalarDistribution` represents an empty series as `n: 0` with every field `null`, because a report always has a slot for a metric it did not measure.
165
- `summarizeNumberSeries` returns `null` for an empty series, because there is no distribution to report and a zero-filled summary would read as a measured all-zero series.
166
-
167
- Both refuse to encode a missing measurement as a zero.
168
- That is the shared rule; the shapes differ because the surfaces differ.
169
-
170
- ---
171
-
172
- ## `costQuality`: cost-vs-quality Pareto
173
-
174
- Always present. `cost.histogram` is the per-run cost distribution; `pareto` is the substrate's `ParetoFigureSpec`.
175
-
176
- ```jsonc
177
- {
178
- "costQuality": {
179
- "cost": {
180
- "mean": 0.024, "p95": 0.041,
181
- "histogram": [/* */]
182
- },
183
- "pareto": {
184
- "kind": "pareto-cost-quality",
185
- "split": "holdout",
186
- "axes": { "x": "costUsd", "y": "score" },
187
- "points": [
188
- { "candidateId": "baseline", "cost": 0.018, "quality": 0.58, "n": 20, "onFrontier": true },
189
- { "candidateId": "winner", "cost": 0.027, "quality": 0.65, "n": 20, "onFrontier": true }
190
- ]
191
- }
192
- }
193
- }
194
- ```
195
-
196
- **Use this when:** comparing prompts, models, or candidate surfaces. The Pareto frontier is your menu of "best you can do at each cost level."
197
-
198
- **Render with:** any chart library: `points` is plain JSON. Hosted-tier dashboards render this as a scatter with the frontier highlighted.
199
-
200
- ---
201
-
202
- ## `judges`: per-judge mean
203
-
204
- Populated when run records carry `outcome.judgeScores`.
205
-
206
- ```jsonc
207
- {
208
- "judges": {
209
- "domain-expert": { "n": 30, "meanScore": 0.71 },
210
- "helpfulness-llm": { "n": 30, "meanScore": 0.62 }
211
- }
212
- }
213
- ```
214
-
215
- The substrate's full judge-calibration suite (positional bias, self-preference, verbosity bias) lives in `/reporting` and operates on **paired-by-condition** inputs that `analyzeRuns` doesn't synthesize from raw `RunRecord[]`. Wire them yourself when you have the paired data; the report's `judges` map is the corpus-level slice.
216
-
217
- **Use this when:** comparing multiple judges over the same corpus. A big gap between two judges' means is the first signal that one of them is mis-calibrated.
218
-
219
- ---
220
-
221
- ## `interRater`: multi-rater agreement and disagreement review
222
-
223
- Populated when `analyzeRuns({ raterScores })` is supplied: typically via `fromFeedbackTable()`.
224
-
225
- ```jsonc
226
- {
227
- "interRater": {
228
- "raters": 3,
229
- "jointlyRated": 30,
230
- "kappa": 0.40,
231
- "icc": 0.42,
232
- "pearson": 0.43,
233
- "spearman": 0.41,
234
- "perPair": {
235
- "alice::bob": 0.53,
236
- "alice::carol": 0.47,
237
- "bob::carol": 0.19
238
- },
239
- "disagreementCases": [
240
- { "runId": "claim-7", "range": 1.00,
241
- "ratings": [{"rater":"alice","score":1},{"rater":"bob","score":1},{"rater":"carol","score":0}] },
242
- { "runId": "claim-13", "range": 1.00,
243
- "ratings": [{"rater":"alice","score":0},{"rater":"bob","score":0},{"rater":"carol","score":1}] }
244
- // ...top 20 by range
245
- ]
246
- }
247
- }
248
- ```
249
-
250
- **Read first:** `kappa` and `icc`, which measure absolute agreement.
251
- Pearson and Spearman measure correlation and can remain high when raters use different score levels.
252
- When absolute agreement is low, review the largest disagreement cases before automating the rubric.
253
-
254
- **Use this when:** building per-rater LLM judges. Each rater's individual scores are the gold signal you calibrate against. Once a calibrated LLM matches the human ≥85%, you can auto-grade and escalate only the disagreement cases.
255
-
256
- ---
257
-
258
- ## `lift`: paired-bootstrap statistical lift
259
-
260
- Populated when baseline + candidate candidates are present (auto-detected from two distinct `candidateId`s, or explicit via `baselineCandidateId` + `candidateCandidateId`).
261
-
262
- ```jsonc
263
- {
264
- "lift": {
265
- "baselineMean": 0.58,
266
- "candidateMean": 0.65,
267
- "delta": 0.07,
268
- "ci95": [0.04, 0.10], // bootstrap CI on the delta
269
- "pValue": 0.0008, // paired t-test; null when the delta is a non-zero constant
270
- "n": 40, // paired observations
271
- "unpairedBaselineRuns": 2,
272
- "unpairedCandidateRuns": 1,
273
- "cohensD": 0.41, // paired Cohen's dz; null when delta variance is zero
274
- "mde": 0.06, // min detectable effect at current n, 80% power
275
- "requiredN": 38 // paired n needed at 80% power; null when dz is undefined
276
- }
277
- }
278
- ```
279
-
280
- Rows pair only when `(experimentId, scenarioId, seed)` matches.
281
- Missing `scenarioId` and duplicate identities fail loudly.
282
- Unmatched rows are reported and excluded from paired statistics.
283
-
284
- **Decision rule:**
285
- - `ci95[0] > threshold` → **SHIP.** Lower bound above your delta threshold means the lift is real at 95% confidence.
286
- - `ci95[0] ≤ threshold < ci95[1]` → **INCONCLUSIVE.** Expand the corpus or wait for more data.
287
- - `ci95[1] ≤ threshold` → **HOLD.** No evidence the candidate is better.
288
-
289
- The `recommendations` array surfaces exactly this decision (`kind: 'ship' | 'hold' | 'expand-corpus'`): that's what consumers should read.
290
-
291
- **Why bootstrap, not t-test alone:** paired bootstrap is distribution-free. Your judge scores are bounded in [0,1] and almost never normal; the bootstrap CI is the honest one.
292
-
293
- ---
294
-
295
- ## `failureClasses`: canonical task-failure counts
296
-
297
- Populated when a run has a non-success `failureClass` or a measured task score below the failure threshold.
298
- Runs without an explicit class are counted as `unknown`.
299
- The optional `failureMode` remains domain-specific detail on the original run and is never used as a second grouping key.
300
-
301
- ```jsonc
302
- {
303
- "failureClasses": [
304
- { "failureClass": "bad_retrieval", "count": 9, "share": 0.28 },
305
- { "failureClass": "instruction_following", "count": 4, "share": 0.13 }
306
- ]
307
- }
308
- ```
309
-
310
- Use this section to compare failure causes across products without a model call.
311
- Use `failureClusters` when you need a semantic diagnosis within those classes.
312
-
313
- ---
314
-
315
- ## `failureClusters`: grouped failure modes
316
-
317
- Populated when an `AnalystRegistry` is passed via `analyzeRuns({ analyst })`. The substrate runs each failed run through the registered analysts and groups findings by `analyst_id` / `area`.
318
-
319
- ```jsonc
320
- {
321
- "failureClusters": {
322
- "totalFailures": 11,
323
- "clusters": [
324
- { "id": "off-topic-drift", "name": "off-topic-drift",
325
- "share": 0.45, "exemplars": ["run-12", "run-19", "run-33"] },
326
- { "id": "over-confidence", "name": "over-confidence",
327
- "share": 0.27, "exemplars": ["run-3", "run-21"] },
328
- { "id": "format-mismatch", "name": "format-mismatch",
329
- "share": 0.18, "exemplars": ["run-41", "run-44"] }
330
- ]
331
- }
332
- }
333
- ```
334
-
335
- **Read first:** the top cluster's `share`. If one cluster is > 40% of failures, fix that pattern before doing anything else.
336
-
337
- **Use this when:** triaging a regression. Failure clusters tell you "fix this kind of thing first."
338
-
339
- **To wire it:** register analysts in `AnalystRegistry`. See `src/analyst/registry.ts` and `src/analyst/kinds/index.ts` for the four built-in kinds (`failure-mode`, `improvement`, `knowledge-gap`, `knowledge-poisoning`).
340
-
341
- ---
342
-
343
- ## `contamination`: canary check
344
-
345
- Populated when canary scenarios are passed via `analyzeRuns({ canaryScenarios })`. Each canary carries a sentinel string the agent should never emit; the report counts leaks.
346
-
347
- ```jsonc
348
- {
349
- "contamination": {
350
- "leaks": 0,
351
- "holdoutAuditPassed": true,
352
- "details": []
353
- }
354
- }
355
- ```
356
-
357
- When `leaks > 0`:
358
-
359
- ```jsonc
360
- {
361
- "contamination": {
362
- "leaks": 2,
363
- "holdoutAuditPassed": false,
364
- "details": [
365
- { "runId": "run-12", "canary": "xyz-secret-canary-123", "matched": "...the secret xyz-secret-canary-123 says..." }
366
- ]
367
- }
368
- }
369
- ```
370
-
371
- **When this fails:** your holdout corpus has leaked into training context. The `lift` number is **unreliable**. Investigate before shipping anything.
372
-
373
- ---
374
-
375
- ## `outcomeCorrelation`: closing the loop on real outcomes
376
-
377
- Populated when `outcomeSignal: { metric, valueByRunId }` is supplied.
378
-
379
- ```jsonc
380
- {
381
- "outcomeCorrelation": {
382
- "metric": "engagement_rate",
383
- "n": 80,
384
- "pearson": 0.72, // linear correlation
385
- "spearman": 0.69, // rank correlation (robust to monotonic nonlinearity)
386
- "rewardModel": {
387
- "intercept": 0.04,
388
- "slope": 1.93,
389
- "r2": 0.52 // share of outcome variance the judge explains
390
- }
391
- }
8
+ import { analyzeRuns } from '@tangle-network/agent-eval/contract'
9
+ import type { RunRecord } from '@tangle-network/agent-eval'
10
+
11
+ export async function compareCapturedRuns(runs: RunRecord[]) {
12
+ return analyzeRuns({
13
+ runs,
14
+ split: 'holdout',
15
+ baselineCandidateId: 'baseline',
16
+ candidateCandidateId: 'candidate',
17
+ decisionThreshold: 0.02,
18
+ })
392
19
  }
393
20
  ```
394
21
 
395
- This is the layer that says **"does my judge's taste actually predict the metric the business cares about?"**
22
+ Pass both candidate IDs to preserve the intended comparison direction, including regressions.
23
+ Without both IDs, the analyzer infers a comparison only when exactly two candidates exist.
24
+ It then treats the lower-scoring candidate as the baseline.
25
+ That inference cannot establish whether a specific change regressed.
396
26
 
397
- **Read first:** `spearman`. If it's < 0.3 in absolute value, your judges are scoring something different from what wins downstream. Refit the judges (use the customer's downstream signal as gold) or change the rubric.
27
+ Use `summarizeExecution({ runs })` when traces contain runtime facts without task-quality labels.
28
+ It returns `execution` and `costProvenance` without interpreting release readiness.
398
29
 
399
- **The reward model** is the simple linear `y = intercept + slope * composite`. Use it to:
400
- - Predict the engagement of a new run from its composite score alone.
401
- - Set a `composite` threshold for "must beat X to ship" based on the engagement equivalent.
30
+ The [offline example](../examples/analyze-existing-runs/) shows a complete call.
31
+ The [report types](../src/contract/insight-report.ts) and [analysis options](../src/contract/analyze-runs.ts) define the current API.
402
32
 
403
- ---
33
+ ## Match sections to evidence
404
34
 
405
- ## `release`: pass/warn/fail axes
406
-
407
- Always present. Roll-up across three axes: quality lift, contamination, composite distribution.
35
+ | Section | Input and scope |
36
+ |---|---|
37
+ | `n` | Number of validated input runs. |
38
+ | `execution` | Recorded durations, tokens, models, execution errors, and terminal outcomes. |
39
+ | `composite` | Finite scores from the selected split. |
40
+ | `perDimension`, `judges` | Recorded `outcome.judgeScores`; empty maps when absent. |
41
+ | `costQuality` | Known observed or estimated costs, their provenance, and candidate cost/quality points. |
42
+ | `lift` | Scored baseline/candidate rows sharing pairing identities. |
43
+ | `interRater` | Supplied `raterScores`, with at least two raters and jointly rated runs. |
44
+ | `failureClasses` | Explicit non-success classes or measured scores below the analyzer's failure threshold. |
45
+ | `failureClusters` | Findings from the supplied `AnalystRegistry` on failed runs. |
46
+ | `contamination` | Supplied `canaryScenarios`, searched within captured text outputs. |
47
+ | `outcomeCorrelation` | Supplied `outcomeSignal`, joined to at least three finite run scores. |
48
+ | `priorPeriodComparison` | Supplied `baselineRuns`, compared as an unpaired prior window. |
49
+ | `release`, `recommendations` | Diagnostic rules applied to the populated sections. |
50
+
51
+ Optional sections are absent when their required inputs are unavailable.
52
+ Some supplied inputs can produce empty sections, such as no failure clusters among successful tasks.
53
+ An absent section does not establish that its check passed.
54
+
55
+ `split: 'auto'` selects holdout if any run has a holdout score; otherwise it selects search.
56
+ Set `split` explicitly when you know which scores the report should use.
57
+ Inspect `composite.n`: it can be smaller than `n` when some runs have no score for that split.
58
+
59
+ ## Read execution and missing values
60
+
61
+ `execution` describes what ran.
62
+ A successful process can still produce an incorrect task result.
63
+ A child tool error can also occur in a run whose root outcome is `succeeded`.
64
+
65
+ `terminalOutcomes` reads `RunRecord.terminalOutcome`.
66
+ Missing terminal evidence counts as `unknown`.
67
+ `executionErrors` reads producer-reported error counts independently.
68
+ Its `fraction` uses `reportingRuns` as the denominator and becomes `null` when no run reports error telemetry.
69
+ The `byTerminalOutcome` table separates reported errors, reported zeroes, and unreported error telemetry for each terminal outcome.
70
+ It describes their co-occurrence; it does not establish recovery or causation.
71
+
72
+ Optional token categories and queue time carry their own distribution counts.
73
+ For any `ScalarDistribution`, `n: 0` means no finite measurement was available.
74
+ Its mean, percentiles, standard deviation, minimum, and maximum are then `null`, with an empty histogram.
75
+ A measured zero has a positive count and a value of zero.
76
+
77
+ Keep orchestration `aggregateUsage` separate from direct token usage.
78
+ An aggregate span can repeat usage already captured in model-call traces.
79
+ Adding both totals can count the same work twice.
80
+
81
+ ## Inspect quality and cost distributions
82
+
83
+ `composite` describes the scored input corpus, including both candidates when both are supplied.
84
+ It is not the candidate's mean alone.
85
+ Use `lift.baselineMean` and `lift.candidateMean` for the paired comparison.
86
+ `composite.tailRuns` identifies the lowest-scoring runs for inspection.
87
+ Histogram peaks can suggest subgroups; inspect cases before attributing them to distinct agent behaviors.
88
+
89
+ Judge details use this recorded shape:
408
90
 
409
- ```jsonc
410
- {
411
- "release": {
412
- "status": "pass",
413
- "axes": [
414
- { "name": "quality-lift", "status": "pass",
415
- "detail": "delta=0.070, CI95=[0.040, 0.100], n=40" },
416
- { "name": "contamination", "status": "pass",
417
- "detail": "0 canary leak(s)" },
418
- { "name": "composite-distribution", "status": "pass",
419
- "detail": "mean=0.683, p50=0.667, p95=1.000 over n=30" }
420
- ],
421
- "issues": []
422
- }
91
+ ```ts
92
+ const judgeScores = {
93
+ perJudge: {
94
+ 'field-check': { accuracy: 0.75 },
95
+ },
96
+ perDimMean: { accuracy: 0.75 },
97
+ composite: 0.75,
423
98
  }
424
99
  ```
425
100
 
426
- Overall `status` is `fail` if any axis fails; `warn` if any warn; `pass` otherwise.
427
-
428
- **Use this when:** wiring agent-eval into CI. A `status === 'pass'` from `analyzeRuns` on the candidate vs baseline is your green-light gate.
429
-
430
- ---
431
-
432
- ## `recommendations`: the actionable layer
101
+ Store it as `RunRecord.outcome.judgeScores` alongside the relevant search or holdout score.
102
+ `perDimension` summarizes dimensions; `judges` reports per-judge counts and means.
103
+ Different judge means can reflect different coverage, scales, or criteria.
104
+ Compare shared cases before attributing a difference to miscalibration.
433
105
 
434
- Always present. Read this first.
106
+ `costQuality.provenance` separates observed USD, estimated USD, and uncaptured costs.
107
+ Uncaptured rows are excluded from the cost distribution and Pareto calculation.
108
+ Read `knownFraction` and `costQuality.degraded` before comparing costs.
109
+ A frontier only compares the observed candidate points; it does not identify the best possible system.
435
110
 
436
- ```jsonc
437
- {
438
- "recommendations": [
439
- { "priority": "critical", "kind": "ship",
440
- "title": "Ship: lift 0.070 (95% CI 0.040..0.100)",
441
- "detail": "Holdout lift exceeds threshold 0.02 with 95% bootstrap confidence (n=40, p=0.0008, d=0.41).",
442
- "evidencePath": "lift" },
443
- { "priority": "high", "kind": "investigate",
444
- "title": "Top failure cluster: off-topic-drift (45% of failures)",
445
- "detail": "11 runs failed. The largest cluster groups 3 exemplars under 'off-topic-drift'.",
446
- "evidencePath": "failureClusters.clusters[0]" }
447
- ]
448
- }
449
- ```
450
-
451
- | `kind` | When emitted |
452
- |---|---|
453
- | `ship` | lift CI lower bound > threshold |
454
- | `hold` | lift CI upper bound ≤ threshold |
455
- | `expand-corpus` | lift CI straddles threshold: more data needed |
456
- | `fix` | canary contamination detected |
457
- | `recalibrate` | inter-rater κ < 0.5, OR outcome correlation < 0.3 |
458
- | `investigate` | top failure cluster > some-share |
111
+ Campaign aggregates use `SeriesDistribution`; insight reports use `ScalarDistribution`.
112
+ The latter adds report fields such as histograms and optional run examples.
113
+ An empty campaign number series returns `null`; an empty report distribution retains its slot with `n: 0` and null statistics.
114
+ Both preserve the distinction between missing measurements and measured zeroes.
459
115
 
460
- `evidencePath` points back into the report (`"lift"`, `"contamination"`, `"failureClusters.clusters[0]"`) so a UI can deep-link from each recommendation to its evidence.
116
+ ## Interpret paired lift
461
117
 
462
- ---
118
+ Rows pair on `(experimentId, scenarioId, seed)`.
119
+ Missing scenario IDs and duplicate identities within an arm fail validation.
120
+ Scored rows without a partner remain in `unpairedBaseline` and `unpairedCandidate` counts and are excluded from the paired statistics.
121
+ Unscored rows are also excluded; check the input and score counts separately.
463
122
 
464
- ## How `analyzeRuns` populates each section
123
+ For repeated tasks from one source, supply `independentUnitByScenarioId` as a map from every scored scenario ID to its independent unit.
124
+ The analyzer pairs runs first, averages matched scores within each declared unit, and weights units equally.
125
+ Raw score distributions and unmatched-run counts remain unchanged.
126
+ Repeated runs measure variation on those tasks; they do not create new independent tasks.
465
127
 
466
- | Section | Required input |
128
+ | Lift field | Meaning |
467
129
  |---|---|
468
- | `composite`, `perDimension`, `costQuality`, `release`, `recommendations` | `runs` |
469
- | `judges` | `runs` with `outcome.judgeScores` |
470
- | `interRater` | `raterScores` (≥ 2 raters jointly rated some runs) |
471
- | `lift` | two distinct `candidateId`s in `runs` (or explicit baseline/candidate ids) |
472
- | `failureClusters` | `analyst` registry passed in |
473
- | `contamination` | `canaryScenarios` passed in |
474
- | `outcomeCorrelation` | `outcomeSignal` passed in |
475
-
476
- All sections beyond the always-present ones are `T | undefined`, never empty objects. If a section is missing, your inputs didn't support it: the report is honest about that.
130
+ | `baselineMean`, `candidateMean`, `delta` | Paired means and candidate-minus-baseline difference after any unit aggregation. |
131
+ | `ci95` | Paired bootstrap interval for the mean difference. |
132
+ | `n` | Paired observations used for inference; independent units when declared. |
133
+ | `pairedRunN`, `independentUnitIds` | Raw matched count and unit IDs, present when units are declared. |
134
+ | `minimumRequired`, `decisionEligible` | Bootstrap sample floor and whether the count reaches it. |
135
+ | `pValue` | Paired t-test diagnostic; `null` for a nonzero constant difference. |
136
+ | `cohensD` | Paired Cohen's dz; `null` when difference variance is zero. |
137
+ | `mde` | Approximate detectable effect in standardized units at 80% power. |
138
+ | `requiredN` | Approximate sample size using the observed standardized effect; `null` when it cannot be estimated. |
139
+
140
+ The analyzer's bootstrap decision floor is 20 paired observations.
141
+ Below it, a positive interval remains descriptive and the lift recommendation requests more evidence.
142
+ Reaching the floor only establishes sample-count eligibility.
143
+ A zero-width interval still cannot produce a lift-based ship recommendation.
144
+ An eligible, nonzero-width interval must exceed `decisionThreshold`, which defaults to `0.02` in score units.
145
+
146
+ Bootstrap inference depends on representative, independent observations and adequate sample size.
147
+ It cannot repair selection bias, leaked final cases, or a miscalibrated judge.
148
+ Do not compare standardized `mde` directly with raw score lift.
149
+ Treat `requiredN` as an exploratory approximation, not a prospective power calculation for a target chosen before the study.
150
+
151
+ ## Use recommendations as diagnostics
152
+
153
+ `recommendations` links findings to report sections through `evidencePath`.
154
+ Its `ship` kind can refer to lift or an improved prior-period metric.
155
+ Other findings can coexist with it, including a failed canary check.
156
+ Read the complete report before acting.
157
+
158
+ `release` rolls up quality lift, canary matches, and composite score thresholds.
159
+ An unavailable axis is `not_evaluated` and makes the overall status at least `warn`.
160
+ The quality-lift axis uses positive lift; recommendation thresholds can differ.
161
+ These built-in thresholds are report heuristics, not your product's complete release policy.
162
+
163
+ For automated promotion, use the campaign gate and inspect its contributing checks.
164
+ `selfImprove().gateDecision` comes from that gate.
165
+ See [concepts](./concepts.md#the-five-release-decisions) and the [held-out gate example](../examples/held-out-gate/).
166
+ A reusable claim can declare independent units and a practical effect.
167
+ Optional final-evidence tracking records fresh confirmation; see [evaluation integrity](./evaluation-integrity.md).
168
+
169
+ ## Investigate failures and disagreement
170
+
171
+ `failureClasses` counts explicit non-success classes and scores below `0.5`.
172
+ A low-scoring run without a non-success class is counted as `unknown`.
173
+ Its `share` uses all input runs as the denominator.
174
+ Domain-specific `failureMode` stays on the original record.
175
+
176
+ `failureClusters` runs registered analysts on those failed runs.
177
+ It groups findings by area, with analyst ID as fallback.
178
+ Each cluster's `share` counts affected failed runs, including those beyond the five displayed exemplars.
179
+ Multiple findings in one cluster count once per run.
180
+ A run can belong to several clusters, so cluster shares can sum above one.
181
+ Cluster shares use `totalFailures`, unlike the corpus denominator in `failureClasses`.
182
+ Empty findings can mean analysts skipped or failed; inspect registry logs and hooks when coverage is uncertain.
183
+ See the [custom analyst example](../examples/custom-trace-analyst/) for registration.
184
+
185
+ `interRater` uses runs scored by every supplied rater.
186
+ Check `jointlyRated` before interpreting agreement over the broader corpus.
187
+ Kappa and ICC assess agreement; Pearson and Spearman assess correlation.
188
+ Review the largest disagreement cases and validate any judge changes on independent examples.
189
+ Choose acceptance thresholds for the decision's actual error costs.
190
+
191
+ ## Check canaries and downstream outcomes
192
+
193
+ The canary check searches strings in `metadata.output`, falling back to `metadata.text`.
194
+ Other output layouts need conversion before analysis.
195
+ A match establishes that captured output contains a sentinel; investigate how it arrived there.
196
+ A zero-leak result does not prove isolation, especially when outputs were not captured.
197
+ The section does not report output-coverage counts.
198
+
199
+ `outcomeCorrelation` joins finite `outcomeSignal.valueByRunId` values to run scores.
200
+ Its Pearson and Spearman values describe association in that supplied sample.
201
+ The linear `rewardModel` is fitted and evaluated on those same observations.
202
+ Validate it on separate data before using it to predict outcomes or set a release threshold.
203
+ Weak correlation can reflect noise, limited range, confounding, or the wrong rubric.
204
+ It does not identify the cause by itself.
205
+
206
+ Use [outcome validity](./outcome-validity.md) for declared outcome directions, explicit exclusions, and association intervals.
207
+ Use `baselineRuns` for an unpaired prior-period comparison of available metrics.
208
+ Period differences can reflect traffic, task mix, or capture changes; they do not isolate the effect of a deployment.