@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -0,0 +1,713 @@
1
+ {
2
+ "reviewedBaseRevision": "fe1cc5111aab5d588bf7db3a3785325635937a91",
3
+ "sourceIdentity": {
4
+ "algorithm": "sha256-canonical-file-manifest",
5
+ "paths": [
6
+ "src",
7
+ "package.json",
8
+ "pnpm-lock.yaml",
9
+ "tsconfig.json"
10
+ ],
11
+ "fileCount": 765,
12
+ "digest": "sha256:d743fe180fb7f89c2add26f3f66344a8586b2798f1edcaf8b881b914c10bc009",
13
+ "dependencyScope": "Records the manifest and lockfile; assumes dependencies were installed from that lockfile."
14
+ },
15
+ "diagnosticIdentity": {
16
+ "path": "docs/design/mlbenchmarks-review/probes.mts",
17
+ "sha256": "29d0bba079d8b76247bd3d92a24ea8b256080e22ee37db9ca28f0ac3330c793a"
18
+ },
19
+ "command": "pnpm exec tsx docs/design/mlbenchmarks-review/probes.mts",
20
+ "paidModelCalls": 0,
21
+ "execution": {
22
+ "kind": "offline deterministic diagnostic",
23
+ "modelCalls": 0,
24
+ "callbackImplementation": "Local arithmetic and string checks only; no provider clients are supplied.",
25
+ "temporaryRunStorage": "Allocated under the OS temporary directory and removed in finally.",
26
+ "outputPolicy": "Preserves current returned measurements; excludes temporary paths, run IDs, and wallclock fields.",
27
+ "assertionPolicy": "No assertions require the observed defects or policy boundaries to persist."
28
+ },
29
+ "probes": {
30
+ "holdoutReuse": {
31
+ "inputs": {
32
+ "calls": 2,
33
+ "trainCases": 6,
34
+ "finalCases": 6,
35
+ "generationsPerCall": 1,
36
+ "populationPerGeneration": 1,
37
+ "replicatesPerCase": 1,
38
+ "sameFinalPayloadsOnBothCalls": true,
39
+ "firstResultFedToSecondProposer": false,
40
+ "baselineSurface": "baseline",
41
+ "proposedSurface": "marker",
42
+ "scoring": "1 if artifact contains marker, otherwise 0",
43
+ "expectUsage": "off"
44
+ },
45
+ "results": {
46
+ "rounds": [
47
+ {
48
+ "round": 1,
49
+ "gateDecision": "ship",
50
+ "finalDispatches": 12,
51
+ "distinctFinalIds": 6,
52
+ "finalSplitDigest": "sha256:fe055aa9390b6483701c8686564c1d4d96426d6a545f59e405fa5350efae84fa",
53
+ "agentCallbacks": 24,
54
+ "judgeCallbacks": 24,
55
+ "proposerCallbacks": 1
56
+ },
57
+ {
58
+ "round": 2,
59
+ "gateDecision": "ship",
60
+ "finalDispatches": 12,
61
+ "distinctFinalIds": 6,
62
+ "finalSplitDigest": "sha256:fe055aa9390b6483701c8686564c1d4d96426d6a545f59e405fa5350efae84fa",
63
+ "agentCallbacks": 24,
64
+ "judgeCallbacks": 24,
65
+ "proposerCallbacks": 1
66
+ }
67
+ ],
68
+ "accessPurposes": [
69
+ "debugging",
70
+ "evaluation"
71
+ ],
72
+ "temporaryRunDirectoriesRemoved": true
73
+ },
74
+ "limitations": [
75
+ "Measures repeated final-set access and debugging access, not empirical overfitting.",
76
+ "The calls use deterministic local callbacks and independent temporary run directories.",
77
+ "Does not estimate false-promotion frequency or test a downstream access-control service."
78
+ ]
79
+ },
80
+ "outcomeKeyOrder": {
81
+ "inputs": {
82
+ "n": 10,
83
+ "rows": [
84
+ {
85
+ "score": 0.1,
86
+ "retention": 0.9,
87
+ "csat": 0.1
88
+ },
89
+ {
90
+ "score": 0.2,
91
+ "retention": 0.8,
92
+ "csat": 0.2
93
+ },
94
+ {
95
+ "score": 0.3,
96
+ "retention": 0.7,
97
+ "csat": 0.3
98
+ },
99
+ {
100
+ "score": 0.4,
101
+ "retention": 0.6,
102
+ "csat": 0.4
103
+ },
104
+ {
105
+ "score": 0.5,
106
+ "retention": 0.5,
107
+ "csat": 0.5
108
+ },
109
+ {
110
+ "score": 0.6,
111
+ "retention": 0.4,
112
+ "csat": 0.6
113
+ },
114
+ {
115
+ "score": 0.7,
116
+ "retention": 0.30000000000000004,
117
+ "csat": 0.7
118
+ },
119
+ {
120
+ "score": 0.8,
121
+ "retention": 0.19999999999999996,
122
+ "csat": 0.8
123
+ },
124
+ {
125
+ "score": 0.9,
126
+ "retention": 0.09999999999999998,
127
+ "csat": 0.9
128
+ },
129
+ {
130
+ "score": 1,
131
+ "retention": 0,
132
+ "csat": 1
133
+ }
134
+ ],
135
+ "outcomeRowsPerRun": 1,
136
+ "requestedMetric": "csat",
137
+ "expectedPearsonForRequestedMetric": 1,
138
+ "expectedSpearmanForRequestedMetric": 1,
139
+ "seed": 1,
140
+ "bootstrapIterations": 500
141
+ },
142
+ "results": [
143
+ {
144
+ "keyOrder": "retention-first",
145
+ "latest": {
146
+ "pairs": [
147
+ {
148
+ "evalMetric": "score",
149
+ "outcomeMetric": "csat",
150
+ "n": 10,
151
+ "pearson": -1,
152
+ "spearman": -1,
153
+ "pearsonCi95": {
154
+ "lower": -1.0000000000000002,
155
+ "upper": -0.9999999999999998
156
+ },
157
+ "verdict": "strong"
158
+ }
159
+ ],
160
+ "joinedSamples": 10,
161
+ "skippedRuns": 0
162
+ },
163
+ "mean": {
164
+ "pairs": [
165
+ {
166
+ "evalMetric": "score",
167
+ "outcomeMetric": "csat",
168
+ "n": 10,
169
+ "pearson": 1,
170
+ "spearman": 1,
171
+ "pearsonCi95": {
172
+ "lower": 1,
173
+ "upper": 1
174
+ },
175
+ "verdict": "strong"
176
+ }
177
+ ],
178
+ "joinedSamples": 10,
179
+ "skippedRuns": 0
180
+ }
181
+ },
182
+ {
183
+ "keyOrder": "csat-first",
184
+ "latest": {
185
+ "pairs": [
186
+ {
187
+ "evalMetric": "score",
188
+ "outcomeMetric": "csat",
189
+ "n": 10,
190
+ "pearson": 1,
191
+ "spearman": 1,
192
+ "pearsonCi95": {
193
+ "lower": 1,
194
+ "upper": 1
195
+ },
196
+ "verdict": "strong"
197
+ }
198
+ ],
199
+ "joinedSamples": 10,
200
+ "skippedRuns": 0
201
+ },
202
+ "mean": {
203
+ "pairs": [
204
+ {
205
+ "evalMetric": "score",
206
+ "outcomeMetric": "csat",
207
+ "n": 10,
208
+ "pearson": 1,
209
+ "spearman": 1,
210
+ "pearsonCi95": {
211
+ "lower": 1,
212
+ "upper": 1
213
+ },
214
+ "verdict": "strong"
215
+ }
216
+ ],
217
+ "joinedSamples": 10,
218
+ "skippedRuns": 0
219
+ }
220
+ }
221
+ ],
222
+ "limitations": [
223
+ "Tests metric selection and JSON key order with constructed data, not deployment validity.",
224
+ "Each run has one outcome row, so latest and mean refer to the same requested observation."
225
+ ]
226
+ },
227
+ "adaptationPairing": {
228
+ "inputs": {
229
+ "scenariosA": [
230
+ {
231
+ "scenarioId": "easy-only",
232
+ "score": 0.9
233
+ }
234
+ ],
235
+ "scenariosB": [
236
+ {
237
+ "scenarioId": "hard-only",
238
+ "score": 0.1
239
+ }
240
+ ],
241
+ "ks": [
242
+ 0,
243
+ 1
244
+ ],
245
+ "reps": 1,
246
+ "observationsPerArm": 2,
247
+ "commonScenarios": 0,
248
+ "sameRunnerForBothArms": true,
249
+ "bootstrapSeed": 1
250
+ },
251
+ "results": {
252
+ "perK": [
253
+ {
254
+ "k": 0,
255
+ "deltaMean": 0.8,
256
+ "aLow": 0.9,
257
+ "aHigh": 0.9,
258
+ "bLow": 0.1,
259
+ "bHigh": 0.1
260
+ },
261
+ {
262
+ "k": 1,
263
+ "deltaMean": 0.8,
264
+ "aLow": 0.9,
265
+ "aHigh": 0.9,
266
+ "bLow": 0.1,
267
+ "bHigh": 0.1
268
+ }
269
+ ],
270
+ "areaDelta": 0.8,
271
+ "firstPassKDelta": null,
272
+ "verdict": "a_better",
273
+ "rationale": "mean per-k delta=0.800, area delta=0.800"
274
+ },
275
+ "limitations": [
276
+ "The two arms differ in task difficulty; zero task identities overlap.",
277
+ "This probes the adaptation helper, not the separately implemented campaign paired comparison."
278
+ ]
279
+ },
280
+ "contaminationDisplay": {
281
+ "inputs": {
282
+ "n": 12,
283
+ "originalScorePerCase": 1,
284
+ "perturbedScorePerCase": 0.4,
285
+ "observationsPerCasePerCondition": 1,
286
+ "modelTrainingExposure": "No model is used; scores are constructed fixture values."
287
+ },
288
+ "results": {
289
+ "perScenario": [
290
+ {
291
+ "scenarioId": "case-0",
292
+ "originalScore": 1,
293
+ "perturbedScore": 0.4,
294
+ "delta": -0.6,
295
+ "qValue": 0.4
296
+ },
297
+ {
298
+ "scenarioId": "case-1",
299
+ "originalScore": 1,
300
+ "perturbedScore": 0.4,
301
+ "delta": -0.6,
302
+ "qValue": 0.4
303
+ },
304
+ {
305
+ "scenarioId": "case-2",
306
+ "originalScore": 1,
307
+ "perturbedScore": 0.4,
308
+ "delta": -0.6,
309
+ "qValue": 0.4
310
+ },
311
+ {
312
+ "scenarioId": "case-3",
313
+ "originalScore": 1,
314
+ "perturbedScore": 0.4,
315
+ "delta": -0.6,
316
+ "qValue": 0.4
317
+ },
318
+ {
319
+ "scenarioId": "case-4",
320
+ "originalScore": 1,
321
+ "perturbedScore": 0.4,
322
+ "delta": -0.6,
323
+ "qValue": 0.4
324
+ },
325
+ {
326
+ "scenarioId": "case-5",
327
+ "originalScore": 1,
328
+ "perturbedScore": 0.4,
329
+ "delta": -0.6,
330
+ "qValue": 0.4
331
+ },
332
+ {
333
+ "scenarioId": "case-6",
334
+ "originalScore": 1,
335
+ "perturbedScore": 0.4,
336
+ "delta": -0.6,
337
+ "qValue": 0.4
338
+ },
339
+ {
340
+ "scenarioId": "case-7",
341
+ "originalScore": 1,
342
+ "perturbedScore": 0.4,
343
+ "delta": -0.6,
344
+ "qValue": 0.4
345
+ },
346
+ {
347
+ "scenarioId": "case-8",
348
+ "originalScore": 1,
349
+ "perturbedScore": 0.4,
350
+ "delta": -0.6,
351
+ "qValue": 0.4
352
+ },
353
+ {
354
+ "scenarioId": "case-9",
355
+ "originalScore": 1,
356
+ "perturbedScore": 0.4,
357
+ "delta": -0.6,
358
+ "qValue": 0.4
359
+ },
360
+ {
361
+ "scenarioId": "case-10",
362
+ "originalScore": 1,
363
+ "perturbedScore": 0.4,
364
+ "delta": -0.6,
365
+ "qValue": 0.4
366
+ },
367
+ {
368
+ "scenarioId": "case-11",
369
+ "originalScore": 1,
370
+ "perturbedScore": 0.4,
371
+ "delta": -0.6,
372
+ "qValue": 0.4
373
+ }
374
+ ],
375
+ "pairedTest": {
376
+ "w": 0,
377
+ "p": 0.00048828125,
378
+ "method": "exact",
379
+ "pFloor": 0.00048828125,
380
+ "nNonZero": 12
381
+ },
382
+ "medianDelta": -0.6,
383
+ "meanDelta": -0.5999999999999999,
384
+ "contaminationSuspected": true,
385
+ "reason": "paired p=0.0005 < 0.05 and median drop -0.6000 ≥ 0.05",
386
+ "n": 12
387
+ },
388
+ "limitations": [
389
+ "The global Wilcoxon test measures the constructed paired difference; it does not identify contamination as its cause.",
390
+ "Per-item qValue uses BH on 1 - abs(delta), without a per-item sampling null; it is a display aid in the inspected source.",
391
+ "The per-item qValues do not drive the global contaminationSuspected result."
392
+ ]
393
+ },
394
+ "negativeOutcomeDirection": {
395
+ "inputs": {
396
+ "n": 8,
397
+ "rows": [
398
+ {
399
+ "quality": 0,
400
+ "successRate": 1
401
+ },
402
+ {
403
+ "quality": 0.14285714285714285,
404
+ "successRate": 0.8571428571428572
405
+ },
406
+ {
407
+ "quality": 0.2857142857142857,
408
+ "successRate": 0.7142857142857143
409
+ },
410
+ {
411
+ "quality": 0.42857142857142855,
412
+ "successRate": 0.5714285714285714
413
+ },
414
+ {
415
+ "quality": 0.5714285714285714,
416
+ "successRate": 0.4285714285714286
417
+ },
418
+ {
419
+ "quality": 0.7142857142857143,
420
+ "successRate": 0.2857142857142857
421
+ },
422
+ {
423
+ "quality": 0.8571428571428571,
424
+ "successRate": 0.1428571428571429
425
+ },
426
+ {
427
+ "quality": 1,
428
+ "successRate": 0
429
+ }
430
+ ],
431
+ "rubric": "quality",
432
+ "outcome": "success_rate",
433
+ "desiredOutcomeDirection": "increase",
434
+ "seed": 1,
435
+ "bootstrapResamples": 100,
436
+ "researcherBootstrapResamples": 500,
437
+ "researcherSeed": "Derived deterministically by the validity helper",
438
+ "researcherFailureThreshold": 0.5
439
+ },
440
+ "results": {
441
+ "report": {
442
+ "pairs": [
443
+ {
444
+ "rubric": "quality",
445
+ "outcome": "success_rate",
446
+ "n": 8,
447
+ "pearson": -1,
448
+ "spearman": -1,
449
+ "ci95": {
450
+ "low": -1.0000000000000002,
451
+ "high": -0.9999999999999998
452
+ },
453
+ "verdict": "load_bearing"
454
+ }
455
+ ],
456
+ "ranked": [
457
+ {
458
+ "rubric": "quality",
459
+ "bestOutcome": "success_rate",
460
+ "spearman": -1,
461
+ "pearson": -1,
462
+ "n": 8,
463
+ "verdict": "load_bearing"
464
+ }
465
+ ],
466
+ "joinedSamples": 8,
467
+ "skippedRuns": 0,
468
+ "rubricsWithoutData": []
469
+ },
470
+ "researcher": {
471
+ "report": {
472
+ "pairs": [
473
+ {
474
+ "rubric": "quality",
475
+ "outcome": "success_rate",
476
+ "n": 8,
477
+ "pearson": -1,
478
+ "spearman": -1,
479
+ "ci95": {
480
+ "low": -1.0000000000000002,
481
+ "high": -0.9999999999999998
482
+ },
483
+ "verdict": "load_bearing"
484
+ }
485
+ ],
486
+ "ranked": [
487
+ {
488
+ "rubric": "quality",
489
+ "bestOutcome": "success_rate",
490
+ "spearman": -1,
491
+ "pearson": -1,
492
+ "n": 8,
493
+ "verdict": "load_bearing"
494
+ }
495
+ ],
496
+ "joinedSamples": 8,
497
+ "skippedRuns": 0,
498
+ "rubricsWithoutData": []
499
+ },
500
+ "failureGroups": 1,
501
+ "failures": [
502
+ {
503
+ "code": "low-score-same-candidate",
504
+ "description": "same-candidate scored < 0.5 on 4 run(s) (mean 0.214)",
505
+ "samples": 4
506
+ }
507
+ ],
508
+ "proposedChanges": [
509
+ {
510
+ "kind": "reviewer_prompt",
511
+ "payload": {
512
+ "rubric": "quality",
513
+ "action": "up-weight",
514
+ "spearman": -1,
515
+ "bestOutcome": "success_rate"
516
+ },
517
+ "rationale": "predictive-validity Spearman=-1.000 vs success_rate (load-bearing); recommend up-weighting",
518
+ "expectedDelta": 0.05
519
+ }
520
+ ]
521
+ }
522
+ },
523
+ "limitations": [
524
+ "Magnitude-based bucketing is intentional in existing tests, despite contradictory interface prose.",
525
+ "A negative association can be desirable for an outcome such as failure rate; direction needs explicit interpretation.",
526
+ "The researcher recommends increasing rubric weight despite its negative association with desired success rate; it does not execute or deploy that recommendation.",
527
+ "Constructed perfect correlation establishes neither causal validity nor held-out predictive performance."
528
+ ]
529
+ },
530
+ "sequentialDependence": {
531
+ "inputs": {
532
+ "alpha": 0.05,
533
+ "minN": 5,
534
+ "maxN": 100,
535
+ "shuffleSeed": 1337,
536
+ "branchesEnumerated": 2,
537
+ "cellsPerBranch": 100,
538
+ "independentRandomSignsPerExperiment": 1,
539
+ "dataGeneratingProcess": "Draw one fair sign Z; set all 100 paired deltas equal to Z.",
540
+ "exchangeable": true,
541
+ "marginalMeanDelta": 0,
542
+ "conditionalMeanAfterFirstObservation": "Z, not necessarily <= 0"
543
+ },
544
+ "results": {
545
+ "branches": [
546
+ {
547
+ "commonDelta": -1,
548
+ "probability": 0.5,
549
+ "result": {
550
+ "decision": "hold",
551
+ "reasons": [
552
+ "sequentialPairedGate: undecided at pre-registered maxN=100 (e-value 1.00 < 1/α=20.00). This is NOT evidence of no effect — the effect may be real but smaller than this budget can detect; re-register with a larger N to test that"
553
+ ],
554
+ "contributingGates": [
555
+ {
556
+ "name": "sequentialPairedGate",
557
+ "status": "fail",
558
+ "detail": {
559
+ "wealth": 1,
560
+ "n": 100,
561
+ "decided": false,
562
+ "alpha": 0.05,
563
+ "maxBet": 0.5,
564
+ "nullMean": 0.5,
565
+ "threshold": 20,
566
+ "sumX": 0,
567
+ "varSum": 0.15877048244745845,
568
+ "decision": "undecided-at-maxN",
569
+ "minN": 5,
570
+ "maxN": 100,
571
+ "scale": 1,
572
+ "shuffleSeed": 1337,
573
+ "direction": "increase",
574
+ "minEffect": 0,
575
+ "pairedN": 100
576
+ }
577
+ }
578
+ ],
579
+ "delta": -1
580
+ }
581
+ },
582
+ {
583
+ "commonDelta": 1,
584
+ "probability": 0.5,
585
+ "result": {
586
+ "decision": "ship",
587
+ "reasons": [
588
+ "sequentialPairedGate: e-value 22.74 ≥ 1/α=20.00 at n=15 (minN=5): the paired improvement exceeds 0 at anytime-valid level α=0.05"
589
+ ],
590
+ "contributingGates": [
591
+ {
592
+ "name": "sequentialPairedGate",
593
+ "status": "pass",
594
+ "detail": {
595
+ "wealth": 22.737367544323206,
596
+ "n": 15,
597
+ "decided": true,
598
+ "alpha": 0.05,
599
+ "maxBet": 0.5,
600
+ "nullMean": 0.5,
601
+ "threshold": 20,
602
+ "decidedAtN": 15,
603
+ "sumX": 15,
604
+ "varSum": 0.14608663336124675,
605
+ "decision": "promote",
606
+ "minN": 5,
607
+ "maxN": 100,
608
+ "scale": 1,
609
+ "shuffleSeed": 1337,
610
+ "direction": "increase",
611
+ "minEffect": 0,
612
+ "pairedN": 100
613
+ }
614
+ }
615
+ ],
616
+ "delta": 1
617
+ }
618
+ }
619
+ ],
620
+ "promotionProbabilityUnderMarginalZeroProcess": 0.5
621
+ },
622
+ "limitations": [
623
+ "Enumerates both equiprobable branches exactly; this is not a Monte Carlo estimate.",
624
+ "The process violates the conditional-mean null required by the e-process core.",
625
+ "This refutes sufficiency of exchangeability and shuffling, not the valid e-process theorem or any measured production dataset."
626
+ ]
627
+ },
628
+ "powerFloor": {
629
+ "inputs": {
630
+ "gate": {
631
+ "kind": "power-floor",
632
+ "target": 0.8,
633
+ "effectGrid": [
634
+ 0.01,
635
+ 1
636
+ ],
637
+ "sim": {
638
+ "trials": 1,
639
+ "resamples": 1,
640
+ "seed": 1
641
+ }
642
+ },
643
+ "curve": [
644
+ {
645
+ "effect": 0.01,
646
+ "power": 0.1
647
+ },
648
+ {
649
+ "effect": 1,
650
+ "power": 1
651
+ }
652
+ ],
653
+ "practicalEffectForInterpretation": 0.01,
654
+ "curveSource": "Supplied deterministic fixture; no power simulation is run."
655
+ },
656
+ "results": {
657
+ "id": "floor",
658
+ "passed": true,
659
+ "evidence": {
660
+ "target": 0.8,
661
+ "maxPower": 1,
662
+ "curve": [
663
+ {
664
+ "effect": 0.01,
665
+ "power": 0.1
666
+ },
667
+ {
668
+ "effect": 1,
669
+ "power": 1
670
+ }
671
+ ]
672
+ }
673
+ },
674
+ "limitations": [
675
+ "The inspected gate documents a maximum-over-grid structural feasibility check; this output matches that contract.",
676
+ "Passing this gate does not establish target power at the practical effect of 0.01.",
677
+ "The fixture powers are inputs, not measured or simulated power estimates."
678
+ ]
679
+ }
680
+ },
681
+ "separateVerificationAtReviewedBase": {
682
+ "provenance": "Historical checks at reviewedBaseRevision; this diagnostic does not rerun them.",
683
+ "sourceRevision": "fe1cc5111aab5d588bf7db3a3785325635937a91",
684
+ "checks": [
685
+ {
686
+ "command": "pnpm typecheck",
687
+ "result": "passed"
688
+ },
689
+ {
690
+ "command": "pnpm build",
691
+ "result": "passed"
692
+ },
693
+ {
694
+ "command": "pnpm verify:package",
695
+ "result": "passed"
696
+ }
697
+ ],
698
+ "tests": {
699
+ "command": "pnpm test -- tests/experiment/preregistration-acceptance.test.ts tests/experiment/power.test.ts tests/contamination-guard.test.ts tests/rl-predictive-validity-researcher.test.ts tests/rubric-predictive-validity.test.ts tests/meta-eval.test.ts",
700
+ "observedScope": "The package command expanded to the full Vitest suite.",
701
+ "files": {
702
+ "passed": 399,
703
+ "skipped": 2
704
+ },
705
+ "tests": {
706
+ "passed": 5876,
707
+ "skipped": 3
708
+ },
709
+ "result": "passed"
710
+ },
711
+ "limitation": "Passing repository checks do not establish correctness of the counterexample behaviors recorded above."
712
+ }
713
+ }