@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,66 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## Unreleased
8
+
9
+ ## [0.181.0] — 2026-09-14
10
+
11
+ ### Changed
12
+
13
+ - README and example guides include verified execution commands, public imports, and explicit limits on fixture results and release evidence.
14
+ The existing-agent quickstart is offline; its guide shows how to meter paid calls with the maintained transport and receipt helpers.
15
+ - The single-optimizer example accepts the same worker `PRICE_*` settings as the method-comparison example.
16
+ Both use one parser to validate endpoint rates before execution.
17
+ - **Breaking:** Root `Scenario`, `JudgeScore`, and `GateDecision` now match `/contract`.
18
+ Product workflows use `ProductScenario`, `DimensionJudgeScore`, and `HeldOutGateDecision`.
19
+ - **Breaking:** Current canonical envelopes and algorithm identifiers are required for seals, attestations, and profile identities.
20
+ Retired digest readers and canonical-JSON waiver paths are removed.
21
+ Historical evidence retains its original identity.
22
+ - **Breaking:** Cluster interval registrations bind their measured `value` field.
23
+ Row execution evidence supplies rows; it cannot select another metric under the same seal.
24
+ - **Breaking:** Power calculations and power-floor gates require `minimumEffect` and assess adequacy at that effect.
25
+ - **Breaking:** Predictive validity requires a declared outcome direction and uses descriptive `aligned`, `inverse`, and `weak` associations.
26
+ Research proposals retain hypotheses instead of invented expected gains.
27
+ - **Breaking:** Adaptation comparisons require matched identified scenario cohorts and report paired uncertainty and inconclusive results.
28
+ Contamination diagnostics use `alpha`; heuristic per-item `qValue` values are removed.
29
+ - Method comparisons expose unit-level scores, raw paired-cell counts, and the deciding statistical evidence.
30
+ `favored: null` replaces the ambiguous `tie` sentinel for inconclusive comparisons.
31
+ Continuous mean decisions cannot use a small-sample sign test as evidence about the mean.
32
+ Binary and explicit median decisions retain their appropriate confidence-dependent observation requirements.
33
+
34
+ ### Added
35
+
36
+ - Optional top-level `claim` metadata declares populations and independent source units without consuming reusable regression evidence.
37
+ New-unit claims keep source families together across automatic partitions.
38
+ Self-improvement retains the selected candidate when release evidence is negative or inconclusive.
39
+ - Optional `finalEvidence` reserves fresh final units before search and records exposure before final measurement.
40
+ The shared journal detects conflicting use, concurrent ownership, corrupted history, and deleted trusted heads.
41
+ - `/meta-eval` exports evaluator admission from actual controls, simultaneous error bounds, and explicit unknown and excluded evidence.
42
+ Existing position and self-preference audits are public alongside calibration and verbosity diagnostics.
43
+ - `calibrationFromPairs()` accepts direct measured rows without requiring trace and outcome stores.
44
+ - Pareto objectives and paired promotion accept `binaryScale` for declared binary outcomes, including zero-only error observations.
45
+ Default Pareto regression tolerances use the declared scale.
46
+ - The [evaluation-integrity guide](docs/evaluation-integrity.md) explains methodology and limits.
47
+ Its offline example composes public imports and exports actual fixture results.
48
+
49
+ ### Fixed
50
+
51
+ - Failure-cluster shares count all affected failed runs independently of the five displayed examples.
52
+ Multiple findings in the same cluster count once per run.
53
+ - Outcome queries select the latest finite requested metric instead of an unrelated latest observation.
54
+ Outcome-store corruption and unavailable evidence remain visible failures.
55
+ - Calibration preserves clipped observations, measures constant predictors, and honors the requested bin count.
56
+ - Registered-unit gates pair complete cells before aggregation; repetitions and source variants cannot multiply independent evidence.
57
+ Cell reduction preserves identical judge scores exactly, including decimal binary scales.
58
+ - Opened experiments and outcome research retain validated snapshots instead of mutable caller-owned rules.
59
+ - Comparisons capture judge configuration and callbacks before asynchronous work.
60
+ Replacing a caller's judge between arms cannot create artificial lift under the original evaluator identity.
61
+ - Pareto promotion applies regression floors to the deciding confidence interval.
62
+ Tied binary outcomes cannot bypass a safety floor through a zero-width diagnostic bootstrap.
63
+ Gate explanations describe failed floors without treating uncertainty as an observed regression.
64
+ Zero-width or non-finite deciding intervals now produce an `indeterminate` axis and `not_evaluated` check.
65
+ They require more evidence before promotion, including undeclared all-zero outcomes and constant continuous differences.
66
+
7
67
  ## [0.180.0] — 2026-09-09
8
68
 
9
69
  - Preserve named-resource receipts in supervisor-run facts, JSON reports, and comparison cells.
package/README.md CHANGED
@@ -1,29 +1,28 @@
1
1
  # `@tangle-network/agent-eval`
2
2
 
3
- Measure agent behavior, compare changes on the same cases, and improve prompts or skills without showing the final test cases to the optimizer.
3
+ Run agent evaluations, compare changes on the same cases, and decide whether a candidate has enough evidence to release.
4
4
 
5
5
  [![npm](https://img.shields.io/npm/v/@tangle-network/agent-eval.svg)](https://www.npmjs.com/package/@tangle-network/agent-eval)
6
6
  [![pypi](https://img.shields.io/pypi/v/agent-eval-rpc.svg)](https://pypi.org/project/agent-eval-rpc/)
7
7
  [![tests](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml/badge.svg)](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
8
8
  [![license: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](./LICENSE)
9
9
 
10
- The evaluation path runs in your TypeScript process.
11
- Model calls happen only through the clients and agents you configure.
12
-
13
- New to the package? Read [concepts](./docs/concepts.md) first — it takes five minutes and defines every word used here.
14
-
15
- Looking for a measured result (a lift, a null, a parity verdict)? The canonical registry is [`evidence/`](./evidence/README.md) — machine-readable records, a generated index, and a freshness gate.
10
+ Eval runs in your TypeScript process.
11
+ You supply agent execution, judges, and model transports.
12
+ It records outputs, failures, costs, and evidence for each comparison.
16
13
 
17
14
  ## Install
18
15
 
16
+ Use Node.js 20.19 or newer.
17
+
19
18
  ```sh
20
19
  pnpm add @tangle-network/agent-eval
21
20
  ```
22
21
 
23
22
  ## Quickstart
24
23
 
25
- This example is offline and complete.
26
- Copy it, run it, then replace the agent and the judge with your product code.
24
+ This complete example runs offline.
25
+ Save it as `eval.mts`.
27
26
 
28
27
  ```ts
29
28
  import { defineAgentEval } from '@tangle-network/agent-eval/contract'
@@ -53,186 +52,147 @@ const evalKit = defineAgentEval<SupportCase, string>({
53
52
  expectUsage: 'off',
54
53
  })
55
54
 
56
- console.log((await evalKit.evaluate()).aggregates.byJudge)
57
- console.log(
58
- (await evalKit.evaluate({ surface: 'Answer politely and cite the ticket id.' })).aggregates
59
- .byJudge,
60
- )
61
- ```
62
-
63
- Each call runs every case, records what the agent produced, applies the same judge, and returns score distributions.
55
+ const baseline = await evalKit.evaluate()
56
+ const candidate = await evalKit.evaluate({
57
+ surface: 'Answer politely and cite the ticket id.',
58
+ })
64
59
 
65
- Three words carry this example.
66
- A **case** is one task the agent must do.
67
- A **surface** is the value being changed: a prompt, a skill, or a serialized configuration.
68
- A **judge** is a function that scores one produced result.
60
+ console.log('baseline:', baseline.aggregates.byJudge['ticket-id']?.mean)
61
+ console.log('candidate:', candidate.aggregates.byJudge['ticket-id']?.mean)
62
+ ```
69
63
 
70
- `expectUsage: 'off'` is set because this agent makes no paid calls.
71
- The default, `'assert'`, fails a run whose cells report no cost receipt.
72
- Keep the default whenever real model calls happen.
64
+ Run it with a TypeScript runner:
73
65
 
74
- Runnable copy: [`examples/evaluate-a-change`](./examples/evaluate-a-change/).
66
+ ```sh
67
+ pnpm add --save-dev tsx
68
+ pnpm exec tsx eval.mts
69
+ ```
75
70
 
76
- ## Auditable optimization history
71
+ ```text
72
+ baseline: 0
73
+ candidate: 1
74
+ ```
77
75
 
78
- Optimization methods may return a bounded `SearchHistoryReceipt` over Eval's canonical hash-chained `SearchLedger`. Existing callers keep working and see missing-history coverage. Autonomous and publication-grade runs set `searchHistoryPolicy: 'require-complete'` to refuse an incomplete planned denominator before the untouched final cases are opened.
76
+ The baseline scores `0`; the candidate scores `1` on all three cases.
77
+ These scores describe the three examples.
78
+ They do not establish a release decision or performance on new tasks.
79
79
 
80
- The receipt is a small proof envelope, not another event log. Exact candidates, attempts, failures, decisions, and missing ids remain in the ledger. See [complete optimization search history](./docs/search-history-receipts.md).
80
+ A **case** is one task.
81
+ A **surface** is the prompt, skill, or configuration being changed.
82
+ A **judge** scores the agent's result.
81
83
 
82
- ## Which Front Door
84
+ `expectUsage: 'off'` applies because this example makes no paid calls.
85
+ Set `expectUsage: 'assert'` for paid agents so missing dispatch receipts become execution failures.
86
+ The [runnable example](./examples/evaluate-a-change/) uses the same evaluation.
87
+ The [existing-agent example](./examples/foreign-agent-quickstart/) shows how to connect your agent and record model usage.
83
88
 
84
- Every row is a function you call. Each links to a runnable example.
89
+ ## Choose a workflow
85
90
 
86
- | When to call it | What you give it | What you get back |
91
+ | Intent | Start with | Result |
87
92
  |---|---|---|
88
- | [`defineAgentEval()`](./examples/evaluate-a-change/) you changed a surface and must know whether it helped | cases, an agent, a judge, a starting surface | `evaluate()` for scores, `improve()` for a search plus a release decision |
89
- | [`selfImprove()`](./examples/selfimprove-quickstart/) you want candidate generation, scoring, and a release decision in one call | cases, an agent, a judge, a starting surface | a report, a winner surface, and a `gateDecision` |
90
- | [`analyzeRuns()`](./examples/analyze-existing-runs/) the runs already happened and no agent needs to run again | `RunRecord[]` | an `InsightReport`: distributions, paired lift, judge agreement, cost, failure clusters |
91
- | `fromFeedbackTable()` ([example](./examples/customer-feedback-loop/)) / `fromOtelSpans()` ([example](./examples/customer-otel-traces/)) your data is in a table or an OTel collector, not in `RunRecord` shape | source rows or spans | `RunRecord[]` ready for `analyzeRuns()` |
92
- | [`planCampaignRun()` / `runCampaign()`](./examples/plan-before-you-spend/) you need direct control of the case grid, or you must see it before paying for it | cases, a dispatch function, judges, a run directory | a per-cell schedule, then a campaign result with cached cells |
93
- | [`loadEvalFixtureScenarios()`](./examples/eval-fixtures-quickstart/) agents should add cases as folders on disk | `evals/<name>/PROMPT.md` plus checks | `Scenario[]` for `runCampaign()` |
94
- | [`compareOptimizationMethods()`](./examples/compare-optimization-methods/) — two search methods must be compared at equal budget | methods, a starting surface, train, selection, and final cases | per-method final lift, intervals, pairwise contrasts, and cost |
95
- | [`gepaOptimizationMethod()` / `skillOptOptimizationMethod()`](./examples/compare-optimization-methods/) official GEPA or Microsoft SkillOpt should own the search | an objective, a recipe or trainer, an optimizer budget | an optimization method for the comparison above |
96
- | [`externalTextOptimizationMethod()`](./examples/adapt-a-text-optimizer/) another package owns text search and you keep the scoring | the package identity, limits, and a `run` callback | the same, with the final cases never exposed |
97
- | [`SurfaceProposer`](./examples/selfimprove-quickstart/) candidate generation belongs to your product | a `propose()` function | candidates the campaign executes, scores, and gates |
98
- | [`runProfileMatrix()`](./examples/profile-matrix/) — the same cases must run across models or profiles | axes of models and profiles, cases | one row per cell, with an explicit `unknown` model rather than an invented one |
99
- | [`ExperimentTracker`](./examples/experiment-evidence/) a candidate must beat its parent across N repetitions | reps with scores, run ids, and evidence references | a KEEP / ITERATE / NOISE / REGRESSION verdict with git provenance |
100
- | [`sealExperiment()` / `openSealedExperiment()`](./examples/sealed-experiment/) — the result must convince someone who does not trust you | arms, an admission funnel, an estimand, an interval, a decision table | a hashed rule tree, and executors that can run no other rule |
101
- | [`runEquivalenceCheck()` / `VERIFICATION_STRATEGIES`](./examples/verify-without-an-answer-key/) the work has no held-out test suite | a claim, two blind arms, an injected checker | a certification that names who vouched and how it can fail |
102
- | [`AnalystRegistry.runExact()`](./examples/custom-trace-analyst/) — a batch of runs failed and you need cited findings | recorded evidence, a declared analyst list | findings with evidence references, an execution plan, and a receipt |
103
- | [`analyzeTraces()`](./docs/trace-analysis.md#answer-one-question) — you have one question about a recorded run ("what first caused this failure?") | stored traces, the question, a DSPy RLM engine with a cost cap | an answer, findings with evidence references, and the investigation trajectory |
104
- | [`runAnalystBenchmark()`](./docs/trace-analysis.md) an analyst's accuracy must be measured, not assumed | labeled issues and exact span locations | scored findings, trace reads, model calls, tokens, cost, and runtime |
105
- | [`deltaRepair()`](./docs/trace-repair-grader.md) — a finding must be graded by executing the repair it proposes | a trajectory, an analyst finding, a sandbox | the repair's measured effect against a no-fix control |
106
- | [`replayVerify()`](./docs/trajectory-replay.md) you must know whether a recorded failure still reproduces | a recorded shell trajectory and its pinned image | a re-execution verdict and the divergences found |
107
- | [`analyzeSupervisorRun()`](./docs/adapters-observability.md) a recursive or supervised run directory must be read | a run directory | counts that stay missing when a measurement is missing, never zero |
108
- | [`plantByPerturbation()` / `seedPlants()` / `catchRate()`](./docs/plants.md) you must know whether the grader catches a wrong answer, not only how the work scored | a grading set, and claims the grader verified | items authored wrong by one value, a sealed manifest, then a catch rate that refuses rather than guessing |
109
- | [`buildRlDataset()`](./examples/publish-rl-dataset/) scored runs should become training data | run records and preferences | reward, preference, and supervised rows |
110
-
111
- ## Configure Model Calls
112
-
113
- Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters all take the same `ChatClient`.
114
- You own model execution, and Agent Eval never goes looking for a credential: it reads no environment variable to find one, and every transport is bound at the call site.
115
-
116
- Bind a transport one of two ways. Pass a `chat` function you wrote:
93
+ | Score one change | [`defineAgentEval()`](./examples/evaluate-a-change/) from `/contract` | Cell results, failures, score distributions, and measured cost. |
94
+ | Search for a better surface | [`selfImprove()`](./examples/selfimprove-quickstart/) from `/contract` | A selected surface, final comparison, and `gateDecision`. |
95
+ | Compare search methods | [`compareOptimizationMethods()`](./examples/compare-optimization-methods/) from `/campaign` | Paired final comparisons, uncertainty, coverage, and costs under declared budgets. |
96
+ | Register evidence and decision rules | [`defineEvaluationClaim()` and `sealExperiment()`](./docs/evaluation-integrity.md) from `/experiment` | A declared population, independent unit, optional practical effect, and sealed rules. |
97
+ | Check the evaluator | [`auditEvaluator()`](./docs/evaluation-integrity.md) and [calibration tools](./docs/outcome-validity.md) from `/meta-eval` | Error rates, admission evidence, bias diagnostics, and outcome associations. |
98
+ | Analyze completed work | [`analyzeRuns()`](./examples/analyze-existing-runs/) from `/contract`; [trace analysts](./docs/trace-analysis.md) from `/analyst` | Comparisons and findings with links to recorded evidence. |
99
+
100
+ `defineAgentEval()` also exposes `improve()` when the same agent, cases, judge, and baseline should share configuration.
101
+ Use direct [campaign controls](./docs/eval-surface-map.md) for scheduling, durable caches, model matrices, or custom release rules.
102
+ The [example index](./examples/README.md) covers fixtures, trace intake, code verification, replay, and training-data exports.
103
+
104
+ ## Make automated improvement accountable
105
+
106
+ Use reusable evaluations for development feedback.
107
+ For a direct edit, compare the baseline and candidate on the same cases.
108
+ Claims, evaluator audits, and final-evidence tracking are optional.
109
+ Add stronger controls when a result must support performance on new tasks or an adaptive release decision.
110
+
111
+ 1. Pass a `claim` describing the population, sampling frame, and independent unit to the comparison.
112
+ Declare `minimumEffect` when the decision concerns a useful improvement.
113
+ 2. When introducing an evaluator, check known good and known bad controls with `auditEvaluator()`.
114
+ 3. Give search separate training and selection cases.
115
+ 4. For fresh confirmation, supply `finalEvidence` with a shared ledger, request ID, and evaluator digest.
116
+ This reserves final units before search and records exposure before measurement.
117
+ 5. Inspect the final comparison, gate contributions, exclusions, uncertainty, cost, and search history before releasing.
118
+
119
+ Repeated attempts on one task do not create new independent tasks.
120
+ The top-level `claim` controls unit aggregation for reusable comparisons.
121
+ Power checks assess the declared minimum effect.
122
+ Optional `finalEvidence` binds fresh confirmation to that claim and refuses reused final units across campaigns sharing the ledger.
123
+
124
+ The host must enforce access isolation and author/auditor separation.
125
+ A digest records identity; it cannot prove secrecy or that a benchmark represents future users.
126
+ Custom gates remain responsible for their decision rules.
127
+ See [evaluation integrity](./docs/evaluation-integrity.md) for the complete API and its boundaries.
128
+
129
+ These controls check the evidence behind a result.
130
+ They do not establish that an optimizer beats a direct edit or simple search.
131
+ The [historical evidence audit](./docs/design/self-improvement-evidence-audit.md) records prior gains, failed transfer, and missing comparisons.
132
+
133
+ Set `searchHistoryPolicy: 'require-complete'` when every attempted search slot must be accounted for before final evidence is exposed.
134
+ The [search-history receipt](./docs/search-history-receipts.md) binds the planned denominator to Eval's existing search ledger.
135
+
136
+ A `gateDecision` is `ship`, `hold`, `need_more_work`, `model_ceiling`, or `arch_ceiling`.
137
+ Gate contributions distinguish missing evidence from measured failures and successful checks.
138
+ [Concepts](./docs/concepts.md) explains these decisions and how gates compose.
139
+
140
+ ## Configure model calls
141
+
142
+ Pass a `ChatClient` to model judges, analysts, and adapters.
143
+ Eval obtains credentials from the values you supply; it does not search your environment.
117
144
 
118
145
  ```ts
119
- import { createChatClient } from '@tangle-network/agent-eval'
146
+ import { createChatClient } from '@tangle-network/agent-eval/contract'
120
147
 
121
148
  const chat = createChatClient({
122
- transport: 'custom',
123
- defaultModel: 'openai/gpt-4.1',
124
- maximumAttempts: 3,
125
- chat: async (request, opts) => myProviderClient(request, opts),
149
+ transport: 'openai-compatible',
150
+ baseUrl: 'https://router.example/v1',
151
+ apiKey: process.env.MY_ROUTER_KEY,
152
+ defaultModel: process.env.EVAL_MODEL_ID,
126
153
  })
127
154
  ```
128
155
 
129
- Or name an OpenAI-compatible endpoint and hand over a credential as values, and Agent Eval drives `POST {baseUrl}/chat/completions` for you:
156
+ Use your deployed model identifier and preserve the returned `servedModel` identity and cost receipt.
157
+ For an existing SDK, use `transport: 'custom'` with your `chat` callback and an explicit `maximumAttempts`.
158
+ Agent Runtime callers can bind `profileChatClient()` from `@tangle-network/agent-runtime/kernel`.
159
+ Eval has no dependency on Runtime.
130
160
 
131
- ```ts
132
- const chat = createChatClient({
133
- transport: 'openai-compatible',
134
- baseUrl: 'https://router.example/v1', // ends at /v1; the path is ours to append
135
- apiKey: process.env.MY_ROUTER_KEY, // or `bearer`, or `authHeader`
136
- defaultModel: 'claude-sonnet-4-6',
137
- })
138
- ```
161
+ Official GEPA, SkillOpt, and DSPy integrations use a Python bridge.
162
+ Their maintained installation instructions and execution contracts are in [campaign proposers](./docs/campaign-proposers.md).
163
+ The [Python client](./clients/python/README.md) and [wire protocol](./docs/wire-protocol.md) support other-language consumers.
164
+
165
+ ## Public imports and evidence
166
+
167
+ Use `/contract` for a product integration, `/campaign` for execution controls, `/experiment` for registered decisions, and `/meta-eval` for evaluator checks.
168
+ Root `Scenario`, `JudgeScore`, and `GateDecision` are the same types as `/contract`.
169
+ Product judging retains the explicit root names `ProductScenario` and `DimensionJudgeScore` beside its functions.
170
+ `HeldOutGate.evaluate()` returns `HeldOutGateDecision`.
171
+
172
+ Specialist subpaths and their examples are listed in the [surface map](./docs/eval-surface-map.md).
173
+ Current canonical envelopes are required for seals, attestations, and profile identities.
174
+ Retired or incomplete formats fail verification; historical reports retain their recorded identities.
139
175
 
140
- Prefer the second over hand-rolling a fetch loop. It carries the retry, degrade, and — load-bearing — the `servedModel` echo that `assertServedModel` and `assertCrossFamilyServed` read; a transport that omits that field makes both checks report `unreported`, so the cross-vendor rules they enforce measure nothing. `baseUrl` and one credential form are required arguments with no default and no fallback: a half-configured client is refused at construction rather than reaching an endpoint you did not name.
141
-
142
- On Agent Runtime, `profileChatClient({ profile, executor, context })` from `@tangle-network/agent-runtime/kernel` is that transport: every call runs one exact `AgentProfile` and reports its measured usage, retries, and served model identity.
143
- Use `sandbox-sdk` for Sandbox and `mock` in tests.
144
- A custom adapter must return a `ChatResponse` and declare `maximumAttempts` before a capped cost account can dispatch it.
145
-
146
- `ChatResponse` carries the whole execution record across that boundary: the served model id, measured input/output/reasoning/cached tokens, billed USD or an explicit unknown, the finish reason, and the per-token log probabilities the expectation judge scores on.
147
-
148
- The official GEPA and SkillOpt optimizers run through a Python bridge.
149
- Install commands, version pins, and the reason for each pin:
150
- [GEPA](./docs/campaign-proposers.md#install-official-gepa),
151
- [SkillOpt](./docs/campaign-proposers.md#install-official-skillopt),
152
- and [DSPy](./docs/campaign-proposers.md#use-official-dspy-optimizers).
153
-
154
- ## Entry Points
155
-
156
- | Import | Use |
157
- |---|---|
158
- | `@tangle-network/agent-eval/contract` | Define an evaluation, run it, improve it, and analyze existing runs. |
159
- | `@tangle-network/agent-eval/campaign` | Campaigns, optimization methods, comparisons, storage, and release rules. |
160
- | `@tangle-network/agent-eval/experiment` | Experiments as sealed objects: registered rules, funnels, estimands, refusals. |
161
- | `@tangle-network/agent-eval/analyst` | Built-in and custom trace analysts, labeled comparison, costs, and reports. |
162
- | `@tangle-network/agent-eval/trace-repair` | Grade one analyst finding by executing the repair it proposes. |
163
- | `@tangle-network/agent-eval/trajectory-replay` | Re-execute a recorded shell trajectory and check whether its failure reproduces. |
164
- | `@tangle-network/agent-eval/traces` | Store, replay, and inspect structured traces. |
165
- | `@tangle-network/agent-eval/reporting` | Statistical comparisons and report rendering. |
166
- | `@tangle-network/agent-eval/supervisor-run` | Read recursive run directories without collapsing missing measurements to zero; `agent-eval supervisor-run report <runDir>` prints one. |
167
- | `@tangle-network/agent-eval/meta-eval` | Measure the grader itself: judge calibration, sentinels, and seeded known-wrong plants. |
168
- | `@tangle-network/agent-eval/profile-cell` | Create and validate portable agent-profile identities. |
169
- | `@tangle-network/agent-eval/ledger-core` | Append-only hash-chained journal with idempotent append and chain verification. |
170
- | `@tangle-network/agent-eval/benchmarks` | Benchmark adapters and retrieval metrics. |
171
- | `@tangle-network/agent-eval/rl` | Export rewards, preferences, and training rows. |
172
- | `@tangle-network/agent-eval/wire` | HTTP and RPC schemas for other languages. |
173
- | `@tangle-network/agent-eval/adapters/http` | Run campaign cells on remote workers over HTTP. |
174
-
175
- Use the root import for common primitives.
176
- Use a subpath when you want an explicit capability boundary.
177
-
178
- ## Documentation
179
-
180
- | Question | Read |
181
- |---|---|
182
- | What do these words mean? | [`docs/concepts.md`](./docs/concepts.md) |
183
- | Why does this package exist, and where is it going? | [`docs/charter.md`](./docs/charter.md) |
184
- | Which `run*` function do I want? | [`docs/eval-surface-map.md`](./docs/eval-surface-map.md) |
185
- | How do I choose a candidate-generation method? | [`docs/campaign-proposers.md`](./docs/campaign-proposers.md) |
186
- | What is in an `InsightReport`? | [`docs/insight-report.md`](./docs/insight-report.md) |
187
- | How do I register an experiment as a sealed object? | [`docs/experiment.md`](./docs/experiment.md) |
188
- | How is something certified without an answer key? | [`docs/verification-strategies.md`](./docs/verification-strategies.md) |
189
- | Where does every verifier land its result? | [`docs/verdicts.md`](./docs/verdicts.md) |
190
- | Does the grader catch a claim that is known to be wrong? | [`docs/plants.md`](./docs/plants.md) |
191
- | How do I turn a coding-agent session log into runs? | [`docs/code-agent-intake.md`](./docs/code-agent-intake.md) |
192
- | How do I score a string from another language? | [`docs/wire-protocol.md`](./docs/wire-protocol.md) |
193
-
194
- The [example index](./examples/README.md) lists every runnable example.
176
+ Published measurements live in the [evidence registry](./evidence/README.md).
177
+ The [benchmark-book review](./docs/design/mlbenchmarks-book-review.md) records the source analysis and reproduced defects behind these integrity changes.
178
+ [The charter](./docs/charter.md) defines package ownership and the remaining research boundaries.
195
179
 
196
180
  ## Development
197
181
 
198
182
  ```sh
199
183
  pnpm install
184
+ pnpm build
200
185
  pnpm typecheck
201
186
  pnpm typecheck:examples
187
+ pnpm typecheck:scripts
188
+ pnpm lint
202
189
  pnpm test
203
- pnpm build
190
+ pnpm verify:package
204
191
  ```
205
192
 
206
- Python compatibility tests use the locked dependencies:
207
-
208
- ```sh
209
- cd clients/python
210
- uv sync --frozen --extra dev --group gepa-release
211
- AGENT_EVAL_EXPECT_GEPA_RELEASE=1 \
212
- uv run --frozen --extra dev --group gepa-release \
213
- pytest tests/test_gepa_release_compatibility.py tests/test_gepa_bridge.py
214
-
215
- uv sync --frozen --extra dev --group skillopt-source --group gepa-source
216
- uv run --frozen pytest
217
-
218
- uv sync --frozen --extra dev --extra dspy
219
- uv run --frozen pytest tests/test_dspy_metric.py
220
- ```
193
+ Build before checking examples because they resolve the package's generated declarations.
194
+ The [Python development guide](./clients/python/README.md#development) gives the locked commands for each optimizer environment.
221
195
 
222
196
  ## License
223
197
 
224
198
  MIT.
225
-
226
-
227
- ## Supervisor-run resource receipts
228
-
229
- The `/supervisor-run` reader preserves named-resource measurements in `economics.resourceRecords`.
230
- Each record identifies its node and source within the normalized journal or terminal result.
231
- Journal row indices refer to parsed rows after reader normalization, not original file line numbers.
232
- The Markdown report renders each resource name, unit, amount, and completeness flag.
233
- Comparison cells retain those same records without combining them.
234
-
235
- A false `known` flag means the amount is a recorded subtotal, not complete usage.
236
- Missing maps, explicit empty maps, and invalid fields remain distinct from measured zero.
237
- Parent settlements and terminal results can include child usage, so these records are not additive totals.
238
- The reporter reads evidence; it does not enforce budgets or infer missing measurements.
@@ -1,5 +1,5 @@
1
- import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-C34V4Vto.js";
2
- import "../index-DxNYmx4a.js";
1
+ import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-CS0qk_Yp.js";
2
+ import "../index-DoykkxW0.js";
3
3
  //#region src/adapters/http.d.ts
4
4
  interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
5
5
  /** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
@@ -1,7 +1,7 @@
1
1
  import { b as CustomTokenPricing, g as CostReceiptInput } from "./cost-ledger-DbQdN3nO.js";
2
- import { D as TraceAnalysisStore, c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef } from "./types-DzuaM493.js";
3
- import { h as ChatResponse, m as ChatRequest } from "./types-gvRsyJLh.js";
4
- import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-ByVld1-5.js";
2
+ import { D as TraceAnalysisStore, c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef } from "./types-D7gEdPoQ.js";
3
+ import { C as ChatResponse, S as ChatRequest } from "./types-CBbLtr2J.js";
4
+ import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-BRbB6Y0v.js";
5
5
  import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
6
6
  //#region src/campaign/external-optimizer-contracts.d.ts
7
7
  interface ExternalOptimizerProcessLimits {
@@ -485,4 +485,4 @@ declare function agentProfileId(profile: AgentProfile): string;
485
485
  declare function agentProfileHash(profile: AgentProfile): string;
486
486
  //#endregion
487
487
  export { runAnalystBenchmark as A, ExternalOptimizerModelBudget as B, AnalystEvidenceResolutionError as C, AnalystLatencyDistribution as D, AnalystIssueExpectation as E, ExternalOptimizerCallbackLimits as F, ExternalOptimizerProcessLimits as G, ExternalOptimizerModelCallRequest as H, ExternalOptimizerChatRequest as I, ExternalOptimizerWireCounts as J, ExternalOptimizerResumeMode as K, ExternalOptimizerEndpointFormat as L, scoreAnalystFindings as M, DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS as N, RunAnalystBenchmarkOptions as O, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS as P, resolveExternalOptimizerProcessLimits as Q, ExternalOptimizerEvaluationObservation as R, AnalystEvidenceResolution as S, AnalystFindingScore as T, ExternalOptimizerModelCallResult as U, ExternalOptimizerModelCall as V, ExternalOptimizerModelExecutionObservation as W, ExternalTextEvaluationRequest as X, ExternalTextCandidate as Y, resolveExternalOptimizerCallbackLimits as Z, AnalystBenchmarkProvenance as _, ProfileAxisSpec as a, AnalystBenchmarkSummary as b, expandProfileAxes as c, AnalystBenchmarkDatasetRef as d, AnalystBenchmarkDescriptor as f, AnalystBenchmarkOutput as g, AnalystBenchmarkObservation as h, HarnessType$1 as i, traceStoreEvidenceResolver as j, registryBenchmarkRunner as k, harnessAxisOf as l, AnalystBenchmarkLabelState as m, CODING_HARNESSES as n, agentProfileHash as o, AnalystBenchmarkError as p, ExternalOptimizerRunnerCommand as q, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t, AnalystBenchmarkCase as u, AnalystBenchmarkResult as v, AnalystEvidenceResolver as w, AnalystEvidenceExpectation as x, AnalystBenchmarkRunner as y, ExternalOptimizerEvaluationRefusalReason as z };
488
- //# sourceMappingURL=agent-profile-B7yErX0q.d.ts.map
488
+ //# sourceMappingURL=agent-profile-CivaSsSy.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"agent-profile-B7yErX0q.d.ts","names":[],"sources":["../src/campaign/external-optimizer-contracts.ts","../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts","../src/agent-profile.ts"],"mappings":";;;;;;UAWiB;;EAEf;;EAEA;;EAEA;;cAGW,2CAA2C,SAAS;UAOhD;;EAEf;;EAEA;;cAGW,4CAA4C,SAAS;UAMjD;EACf;EACA;EACA,MAAM,OAAO;;EAEb,SAAS,QAAQ;;KAGP;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;iBAUc,sCACd,OAAO,QAAQ,6CACf,iBACC;iBAmBa,uCACd,OAAO,QAAQ,8CACf,iBACC;KAkBE,aAAa,KAAK,eAAc,6BACjC,IACA,0BAA0B,gBACf,aAAa,OACtB,+BACc,WAAW,IAAI,aAAa,EAAE,SAC1C;;KAGI,+BAA+B,aACzC,KAAK;EAA0B;;KAGrB;;UAMK;;WAEN;;WAEA,SAAS;;WAET,iBAAiB;WACjB,QAAQ;;;KAIP;WAEG;;WAEA,UAAU;;;;;;;WAOV,SAPU;;WASV;;WAGA;;WAEA;;WAEA,SATkC;;WAWlC;;;;;;;;;;;;KAaH,8BACV,SAAS,sCACN,QAAQ;;KAGD;WAEG;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;WAGA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;KAGH;;;;;;KAUA;WAEG;WACA;WACA,WAAW;WACX;;WAGA;WACA;WACA,WAAW;WACX;WACA;;WAEA;WACA;;WAGA;WACA;WACA,QAAQ;WACR,YAAY;WACZ;WACA;;UAGE;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;EAUA;;;;;;EAMA,UAAU;;EAEV;;;UAIe;;EAEf;;EAEA;;;;iBCtQc,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB;;;;;;;;;;cCpUd,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
1
+ {"version":3,"file":"agent-profile-CivaSsSy.d.ts","names":[],"sources":["../src/campaign/external-optimizer-contracts.ts","../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts","../src/agent-profile.ts"],"mappings":";;;;;;UAWiB;;EAEf;;EAEA;;EAEA;;cAGW,2CAA2C,SAAS;UAOhD;;EAEf;;EAEA;;cAGW,4CAA4C,SAAS;UAMjD;EACf;EACA;EACA,MAAM,OAAO;;EAEb,SAAS,QAAQ;;KAGP;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;iBAUc,sCACd,OAAO,QAAQ,6CACf,iBACC;iBAmBa,uCACd,OAAO,QAAQ,8CACf,iBACC;KAkBE,aAAa,KAAK,eAAc,6BACjC,IACA,0BAA0B,gBACf,aAAa,OACtB,+BACc,WAAW,IAAI,aAAa,EAAE,SAC1C;;KAGI,+BAA+B,aACzC,KAAK;EAA0B;;KAGrB;;UAMK;;WAEN;;WAEA,SAAS;;WAET,iBAAiB;WACjB,QAAQ;;;KAIP;WAEG;;WAEA,UAAU;;;;;;;WAOV,SAPU;;WASV;;WAGA;;WAEA;;WAEA,SATkC;;WAWlC;;;;;;;;;;;;KAaH,8BACV,SAAS,sCACN,QAAQ;;KAGD;WAEG;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;WAGA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;KAGH;;;;;;KAUA;WAEG;WACA;WACA,WAAW;WACX;;WAGA;WACA;WACA,WAAW;WACX;WACA;;WAEA;WACA;;WAGA;WACA;WACA,QAAQ;WACR,YAAY;WACZ;WACA;;UAGE;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;EAUA;;;;;;EAMA,UAAU;;EAEV;;;UAIe;;EAEf;;EAEA;;;;iBCtQc,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB;;;;;;;;;;cCpUd,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
@@ -1,6 +1,5 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { a as hashCanonical, r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
- import { createHash } from "node:crypto";
2
+ import { a as hashCanonical } from "./canonical-DPyQ_rpt.js";
4
3
  //#region src/pre-registration.ts
5
4
  /**
6
5
  * Pre-registered hypotheses — declare what you're testing BEFORE the
@@ -12,10 +11,8 @@ import { createHash } from "node:crypto";
12
11
  * evaluate the manifest against observed results — the library refuses
13
12
  * to let you re-interpret a different metric as the declared one.
14
13
  *
15
- * A signed manifest is a portable record: it is written once and verified
16
- * later, possibly by a different release. `algo` names the digest scheme it
17
- * was signed under, and verification selects the encoder by that field, so a
18
- * manifest signed by an earlier release still verifies.
14
+ * A signed manifest carries its required digest scheme. Verification accepts
15
+ * only RFC 8785 canonical JSON, using the same encoder as every new identity.
19
16
  */
20
17
  /**
21
18
  * SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of
@@ -36,31 +33,13 @@ async function hashJson(obj) {
36
33
  return hashCanonical(obj).slice(7);
37
34
  }
38
35
  /**
39
- * Key-sorted `JSON.stringify` digest. Private and read-only: it exists so a
40
- * manifest signed under `'sha256-content'` still verifies, and nothing that
41
- * WRITES a digest may call it.
42
- */
43
- function legacyContentDigest(value) {
44
- return createHash("sha256").update(JSON.stringify(sortKeysDeep$1(value)), "utf8").digest("hex");
45
- }
46
- function sortKeysDeep$1(value) {
47
- if (value === null || typeof value !== "object") return value;
48
- if (Array.isArray(value)) return value.map(sortKeysDeep$1);
49
- const out = {};
50
- for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep$1(value[key]);
51
- return out;
52
- }
53
- /**
54
- * Digest of a manifest under its own declared scheme, with `contentHash` and
55
- * `algo` stripped. Synchronous, so a caller that must fail before consuming an
56
- * observation does not have to await. Throws on an `algo` this release does
57
- * not know — an unverifiable manifest must not read as a valid one.
36
+ * Digest a manifest after validating its scheme, excluding `contentHash` and
37
+ * `algo`. This synchronous check can refuse a manifest before consuming data.
58
38
  */
59
39
  function manifestContentDigest(manifest) {
60
40
  const { contentHash: _contentHash, algo, ...rest } = manifest;
61
- if (algo === void 0 || algo === "sha256-content") return legacyContentDigest(rest);
62
- if (algo === "sha256-rfc8785") return createHash("sha256").update(canonicalString(rest), "utf8").digest("hex");
63
- throw new Error(`pre-registration: unrecognized manifest hash algo '${String(algo)}'`);
41
+ if (algo !== "sha256-rfc8785") throw new Error(`pre-registration: unsupported manifest hash algo '${String(algo)}'`);
42
+ return hashCanonical(rest).slice(7);
64
43
  }
65
44
  /**
66
45
  * Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical
@@ -83,14 +62,18 @@ async function signManifest(m) {
83
62
  * the manifest itself declares.
84
63
  */
85
64
  async function verifyManifest(m) {
86
- return manifestContentDigest(m) === m.contentHash;
65
+ try {
66
+ return manifestContentDigest(m) === m.contentHash;
67
+ } catch {
68
+ return false;
69
+ }
87
70
  }
88
71
  /**
89
72
  * Evaluate a pre-registered hypothesis against observed results.
90
73
  * Mechanical — no re-interpretation permitted.
91
74
  */
92
75
  async function evaluateHypothesis(manifest, observed) {
93
- if (!await verifyManifest(manifest)) throw new Error("evaluateHypothesis: manifest content hash mismatch (tampered)");
76
+ if (!await verifyManifest(manifest)) throw new Error("evaluateHypothesis: unsupported manifest hash scheme or content hash mismatch");
94
77
  const reasons = [];
95
78
  if (!(manifest.direction === "increase" ? observed.effect > 0 : observed.effect < 0)) reasons.push("wrong_direction");
96
79
  if (Math.abs(observed.effect) < manifest.minEffect) reasons.push("effect_too_small");
@@ -115,14 +98,9 @@ var AgentProfileCellValidationError = class extends ValidationError {
115
98
  }
116
99
  };
117
100
  const SHA256_HEX = /^[0-9a-f]{64}$/;
118
- /**
119
- * A cell id names the digest scheme that produced it. `sha256-rfc8785` is what
120
- * {@link buildAgentProfileCell} mints; the bare `sha256` form is read-only,
121
- * carried by cells built under an earlier release, and still verifies.
122
- */
123
- const CELL_ID = /^agent-profile-cell:sha256(?:-rfc8785)?:[0-9a-f]{64}$/;
101
+ /** A cell id carries the canonical digest scheme required for verification. */
102
+ const CELL_ID = /^agent-profile-cell:sha256-rfc8785:[0-9a-f]{64}$/;
124
103
  const CELL_ID_PREFIX = "agent-profile-cell:sha256-rfc8785:";
125
- const LEGACY_CELL_ID_PREFIX = "agent-profile-cell:sha256:";
126
104
  async function buildAgentProfileCell(input) {
127
105
  const material = await normalizeAgentProfileCellInput(input);
128
106
  const cellId = `${CELL_ID_PREFIX}${await hashJson(material)}`;
@@ -137,34 +115,19 @@ function agentProfileCellHashMaterial(cell) {
137
115
  }
138
116
  /**
139
117
  * Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material
140
- * fields, confirming the record has not been tampered with. The id names its own
141
- * digest scheme, so a cell minted by an earlier release verifies under that scheme.
118
+ * fields, confirming the record has not been tampered with. Unsupported digest
119
+ * schemes are refused before comparing the material.
142
120
  */
143
121
  async function verifyAgentProfileCell(cell) {
144
122
  validateAgentProfileCell(cell);
145
123
  const material = agentProfileCellHashMaterial(cell);
146
- if (cell.cellId.startsWith(CELL_ID_PREFIX)) return cell.cellId === `${CELL_ID_PREFIX}${await hashJson(material)}`;
147
- return cell.cellId === `${LEGACY_CELL_ID_PREFIX}${legacyCellDigest(material)}`;
148
- }
149
- /**
150
- * Key-sorted `JSON.stringify` digest. Private and read-only: it verifies a cell
151
- * id minted before the RFC 8785 scheme, and no path that MINTS an id calls it.
152
- */
153
- function legacyCellDigest(value) {
154
- return createHash("sha256").update(JSON.stringify(sortKeysDeep(value)), "utf8").digest("hex");
155
- }
156
- function sortKeysDeep(value) {
157
- if (value === null || typeof value !== "object") return value;
158
- if (Array.isArray(value)) return value.map(sortKeysDeep);
159
- const out = {};
160
- for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep(value[key]);
161
- return out;
124
+ return cell.cellId === `${CELL_ID_PREFIX}${await hashJson(material)}`;
162
125
  }
163
126
  function validateAgentProfileCell(input) {
164
127
  if (input === null || typeof input !== "object") throw new AgentProfileCellValidationError("expected object");
165
128
  const obj = input;
166
129
  expectLiteral(obj.schemaVersion, "agent-profile-cell/v1", "schemaVersion");
167
- if (typeof obj.cellId !== "string" || !CELL_ID.test(obj.cellId)) throw new AgentProfileCellValidationError("cellId must match agent-profile-cell:sha256:<64 lowercase hex chars>", "cellId");
130
+ if (typeof obj.cellId !== "string" || !CELL_ID.test(obj.cellId)) throw new AgentProfileCellValidationError("cellId must match agent-profile-cell:sha256-rfc8785:<64 lowercase hex chars>", "cellId");
168
131
  expectString(obj.profileId, "profileId");
169
132
  validateSource(obj.sourceProfile, "sourceProfile");
170
133
  if (obj.harness !== void 0) validateHarness(obj.harness, "harness");
@@ -371,4 +334,4 @@ async function buildAgentInterfaceProfileCell(profile, input) {
371
334
  //#endregion
372
335
  export { verifyManifest as _, assertRunAgentProfileCell as a, groupRunsByAgentProfileCell as c, validateAgentProfileCell as d, verifyAgentProfileCell as f, signManifest as g, manifestContentDigest as h, agentProfileCellKey as i, requireAgentProfileCell as l, hashJson as m, AgentProfileCellValidationError as n, buildAgentInterfaceProfileCell as o, evaluateHypothesis as p, agentProfileCellHashMaterial as r, buildAgentProfileCell as s, AGENT_PROFILE_KINDS as t, toAgentProfileJson as u };
373
336
 
374
- //# sourceMappingURL=agent-profile-cell-0gSi5ffD.js.map
337
+ //# sourceMappingURL=agent-profile-cell-Cv6UA-W_.js.map