@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,72 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## Unreleased
|
|
8
|
+
|
|
9
|
+
## [0.181.0] — 2026-09-14
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- README and example guides include verified execution commands, public imports, and explicit limits on fixture results and release evidence.
|
|
14
|
+
The existing-agent quickstart is offline; its guide shows how to meter paid calls with the maintained transport and receipt helpers.
|
|
15
|
+
- The single-optimizer example accepts the same worker `PRICE_*` settings as the method-comparison example.
|
|
16
|
+
Both use one parser to validate endpoint rates before execution.
|
|
17
|
+
- **Breaking:** Root `Scenario`, `JudgeScore`, and `GateDecision` now match `/contract`.
|
|
18
|
+
Product workflows use `ProductScenario`, `DimensionJudgeScore`, and `HeldOutGateDecision`.
|
|
19
|
+
- **Breaking:** Current canonical envelopes and algorithm identifiers are required for seals, attestations, and profile identities.
|
|
20
|
+
Retired digest readers and canonical-JSON waiver paths are removed.
|
|
21
|
+
Historical evidence retains its original identity.
|
|
22
|
+
- **Breaking:** Cluster interval registrations bind their measured `value` field.
|
|
23
|
+
Row execution evidence supplies rows; it cannot select another metric under the same seal.
|
|
24
|
+
- **Breaking:** Power calculations and power-floor gates require `minimumEffect` and assess adequacy at that effect.
|
|
25
|
+
- **Breaking:** Predictive validity requires a declared outcome direction and uses descriptive `aligned`, `inverse`, and `weak` associations.
|
|
26
|
+
Research proposals retain hypotheses instead of invented expected gains.
|
|
27
|
+
- **Breaking:** Adaptation comparisons require matched identified scenario cohorts and report paired uncertainty and inconclusive results.
|
|
28
|
+
Contamination diagnostics use `alpha`; heuristic per-item `qValue` values are removed.
|
|
29
|
+
- Method comparisons expose unit-level scores, raw paired-cell counts, and the deciding statistical evidence.
|
|
30
|
+
`favored: null` replaces the ambiguous `tie` sentinel for inconclusive comparisons.
|
|
31
|
+
Continuous mean decisions cannot use a small-sample sign test as evidence about the mean.
|
|
32
|
+
Binary and explicit median decisions retain their appropriate confidence-dependent observation requirements.
|
|
33
|
+
|
|
34
|
+
### Added
|
|
35
|
+
|
|
36
|
+
- Optional top-level `claim` metadata declares populations and independent source units without consuming reusable regression evidence.
|
|
37
|
+
New-unit claims keep source families together across automatic partitions.
|
|
38
|
+
Self-improvement retains the selected candidate when release evidence is negative or inconclusive.
|
|
39
|
+
- Optional `finalEvidence` reserves fresh final units before search and records exposure before final measurement.
|
|
40
|
+
The shared journal detects conflicting use, concurrent ownership, corrupted history, and deleted trusted heads.
|
|
41
|
+
- `/meta-eval` exports evaluator admission from actual controls, simultaneous error bounds, and explicit unknown and excluded evidence.
|
|
42
|
+
Existing position and self-preference audits are public alongside calibration and verbosity diagnostics.
|
|
43
|
+
- `calibrationFromPairs()` accepts direct measured rows without requiring trace and outcome stores.
|
|
44
|
+
- Pareto objectives and paired promotion accept `binaryScale` for declared binary outcomes, including zero-only error observations.
|
|
45
|
+
Default Pareto regression tolerances use the declared scale.
|
|
46
|
+
- The [evaluation-integrity guide](docs/evaluation-integrity.md) explains methodology and limits.
|
|
47
|
+
Its offline example composes public imports and exports actual fixture results.
|
|
48
|
+
|
|
49
|
+
### Fixed
|
|
50
|
+
|
|
51
|
+
- Failure-cluster shares count all affected failed runs independently of the five displayed examples.
|
|
52
|
+
Multiple findings in the same cluster count once per run.
|
|
53
|
+
- Outcome queries select the latest finite requested metric instead of an unrelated latest observation.
|
|
54
|
+
Outcome-store corruption and unavailable evidence remain visible failures.
|
|
55
|
+
- Calibration preserves clipped observations, measures constant predictors, and honors the requested bin count.
|
|
56
|
+
- Registered-unit gates pair complete cells before aggregation; repetitions and source variants cannot multiply independent evidence.
|
|
57
|
+
Cell reduction preserves identical judge scores exactly, including decimal binary scales.
|
|
58
|
+
- Opened experiments and outcome research retain validated snapshots instead of mutable caller-owned rules.
|
|
59
|
+
- Comparisons capture judge configuration and callbacks before asynchronous work.
|
|
60
|
+
Replacing a caller's judge between arms cannot create artificial lift under the original evaluator identity.
|
|
61
|
+
- Pareto promotion applies regression floors to the deciding confidence interval.
|
|
62
|
+
Tied binary outcomes cannot bypass a safety floor through a zero-width diagnostic bootstrap.
|
|
63
|
+
Gate explanations describe failed floors without treating uncertainty as an observed regression.
|
|
64
|
+
Zero-width or non-finite deciding intervals now produce an `indeterminate` axis and `not_evaluated` check.
|
|
65
|
+
They require more evidence before promotion, including undeclared all-zero outcomes and constant continuous differences.
|
|
66
|
+
|
|
67
|
+
## [0.180.0] — 2026-09-09
|
|
68
|
+
|
|
69
|
+
- Preserve named-resource receipts in supervisor-run facts, JSON reports, and comparison cells.
|
|
70
|
+
- Render each receipt's source, node, name, unit, amount, and completeness without adding inclusive parent and child totals.
|
|
71
|
+
- Distinguish missing maps, recorded empty maps, malformed measurements, and measured zero.
|
|
72
|
+
|
|
7
73
|
## [0.179.0] — 2026-09-08
|
|
8
74
|
|
|
9
75
|
### Added
|
package/README.md
CHANGED
|
@@ -1,29 +1,28 @@
|
|
|
1
1
|
# `@tangle-network/agent-eval`
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Run agent evaluations, compare changes on the same cases, and decide whether a candidate has enough evidence to release.
|
|
4
4
|
|
|
5
5
|
[](https://www.npmjs.com/package/@tangle-network/agent-eval)
|
|
6
6
|
[](https://pypi.org/project/agent-eval-rpc/)
|
|
7
7
|
[](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
|
|
8
8
|
[](./LICENSE)
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
New to the package? Read [concepts](./docs/concepts.md) first — it takes five minutes and defines every word used here.
|
|
14
|
-
|
|
15
|
-
Looking for a measured result (a lift, a null, a parity verdict)? The canonical registry is [`evidence/`](./evidence/README.md) — machine-readable records, a generated index, and a freshness gate.
|
|
10
|
+
Eval runs in your TypeScript process.
|
|
11
|
+
You supply agent execution, judges, and model transports.
|
|
12
|
+
It records outputs, failures, costs, and evidence for each comparison.
|
|
16
13
|
|
|
17
14
|
## Install
|
|
18
15
|
|
|
16
|
+
Use Node.js 20.19 or newer.
|
|
17
|
+
|
|
19
18
|
```sh
|
|
20
19
|
pnpm add @tangle-network/agent-eval
|
|
21
20
|
```
|
|
22
21
|
|
|
23
22
|
## Quickstart
|
|
24
23
|
|
|
25
|
-
This example
|
|
26
|
-
|
|
24
|
+
This complete example runs offline.
|
|
25
|
+
Save it as `eval.mts`.
|
|
27
26
|
|
|
28
27
|
```ts
|
|
29
28
|
import { defineAgentEval } from '@tangle-network/agent-eval/contract'
|
|
@@ -53,173 +52,147 @@ const evalKit = defineAgentEval<SupportCase, string>({
|
|
|
53
52
|
expectUsage: 'off',
|
|
54
53
|
})
|
|
55
54
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
)
|
|
61
|
-
```
|
|
62
|
-
|
|
63
|
-
Each call runs every case, records what the agent produced, applies the same judge, and returns score distributions.
|
|
55
|
+
const baseline = await evalKit.evaluate()
|
|
56
|
+
const candidate = await evalKit.evaluate({
|
|
57
|
+
surface: 'Answer politely and cite the ticket id.',
|
|
58
|
+
})
|
|
64
59
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
A **judge** is a function that scores one produced result.
|
|
60
|
+
console.log('baseline:', baseline.aggregates.byJudge['ticket-id']?.mean)
|
|
61
|
+
console.log('candidate:', candidate.aggregates.byJudge['ticket-id']?.mean)
|
|
62
|
+
```
|
|
69
63
|
|
|
70
|
-
|
|
71
|
-
The default, `'assert'`, fails a run whose cells report no cost receipt.
|
|
72
|
-
Keep the default whenever real model calls happen.
|
|
64
|
+
Run it with a TypeScript runner:
|
|
73
65
|
|
|
74
|
-
|
|
66
|
+
```sh
|
|
67
|
+
pnpm add --save-dev tsx
|
|
68
|
+
pnpm exec tsx eval.mts
|
|
69
|
+
```
|
|
75
70
|
|
|
76
|
-
|
|
71
|
+
```text
|
|
72
|
+
baseline: 0
|
|
73
|
+
candidate: 1
|
|
74
|
+
```
|
|
77
75
|
|
|
78
|
-
|
|
76
|
+
The baseline scores `0`; the candidate scores `1` on all three cases.
|
|
77
|
+
These scores describe the three examples.
|
|
78
|
+
They do not establish a release decision or performance on new tasks.
|
|
79
79
|
|
|
80
|
-
|
|
80
|
+
A **case** is one task.
|
|
81
|
+
A **surface** is the prompt, skill, or configuration being changed.
|
|
82
|
+
A **judge** scores the agent's result.
|
|
81
83
|
|
|
82
|
-
|
|
84
|
+
`expectUsage: 'off'` applies because this example makes no paid calls.
|
|
85
|
+
Set `expectUsage: 'assert'` for paid agents so missing dispatch receipts become execution failures.
|
|
86
|
+
The [runnable example](./examples/evaluate-a-change/) uses the same evaluation.
|
|
87
|
+
The [existing-agent example](./examples/foreign-agent-quickstart/) shows how to connect your agent and record model usage.
|
|
83
88
|
|
|
84
|
-
|
|
89
|
+
## Choose a workflow
|
|
85
90
|
|
|
86
|
-
|
|
|
91
|
+
| Intent | Start with | Result |
|
|
87
92
|
|---|---|---|
|
|
88
|
-
| [`defineAgentEval()`](./examples/evaluate-a-change/)
|
|
89
|
-
| [`selfImprove()`](./examples/selfimprove-quickstart/)
|
|
90
|
-
| [`
|
|
91
|
-
| `
|
|
92
|
-
|
|
|
93
|
-
| [`
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
93
|
+
| Score one change | [`defineAgentEval()`](./examples/evaluate-a-change/) from `/contract` | Cell results, failures, score distributions, and measured cost. |
|
|
94
|
+
| Search for a better surface | [`selfImprove()`](./examples/selfimprove-quickstart/) from `/contract` | A selected surface, final comparison, and `gateDecision`. |
|
|
95
|
+
| Compare search methods | [`compareOptimizationMethods()`](./examples/compare-optimization-methods/) from `/campaign` | Paired final comparisons, uncertainty, coverage, and costs under declared budgets. |
|
|
96
|
+
| Register evidence and decision rules | [`defineEvaluationClaim()` and `sealExperiment()`](./docs/evaluation-integrity.md) from `/experiment` | A declared population, independent unit, optional practical effect, and sealed rules. |
|
|
97
|
+
| Check the evaluator | [`auditEvaluator()`](./docs/evaluation-integrity.md) and [calibration tools](./docs/outcome-validity.md) from `/meta-eval` | Error rates, admission evidence, bias diagnostics, and outcome associations. |
|
|
98
|
+
| Analyze completed work | [`analyzeRuns()`](./examples/analyze-existing-runs/) from `/contract`; [trace analysts](./docs/trace-analysis.md) from `/analyst` | Comparisons and findings with links to recorded evidence. |
|
|
99
|
+
|
|
100
|
+
`defineAgentEval()` also exposes `improve()` when the same agent, cases, judge, and baseline should share configuration.
|
|
101
|
+
Use direct [campaign controls](./docs/eval-surface-map.md) for scheduling, durable caches, model matrices, or custom release rules.
|
|
102
|
+
The [example index](./examples/README.md) covers fixtures, trace intake, code verification, replay, and training-data exports.
|
|
103
|
+
|
|
104
|
+
## Make automated improvement accountable
|
|
105
|
+
|
|
106
|
+
Use reusable evaluations for development feedback.
|
|
107
|
+
For a direct edit, compare the baseline and candidate on the same cases.
|
|
108
|
+
Claims, evaluator audits, and final-evidence tracking are optional.
|
|
109
|
+
Add stronger controls when a result must support performance on new tasks or an adaptive release decision.
|
|
110
|
+
|
|
111
|
+
1. Pass a `claim` describing the population, sampling frame, and independent unit to the comparison.
|
|
112
|
+
Declare `minimumEffect` when the decision concerns a useful improvement.
|
|
113
|
+
2. When introducing an evaluator, check known good and known bad controls with `auditEvaluator()`.
|
|
114
|
+
3. Give search separate training and selection cases.
|
|
115
|
+
4. For fresh confirmation, supply `finalEvidence` with a shared ledger, request ID, and evaluator digest.
|
|
116
|
+
This reserves final units before search and records exposure before measurement.
|
|
117
|
+
5. Inspect the final comparison, gate contributions, exclusions, uncertainty, cost, and search history before releasing.
|
|
118
|
+
|
|
119
|
+
Repeated attempts on one task do not create new independent tasks.
|
|
120
|
+
The top-level `claim` controls unit aggregation for reusable comparisons.
|
|
121
|
+
Power checks assess the declared minimum effect.
|
|
122
|
+
Optional `finalEvidence` binds fresh confirmation to that claim and refuses reused final units across campaigns sharing the ledger.
|
|
123
|
+
|
|
124
|
+
The host must enforce access isolation and author/auditor separation.
|
|
125
|
+
A digest records identity; it cannot prove secrecy or that a benchmark represents future users.
|
|
126
|
+
Custom gates remain responsible for their decision rules.
|
|
127
|
+
See [evaluation integrity](./docs/evaluation-integrity.md) for the complete API and its boundaries.
|
|
128
|
+
|
|
129
|
+
These controls check the evidence behind a result.
|
|
130
|
+
They do not establish that an optimizer beats a direct edit or simple search.
|
|
131
|
+
The [historical evidence audit](./docs/design/self-improvement-evidence-audit.md) records prior gains, failed transfer, and missing comparisons.
|
|
132
|
+
|
|
133
|
+
Set `searchHistoryPolicy: 'require-complete'` when every attempted search slot must be accounted for before final evidence is exposed.
|
|
134
|
+
The [search-history receipt](./docs/search-history-receipts.md) binds the planned denominator to Eval's existing search ledger.
|
|
135
|
+
|
|
136
|
+
A `gateDecision` is `ship`, `hold`, `need_more_work`, `model_ceiling`, or `arch_ceiling`.
|
|
137
|
+
Gate contributions distinguish missing evidence from measured failures and successful checks.
|
|
138
|
+
[Concepts](./docs/concepts.md) explains these decisions and how gates compose.
|
|
139
|
+
|
|
140
|
+
## Configure model calls
|
|
141
|
+
|
|
142
|
+
Pass a `ChatClient` to model judges, analysts, and adapters.
|
|
143
|
+
Eval obtains credentials from the values you supply; it does not search your environment.
|
|
117
144
|
|
|
118
145
|
```ts
|
|
119
|
-
import { createChatClient } from '@tangle-network/agent-eval'
|
|
146
|
+
import { createChatClient } from '@tangle-network/agent-eval/contract'
|
|
120
147
|
|
|
121
148
|
const chat = createChatClient({
|
|
122
|
-
transport: '
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
149
|
+
transport: 'openai-compatible',
|
|
150
|
+
baseUrl: 'https://router.example/v1',
|
|
151
|
+
apiKey: process.env.MY_ROUTER_KEY,
|
|
152
|
+
defaultModel: process.env.EVAL_MODEL_ID,
|
|
126
153
|
})
|
|
127
154
|
```
|
|
128
155
|
|
|
129
|
-
|
|
156
|
+
Use your deployed model identifier and preserve the returned `servedModel` identity and cost receipt.
|
|
157
|
+
For an existing SDK, use `transport: 'custom'` with your `chat` callback and an explicit `maximumAttempts`.
|
|
158
|
+
Agent Runtime callers can bind `profileChatClient()` from `@tangle-network/agent-runtime/kernel`.
|
|
159
|
+
Eval has no dependency on Runtime.
|
|
130
160
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
161
|
+
Official GEPA, SkillOpt, and DSPy integrations use a Python bridge.
|
|
162
|
+
Their maintained installation instructions and execution contracts are in [campaign proposers](./docs/campaign-proposers.md).
|
|
163
|
+
The [Python client](./clients/python/README.md) and [wire protocol](./docs/wire-protocol.md) support other-language consumers.
|
|
164
|
+
|
|
165
|
+
## Public imports and evidence
|
|
166
|
+
|
|
167
|
+
Use `/contract` for a product integration, `/campaign` for execution controls, `/experiment` for registered decisions, and `/meta-eval` for evaluator checks.
|
|
168
|
+
Root `Scenario`, `JudgeScore`, and `GateDecision` are the same types as `/contract`.
|
|
169
|
+
Product judging retains the explicit root names `ProductScenario` and `DimensionJudgeScore` beside its functions.
|
|
170
|
+
`HeldOutGate.evaluate()` returns `HeldOutGateDecision`.
|
|
139
171
|
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
The official GEPA and SkillOpt optimizers run through a Python bridge.
|
|
149
|
-
Install commands, version pins, and the reason for each pin:
|
|
150
|
-
[GEPA](./docs/campaign-proposers.md#install-official-gepa),
|
|
151
|
-
[SkillOpt](./docs/campaign-proposers.md#install-official-skillopt),
|
|
152
|
-
and [DSPy](./docs/campaign-proposers.md#use-official-dspy-optimizers).
|
|
153
|
-
|
|
154
|
-
## Entry Points
|
|
155
|
-
|
|
156
|
-
| Import | Use |
|
|
157
|
-
|---|---|
|
|
158
|
-
| `@tangle-network/agent-eval/contract` | Define an evaluation, run it, improve it, and analyze existing runs. |
|
|
159
|
-
| `@tangle-network/agent-eval/campaign` | Campaigns, optimization methods, comparisons, storage, and release rules. |
|
|
160
|
-
| `@tangle-network/agent-eval/experiment` | Experiments as sealed objects: registered rules, funnels, estimands, refusals. |
|
|
161
|
-
| `@tangle-network/agent-eval/analyst` | Built-in and custom trace analysts, labeled comparison, costs, and reports. |
|
|
162
|
-
| `@tangle-network/agent-eval/trace-repair` | Grade one analyst finding by executing the repair it proposes. |
|
|
163
|
-
| `@tangle-network/agent-eval/trajectory-replay` | Re-execute a recorded shell trajectory and check whether its failure reproduces. |
|
|
164
|
-
| `@tangle-network/agent-eval/traces` | Store, replay, and inspect structured traces. |
|
|
165
|
-
| `@tangle-network/agent-eval/reporting` | Statistical comparisons and report rendering. |
|
|
166
|
-
| `@tangle-network/agent-eval/supervisor-run` | Read recursive run directories without collapsing missing measurements to zero; `agent-eval supervisor-run report <runDir>` prints one. |
|
|
167
|
-
| `@tangle-network/agent-eval/meta-eval` | Measure the grader itself: judge calibration, sentinels, and seeded known-wrong plants. |
|
|
168
|
-
| `@tangle-network/agent-eval/profile-cell` | Create and validate portable agent-profile identities. |
|
|
169
|
-
| `@tangle-network/agent-eval/ledger-core` | Append-only hash-chained journal with idempotent append and chain verification. |
|
|
170
|
-
| `@tangle-network/agent-eval/benchmarks` | Benchmark adapters and retrieval metrics. |
|
|
171
|
-
| `@tangle-network/agent-eval/rl` | Export rewards, preferences, and training rows. |
|
|
172
|
-
| `@tangle-network/agent-eval/wire` | HTTP and RPC schemas for other languages. |
|
|
173
|
-
| `@tangle-network/agent-eval/adapters/http` | Run campaign cells on remote workers over HTTP. |
|
|
174
|
-
|
|
175
|
-
Use the root import for common primitives.
|
|
176
|
-
Use a subpath when you want an explicit capability boundary.
|
|
177
|
-
|
|
178
|
-
## Documentation
|
|
179
|
-
|
|
180
|
-
| Question | Read |
|
|
181
|
-
|---|---|
|
|
182
|
-
| What do these words mean? | [`docs/concepts.md`](./docs/concepts.md) |
|
|
183
|
-
| Why does this package exist, and where is it going? | [`docs/charter.md`](./docs/charter.md) |
|
|
184
|
-
| Which `run*` function do I want? | [`docs/eval-surface-map.md`](./docs/eval-surface-map.md) |
|
|
185
|
-
| How do I choose a candidate-generation method? | [`docs/campaign-proposers.md`](./docs/campaign-proposers.md) |
|
|
186
|
-
| What is in an `InsightReport`? | [`docs/insight-report.md`](./docs/insight-report.md) |
|
|
187
|
-
| How do I register an experiment as a sealed object? | [`docs/experiment.md`](./docs/experiment.md) |
|
|
188
|
-
| How is something certified without an answer key? | [`docs/verification-strategies.md`](./docs/verification-strategies.md) |
|
|
189
|
-
| Where does every verifier land its result? | [`docs/verdicts.md`](./docs/verdicts.md) |
|
|
190
|
-
| Does the grader catch a claim that is known to be wrong? | [`docs/plants.md`](./docs/plants.md) |
|
|
191
|
-
| How do I turn a coding-agent session log into runs? | [`docs/code-agent-intake.md`](./docs/code-agent-intake.md) |
|
|
192
|
-
| How do I score a string from another language? | [`docs/wire-protocol.md`](./docs/wire-protocol.md) |
|
|
193
|
-
|
|
194
|
-
The [example index](./examples/README.md) lists every runnable example.
|
|
172
|
+
Specialist subpaths and their examples are listed in the [surface map](./docs/eval-surface-map.md).
|
|
173
|
+
Current canonical envelopes are required for seals, attestations, and profile identities.
|
|
174
|
+
Retired or incomplete formats fail verification; historical reports retain their recorded identities.
|
|
175
|
+
|
|
176
|
+
Published measurements live in the [evidence registry](./evidence/README.md).
|
|
177
|
+
The [benchmark-book review](./docs/design/mlbenchmarks-book-review.md) records the source analysis and reproduced defects behind these integrity changes.
|
|
178
|
+
[The charter](./docs/charter.md) defines package ownership and the remaining research boundaries.
|
|
195
179
|
|
|
196
180
|
## Development
|
|
197
181
|
|
|
198
182
|
```sh
|
|
199
183
|
pnpm install
|
|
184
|
+
pnpm build
|
|
200
185
|
pnpm typecheck
|
|
201
186
|
pnpm typecheck:examples
|
|
187
|
+
pnpm typecheck:scripts
|
|
188
|
+
pnpm lint
|
|
202
189
|
pnpm test
|
|
203
|
-
pnpm
|
|
190
|
+
pnpm verify:package
|
|
204
191
|
```
|
|
205
192
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
```sh
|
|
209
|
-
cd clients/python
|
|
210
|
-
uv sync --frozen --extra dev --group gepa-release
|
|
211
|
-
AGENT_EVAL_EXPECT_GEPA_RELEASE=1 \
|
|
212
|
-
uv run --frozen --extra dev --group gepa-release \
|
|
213
|
-
pytest tests/test_gepa_release_compatibility.py tests/test_gepa_bridge.py
|
|
214
|
-
|
|
215
|
-
uv sync --frozen --extra dev --group skillopt-source --group gepa-source
|
|
216
|
-
uv run --frozen pytest
|
|
217
|
-
|
|
218
|
-
uv sync --frozen --extra dev --extra dspy
|
|
219
|
-
uv run --frozen pytest tests/test_dspy_metric.py
|
|
220
|
-
```
|
|
193
|
+
Build before checking examples because they resolve the package's generated declarations.
|
|
194
|
+
The [Python development guide](./clients/python/README.md#development) gives the locked commands for each optimizer environment.
|
|
221
195
|
|
|
222
196
|
## License
|
|
223
197
|
|
|
224
198
|
MIT.
|
|
225
|
-
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-
|
|
2
|
-
import "../index-
|
|
1
|
+
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-CS0qk_Yp.js";
|
|
2
|
+
import "../index-DoykkxW0.js";
|
|
3
3
|
//#region src/adapters/http.d.ts
|
|
4
4
|
interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
|
|
5
5
|
/** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { b as CustomTokenPricing, g as CostReceiptInput } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { D as TraceAnalysisStore, c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef } from "./types-
|
|
3
|
-
import {
|
|
4
|
-
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-
|
|
2
|
+
import { D as TraceAnalysisStore, c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef } from "./types-D7gEdPoQ.js";
|
|
3
|
+
import { C as ChatResponse, S as ChatRequest } from "./types-CBbLtr2J.js";
|
|
4
|
+
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-BRbB6Y0v.js";
|
|
5
5
|
import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
|
|
6
6
|
//#region src/campaign/external-optimizer-contracts.d.ts
|
|
7
7
|
interface ExternalOptimizerProcessLimits {
|
|
@@ -485,4 +485,4 @@ declare function agentProfileId(profile: AgentProfile): string;
|
|
|
485
485
|
declare function agentProfileHash(profile: AgentProfile): string;
|
|
486
486
|
//#endregion
|
|
487
487
|
export { runAnalystBenchmark as A, ExternalOptimizerModelBudget as B, AnalystEvidenceResolutionError as C, AnalystLatencyDistribution as D, AnalystIssueExpectation as E, ExternalOptimizerCallbackLimits as F, ExternalOptimizerProcessLimits as G, ExternalOptimizerModelCallRequest as H, ExternalOptimizerChatRequest as I, ExternalOptimizerWireCounts as J, ExternalOptimizerResumeMode as K, ExternalOptimizerEndpointFormat as L, scoreAnalystFindings as M, DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS as N, RunAnalystBenchmarkOptions as O, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS as P, resolveExternalOptimizerProcessLimits as Q, ExternalOptimizerEvaluationObservation as R, AnalystEvidenceResolution as S, AnalystFindingScore as T, ExternalOptimizerModelCallResult as U, ExternalOptimizerModelCall as V, ExternalOptimizerModelExecutionObservation as W, ExternalTextEvaluationRequest as X, ExternalTextCandidate as Y, resolveExternalOptimizerCallbackLimits as Z, AnalystBenchmarkProvenance as _, ProfileAxisSpec as a, AnalystBenchmarkSummary as b, expandProfileAxes as c, AnalystBenchmarkDatasetRef as d, AnalystBenchmarkDescriptor as f, AnalystBenchmarkOutput as g, AnalystBenchmarkObservation as h, HarnessType$1 as i, traceStoreEvidenceResolver as j, registryBenchmarkRunner as k, harnessAxisOf as l, AnalystBenchmarkLabelState as m, CODING_HARNESSES as n, agentProfileHash as o, AnalystBenchmarkError as p, ExternalOptimizerRunnerCommand as q, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t, AnalystBenchmarkCase as u, AnalystBenchmarkResult as v, AnalystEvidenceResolver as w, AnalystEvidenceExpectation as x, AnalystBenchmarkRunner as y, ExternalOptimizerEvaluationRefusalReason as z };
|
|
488
|
-
//# sourceMappingURL=agent-profile-
|
|
488
|
+
//# sourceMappingURL=agent-profile-CivaSsSy.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"agent-profile-
|
|
1
|
+
{"version":3,"file":"agent-profile-CivaSsSy.d.ts","names":[],"sources":["../src/campaign/external-optimizer-contracts.ts","../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts","../src/agent-profile.ts"],"mappings":";;;;;;UAWiB;;EAEf;;EAEA;;EAEA;;cAGW,2CAA2C,SAAS;UAOhD;;EAEf;;EAEA;;cAGW,4CAA4C,SAAS;UAMjD;EACf;EACA;EACA,MAAM,OAAO;;EAEb,SAAS,QAAQ;;KAGP;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;iBAUc,sCACd,OAAO,QAAQ,6CACf,iBACC;iBAmBa,uCACd,OAAO,QAAQ,8CACf,iBACC;KAkBE,aAAa,KAAK,eAAc,6BACjC,IACA,0BAA0B,gBACf,aAAa,OACtB,+BACc,WAAW,IAAI,aAAa,EAAE,SAC1C;;KAGI,+BAA+B,aACzC,KAAK;EAA0B;;KAGrB;;UAMK;;WAEN;;WAEA,SAAS;;WAET,iBAAiB;WACjB,QAAQ;;;KAIP;WAEG;;WAEA,UAAU;;;;;;;WAOV,SAPU;;WASV;;WAGA;;WAEA;;WAEA,SATkC;;WAWlC;;;;;;;;;;;;KAaH,8BACV,SAAS,sCACN,QAAQ;;KAGD;WAEG;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;WAGA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;KAGH;;;;;;KAUA;WAEG;WACA;WACA,WAAW;WACX;;WAGA;WACA;WACA,WAAW;WACX;WACA;;WAEA;WACA;;WAGA;WACA;WACA,QAAQ;WACR,YAAY;WACZ;WACA;;UAGE;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;EAUA;;;;;;EAMA,UAAU;;EAEV;;;UAIe;;EAEf;;EAEA;;;;iBCtQc,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB;;;;;;;;;;cCpUd,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
|
-
import { a as hashCanonical
|
|
3
|
-
import { createHash } from "node:crypto";
|
|
2
|
+
import { a as hashCanonical } from "./canonical-DPyQ_rpt.js";
|
|
4
3
|
//#region src/pre-registration.ts
|
|
5
4
|
/**
|
|
6
5
|
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
@@ -12,10 +11,8 @@ import { createHash } from "node:crypto";
|
|
|
12
11
|
* evaluate the manifest against observed results — the library refuses
|
|
13
12
|
* to let you re-interpret a different metric as the declared one.
|
|
14
13
|
*
|
|
15
|
-
* A signed manifest
|
|
16
|
-
*
|
|
17
|
-
* was signed under, and verification selects the encoder by that field, so a
|
|
18
|
-
* manifest signed by an earlier release still verifies.
|
|
14
|
+
* A signed manifest carries its required digest scheme. Verification accepts
|
|
15
|
+
* only RFC 8785 canonical JSON, using the same encoder as every new identity.
|
|
19
16
|
*/
|
|
20
17
|
/**
|
|
21
18
|
* SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of
|
|
@@ -36,31 +33,13 @@ async function hashJson(obj) {
|
|
|
36
33
|
return hashCanonical(obj).slice(7);
|
|
37
34
|
}
|
|
38
35
|
/**
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
* WRITES a digest may call it.
|
|
42
|
-
*/
|
|
43
|
-
function legacyContentDigest(value) {
|
|
44
|
-
return createHash("sha256").update(JSON.stringify(sortKeysDeep$1(value)), "utf8").digest("hex");
|
|
45
|
-
}
|
|
46
|
-
function sortKeysDeep$1(value) {
|
|
47
|
-
if (value === null || typeof value !== "object") return value;
|
|
48
|
-
if (Array.isArray(value)) return value.map(sortKeysDeep$1);
|
|
49
|
-
const out = {};
|
|
50
|
-
for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep$1(value[key]);
|
|
51
|
-
return out;
|
|
52
|
-
}
|
|
53
|
-
/**
|
|
54
|
-
* Digest of a manifest under its own declared scheme, with `contentHash` and
|
|
55
|
-
* `algo` stripped. Synchronous, so a caller that must fail before consuming an
|
|
56
|
-
* observation does not have to await. Throws on an `algo` this release does
|
|
57
|
-
* not know — an unverifiable manifest must not read as a valid one.
|
|
36
|
+
* Digest a manifest after validating its scheme, excluding `contentHash` and
|
|
37
|
+
* `algo`. This synchronous check can refuse a manifest before consuming data.
|
|
58
38
|
*/
|
|
59
39
|
function manifestContentDigest(manifest) {
|
|
60
40
|
const { contentHash: _contentHash, algo, ...rest } = manifest;
|
|
61
|
-
if (algo
|
|
62
|
-
|
|
63
|
-
throw new Error(`pre-registration: unrecognized manifest hash algo '${String(algo)}'`);
|
|
41
|
+
if (algo !== "sha256-rfc8785") throw new Error(`pre-registration: unsupported manifest hash algo '${String(algo)}'`);
|
|
42
|
+
return hashCanonical(rest).slice(7);
|
|
64
43
|
}
|
|
65
44
|
/**
|
|
66
45
|
* Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical
|
|
@@ -83,14 +62,18 @@ async function signManifest(m) {
|
|
|
83
62
|
* the manifest itself declares.
|
|
84
63
|
*/
|
|
85
64
|
async function verifyManifest(m) {
|
|
86
|
-
|
|
65
|
+
try {
|
|
66
|
+
return manifestContentDigest(m) === m.contentHash;
|
|
67
|
+
} catch {
|
|
68
|
+
return false;
|
|
69
|
+
}
|
|
87
70
|
}
|
|
88
71
|
/**
|
|
89
72
|
* Evaluate a pre-registered hypothesis against observed results.
|
|
90
73
|
* Mechanical — no re-interpretation permitted.
|
|
91
74
|
*/
|
|
92
75
|
async function evaluateHypothesis(manifest, observed) {
|
|
93
|
-
if (!await verifyManifest(manifest)) throw new Error("evaluateHypothesis: manifest content hash mismatch
|
|
76
|
+
if (!await verifyManifest(manifest)) throw new Error("evaluateHypothesis: unsupported manifest hash scheme or content hash mismatch");
|
|
94
77
|
const reasons = [];
|
|
95
78
|
if (!(manifest.direction === "increase" ? observed.effect > 0 : observed.effect < 0)) reasons.push("wrong_direction");
|
|
96
79
|
if (Math.abs(observed.effect) < manifest.minEffect) reasons.push("effect_too_small");
|
|
@@ -115,14 +98,9 @@ var AgentProfileCellValidationError = class extends ValidationError {
|
|
|
115
98
|
}
|
|
116
99
|
};
|
|
117
100
|
const SHA256_HEX = /^[0-9a-f]{64}$/;
|
|
118
|
-
/**
|
|
119
|
-
|
|
120
|
-
* {@link buildAgentProfileCell} mints; the bare `sha256` form is read-only,
|
|
121
|
-
* carried by cells built under an earlier release, and still verifies.
|
|
122
|
-
*/
|
|
123
|
-
const CELL_ID = /^agent-profile-cell:sha256(?:-rfc8785)?:[0-9a-f]{64}$/;
|
|
101
|
+
/** A cell id carries the canonical digest scheme required for verification. */
|
|
102
|
+
const CELL_ID = /^agent-profile-cell:sha256-rfc8785:[0-9a-f]{64}$/;
|
|
124
103
|
const CELL_ID_PREFIX = "agent-profile-cell:sha256-rfc8785:";
|
|
125
|
-
const LEGACY_CELL_ID_PREFIX = "agent-profile-cell:sha256:";
|
|
126
104
|
async function buildAgentProfileCell(input) {
|
|
127
105
|
const material = await normalizeAgentProfileCellInput(input);
|
|
128
106
|
const cellId = `${CELL_ID_PREFIX}${await hashJson(material)}`;
|
|
@@ -137,34 +115,19 @@ function agentProfileCellHashMaterial(cell) {
|
|
|
137
115
|
}
|
|
138
116
|
/**
|
|
139
117
|
* Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material
|
|
140
|
-
* fields, confirming the record has not been tampered with.
|
|
141
|
-
*
|
|
118
|
+
* fields, confirming the record has not been tampered with. Unsupported digest
|
|
119
|
+
* schemes are refused before comparing the material.
|
|
142
120
|
*/
|
|
143
121
|
async function verifyAgentProfileCell(cell) {
|
|
144
122
|
validateAgentProfileCell(cell);
|
|
145
123
|
const material = agentProfileCellHashMaterial(cell);
|
|
146
|
-
|
|
147
|
-
return cell.cellId === `${LEGACY_CELL_ID_PREFIX}${legacyCellDigest(material)}`;
|
|
148
|
-
}
|
|
149
|
-
/**
|
|
150
|
-
* Key-sorted `JSON.stringify` digest. Private and read-only: it verifies a cell
|
|
151
|
-
* id minted before the RFC 8785 scheme, and no path that MINTS an id calls it.
|
|
152
|
-
*/
|
|
153
|
-
function legacyCellDigest(value) {
|
|
154
|
-
return createHash("sha256").update(JSON.stringify(sortKeysDeep(value)), "utf8").digest("hex");
|
|
155
|
-
}
|
|
156
|
-
function sortKeysDeep(value) {
|
|
157
|
-
if (value === null || typeof value !== "object") return value;
|
|
158
|
-
if (Array.isArray(value)) return value.map(sortKeysDeep);
|
|
159
|
-
const out = {};
|
|
160
|
-
for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep(value[key]);
|
|
161
|
-
return out;
|
|
124
|
+
return cell.cellId === `${CELL_ID_PREFIX}${await hashJson(material)}`;
|
|
162
125
|
}
|
|
163
126
|
function validateAgentProfileCell(input) {
|
|
164
127
|
if (input === null || typeof input !== "object") throw new AgentProfileCellValidationError("expected object");
|
|
165
128
|
const obj = input;
|
|
166
129
|
expectLiteral(obj.schemaVersion, "agent-profile-cell/v1", "schemaVersion");
|
|
167
|
-
if (typeof obj.cellId !== "string" || !CELL_ID.test(obj.cellId)) throw new AgentProfileCellValidationError("cellId must match agent-profile-cell:sha256:<64 lowercase hex chars>", "cellId");
|
|
130
|
+
if (typeof obj.cellId !== "string" || !CELL_ID.test(obj.cellId)) throw new AgentProfileCellValidationError("cellId must match agent-profile-cell:sha256-rfc8785:<64 lowercase hex chars>", "cellId");
|
|
168
131
|
expectString(obj.profileId, "profileId");
|
|
169
132
|
validateSource(obj.sourceProfile, "sourceProfile");
|
|
170
133
|
if (obj.harness !== void 0) validateHarness(obj.harness, "harness");
|
|
@@ -371,4 +334,4 @@ async function buildAgentInterfaceProfileCell(profile, input) {
|
|
|
371
334
|
//#endregion
|
|
372
335
|
export { verifyManifest as _, assertRunAgentProfileCell as a, groupRunsByAgentProfileCell as c, validateAgentProfileCell as d, verifyAgentProfileCell as f, signManifest as g, manifestContentDigest as h, agentProfileCellKey as i, requireAgentProfileCell as l, hashJson as m, AgentProfileCellValidationError as n, buildAgentInterfaceProfileCell as o, evaluateHypothesis as p, agentProfileCellHashMaterial as r, buildAgentProfileCell as s, AGENT_PROFILE_KINDS as t, toAgentProfileJson as u };
|
|
373
336
|
|
|
374
|
-
//# sourceMappingURL=agent-profile-cell-
|
|
337
|
+
//# sourceMappingURL=agent-profile-cell-Cv6UA-W_.js.map
|