@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
# Evaluation claims and automated improvement
|
|
2
|
+
|
|
3
|
+
Use reusable evaluations to guide development and select candidates.
|
|
4
|
+
Declare stronger evidence requirements when a result must support a broader claim.
|
|
5
|
+
`selfImprove()` returns its selected surface and measured lift even when its release gate remains inconclusive.
|
|
6
|
+
|
|
7
|
+
The package separates three questions:
|
|
8
|
+
|
|
9
|
+
| Question | Evidence | Public entry |
|
|
10
|
+
| --- | --- | --- |
|
|
11
|
+
| Did this change help on these cases? | Paired scores, failures, cost, and case coverage. | `defineAgentEval()` or `selfImprove()` from `/contract`. |
|
|
12
|
+
| Does the improvement extend to new tasks? | Representative independent tasks and an appropriate comparison; a useful effect when making an improvement decision. | Optional `claim` on campaign comparisons; registered rules from `/experiment`. |
|
|
13
|
+
| Can fresh final evidence support this adaptive decision? | A frozen comparison, retained access boundaries, and a durable exposure record. | Optional `finalEvidence` on the same comparison. |
|
|
14
|
+
|
|
15
|
+
These controls reuse the existing execution path, paired estimators, sealed experiments, and locked journal.
|
|
16
|
+
They do not add another optimizer or agent runner.
|
|
17
|
+
|
|
18
|
+
## Declare the comparison without consuming data
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
import { defineEvaluationClaim } from '@tangle-network/agent-eval/experiment'
|
|
22
|
+
|
|
23
|
+
const claim = defineEvaluationClaim({
|
|
24
|
+
use: 'comparison',
|
|
25
|
+
population: {
|
|
26
|
+
id: 'support-incidents',
|
|
27
|
+
description: 'Support incidents from the deployed product',
|
|
28
|
+
},
|
|
29
|
+
samplingFrame: 'A random sample of incidents from the declared collection window',
|
|
30
|
+
independentUnit: 'source.incidentId',
|
|
31
|
+
generalization: 'new-units',
|
|
32
|
+
minimumEffect: 0.05,
|
|
33
|
+
})
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Pass `claim` to `selfImprove()`, `runImprovementLoop()`, or `compareOptimizationMethods()`.
|
|
37
|
+
It remains independent of final-evidence storage.
|
|
38
|
+
`minimumEffect` is optional because development reports and absolute-rate measurements need not test an improvement threshold.
|
|
39
|
+
When omitted, the default `selfImprove()` gate uses a 0.05 gain threshold; method comparison uses 0.
|
|
40
|
+
Custom gates choose their own thresholds.
|
|
41
|
+
|
|
42
|
+
`independentUnit` names a field path in each scenario or evidence row.
|
|
43
|
+
Variants from one incident must carry the same source identity.
|
|
44
|
+
For new-unit claims, automatic `selfImprove()` partitions keep those variants together.
|
|
45
|
+
Fixed-roster claims retain source units in reports without requiring disjoint sources between development partitions.
|
|
46
|
+
Explicit partitions for new-unit claims must not share source units between development and final evaluation.
|
|
47
|
+
|
|
48
|
+
The default self-improvement gate averages paired cells within registered units.
|
|
49
|
+
Method comparison first averages repetitions within each scenario, then averages scenarios within each source unit.
|
|
50
|
+
It weights source units equally.
|
|
51
|
+
Results retain `scenarioScores`, `unitScores`, `units`, and `pairedCellN` so callers can inspect each denominator.
|
|
52
|
+
Custom improvement gates receive raw cell scores and scenarios.
|
|
53
|
+
Pass the claim into your gate's configuration and apply its grouping and decision rules there.
|
|
54
|
+
|
|
55
|
+
`fixed-roster` describes the specified cases.
|
|
56
|
+
Repeated executions can measure execution variability on that roster.
|
|
57
|
+
They do not establish task diversity or performance on unseen users.
|
|
58
|
+
Claim metadata records the intended scope; it does not authenticate sampling or turn an exploratory result into certification.
|
|
59
|
+
|
|
60
|
+
## Interpret small and inconclusive results
|
|
61
|
+
|
|
62
|
+
There is no universal task count that proves an improvement.
|
|
63
|
+
The effect, outcome type, dependence, confidence level, and decision procedure determine what the evidence supports.
|
|
64
|
+
|
|
65
|
+
Paired binary decisions use the shared score interval.
|
|
66
|
+
At nonnegative gain thresholds, the exact discordance check can also veto promotion.
|
|
67
|
+
Negative thresholds ask whether a regression stays within a tolerance and do not use that veto.
|
|
68
|
+
A sufficiently large binary gain can pass with fewer than 20 independent pairs.
|
|
69
|
+
Continuous mean decisions require the existing bootstrap path's 20-pair eligibility threshold.
|
|
70
|
+
That implementation threshold does not establish adequate power or guarantee interval coverage for every distribution.
|
|
71
|
+
The smaller-sample sign test answers a different question about directional or median change.
|
|
72
|
+
|
|
73
|
+
Method rankings describe observed lift.
|
|
74
|
+
`favored: null` means the evidence does not establish a favored method; it does not establish equivalence.
|
|
75
|
+
Each score and pairwise contrast retains its full `decision`, including the estimator, threshold, minimum, and sufficiency.
|
|
76
|
+
An inconclusive gate leaves the selected candidate available for further development or a narrower evaluation.
|
|
77
|
+
|
|
78
|
+
For a design that matches its outcome model, `clusteredPower()` simulates power at the declared `minimumEffect`.
|
|
79
|
+
A registered `power-floor` gate checks a supplied power curve against its minimum effect and target power.
|
|
80
|
+
These optional design checks do not run automatically when you pass a `claim`.
|
|
81
|
+
High power at a much larger effect cannot substitute for power at the improvement that matters.
|
|
82
|
+
|
|
83
|
+
## Opt into fresh final evidence
|
|
84
|
+
|
|
85
|
+
```ts
|
|
86
|
+
import { openFinalEvidenceLedger } from '@tangle-network/agent-eval/experiment'
|
|
87
|
+
|
|
88
|
+
const finalEvidence = {
|
|
89
|
+
ledger: openFinalEvidenceLedger({ path: '.agent-eval/final-evidence.jsonl' }),
|
|
90
|
+
requestId: 'support-comparison-2026-09-13',
|
|
91
|
+
evaluatorDigest, // Content identity of the actual evaluator and its configuration.
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// Pass both claim and finalEvidence to the existing comparison entrypoint.
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Ordinary regression and development evaluations remain reusable.
|
|
98
|
+
Supply this policy when freshness is part of the evidence supporting a particular final comparison.
|
|
99
|
+
It requires a comparison or certification claim and measured final execution.
|
|
100
|
+
|
|
101
|
+
Comparison entrypoints capture judge configuration and callbacks before search or final exposure.
|
|
102
|
+
The host must keep callback receiver and closed-over state stable throughout the comparison.
|
|
103
|
+
|
|
104
|
+
The campaign reserves source units before candidate search.
|
|
105
|
+
It records exposure before dispatching the baseline, selected candidates, or an optional neutralized control.
|
|
106
|
+
Failed or interrupted final execution still consumes that evidence.
|
|
107
|
+
An already exposed request cannot start another final measurement.
|
|
108
|
+
|
|
109
|
+
The ledger permits exact retries of reservation and exposure writes.
|
|
110
|
+
Campaign entrypoints refuse a replayed exposure so competing workers cannot each start a new measurement.
|
|
111
|
+
Read retained campaign artifacts after exposure; a fresh request ID does not restore freshness.
|
|
112
|
+
|
|
113
|
+
Source unit identities are unique across the shared ledger, including across population labels.
|
|
114
|
+
The ledger also rejects the same dataset digest under another request.
|
|
115
|
+
Use one persistent ledger for related decisions and preserve stable source identities.
|
|
116
|
+
Opening another empty ledger or inventing new lineage identities cannot establish independent evidence.
|
|
117
|
+
|
|
118
|
+
`FinalEvidenceLedger` returns typed outcomes.
|
|
119
|
+
Inspect `succeeded` before reading `value`.
|
|
120
|
+
Failures distinguish conflicting use, invalid input, and unavailable or damaged storage.
|
|
121
|
+
Campaign errors preserve these categories through `FinalEvidenceError.kind`.
|
|
122
|
+
|
|
123
|
+
The filesystem implementation uses the existing hash-chained journal, process locks, durable writes, and required trusted head.
|
|
124
|
+
Preserve both the journal and its `.head` file.
|
|
125
|
+
The head detects truncation while it remains trusted.
|
|
126
|
+
An actor who can replace both files can replace the recorded history.
|
|
127
|
+
|
|
128
|
+
The host owns answer-file permissions, model context, credentials, and author/evaluator separation.
|
|
129
|
+
The ledger records exposure; it cannot prove that earlier undisclosed access never occurred.
|
|
130
|
+
|
|
131
|
+
## Seal the rule and measured field
|
|
132
|
+
|
|
133
|
+
Attach the same `claim` to `defineExperiment()` before calling `sealExperiment()`.
|
|
134
|
+
Cluster intervals register both the source unit and measured field:
|
|
135
|
+
|
|
136
|
+
```ts
|
|
137
|
+
const interval = {
|
|
138
|
+
kind: 'cluster-bootstrap' as const,
|
|
139
|
+
clusterBy: 'source.incidentId',
|
|
140
|
+
value: 'pairedDelta',
|
|
141
|
+
resamples: 2000,
|
|
142
|
+
seed: 7,
|
|
143
|
+
level: 0.95,
|
|
144
|
+
method: 'percentile' as const,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// Register interval in the experiment's intervals map before sealing.
|
|
148
|
+
// Then execute registered.interval('lift', { kind: 'rows', rows }).
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
For new-unit claims, each cluster interval's `clusterBy` must equal `claim.independentUnit`.
|
|
152
|
+
Registered binomial intervals require one unique `unitId` per trial for new-unit claims.
|
|
153
|
+
Opened seals capture validated rules before asynchronous execution.
|
|
154
|
+
Caller mutation cannot change the opened experiment's rules.
|
|
155
|
+
|
|
156
|
+
Only current canonical digest schemes can execute.
|
|
157
|
+
Retain historical artifacts with their original identities; re-register current work under the supported format.
|
|
158
|
+
See [registered experiments](./experiment.md) for the complete rule language.
|
|
159
|
+
|
|
160
|
+
## Audit an evaluator's errors
|
|
161
|
+
|
|
162
|
+
Use `auditEvaluator()` from `/meta-eval` when admitting a new checker or model judge.
|
|
163
|
+
Provide actual judgments of independently verified good and bad controls.
|
|
164
|
+
Each observation names its source unit, evidence reference, expected decision, observed decision, and development exposure.
|
|
165
|
+
This audit is separate from `selfImprove()`; the caller decides whether to require admission before search or release.
|
|
166
|
+
|
|
167
|
+
The audit measures false acceptance and false rejection separately.
|
|
168
|
+
A source unit fails a class when any variant in that class is misjudged.
|
|
169
|
+
Repeated variants increase case coverage without increasing the independent-unit count.
|
|
170
|
+
If a source appeared during evaluator development, every supplied variant from that source is excluded from fresh audit evidence.
|
|
171
|
+
|
|
172
|
+
The audit uses exact binomial bounds and adjusts the two intervals for simultaneous confidence.
|
|
173
|
+
Unknown judgments remain visible and contribute their most adverse possible outcomes to each upper bound.
|
|
174
|
+
Admission is possible when both worst-case upper bounds meet policy.
|
|
175
|
+
If unresolved outcomes affect the measured error rate, that rate is `null`.
|
|
176
|
+
An always-accept checker fails false acceptance; an always-reject checker fails false rejection.
|
|
177
|
+
|
|
178
|
+
Reports retain inputs, source coverage, exclusions, unknowns, limits, and content digests.
|
|
179
|
+
The declared audit authority must differ from the evaluator author.
|
|
180
|
+
Different identifiers alone do not prove independence; the host must enforce and record separation.
|
|
181
|
+
Audit cases must represent the stated population and a consistent control-generation procedure.
|
|
182
|
+
Changing the number or kind of variants changes the meaning of an any-variant error rate.
|
|
183
|
+
|
|
184
|
+
`auditEvaluator()` measures supplied judgments.
|
|
185
|
+
It does not execute models or automatically approve a deployment.
|
|
186
|
+
For outcome associations and direct score calibration, use [the outcome-validity tools](./outcome-validity.md).
|
|
187
|
+
|
|
188
|
+
## Test whether self-improvement is useful
|
|
189
|
+
|
|
190
|
+
These integrity checks establish execution and measurement behavior.
|
|
191
|
+
They do not establish that a particular optimizer improves agents across domains.
|
|
192
|
+
The [historical evidence audit](./design/self-improvement-evidence-audit.md) records prior gains, nulls, regressions, and their limits.
|
|
193
|
+
|
|
194
|
+
For a benefit experiment, define the user behavior and useful effect before search.
|
|
195
|
+
Compare the starting agent, a direct edit or simple search baseline, and the proposed improvement method at equal actual resources.
|
|
196
|
+
Give every method the same allowed preparation, feedback, tools, and candidate surface.
|
|
197
|
+
Retain all attempts, costs, failures, selected candidates, and final comparisons.
|
|
198
|
+
|
|
199
|
+
Before interpreting a null result, verify that candidate generation executed and the evaluator distinguishes plausible improvements from regressions.
|
|
200
|
+
Measure improvement under the conditions where the method claims an advantage.
|
|
201
|
+
Use fresh tasks when the conclusion concerns unseen tasks.
|
|
202
|
+
Treat a result on a fixed product workflow as evidence for that workflow.
|
|
203
|
+
Repeat across distinct domains before making a broad claim.
|
|
204
|
+
|
|
205
|
+
The [offline example](../examples/evaluation-integrity/) exercises these public APIs and exports its report without paid calls.
|
|
206
|
+
It verifies the integration with deterministic fixtures; it is not an optimizer-benefit study.
|
|
207
|
+
|
|
208
|
+
## Source and design rationale
|
|
209
|
+
|
|
210
|
+
The book motivates the distinctions; the API and policies are project design choices.
|
|
211
|
+
|
|
212
|
+
| Source | Applied idea |
|
|
213
|
+
| --- | --- |
|
|
214
|
+
| [Chapter 4: purposes of holdout](https://mlbenchmarks.org/04-holdout-method.html#whats-the-holdout-method-for) | Development feedback, selection, and capability measurement require different evidence. |
|
|
215
|
+
| [Chapter 3: detecting differences](https://mlbenchmarks.org/03-detecting-differences.html#comparing-similar-models) | Pair comparisons and evaluate precision against the actual effect and independent observations. |
|
|
216
|
+
| [Chapter 5: test-set reuse](https://mlbenchmarks.org/05-test-set-reuse.html) | Preserve development feedback while tracking adaptive final-data exposure. |
|
|
217
|
+
| [Chapter 11: confounded evaluations](https://mlbenchmarks.org/11-evaluating-language-models.html#confounded-evaluations) | Give methods comparable preparation before judging their adaptation potential. |
|
|
218
|
+
| [Chapter 14: judge agreement](https://mlbenchmarks.org/14-evaluation-frontier.html#agreement-alone-is-not-enough) | Measure consequential evaluator errors; agreement alone does not establish correct rankings. |
|
|
219
|
+
|
|
220
|
+
The [complete review](./design/mlbenchmarks-book-review.md) records all available chapters, repository evidence, and remaining research questions.
|
package/docs/experiment.md
CHANGED
|
@@ -1,31 +1,31 @@
|
|
|
1
1
|
# The experiment subpath
|
|
2
2
|
|
|
3
|
-
`@tangle-network/agent-eval/experiment`
|
|
3
|
+
`@tangle-network/agent-eval/experiment` registers decision rules as data and provides interpreters bound to a verified seal.
|
|
4
4
|
|
|
5
|
-
The
|
|
5
|
+
The [`evidence/` registry](../evidence/README.md) stores published measurements and their experiment identities.
|
|
6
|
+
Sealing does not execute an agent or publish an evidence record.
|
|
6
7
|
|
|
7
|
-
##
|
|
8
|
+
## What a seal enforces
|
|
8
9
|
|
|
9
|
-
1.
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
The digest history is the audit trail; changing what is decided without a new digest is impossible.
|
|
10
|
+
1. `sealExperiment()` validates and hashes the specification, including its registered rules and optional claim.
|
|
11
|
+
2. `openSealedExperiment()` verifies that digest and captures the rules before returning the execution handle.
|
|
12
|
+
Its decision, interval, gate, and budget methods read their rules from that captured specification.
|
|
13
|
+
3. `amendExperiment()` verifies the current seal, validates the replacement specification, and records its digest, reason, time, and declared blindness.
|
|
14
|
+
|
|
15
|
+
A seal verifies the current specification's identity.
|
|
16
|
+
It does not authenticate registration time, amendment history, or the origin of supplied measurements.
|
|
17
|
+
The host must retain evidence, invoke the required checks, and honor their refusal results.
|
|
18
|
+
Calling `registered.decide()` does not automatically run admission, power, budget, or halt checks.
|
|
19
19
|
|
|
20
20
|
## The objects
|
|
21
21
|
|
|
22
|
-
|
|
|
22
|
+
| Object | Entry point | Purpose |
|
|
23
23
|
| --- | --- | --- |
|
|
24
|
-
| Registered
|
|
25
|
-
| Define / seal / execute | `defineExperiment`, `sealExperiment`, `amendExperiment`, `openSealedExperiment` |
|
|
26
|
-
| Cluster-aware power | `clusteredPower`, `assertDesignAdequate` |
|
|
27
|
-
| Denominator chain | `buildFunnel`, `executeAdmissionRule`, `composeFunnels`, `renderFunnelTable` |
|
|
28
|
-
| Matched budgets | `verifyMatchedBudgets`, `assertMatchedBudgets` |
|
|
24
|
+
| Registered rules | [Rule types](../src/experiment/ast.ts) and [`ExperimentSpec`](../src/experiment/define.ts) | Describe admission, estimation, intervals, and decisions as data. |
|
|
25
|
+
| Define / seal / execute | `defineExperiment`, `sealExperiment`, `amendExperiment`, `openSealedExperiment` | Validate, identify, and execute the registered rules. |
|
|
26
|
+
| Cluster-aware power | `clusteredPower`, `assertDesignAdequate` | Assess a declared effect under a simulated outcome model and cluster-count policy. |
|
|
27
|
+
| Denominator chain | `buildFunnel`, `executeAdmissionRule`, `composeFunnels`, `renderFunnelTable` | Reconcile retained and excluded evidence. |
|
|
28
|
+
| Matched budgets | `verifyMatchedBudgets`, `assertMatchedBudgets` | Check realized tokens against a declared tolerance. |
|
|
29
29
|
|
|
30
30
|
### Sealing and execution
|
|
31
31
|
|
|
@@ -35,67 +35,123 @@ import {
|
|
|
35
35
|
sealExperiment,
|
|
36
36
|
} from '@tangle-network/agent-eval/experiment'
|
|
37
37
|
|
|
38
|
-
const sealed = await sealExperiment(spec)
|
|
39
|
-
const registered = await openSealedExperiment(sealed)
|
|
38
|
+
const sealed = await sealExperiment(spec)
|
|
39
|
+
const registered = await openSealedExperiment(sealed)
|
|
40
40
|
|
|
41
|
-
const admission = registered.admit(rows)
|
|
41
|
+
const admission = registered.admit(rows)
|
|
42
42
|
const gate = registered.gate('power-floor', { kind: 'power-floor', curve })
|
|
43
|
-
const halt = registered.halt([gate])
|
|
44
|
-
|
|
43
|
+
const halt = registered.halt([gate])
|
|
44
|
+
if (halt.fired) throw new Error(`Experiment halted: ${halt.failedGates.join(', ')}`)
|
|
45
|
+
// Compute quantities from the admitted evidence, then call registered.decide(quantities).
|
|
45
46
|
```
|
|
46
47
|
|
|
47
|
-
|
|
48
|
+
This fragment assumes `spec` registers admission, the named gate, and a halt rule.
|
|
49
|
+
Malformed specifications and unusable evidence throw typed errors; decision and validity refusals remain in returned artifacts.
|
|
48
50
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
51
|
+
Cluster intervals register both `clusterBy` and `value` inside the sealed `IntervalSpec`.
|
|
52
|
+
Call `registered.interval('gain95', { kind: 'rows', rows })` to apply those fields.
|
|
53
|
+
The row evidence cannot override the registered value field.
|
|
54
|
+
Changing the measured field requires a new seal.
|
|
53
55
|
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
56
|
+
For a paired contrast, prepare one difference per pair and register that difference field as `value`.
|
|
57
|
+
A pooled pass rate from both arms measures a different quantity.
|
|
58
|
+
The [runnable sealed experiment](../examples/sealed-experiment/index.ts) demonstrates the paired path.
|
|
59
|
+
Confidence levels, field paths, seeds, and resample counts are validated before sealing and direct computation.
|
|
60
|
+
|
|
61
|
+
Older cluster interval registrations omitted `value` and require their original package version for execution.
|
|
62
|
+
Retain their original bytes and evidence; create a new registration for subsequent measurements.
|
|
58
63
|
|
|
59
|
-
|
|
60
|
-
is unreachable from any path that writes a digest. Retire it once no record
|
|
61
|
-
carrying that tag needs to verify; until then, deleting it would make those
|
|
62
|
-
records unverifiable rather than invalid.
|
|
64
|
+
#### Canonical identities and migration
|
|
63
65
|
|
|
64
|
-
|
|
65
|
-
|
|
66
|
+
Readers and writers use RFC 8785 canonical JSON from `ledger-core/canonical`.
|
|
67
|
+
Verification refuses missing or unsupported digest schemes.
|
|
68
|
+
|
|
69
|
+
| Artifact | Required identity | Refusal |
|
|
70
|
+
|---|---|---|
|
|
71
|
+
| Sealed experiment | `algo: 'sha256-rfc8785'` | `verifySealedExperiment()` returns `false`; `openSealedExperiment()` refuses execution |
|
|
72
|
+
| Signed hypothesis | `algo: 'sha256-rfc8785'` | `verifyManifest()` returns `false`; synchronous digest checks and hypothesis evaluation refuse the record |
|
|
73
|
+
| Agent profile cell | `agent-profile-cell:sha256-rfc8785:<digest>` | Cell validation refuses any other scheme |
|
|
74
|
+
| Report attestation | Report hash and required `envelopeHash` over its provenance | `verifyAttestation()` returns an invalid result with a reason |
|
|
75
|
+
|
|
76
|
+
The package no longer verifies `sha256-content` records, untagged manifests, bare `agent-profile-cell:sha256:` identifiers, or attestations without provenance envelopes.
|
|
77
|
+
Keep those records unchanged as historical artifacts with their original package version.
|
|
78
|
+
Create new registrations with `sealExperiment()` or `signManifest()` before collecting new decision evidence.
|
|
79
|
+
Use `buildAgentProfileCell()` and `attest()` to produce current identities from independently verified source material.
|
|
80
|
+
Never relabel an existing digest or reconstruct a provenance envelope from unverified metadata.
|
|
81
|
+
A new digest cannot establish that a registration existed before its evidence was observed.
|
|
82
|
+
|
|
83
|
+
Use the opened handle when the result must follow a particular registration.
|
|
84
|
+
Direct helpers such as `computeInterval()` and `executeDecisionRule()` also accept unsealed rules for development.
|
|
85
|
+
They do not establish a link to a registered experiment.
|
|
66
86
|
|
|
67
87
|
### Cluster-aware power refusal
|
|
68
88
|
|
|
69
|
-
|
|
89
|
+
`clusteredPower()` combines a cluster-count policy with a simulated power curve:
|
|
90
|
+
|
|
91
|
+
- The exact whole-cluster sign-flip test has a minimum two-sided p-value of `2^(1-C)` for `C` independent clusters.
|
|
92
|
+
At alpha 0.05, this policy requires at least six clusters; four give 0.125 and three give 0.25.
|
|
93
|
+
- Seeded simulations draw paired contrasts under the configured win/loss model and apply a whole-cluster percentile bootstrap.
|
|
94
|
+
Power is the fraction of simulated intervals that exclude zero.
|
|
70
95
|
|
|
71
|
-
-
|
|
72
|
-
|
|
73
|
-
Six clusters is the smallest certifiable count at 0.05.
|
|
74
|
-
- **Seeded simulation.** Per-row paired contrasts are drawn under a registered effect model (base win/loss rates, optional noisy clusters), each trial takes a whole-cluster percentile bootstrap, and power is the fraction of trials whose interval excludes zero.
|
|
96
|
+
The six-cluster floor is a policy for this helper, not a universal requirement for every estimator or fixed-roster evaluation.
|
|
97
|
+
The helper computes both results; it does not skip simulation when the cluster-count policy fails.
|
|
75
98
|
|
|
76
99
|
The refusal is a verdict inside the returned artifact (`result.refusal`), with `assertDesignAdequate` as the throwing form.
|
|
77
|
-
|
|
100
|
+
Both `clusteredPower` and the registered `power-floor` gate require `minimumEffect`.
|
|
101
|
+
The gate evaluates a supplied curve; it does not run the simulation itself.
|
|
102
|
+
The effect must appear exactly in the supplied grid; the API does not interpolate.
|
|
103
|
+
Adequacy requires target power at that effect.
|
|
104
|
+
`maxPower` describes the grid and cannot establish adequacy at a smaller effect.
|
|
105
|
+
|
|
106
|
+
```ts
|
|
107
|
+
import { assertDesignAdequate, clusteredPower } from '@tangle-network/agent-eval/experiment'
|
|
108
|
+
|
|
109
|
+
const power = clusteredPower({
|
|
110
|
+
clusterSizes: Array.from({ length: 24 }, () => 3),
|
|
111
|
+
effects: [0.05, 0.1, 0.2],
|
|
112
|
+
minimumEffect: 0.1,
|
|
113
|
+
targetPower: 0.8,
|
|
114
|
+
seed: 17,
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
assertDesignAdequate(power)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Simulation effects are expected paired contrasts within signal clusters.
|
|
121
|
+
The zero-effect model requires equal `baseWinRate` and `baseLossRate`.
|
|
122
|
+
Configured noisy clusters retain zero expected contrast, so they dilute the pooled population effect.
|
|
123
|
+
Power remains conditional on this outcome model, the registered sampling structure, and the simulated test.
|
|
124
|
+
The [statistical evidence guide](./statistical-evidence.md) explains unit counts, adaptation comparisons, and sequential assumptions.
|
|
78
125
|
|
|
79
126
|
### The funnel
|
|
80
127
|
|
|
81
128
|
`buildFunnel` refuses a stage that gains rows, named exclusions that do not sum, and partitions that overdraw their source stage.
|
|
82
|
-
`
|
|
83
|
-
|
|
129
|
+
`registered.admit(rows)` applies the sealed admission rule and returns the funnel, survivors, and partition rows together.
|
|
130
|
+
The standalone `executeAdmissionRule(rule, rows)` also accepts an unsealed rule.
|
|
131
|
+
Partitions carry `pooling: 'never'`: report each secondary set separately from the primary chain.
|
|
84
132
|
The object is its own JSON render; `renderFunnelTable` prints the text table with the reconciliation line (`input = surviving + excluded`).
|
|
85
133
|
|
|
86
134
|
### Matched budgets
|
|
87
135
|
|
|
88
136
|
`verifyMatchedBudgets` compares realized per-arm tokens under the registered tolerance and returns a verdict whose `refusal` field carries `onFail: 'refuse-contrast'` when arms diverge.
|
|
89
|
-
|
|
137
|
+
Use this check when the claim requires matched token use.
|
|
138
|
+
An unequal-budget comparison answers a different question and must retain the resource difference in its interpretation.
|
|
90
139
|
|
|
91
140
|
## Acceptance: the three preregistrations
|
|
92
141
|
|
|
93
|
-
The
|
|
142
|
+
The [acceptance suite](../tests/experiment/preregistration-acceptance.test.ts) encodes three historical preregistrations as sealed specifications.
|
|
143
|
+
It checks their recorded decisions against fixed evidence:
|
|
94
144
|
|
|
95
|
-
- **killtest-20260810
|
|
145
|
+
- **killtest-20260810**: all four validity gates fail on the recorded evidence.
|
|
146
|
+
The failures are the rep-4 oracle flip, 2-row population drift, zero-call control, and 0.692 power ceiling.
|
|
147
|
+
The halt rule refuses spend, matching the recorded `$0.00, contrast never run`.
|
|
96
148
|
The obligation node routes a positive interval without the registered control to `blocked-pending-registered-control`, never to `thesis-survives`.
|
|
97
|
-
- **freelunch-20260810
|
|
98
|
-
|
|
149
|
+
- **freelunch-20260810**: the admission funnel reproduces `48 > 43 > 35 > 35 > 32` with the 3-row secondary partition.
|
|
150
|
+
The uniform-pass budget reproduces uniform n=2; the amendment-6 ledger under the same sealed rule refuses pass 2.
|
|
151
|
+
The report-only decision reproduces `3/64` and `2/32`.
|
|
152
|
+
- **tbench-20260808 milestone 2**: round-robin selection reproduces the recorded 20-row subset in pick order.
|
|
153
|
+
The m3 subset filters the sealed m2 draw to 16 rows.
|
|
154
|
+
The decision table on the recorded interval reproduces `not-certified-at-this-n`.
|
|
99
155
|
|
|
100
156
|
## What is composed, not duplicated
|
|
101
157
|
|
|
@@ -103,7 +159,7 @@ The statistical machinery underneath is re-exported from its existing homes; thi
|
|
|
103
159
|
|
|
104
160
|
| family | home |
|
|
105
161
|
| --- | --- |
|
|
106
|
-
| `pairedBootstrap`, `mcnemar`/`mcnemarPower`/`mcnemarRequiredN`, `pairedRiskDifference*`, `holm`, `benjaminiHochberg`, `eProcess`, `wilson`, `mulberry32`, sample-size helpers | `src/statistics.ts` |
|
|
162
|
+
| `pairedBootstrap`, `mcnemar`/`mcnemarPower`/`mcnemarRequiredN`, `pairedRiskDifference*`, `holm`, `benjaminiHochberg`, `eProcess`, `wilson`, `mulberry32`, sample-size helpers | [`src/statistics/index.ts`](../src/statistics/index.ts) |
|
|
107
163
|
| `pairedEvalueSequence` (anytime-valid) | `src/sequential.ts` |
|
|
108
164
|
| `powerPreflight` (variance-based MDE refusal) | `src/campaign/gates/power-preflight.ts` |
|
|
109
165
|
| `sequentialPairedGate`, `sequentialDecide` (manifest-bound) | `src/campaign/gates/sequential.ts` |
|
|
@@ -118,5 +174,5 @@ The trace-repair admission machinery (`buildDenominatorChain`, oracle determinis
|
|
|
118
174
|
|
|
119
175
|
## Where this sits
|
|
120
176
|
|
|
121
|
-
|
|
122
|
-
|
|
177
|
+
The [charter](./charter.md) describes current package ownership and host responsibilities.
|
|
178
|
+
Use [evaluation claims and final evidence](./evaluation-integrity.md) when connecting a registration to an automated improvement workflow.
|
package/docs/feature-guide.md
CHANGED
|
@@ -55,10 +55,10 @@ user intent
|
|
|
55
55
|
-> datasets and optimizers replay the same adapter
|
|
56
56
|
```
|
|
57
57
|
|
|
58
|
-
|
|
59
|
-
adapter can
|
|
60
|
-
|
|
61
|
-
|
|
58
|
+
Keep the production state, validators, actions, budgets, and stop policies in the evaluation path.
|
|
59
|
+
The adapter can supply a real user session, replay fixture, or sandbox.
|
|
60
|
+
This tests the behavior that production executes.
|
|
61
|
+
Transfer to future tasks still requires representative evaluation data and a measured comparison.
|
|
62
62
|
|
|
63
63
|
### Agent Runtime Integration
|
|
64
64
|
|
|
@@ -82,8 +82,7 @@ Implementation ownership:
|
|
|
82
82
|
assignment, and optimizer row conversion in `agent-eval`.
|
|
83
83
|
- Put product state readers, action executors, approval policy, credentials,
|
|
84
84
|
workspace paths, and UI-specific storage in the downstream repo.
|
|
85
|
-
-
|
|
86
|
-
need the same adapter shape.
|
|
85
|
+
- Keep product execution adapters in the consuming repository.
|
|
87
86
|
|
|
88
87
|
### Code Generator
|
|
89
88
|
|
|
@@ -178,17 +178,10 @@ It is not production storage because process restart clears its in-memory data.
|
|
|
178
178
|
TENANT_KEY=dev-token TENANT_ID=acme pnpm tsx examples/hosted-ingest-server/server.ts
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
HOSTED_TENANT_KEY=dev-token \
|
|
186
|
-
HOSTED_TENANT_ID=acme \
|
|
187
|
-
pnpm tsx examples/foreign-agent-quickstart/index.ts
|
|
188
|
-
```
|
|
189
|
-
|
|
190
|
-
The quickstart's eval-run gets POSTed to the reference receiver; the
|
|
191
|
-
receiver's `GET /v1/runs` lists it back.
|
|
181
|
+
Send events with [`createHostedClient`](../src/hosted/client.ts) from `@tangle-network/agent-eval/hosted`.
|
|
182
|
+
Call `client.ingestEvalRun(event)` with a valid `EvalRunEvent` after configuring the client's endpoint, tenant ID, and API key.
|
|
183
|
+
For `selfImprove`, configure `hostedTenant` as shown in the [receiver example](../examples/hosted-ingest-server/).
|
|
184
|
+
The receiver's authenticated `GET /v1/runs` endpoint lists ingested runs.
|
|
192
185
|
|
|
193
186
|
---
|
|
194
187
|
|